@hamza1331/aieval 0.0.0-stage → 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md ADDED
@@ -0,0 +1,28 @@
1
+ # Changelog
2
+
3
+ All notable changes to this project are documented here. The format is based on
4
+ [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and this project adheres to
5
+ [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
6
+
7
+ ## [Unreleased]
8
+
9
+ ## [0.1.0]
10
+
11
+ First release.
12
+
13
+ ### Added
14
+
15
+ - Core API: `evaluate`, `validate`, `assert`, and the `Check`, `Failure`, `Severity` and `EvaluationResult` model.
16
+ - Severity-based routing with a `failOn` threshold (default `"error"`); lower severities are reported as warnings.
17
+ - Checks: `schemaCheck` (Zod), `requiredFields`, `noExtraFields`, `enumCheck`, `fieldTypeCheck`, `jsonCheck`,
18
+ `nonEmpty`, `regexCheck`, `rangeCheck`, `toolCallCheck`, `toolCallCheck.oneOf`, `customCheck`.
19
+ - Tool-call validation for `{ name, args }` and OpenAI-style `{ name, arguments: "<json>" }` calls.
20
+ - Optional semantic judging: `semanticCheck`, the `SemanticJudge` interface and `mockJudge`.
21
+ - JSON check registry (`buildChecks`) and fixture runner (`loadFixtures`, `runFixture`, `runAll`).
22
+ - CLI: `aieval run` and `aieval check`, with `--fail-on`, `--format text|json` and stable exit codes.
23
+ - Per-check metadata in `result.metadata.checks` and `durationMs`.
24
+ - `EvaluationError` thrown by `assert`, carrying the full result.
25
+ - Checks that throw are converted to `critical` failures instead of crashing the run.
26
+
27
+ [Unreleased]: https://github.com/hamza1331/aieval/compare/v0.1.0...HEAD
28
+ [0.1.0]: https://github.com/hamza1331/aieval/releases/tag/v0.1.0
package/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
package/README.md CHANGED
@@ -1,3 +1,226 @@
1
- # Temporary Holding Version
1
+ # aieval
2
2
 
3
- This version is a temporary placeholder for this package. An operational version to replace this has been submitted for review and is awaiting a staged release.
3
+ [![CI](https://github.com/hamza1331/aieval/actions/workflows/ci.yml/badge.svg)](https://github.com/hamza1331/aieval/actions/workflows/ci.yml)
4
+ [![npm version](https://img.shields.io/npm/v/@hamza1331/aieval.svg)](https://www.npmjs.com/package/@hamza1331/aieval)
5
+ [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](https://opensource.org/licenses/MIT)
6
+
7
+ TypeScript-first, deterministic-first validation for LLM outputs and agent tool calls.
8
+
9
+ Catch the failures you can detect with code — malformed JSON, missing fields, enum drift, bad tool-call arguments,
10
+ policy violations — **before** you spend time and money on LLM-as-a-judge scoring. Failures are structured and
11
+ machine-readable, so they are easy to log, assert on, or feed back to a model for a retry.
12
+
13
+ - Small and composable: plain functions, no framework, no hosted service
14
+ - Strong typing and no runtime dependencies (`zod` is a peer dependency, used by `schemaCheck`)
15
+ - A fixture-driven CLI for local checks and CI
16
+ - Optional, clearly-labelled semantic (LLM) judging behind a provider-agnostic interface
17
+
18
+ ## Install
19
+
20
+ ```bash
21
+ npm install @hamza1331/aieval zod
22
+ ```
23
+
24
+ Requires Node.js 20+. `zod` (v4) is a peer dependency used by `schemaCheck`.
25
+
26
+ ## Quick start
27
+
28
+ ```ts
29
+ import { z } from "zod";
30
+ import { evaluate, requiredFields, schemaCheck } from "@hamza1331/aieval";
31
+
32
+ const schema = z.object({ name: z.string(), email: z.string() });
33
+
34
+ const result = await evaluate({
35
+ output: { name: "Ada Lovelace", email: "ada@example.com" },
36
+ checks: [schemaCheck(schema), requiredFields(["name", "email"])],
37
+ });
38
+
39
+ console.log(result.passed); // true
40
+ console.log(result.failures); // []
41
+ ```
42
+
43
+ Other entry points:
44
+
45
+ ```ts
46
+ import { assert, validate } from "@hamza1331/aieval";
47
+
48
+ const result = await validate(output, checks); // same as evaluate({ output, checks })
49
+ await assert(output, checks); // throws EvaluationError (with `.result`) if validation fails
50
+ ```
51
+
52
+ ## Results and severity
53
+
54
+ Every issue is a `Failure`:
55
+
56
+ ```ts
57
+ interface Failure {
58
+ code: FailureCode; // "missing_field" | "invalid_type" | "invalid_enum" | "schema_violation" | "invalid_json" | ...
59
+ message: string;
60
+ severity: "info" | "warning" | "error" | "critical";
61
+ path?: string; // e.g. "user.age" or "args.city"
62
+ metadata?: Record<string, unknown>; // always includes `checkName`
63
+ suggestedRepair?: string;
64
+ }
65
+ ```
66
+
67
+ **Severity decides what fails.** Issues at or above `failOn` (default `"error"`) go to `result.failures` and make
68
+ `result.passed` false. Lower-severity issues go to `result.warnings` and do not fail the run.
69
+
70
+ ```ts
71
+ await evaluate({ output, checks }); // warnings are reported, run still passes
72
+ await evaluate({ output, checks, failOn: "warning" }); // strict mode: warnings fail the run
73
+ ```
74
+
75
+ If a check throws, `evaluate` records a `critical` failure and keeps going. `result.metadata` carries `durationMs`
76
+ and, for checks that report data (like semantic scores), `checks[checkName]`.
77
+
78
+ ## Built-in checks
79
+
80
+ | Check | What it verifies | Failure code(s) |
81
+ | --------------------------------------- | ---------------------------------------------------------- | ------------------------------------------ |
82
+ | `schemaCheck(zodSchema)` | Output matches a Zod schema | `schema_violation` |
83
+ | `requiredFields(fields)` | Object has the given keys | `missing_field` |
84
+ | `noExtraFields(allowed, { severity? })` | No keys outside the allowed set (default severity `error`) | `schema_violation` |
85
+ | `enumCheck(field, values)` | Field is one of the allowed values | `invalid_enum`, `missing_field` |
86
+ | `fieldTypeCheck(field, type)` | Field has the runtime type | `invalid_type` |
87
+ | `jsonCheck()` | String output parses as JSON | `invalid_json` |
88
+ | `nonEmpty(field)` | Field is not blank/empty | `policy_violation`, `missing_field` |
89
+ | `regexCheck(field, pattern)` | String field matches a regex | `policy_violation`, `invalid_type` |
90
+ | `rangeCheck(field, { min?, max? })` | Number within bounds | `policy_violation`, `invalid_type` |
91
+ | `toolCallCheck(definition)` | Tool call matches a contract | `tool_call_mismatch`, `missing_field`, ... |
92
+ | `toolCallCheck.oneOf(definitions)` | Tool call matches one of several allowed tools | `tool_call_mismatch`, ... |
93
+ | `semanticCheck(judge, options)` | Optional LLM-judged score (advisory by default) | `unsupported_claim` |
94
+ | `customCheck(name, fn)` | Your own sync or async logic | anything |
95
+
96
+ ### Tool-call validation
97
+
98
+ Validate a model's tool call before executing it. A call is `{ name, args }`, or `{ name, arguments: "<json string>" }`
99
+ as returned by OpenAI-style APIs.
100
+
101
+ ```ts
102
+ import { evaluate, toolCallCheck } from "@hamza1331/aieval";
103
+
104
+ const result = await evaluate({
105
+ output: { name: "get_weather", args: { city: "London", unit: "celsius" } },
106
+ checks: [
107
+ toolCallCheck({
108
+ name: "get_weather",
109
+ requiredArgs: ["city"],
110
+ allowedArgs: ["city", "unit"],
111
+ argTypes: { city: "string", unit: "string" },
112
+ }),
113
+ ],
114
+ });
115
+ ```
116
+
117
+ Unknown arguments are errors unless you set `allowExtraArgs: true`. To allow several tools and reject everything
118
+ else, use `toolCallCheck.oneOf([...definitions])`.
119
+
120
+ ### Custom checks
121
+
122
+ ```ts
123
+ import { customCheck } from "@hamza1331/aieval";
124
+
125
+ const evenScore = customCheck<{ score: number }>(
126
+ "evenScore",
127
+ async (value) => ({
128
+ passed: value.score % 2 === 0,
129
+ failures:
130
+ value.score % 2 === 0
131
+ ? []
132
+ : [
133
+ {
134
+ code: "policy_violation",
135
+ message: "score must be even",
136
+ severity: "error",
137
+ path: "score",
138
+ },
139
+ ],
140
+ warnings: [],
141
+ }),
142
+ );
143
+ ```
144
+
145
+ ## Optional: semantic judging
146
+
147
+ Deterministic checks cover what code can verify. For subjective quality you can plug in an LLM judge — a plain
148
+ function you provide, so `aieval` has no provider SDK or network dependency.
149
+
150
+ ```ts
151
+ import { evaluate, semanticCheck, type SemanticJudge } from "@hamza1331/aieval";
152
+
153
+ const judge: SemanticJudge = async ({ input, output, criteria }) => {
154
+ // call your model here and return { score: 0..1, reasoning? }
155
+ return { score: 0.9, reasoning: "Grounded in the context." };
156
+ };
157
+
158
+ const result = await evaluate({
159
+ input: question,
160
+ output: answer,
161
+ checks: [
162
+ semanticCheck(judge, {
163
+ criteria: "The answer is grounded in the context",
164
+ threshold: 0.7,
165
+ }),
166
+ ],
167
+ });
168
+ ```
169
+
170
+ Semantic scores are **advisory by default**: a low score is a `warning`, not a failure. Raise `severity` or use
171
+ `failOn: "warning"` to enforce them. Scores appear in `result.metadata.checks`. Use `mockJudge` in tests. See
172
+ [`examples/openai-judge.ts`](examples/openai-judge.ts) for a real provider via `fetch`.
173
+
174
+ ## CLI
175
+
176
+ ```bash
177
+ # Run fixture files or directories of them
178
+ aieval run examples/fixtures
179
+ aieval run fixtures/ --fail-on warning --format json
180
+
181
+ # Validate one output (file or stdin) against a list of check specs
182
+ echo '{"name":"Ada"}' | aieval check --checks checks.json
183
+ ```
184
+
185
+ A fixture is JSON (a single object or an array):
186
+
187
+ ```json
188
+ {
189
+ "name": "model invented a priority",
190
+ "output": { "title": "Login fails", "priority": "urgent" },
191
+ "checks": [
192
+ { "type": "enum", "field": "priority", "values": ["low", "medium", "high"] }
193
+ ],
194
+ "expected": { "passed": false, "codes": ["invalid_enum"] }
195
+ }
196
+ ```
197
+
198
+ - With `expected`, the case is a **regression test**: the result must match. Without it, the case passes if validation
199
+ passes.
200
+ - Check types: `requiredFields`, `noExtraFields`, `enum`, `fieldType`, `toolCall`, `toolCallOneOf`, `json`,
201
+ `nonEmpty`, `regex`, `range`. (`schemaCheck` and `semanticCheck` are code-only.)
202
+ - Exit codes: `0` all passed, `1` at least one case failed, `2` usage or input error.
203
+
204
+ The same machinery is available programmatically: `buildChecks`, `loadFixtures`, `runFixture`, `runAll`.
205
+
206
+ ## Design principles
207
+
208
+ - Fail fast on what you can check deterministically; keep LLM evaluation optional and separate.
209
+ - Make failures explainable and machine-readable.
210
+ - Stay small: no dashboard, hosted service, agent framework or dataset platform.
211
+
212
+ See [`PROJECT_DISCOVERY.md`](PROJECT_DISCOVERY.md) for the research and rationale.
213
+
214
+ ## Roadmap
215
+
216
+ - Repair and retry strategy helpers
217
+ - Adapter packages for OpenAI and LangChain (only if there is demand)
218
+ - Trace metadata export (OpenTelemetry / Langfuse)
219
+
220
+ ## Contributing
221
+
222
+ See [CONTRIBUTING.md](CONTRIBUTING.md).
223
+
224
+ ## License
225
+
226
+ MIT
@@ -0,0 +1,39 @@
1
+ import { z } from "zod";
2
+ import type { Check, CheckContext, CheckResult, Failure, Severity } from "./types.js";
3
+ export declare const makeFailure: (code: Failure["code"], message: string, severity?: Severity, options?: Partial<Pick<Failure, "path" | "metadata" | "suggestedRepair">>) => Failure;
4
+ export declare const requiredFields: (fields: string[]) => Check<Record<string, unknown>>;
5
+ export declare const noExtraFields: (allowedFields: string[], options?: {
6
+ severity?: Severity;
7
+ }) => Check<Record<string, unknown>>;
8
+ export declare const enumCheck: <T extends string | number>(field: string, allowedValues: readonly T[]) => Check<Record<string, unknown>>;
9
+ export declare const fieldTypeCheck: (field: string, expectedType: "string" | "number" | "boolean" | "object" | "array") => Check<Record<string, unknown>>;
10
+ export declare const schemaCheck: <T>(schema: z.ZodType<T>) => Check<T>;
11
+ type ValueType = "string" | "number" | "boolean" | "object" | "array";
12
+ export interface ToolCallDefinition {
13
+ name: string;
14
+ requiredArgs?: string[];
15
+ allowedArgs?: string[];
16
+ argTypes?: Record<string, ValueType>;
17
+ /** When true, arguments not listed in allowedArgs/argTypes are tolerated. Defaults to false. */
18
+ allowExtraArgs?: boolean;
19
+ }
20
+ /**
21
+ * Validates a tool call shaped `{ name, args }` (or `{ name, arguments: "<json>" }`).
22
+ */
23
+ export declare const toolCallCheck: {
24
+ <T = unknown>(definition: ToolCallDefinition): Check<T>;
25
+ oneOf<T = unknown>(definitions: ToolCallDefinition[]): Check<T>;
26
+ };
27
+ /**
28
+ * Parses a JSON string output and emits `invalid_json` on failure.
29
+ * Use to guard string outputs before running object-level checks.
30
+ */
31
+ export declare const jsonCheck: () => Check<unknown>;
32
+ export declare const nonEmpty: (field: string) => Check<Record<string, unknown>>;
33
+ export declare const regexCheck: (field: string, pattern: RegExp) => Check<Record<string, unknown>>;
34
+ export declare const rangeCheck: (field: string, bounds: {
35
+ min?: number;
36
+ max?: number;
37
+ }) => Check<Record<string, unknown>>;
38
+ export declare const customCheck: <T>(name: string, fn: (value: T, context: CheckContext) => CheckResult | Promise<CheckResult>) => Check<T>;
39
+ export {};