openmerit 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +270 -0
- package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +38 -0
- package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +32 -0
- package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +26 -0
- package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +26 -0
- package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +38 -0
- package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +38 -0
- package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +20 -0
- package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +38 -0
- package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +26 -0
- package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +20 -0
- package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
- package/benchmark/invoice_ocr/data/manifest.json +97 -0
- package/dist/benchmarks.js +98 -0
- package/dist/catalog.js +51 -0
- package/dist/cli.js +261 -0
- package/dist/daemon.js +305 -0
- package/dist/frontier.js +49 -0
- package/dist/invoice-eval.js +33 -0
- package/dist/invoice-score.js +124 -0
- package/dist/judge.js +43 -0
- package/dist/llm.js +66 -0
- package/dist/pi-trials.js +224 -0
- package/dist/policy.js +110 -0
- package/dist/recommend.js +63 -0
- package/dist/store.js +55 -0
- package/dist/strategist.js +64 -0
- package/dist/task-input.js +30 -0
- package/dist/traces.js +123 -0
- package/dist/trials.js +101 -0
- package/dist/types.js +2 -0
- package/examples/invoice-prompt.txt +19 -0
- package/examples/task.example.json +6 -0
- package/extension/openmerit.ts +705 -0
- package/instructions/OPENMERIT.md +51 -0
- package/instructions/openmerit.policy.json +33 -0
- package/package.json +77 -0
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# OpenMerit — Model Merit Harness
|
|
2
|
+
|
|
3
|
+
You are running under a main agent harness that is observed by **OpenMerit**, an
|
|
4
|
+
external merit harness. OpenMerit's job is to make sure you are always running on
|
|
5
|
+
the best model for the task at hand, at the best price, with a vetted fallback.
|
|
6
|
+
|
|
7
|
+
## What OpenMerit does
|
|
8
|
+
|
|
9
|
+
1. **Observes** this harness's session traces (models used, tokens, cost,
|
|
10
|
+
latency, errors) without intercepting or slowing down your work.
|
|
11
|
+
2. **Evaluates** candidate models for each completed text or image task sequentially
|
|
12
|
+
through pi and scores their outputs with a judge model.
|
|
13
|
+
3. **Maintains a pareto frontier** per task (quality vs. cost vs. latency) and
|
|
14
|
+
an aggregate frontier across all of your tasks.
|
|
15
|
+
4. **Watches for new model releases** (provider catalogs + public benchmarks)
|
|
16
|
+
and queues promising releases for trial against your existing frontier.
|
|
17
|
+
5. **Recommends or applies model swaps**, with reasoning, and keeps your
|
|
18
|
+
fallback model up to date.
|
|
19
|
+
|
|
20
|
+
## What you should do
|
|
21
|
+
|
|
22
|
+
- **Treat model changes as explicit routing decisions.** If you notice your
|
|
23
|
+
model identity change, it was approved by the user or passed their opt-in
|
|
24
|
+
auto-apply policy. Continue the task; your instructions and context are
|
|
25
|
+
unchanged.
|
|
26
|
+
- **Respect the fallback.** If your current model errors, rate-limits, or is
|
|
27
|
+
retired, OpenMerit can offer the current fallback or apply it under an
|
|
28
|
+
opt-in automatic policy. Treat an approved fallback as normal operation.
|
|
29
|
+
- **State your task clearly in your first message** of a session when possible.
|
|
30
|
+
OpenMerit keys its pareto frontier to task signatures; clear task statements
|
|
31
|
+
produce better model choices for you.
|
|
32
|
+
- **Do not edit OpenMerit state files** (`~/.openmerit/`). If something looks
|
|
33
|
+
wrong, tell the user.
|
|
34
|
+
|
|
35
|
+
## What you can ask the user for
|
|
36
|
+
|
|
37
|
+
- `/openmerit` — show current model, fallback, frontier position, and pending
|
|
38
|
+
recommendations.
|
|
39
|
+
- `/openmerit apply` / `/openmerit dismiss` — act on a pending recommendation
|
|
40
|
+
when automatic application is declined by the policy gate.
|
|
41
|
+
- Policy changes (auto-apply thresholds, budgets, provider allow-lists) are
|
|
42
|
+
made by the user in `~/.openmerit/policy.json`, not by you.
|
|
43
|
+
|
|
44
|
+
## Guarantees
|
|
45
|
+
|
|
46
|
+
- Per-task comparisons send the task text, uploaded images (when present), and
|
|
47
|
+
candidate answers through pi/OpenRouter for model runs and judging. Use
|
|
48
|
+
non-sensitive examples while evaluating this alpha.
|
|
49
|
+
- The shipped policy is supervised. Swaps happen automatically only after the
|
|
50
|
+
user opts in and the configured quality/cost guardrails pass; otherwise they
|
|
51
|
+
remain recommendations for a human to approve.
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
{
|
|
2
|
+
"version": 1,
|
|
3
|
+
"mode": "recommend",
|
|
4
|
+
"auto_apply": {
|
|
5
|
+
"enabled": false,
|
|
6
|
+
"min_score_gain": 0.1,
|
|
7
|
+
"max_price_ratio": 1.5,
|
|
8
|
+
"require_frontier": true
|
|
9
|
+
},
|
|
10
|
+
"budgets": {
|
|
11
|
+
"max_usd_per_trial": 0.25,
|
|
12
|
+
"max_trials_per_day": 20,
|
|
13
|
+
"max_usd_per_day": 5.0
|
|
14
|
+
},
|
|
15
|
+
"providers": {
|
|
16
|
+
"allow": ["*"],
|
|
17
|
+
"deny": []
|
|
18
|
+
},
|
|
19
|
+
"watch": {
|
|
20
|
+
"catalog_interval_min": 360,
|
|
21
|
+
"traces_interval_sec": 5,
|
|
22
|
+
"trial_interval_min": 30,
|
|
23
|
+
"models_per_task": 3
|
|
24
|
+
},
|
|
25
|
+
"fallback": {
|
|
26
|
+
"auto_update": true,
|
|
27
|
+
"min_score": 0.6,
|
|
28
|
+
"apply_on_error": true
|
|
29
|
+
},
|
|
30
|
+
"judge_model": null,
|
|
31
|
+
"strategist_model": null,
|
|
32
|
+
"max_usd_per_m": 20.0
|
|
33
|
+
}
|
package/package.json
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "openmerit",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "Find better models for each pi task by comparing quality, cost, and latency.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"keywords": [
|
|
7
|
+
"pi-package",
|
|
8
|
+
"pi-extension",
|
|
9
|
+
"pi-coding-agent",
|
|
10
|
+
"model-routing",
|
|
11
|
+
"llm",
|
|
12
|
+
"openrouter",
|
|
13
|
+
"evaluation"
|
|
14
|
+
],
|
|
15
|
+
"files": [
|
|
16
|
+
"dist/",
|
|
17
|
+
"extension/openmerit.ts",
|
|
18
|
+
"instructions/",
|
|
19
|
+
"examples/",
|
|
20
|
+
"benchmark/invoice_ocr/data/"
|
|
21
|
+
],
|
|
22
|
+
"pi": {
|
|
23
|
+
"extensions": ["./extension/openmerit.ts"]
|
|
24
|
+
},
|
|
25
|
+
"bin": {
|
|
26
|
+
"openmerit": "dist/cli.js"
|
|
27
|
+
},
|
|
28
|
+
"engines": {
|
|
29
|
+
"node": ">=22.18.0"
|
|
30
|
+
},
|
|
31
|
+
"repository": {
|
|
32
|
+
"type": "git",
|
|
33
|
+
"url": "git+https://github.com/laz-aslam/openmerit.git"
|
|
34
|
+
},
|
|
35
|
+
"homepage": "https://github.com/laz-aslam/openmerit#readme",
|
|
36
|
+
"bugs": {
|
|
37
|
+
"url": "https://github.com/laz-aslam/openmerit/issues"
|
|
38
|
+
},
|
|
39
|
+
"publishConfig": {
|
|
40
|
+
"access": "public"
|
|
41
|
+
},
|
|
42
|
+
"scripts": {
|
|
43
|
+
"prepare": "npm run build",
|
|
44
|
+
"build": "tsc -p tsconfig.json",
|
|
45
|
+
"typecheck": "tsc --noEmit && tsc -p benchmark/tsconfig.json",
|
|
46
|
+
"typecheck:extension": "tsc --noEmit --strict --target ES2022 --module NodeNext --moduleResolution NodeNext --skipLibCheck extension/openmerit.ts",
|
|
47
|
+
"test": "vitest run",
|
|
48
|
+
"check": "npm run typecheck && npm run typecheck:extension && npm test && npm run build && npm run benchmark:verify-recorded",
|
|
49
|
+
"prepublishOnly": "npm run check",
|
|
50
|
+
"benchmark:invoice:prepare": "node benchmark/invoice_ocr/prepare_data.ts",
|
|
51
|
+
"benchmark:invoice": "node benchmark/invoice_ocr/benchmark.ts",
|
|
52
|
+
"benchmark:invoice:audit": "node benchmark/invoice_ocr/audit_results.ts",
|
|
53
|
+
"benchmark:invoice:round2": "node benchmark/invoice_ocr/benchmark_round2.ts",
|
|
54
|
+
"benchmark:invoice:round2:audit": "node benchmark/invoice_ocr/audit_round2.ts",
|
|
55
|
+
"benchmark:text-to-sql": "node benchmark/text_to_sql/benchmark.ts",
|
|
56
|
+
"benchmark:text-to-sql:audit": "node benchmark/text_to_sql/audit_results.ts",
|
|
57
|
+
"benchmark:policy": "node benchmark/policy_adjudication/benchmark.ts",
|
|
58
|
+
"benchmark:policy:audit": "node benchmark/policy_adjudication/audit_results.ts",
|
|
59
|
+
"benchmark:policy:validate": "node benchmark/policy_adjudication/validate_eval.ts",
|
|
60
|
+
"benchmark:verify-recorded": "node benchmark/verify_recorded.ts"
|
|
61
|
+
},
|
|
62
|
+
"license": "MIT",
|
|
63
|
+
"peerDependencies": {
|
|
64
|
+
"@earendil-works/pi-coding-agent": "*"
|
|
65
|
+
},
|
|
66
|
+
"peerDependenciesMeta": {
|
|
67
|
+
"@earendil-works/pi-coding-agent": {
|
|
68
|
+
"optional": true
|
|
69
|
+
}
|
|
70
|
+
},
|
|
71
|
+
"devDependencies": {
|
|
72
|
+
"@earendil-works/pi-coding-agent": "^0.85.1",
|
|
73
|
+
"@types/node": "^22.10.0",
|
|
74
|
+
"typescript": "^5.7.0",
|
|
75
|
+
"vitest": "^2.1.0"
|
|
76
|
+
}
|
|
77
|
+
}
|