openmerit 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. package/LICENSE +21 -0
  2. package/README.md +270 -0
  3. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +38 -0
  4. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  5. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +32 -0
  6. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  7. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +26 -0
  8. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  9. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +26 -0
  10. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  11. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +38 -0
  12. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  13. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +38 -0
  14. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  15. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +20 -0
  16. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  17. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +38 -0
  18. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  19. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +26 -0
  20. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  21. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +20 -0
  22. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  23. package/benchmark/invoice_ocr/data/manifest.json +97 -0
  24. package/dist/benchmarks.js +98 -0
  25. package/dist/catalog.js +51 -0
  26. package/dist/cli.js +261 -0
  27. package/dist/daemon.js +305 -0
  28. package/dist/frontier.js +49 -0
  29. package/dist/invoice-eval.js +33 -0
  30. package/dist/invoice-score.js +124 -0
  31. package/dist/judge.js +43 -0
  32. package/dist/llm.js +66 -0
  33. package/dist/pi-trials.js +224 -0
  34. package/dist/policy.js +110 -0
  35. package/dist/recommend.js +63 -0
  36. package/dist/store.js +55 -0
  37. package/dist/strategist.js +64 -0
  38. package/dist/task-input.js +30 -0
  39. package/dist/traces.js +123 -0
  40. package/dist/trials.js +101 -0
  41. package/dist/types.js +2 -0
  42. package/examples/invoice-prompt.txt +19 -0
  43. package/examples/task.example.json +6 -0
  44. package/extension/openmerit.ts +705 -0
  45. package/instructions/OPENMERIT.md +51 -0
  46. package/instructions/openmerit.policy.json +33 -0
  47. package/package.json +77 -0
@@ -0,0 +1,51 @@
1
+ # OpenMerit — Model Merit Harness
2
+
3
+ You are running under a main agent harness that is observed by **OpenMerit**, an
4
+ external merit harness. OpenMerit's job is to make sure you are always running on
5
+ the best model for the task at hand, at the best price, with a vetted fallback.
6
+
7
+ ## What OpenMerit does
8
+
9
+ 1. **Observes** this harness's session traces (models used, tokens, cost,
10
+ latency, errors) without intercepting or slowing down your work.
11
+ 2. **Evaluates** candidate models for each completed text or image task sequentially
12
+ through pi and scores their outputs with a judge model.
13
+ 3. **Maintains a pareto frontier** per task (quality vs. cost vs. latency) and
14
+ an aggregate frontier across all of your tasks.
15
+ 4. **Watches for new model releases** (provider catalogs + public benchmarks)
16
+ and queues promising releases for trial against your existing frontier.
17
+ 5. **Recommends or applies model swaps**, with reasoning, and keeps your
18
+ fallback model up to date.
19
+
20
+ ## What you should do
21
+
22
+ - **Treat model changes as explicit routing decisions.** If you notice your
23
+ model identity change, it was approved by the user or passed their opt-in
24
+ auto-apply policy. Continue the task; your instructions and context are
25
+ unchanged.
26
+ - **Respect the fallback.** If your current model errors, rate-limits, or is
27
+ retired, OpenMerit can offer the current fallback or apply it under an
28
+ opt-in automatic policy. Treat an approved fallback as normal operation.
29
+ - **State your task clearly in your first message** of a session when possible.
30
+ OpenMerit keys its pareto frontier to task signatures; clear task statements
31
+ produce better model choices for you.
32
+ - **Do not edit OpenMerit state files** (`~/.openmerit/`). If something looks
33
+ wrong, tell the user.
34
+
35
+ ## What you can ask the user for
36
+
37
+ - `/openmerit` — show current model, fallback, frontier position, and pending
38
+ recommendations.
39
+ - `/openmerit apply` / `/openmerit dismiss` — act on a pending recommendation
40
+ when automatic application is declined by the policy gate.
41
+ - Policy changes (auto-apply thresholds, budgets, provider allow-lists) are
42
+ made by the user in `~/.openmerit/policy.json`, not by you.
43
+
44
+ ## Guarantees
45
+
46
+ - Per-task comparisons send the task text, uploaded images (when present), and
47
+ candidate answers through pi/OpenRouter for model runs and judging. Use
48
+ non-sensitive examples while evaluating this alpha.
49
+ - The shipped policy is supervised. Swaps happen automatically only after the
50
+ user opts in and the configured quality/cost guardrails pass; otherwise they
51
+ remain recommendations for a human to approve.
@@ -0,0 +1,33 @@
1
+ {
2
+ "version": 1,
3
+ "mode": "recommend",
4
+ "auto_apply": {
5
+ "enabled": false,
6
+ "min_score_gain": 0.1,
7
+ "max_price_ratio": 1.5,
8
+ "require_frontier": true
9
+ },
10
+ "budgets": {
11
+ "max_usd_per_trial": 0.25,
12
+ "max_trials_per_day": 20,
13
+ "max_usd_per_day": 5.0
14
+ },
15
+ "providers": {
16
+ "allow": ["*"],
17
+ "deny": []
18
+ },
19
+ "watch": {
20
+ "catalog_interval_min": 360,
21
+ "traces_interval_sec": 5,
22
+ "trial_interval_min": 30,
23
+ "models_per_task": 3
24
+ },
25
+ "fallback": {
26
+ "auto_update": true,
27
+ "min_score": 0.6,
28
+ "apply_on_error": true
29
+ },
30
+ "judge_model": null,
31
+ "strategist_model": null,
32
+ "max_usd_per_m": 20.0
33
+ }
package/package.json ADDED
@@ -0,0 +1,77 @@
1
+ {
2
+ "name": "openmerit",
3
+ "version": "0.1.0",
4
+ "description": "Find better models for each pi task by comparing quality, cost, and latency.",
5
+ "type": "module",
6
+ "keywords": [
7
+ "pi-package",
8
+ "pi-extension",
9
+ "pi-coding-agent",
10
+ "model-routing",
11
+ "llm",
12
+ "openrouter",
13
+ "evaluation"
14
+ ],
15
+ "files": [
16
+ "dist/",
17
+ "extension/openmerit.ts",
18
+ "instructions/",
19
+ "examples/",
20
+ "benchmark/invoice_ocr/data/"
21
+ ],
22
+ "pi": {
23
+ "extensions": ["./extension/openmerit.ts"]
24
+ },
25
+ "bin": {
26
+ "openmerit": "dist/cli.js"
27
+ },
28
+ "engines": {
29
+ "node": ">=22.18.0"
30
+ },
31
+ "repository": {
32
+ "type": "git",
33
+ "url": "git+https://github.com/laz-aslam/openmerit.git"
34
+ },
35
+ "homepage": "https://github.com/laz-aslam/openmerit#readme",
36
+ "bugs": {
37
+ "url": "https://github.com/laz-aslam/openmerit/issues"
38
+ },
39
+ "publishConfig": {
40
+ "access": "public"
41
+ },
42
+ "scripts": {
43
+ "prepare": "npm run build",
44
+ "build": "tsc -p tsconfig.json",
45
+ "typecheck": "tsc --noEmit && tsc -p benchmark/tsconfig.json",
46
+ "typecheck:extension": "tsc --noEmit --strict --target ES2022 --module NodeNext --moduleResolution NodeNext --skipLibCheck extension/openmerit.ts",
47
+ "test": "vitest run",
48
+ "check": "npm run typecheck && npm run typecheck:extension && npm test && npm run build && npm run benchmark:verify-recorded",
49
+ "prepublishOnly": "npm run check",
50
+ "benchmark:invoice:prepare": "node benchmark/invoice_ocr/prepare_data.ts",
51
+ "benchmark:invoice": "node benchmark/invoice_ocr/benchmark.ts",
52
+ "benchmark:invoice:audit": "node benchmark/invoice_ocr/audit_results.ts",
53
+ "benchmark:invoice:round2": "node benchmark/invoice_ocr/benchmark_round2.ts",
54
+ "benchmark:invoice:round2:audit": "node benchmark/invoice_ocr/audit_round2.ts",
55
+ "benchmark:text-to-sql": "node benchmark/text_to_sql/benchmark.ts",
56
+ "benchmark:text-to-sql:audit": "node benchmark/text_to_sql/audit_results.ts",
57
+ "benchmark:policy": "node benchmark/policy_adjudication/benchmark.ts",
58
+ "benchmark:policy:audit": "node benchmark/policy_adjudication/audit_results.ts",
59
+ "benchmark:policy:validate": "node benchmark/policy_adjudication/validate_eval.ts",
60
+ "benchmark:verify-recorded": "node benchmark/verify_recorded.ts"
61
+ },
62
+ "license": "MIT",
63
+ "peerDependencies": {
64
+ "@earendil-works/pi-coding-agent": "*"
65
+ },
66
+ "peerDependenciesMeta": {
67
+ "@earendil-works/pi-coding-agent": {
68
+ "optional": true
69
+ }
70
+ },
71
+ "devDependencies": {
72
+ "@earendil-works/pi-coding-agent": "^0.85.1",
73
+ "@types/node": "^22.10.0",
74
+ "typescript": "^5.7.0",
75
+ "vitest": "^2.1.0"
76
+ }
77
+ }