openmerit 0.1.3 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/CHANGELOG.md +26 -0
  2. package/README.md +83 -312
  3. package/dist/core/src/index.d.ts +90 -0
  4. package/dist/core/src/index.js +1137 -0
  5. package/dist/core/src/store.d.ts +35 -0
  6. package/dist/core/src/store.js +102 -0
  7. package/dist/pi/src/index.d.ts +15 -0
  8. package/dist/pi/src/index.js +423 -0
  9. package/dist/protocol/src/index.d.ts +402 -0
  10. package/dist/protocol/src/index.js +47 -0
  11. package/dist/protocol/src/schemas.d.ts +450 -0
  12. package/dist/protocol/src/schemas.js +224 -0
  13. package/docs/adapter-guide.md +172 -0
  14. package/docs/architecture.md +55 -0
  15. package/docs/automation.md +66 -0
  16. package/docs/getting-started.md +55 -0
  17. package/docs/lifecycle.md +30 -0
  18. package/docs/metrics-and-evidence.md +40 -0
  19. package/docs/operations.md +31 -0
  20. package/docs/pareto-spec.md +76 -0
  21. package/docs/pi-extension.md +44 -0
  22. package/docs/roadmap.md +26 -0
  23. package/docs/security.md +23 -0
  24. package/docs/testing.md +36 -0
  25. package/docs/ux-reference.md +32 -0
  26. package/package.json +45 -54
  27. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
  28. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  29. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
  30. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  31. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
  32. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  33. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
  34. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  35. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
  36. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  37. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
  38. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  39. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
  40. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  41. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
  42. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  43. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
  44. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  45. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
  46. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  47. package/benchmark/invoice_ocr/data/manifest.json +0 -97
  48. package/dist/benchmarks.js +0 -98
  49. package/dist/catalog.js +0 -61
  50. package/dist/cli.js +0 -188
  51. package/dist/daemon.js +0 -388
  52. package/dist/diagnostics.js +0 -194
  53. package/dist/frontier.js +0 -56
  54. package/dist/harness.js +0 -1
  55. package/dist/integrations.js +0 -19
  56. package/dist/invoice-eval.js +0 -33
  57. package/dist/invoice-score.js +0 -124
  58. package/dist/judge.js +0 -43
  59. package/dist/llm.js +0 -203
  60. package/dist/pi-trials.js +0 -366
  61. package/dist/policy.js +0 -115
  62. package/dist/providers.js +0 -1
  63. package/dist/recommend.js +0 -76
  64. package/dist/routes.js +0 -59
  65. package/dist/standalone.js +0 -220
  66. package/dist/store.js +0 -89
  67. package/dist/strategist.js +0 -68
  68. package/dist/task-input.js +0 -54
  69. package/dist/traces.js +0 -127
  70. package/dist/trials.js +0 -140
  71. package/dist/types.js +0 -2
  72. package/examples/invoice-prompt.txt +0 -19
  73. package/examples/task.example.json +0 -7
  74. package/extension/openmerit.ts +0 -820
  75. package/instructions/OPENMERIT.md +0 -54
  76. package/instructions/openmerit.policy.json +0 -33
  77. package/rules.md +0 -39
package/CHANGELOG.md ADDED
@@ -0,0 +1,26 @@
1
+ # Changelog
2
+
3
+ ## 0.1.5 — 2026-09-23
4
+
5
+ OpenMerit now ships as one npm package with a Pi extension, a reusable
6
+ coordinator, and a harness-neutral protocol.
7
+
8
+ ### Changed
9
+
10
+ - Pi issues typed, bounded work to the coding harness and checks structured
11
+ results and durable evidence before advancing the model-improvement lifecycle.
12
+ - Project state, audit events, and evidence manifests live under `.openmerit/`
13
+ in the product repository.
14
+ - Baseline assessment, candidate trials, frontier verification, supervised
15
+ swaps, post-swap verification, and rollback follow a confirmed project policy.
16
+ - The package exposes `openmerit/core`, `openmerit/protocol`, and `openmerit/pi`.
17
+ - Node.js 22.19 or newer and Pi 0.87 are required for the Pi integration.
18
+
19
+ ### Removed from the 0.1.4 package
20
+
21
+ - The background watcher, direct provider clients, standalone CLI, and
22
+ session-bound route comparison flow.
23
+ - Automatic use of the earlier home-directory state. The new project-local
24
+ state starts with a confirmed task profile and leaves older files untouched.
25
+
26
+ See [the README](README.md) for installation and the current command surface.
package/README.md CHANGED
@@ -1,342 +1,113 @@
1
- # openmerit
1
+ # OpenMerit
2
2
 
3
- An **experimental model-merit harness** for [pi](https://pi.dev). Start
4
- pi with model A, and the OpenMerit extension observes
5
- its completed trace, trials B and C sequentially through pi, compares quality,
6
- price, and latency, and recommends a model for that exact pi session. The
7
- shipped policy is supervised: you review and apply a recommendation with
8
- `/openmerit apply`. Automatic swaps are available as an explicit opt-in.
3
+ OpenMerit is a harness-neutral orchestration layer that helps a coding harness continuously find and maintain the Pareto frontier of models for a real product task.
9
4
 
10
- ```text
11
- pi task A → saved pi trace → extension trial job → pi trials B, C → scored frontier
12
- ↑ │
13
- └──── pi.setModel() ← approval/policy ← recommendation ┘
14
- ```
15
-
16
- ## How it maps to the design
17
-
18
- 1. **Instruction file + policy file** — [`instructions/OPENMERIT.md`](instructions/OPENMERIT.md)
19
- goes into the main harness's context (AGENTS.md-style);
20
- [`instructions/openmerit.policy.json`](instructions/openmerit.policy.json)
21
- (copied to `~/.openmerit/policy.json` by `openmerit init`) gates autonomy:
22
- supervised recommendations by default, optional auto-apply thresholds,
23
- budgets, and provider allow-lists. Public-benchmark relevance lives in
24
- [`src/benchmarks.ts`](src/benchmarks.ts) (task category → benchmarks that
25
- matter) plus a refreshable digest at `~/.openmerit/benchmarks/digest.json`.
26
- 2. **Trial workflow** — the pi extension starts a session-bound comparison
27
- after a supported task settles. It runs candidate trials sequentially under
28
- configured candidate-trial caps. `openmerit watch` remains an optional CLI
29
- mode. Pi's traces and JSON event stream supply results, cost,
30
- latency, and errors; a judge or known-ground-truth scorer measures quality.
31
- 3. **Per-task pareto frontier** — trials are keyed by task signature
32
- ([`src/frontier.ts`](src/frontier.ts), 3 objectives: quality ↑, price ↓,
33
- latency ↓). The aggregate across all of an agent's tasks is the union of
34
- its task frontiers (`openmerit frontier`).
35
- 4. **Model discovery** — the active Pi session's scoped models (or Pi's
36
- authenticated available-model registry when unscoped) are the candidate
37
- pool. OpenRouter is an optional route and enrichment source: its catalog can
38
- be snapshotted and its benchmark signals can improve the shortlist
39
- ([`src/catalog.ts`](src/catalog.ts) + [`src/daemon.ts`](src/daemon.ts)).
40
- 5. **Informing the main harness** — recommendations land in
41
- `~/.openmerit/recommendations.jsonl`. The pi extension
42
- ([`extension/openmerit.ts`](extension/openmerit.ts)) reports progress, polls it, and applies
43
- swaps with `pi.setModel()` — automatically when the policy gate passes,
44
- otherwise after user approval (`/openmerit`, `/openmerit apply`). The same
45
- recommendation carries the updated fallback, used when the active model
46
- starts failing.
47
-
48
- ## Install from npm
49
-
50
- You need Node 22.18+, pi 0.85.1+, and at least two eligible model routes
51
- authenticated in Pi. OpenRouter is supported but not required. Install the
52
- package into pi, then initialize its policy and state:
53
-
54
- ```bash
55
- pi install npm:openmerit
56
- npx --yes openmerit init
57
- npx --yes openmerit verify
58
- pi list
59
- ```
60
-
61
- `pi install` makes the extension and its trial engine available to pi. The
62
- one-off `npx` command creates `~/.openmerit/policy.json`; install OpenMerit
63
- globally with `npm install --global openmerit` only if you also want persistent
64
- shell access to `openmerit status`, `doctor`, `verify`, `frontier`, or the optional watcher. Install
65
- the extension from only one source—remove any older copied `openmerit.ts` first
66
- so pi does not load it twice.
67
-
68
- For development from a local checkout instead, run:
69
-
70
- ```bash
71
- npm ci
72
- node dist/cli.js init
73
- pi install "$PWD"
74
- ```
75
-
76
- Keep that checkout available because pi loads a local package from its path.
77
-
78
- ## Run OpenMerit with pi
79
-
80
- 1. Initialize OpenMerit if you did not do so during installation:
81
-
82
- ```bash
83
- npx --yes openmerit init
84
- ```
85
-
86
- `init` creates `~/.openmerit/policy.json` if missing and **keeps an existing
87
- policy**. The shipped policy uses `"mode": "recommend"` and does not change
88
- the active model without approval.
5
+ OpenMerit does not perform the intelligent work. It nudges a connected coding harness, such as Pi, with typed outcome requests. The harness understands the project, establishes evaluations and observability, gathers evidence, discovers and tests candidates, calculates the Pareto frontier, and performs an authorized model change. OpenMerit validates the returned contracts, preserves evidence and preferences, and advances the approved lifecycle.
89
6
 
90
- 2. Authenticate the providers you want to compare in Pi, using `/login` or
91
- Pi's normal environment/model configuration. OpenMerit takes its eligible
92
- model routes, capabilities, prices, and credentials from Pi; judge and
93
- strategist calls use the same routes and do not require duplicate keys.
7
+ The npm package `openmerit` contains the Pi extension, the reusable coordinator,
8
+ and the protocol in one install. The Pi adapter is the first integration; other
9
+ harnesses can use the same core and protocol exports.
94
10
 
95
- For example, verify one provider without printing its credential:
11
+ ## Why OpenMerit
96
12
 
97
- ```bash
98
- pi auth check --provider openai --json
99
- ```
13
+ The cheapest model, fastest model, and highest-quality model are often different. Public benchmarks are useful before product evidence exists, but they cannot establish which model is best for a particular workload. OpenMerit connects the initial choice to production evidence and repeated reassessment.
100
14
 
101
- Add an OpenRouter route the same way if you want its routed catalog:
15
+ It distinguishes:
102
16
 
103
- ```bash
104
- pi auth check --provider openrouter --json
105
- ```
17
+ - one-run observations from rate, percentile, consistency, and reliability claims;
18
+ - missing evidence from poor performance;
19
+ - feasible candidates from dominated, unresolved, or ineligible candidates;
20
+ - an automatic nudge from an intelligent action performed by the harness;
21
+ - a recommendation from an authorized, verified model swap.
106
22
 
107
- An `OPENROUTER_API_KEY` in the process environment or
108
- `~/.openmerit/.env` additionally enables OpenRouter catalog and public-
109
- benchmark enrichment. It is optional for native-provider comparisons. If
110
- using the file, protect it:
23
+ ## Current capabilities
111
24
 
112
- ```bash
113
- nano ~/.openmerit/.env
114
- chmod 600 ~/.openmerit/.env
115
- ```
25
+ - Pi 0.87 extension with a single `/openmerit` command surface.
26
+ - Confirmed task profiles, objectives, constraints, sample requirements, and evaluation budgets.
27
+ - The complete initial quality, reliability, cost, latency, efficiency, and human-intervention metric catalogue.
28
+ - Baseline-first evidence collection followed by budgeted candidate discovery and controlled challenger trials.
29
+ - A normative, uncertainty-aware Pareto definition owned by OpenMerit.
30
+ - Harness-calculated frontier results independently checked by OpenMerit's conformance verifier.
31
+ - Versioned result schemas for all ten intents, reusable across harness adapters and exposed dynamically by Pi.
32
+ - Supervised graduation: early swaps require approval, every swap requires verification, and regressions can trigger rollback.
33
+ - Project-local snapshots, redacted append-only audit events, evidence manifests, and optional best-effort event exporters under `.openmerit/`.
34
+ - Model-catalog change detection that can nudge the harness to reassess a mature baseline.
35
+ - User-confirmed task-count, elapsed-time, regression, catalogue-change, and post-swap check policies.
36
+ - Idempotent harness signals, durable pending work, native wakeup planning, and explicit scheduler capability gaps.
116
37
 
117
- Pi does not read OpenMerit's `.env` file for its ordinary sessions, so an
118
- OpenRouter route still needs Pi authentication. Older `model_search/.env`
119
- files are not read.
38
+ ## Responsibility boundary
120
39
 
121
- 3. Verify the extension appears in `pi list`. The instruction file
122
- [`instructions/OPENMERIT.md`](instructions/OPENMERIT.md) can be added to a
123
- pi project's AGENTS.md for agent context, but the extension does
124
- not require it. Run `npx openmerit verify` for a provider-free core self-test,
125
- then `npx openmerit doctor` after starting Pi once to inspect the eligible
126
- route snapshot and configuration without printing secrets.
40
+ | OpenMerit | Coding harness |
41
+ | --- | --- |
42
+ | Defines typed outcomes and Pareto conformance rules | Understands the product and user task |
43
+ | Tracks lifecycle, policy, budget, and evidence references | Builds evals and observability |
44
+ | Determines which approved nudge is due | Collects and calculates metrics |
45
+ | Verifies returned structure and frontier conformance | Discovers candidates and calculates the frontier |
46
+ | Enforces confirmation and rollback policy | Applies and verifies an authorized model change |
127
47
 
128
- 4. Start pi in one terminal with any configured model. For example:
48
+ Jev System One, when configured, belongs to the coding harness. Pi may use it aggressively for bounded ranking, triage, and fast judgment. OpenMerit contains no Jev client and accepts no Jev credential.
129
49
 
130
- ```bash
131
- pi --provider openrouter --model openai/gpt-4o-mini
132
- ```
50
+ ## Install with Pi
133
51
 
134
- For a first text task, ask: “Give the shortest valid word ladder from cat
135
- to dog. Each step changes one letter and must be a common English word.
136
- Return only the path.” Wait for A to finish and leave the pi session open.
137
- The extension automatically starts B and C **sequentially through their
138
- selected Pi routes**,
139
- reports each score in Pi, and writes a recommendation for this exact
140
- session. With the shipped supervised policy, use
141
- `/openmerit` inside pi to inspect the evidence and `/openmerit apply` to
142
- switch. Then send a second message to see which model actually handles it.
143
- You do not need a watcher terminal or the standalone `trial` command.
52
+ Requirements:
144
53
 
145
- Candidate comparisons have no tools by default, even when the observed task
146
- used tools. See **Alpha boundaries** before explicitly enabling candidate
147
- tools.
54
+ - Node.js 22.19 or newer
55
+ - Pi 0.87 with a supported model provider configured
148
56
 
149
- To opt in to automatic swaps, edit `~/.openmerit/policy.json`, set
150
- `"mode": "auto"` and `"auto_apply.enabled": true`, review the score-gain
151
- and price-ratio thresholds, and run the next task.
152
-
153
- ### Try an invoice-to-JSON task
154
-
155
- Use the same one-terminal setup. Start pi with a vision-capable model, for
156
- example `openai/gpt-4o-mini`. In pi, type `@` to select
157
- [`benchmark/invoice_ocr/data/invoice_01_row_2.jpg`](benchmark/invoice_ocr/data/invoice_01_row_2.jpg)
158
- and paste the **entire, unchanged** text from
159
- [`examples/invoice-prompt.txt`](examples/invoice-prompt.txt) into the same
160
- message. Pi also accepts pasted or dragged images. Wait for the JSON answer;
161
- keep pi open while the extension trials two other vision models. OpenMerit uses
162
- the existing benchmark's visibility-audited exact-field scorer when the saved
163
- task contains the exact example prompt and original bytes of a pinned JPEG.
164
- Pi may resize an attached image or add a file header to the prompt; in that
165
- case, the run uses the vision judge that sees the image and answer. Other
166
- uploaded invoices also use that judge because they have no ground truth.
167
-
168
- A real pinned-invoice run produced this trial output (models and scores will
169
- vary between runs):
170
-
171
- ```text
172
- [openmerit] task 0e7642522dc5: A=openai/gpt-4o-mini score=1.00 from pi trace
173
- [openmerit] task 0e7642522dc5: google/gemma-3-12b-it score=0.31 cost=$0.0001
174
- [openmerit] task 0e7642522dc5: mistralai/ministral-14b-2512 score=1.00 cost=$0.0007
175
- [openmerit] task 0e7642522dc5: selected mistralai/ministral-14b-2512; auto=false
176
- ```
177
-
178
- Pi candidate runs rely on the explicit invoice prompt for JSON structure.
179
- The standalone invoice benchmark sends a strict OpenRouter `response_format`
180
- schema, so its published scores are not directly comparable to these pi runs.
181
-
182
- ### Inspect or troubleshoot a run
183
-
184
- `npx openmerit doctor` checks Pi, policy, eligible routes and prices, saved
185
- session state, duplicate package sources, append-only files, and optional
186
- OpenRouter enrichment. Add `--json` for a sanitized diagnostic report suitable
187
- for a bug report; it contains no credentials, prompts, or trace contents.
188
- `npx openmerit verify` runs an offline self-test of atomic state, JSONL recovery,
189
- route preservation, policy evidence, and neutral events without contacting a
190
- provider. `npx openmerit status` shows the latest pi model and pending recommendations;
191
- `npx openmerit frontier` shows measured quality, blended price, latency, and
192
- the chosen frontier per task. `/openmerit` inside pi shows the current model,
193
- fallback, the model currently being compared, completed models with quality,
194
- cost, and latency, trial budget, pending recommendations, and any exact-task result
195
- measured in another session. When the gate declines an automatic swap, it
196
- prints reasons and `/openmerit apply` remains available.
197
-
198
- The daily trial-count and dollar limits come from `~/.openmerit/policy.json`.
199
- If a limit is reached, the extension reports why it skipped the comparison;
200
- the count resets at midnight UTC. You may raise `budgets.max_trials_per_day`
201
- for local experiments while keeping `budgets.max_usd_per_day` as a conservative
202
- candidate-spend threshold.
203
-
204
- If no comparison starts after Pi settles, confirm that pi loaded the extension
205
- (`pi list`), the task finished, and the pi session is saved (do not use
206
- `--no-session`). Image candidates must advertise image input in Pi's model
207
- registry. The extension queues
208
- completed tasks from its current session and runs one comparison at a time.
209
- Closing or switching the session cancels the active job.
210
- Candidate runs use the same text and uploaded image bytes, but they do not
211
- replay earlier answers or file changes. An exact task in another session
212
- (including identical image bytes) appears as **advice**, not a pending swap;
213
- the new session still gets its own comparison.
214
-
215
- The optional `trial` command runs controlled text-task comparisons through the
216
- same exact Pi provider routes and credential store as the automatic session
217
- path:
218
-
219
- ```bash
220
- npx openmerit trial examples/task.example.json --rounds 3
57
+ ```sh
58
+ pi install npm:openmerit
59
+ pi list
221
60
  ```
222
61
 
223
- `initial_model` is the stable `vendor/model` identity. When Pi exposes that
224
- model through more than one provider, set `initial_route` to
225
- `provider:model-id` (for example `openai:gpt-4o-mini` or
226
- `openrouter:openai/gpt-4o-mini`). Judge and strategist calls also run through
227
- Pi. An OpenRouter key only adds optional catalog and public-benchmark metadata.
62
+ When Pi enters an unconfigured product with an interactive UI, OpenMerit automatically issues the setup nudge. Pi infers a task profile, asks the user to confirm requirements and an evaluation budget, establishes task-specific eval and observability artifacts, verifies them, and reports evidence to OpenMerit. `/openmerit setup` remains available as an explicit recovery or rerun command.
228
63
 
229
- ## Alpha boundaries
64
+ Run `/openmerit doctor` inside Pi to check the installation. OpenMerit keeps
65
+ project evidence under `.openmerit/` in the product repository. Version 0.1.5
66
+ replaces the 0.1.4 background watcher and standalone CLI with this
67
+ harness-driven lifecycle; it does not migrate the earlier home-directory state.
230
68
 
231
- - With the extension installed and at least two eligible Pi routes, each
232
- supported settled task can start comparison calls automatically. Candidate,
233
- judge, and strategist requests send the task text, attached files or images,
234
- and candidate output to the configured providers and can incur charges. Use
235
- non-sensitive test tasks and conservative account limits while evaluating
236
- this alpha.
237
- - The extension queues tasks from its current pi session and compares one at a
238
- time. Candidate runs use isolated temporary copies of the original working
239
- directory and have no Pi tools by default. Their temporary changes are
240
- discarded and recorded as counts, and raw Pi JSON event streams are saved
241
- under `~/.openmerit/traces/trials/`.
242
- Candidate sessions replay the task text and images, not earlier conversation
243
- context or workspace changes.
244
- - Set `OPENMERIT_PI_TRIAL_TOOLS` to an explicit comma-separated allowlist such
245
- as `read,grep,find,ls` to let candidate models use tools; `none` keeps them
246
- disabled. Any tool access is an advanced opt-in: the copied working directory
247
- prevents ordinary project writes from touching the original, but it is not an
248
- OS sandbox. Tools may accept absolute paths, and shell tools may access the
249
- network. Keep the no-tools default for untrusted or sensitive projects.
250
- - File attachments that Pi records as `<file name="…">` are copied into every
251
- candidate sandbox and passed back to Pi as `@` file inputs. This covers PDFs,
252
- CSVs, spreadsheets, and other files that Pi can open; OpenMerit does not
253
- implement a separate parser for them.
254
- - `ledger.json` counts reported candidate, rubric, judge, and strategist spend
255
- for automatic session comparisons plus the daily candidate count.
256
- `max_usd_per_trial` is a conservative admission estimate based on known Pi
257
- prices and a 4K answer; it is not a provider-side hard cap. Routes without
258
- known pricing are excluded from automatic comparisons and cannot auto-apply.
259
- - Each comparison has one observed baseline plus a small candidate slate and
260
- one quality score per answer. Treat recommendations as experimental evidence,
261
- not a universal model ranking.
262
- - Candidate execution, judging, and strategy use the provider/model routes
263
- exposed by Pi. OpenRouter remains an optional route plus catalog/benchmark
264
- enrichment source. `HarnessAdapter`, `ModelProviderAdapter`,
265
- `ObservationSource`, and `EventSink` remain separate integration boundaries;
266
- Pi and local JSONL are the implementations shipped in this release.
267
- - Candidate subprocesses can use Pi built-ins and custom/local routes available
268
- without loading extensions (for example routes from Pi's model
269
- configuration). A provider registered only at runtime by another extension
270
- is visible in the route snapshot but cannot yet be executed by the isolated
271
- subprocess.
272
- - JSON state snapshots are replaced atomically. Append-only readers skip and
273
- report malformed or interrupted lines while retaining later valid records.
274
- A job owned by a crashed process is reclaimable instead of remaining stuck
275
- in `running`; completed jobs remain final.
69
+ ## Commands
276
70
 
277
- ## State layout (`~/.openmerit/`)
71
+ - `/openmerit` or `/openmerit status` — show evidence readiness and lifecycle stage.
72
+ - `/openmerit setup` — explicitly rerun or recover automatic setup.
73
+ - `/openmerit assess` — request an evidence-sufficiency assessment now.
74
+ - `/openmerit logs` — show the canonical audit, exporter-error, evidence, and state paths.
75
+ - `/openmerit frontier` — ask the harness to calculate a frontier now; OpenMerit verifies it.
76
+ - `/openmerit approve` — approve a verified proposal during supervised graduation.
77
+ - `/openmerit pause` and `/openmerit resume` — control proactive nudges.
78
+ - `/openmerit doctor` — show sanitized adapter diagnostics.
278
79
 
279
- | file | contents |
280
- |---|---|
281
- | `policy.json` | the policy file (gate thresholds, budgets, intervals) |
282
- | `harness-state.json` | current/fallback routes, Pi's eligible route snapshot, latest session and settled task |
283
- | `recommendations.jsonl` | append-only session-bound recommendations, routes, evidence, gate reasons, and status updates |
284
- | `trials.jsonl` | every model trial point, including its provider route when known |
285
- | `traces/observations.jsonl` | task observations extracted from session traces |
286
- | `events.jsonl` | versioned provider-neutral observation, trial, and recommendation events for future sinks |
287
- | `traces/trials/*.jsonl` | raw Pi JSON event streams for candidate trials |
288
- | `catalog/snapshot.json` + `candidates.json` | catalog snapshot (including input modalities) + new-model queue |
289
- | `benchmarks/digest.json` | public-benchmark scores per model (seed + refresh) |
290
- | `ledger.json` | daily trial spend (budget enforcement) |
291
- | `watch/processed.json` | latest task handled by optional CLI watcher |
292
- | `watch/jobs/*.json` | per-session comparison status and retry marker |
80
+ ## Documentation
293
81
 
294
- ## Benchmarks
82
+ - [Getting started](docs/getting-started.md)
83
+ - [Architecture and responsibility boundary](docs/architecture.md)
84
+ - [Metrics and evidence](docs/metrics-and-evidence.md)
85
+ - [Improvement lifecycle](docs/lifecycle.md)
86
+ - [Harness-neutral automation](docs/automation.md)
87
+ - [Pi extension guide](docs/pi-extension.md)
88
+ - [Building another harness adapter](docs/adapter-guide.md)
89
+ - [Pareto conformance specification](docs/pareto-spec.md)
90
+ - [Project state and operations](docs/operations.md)
91
+ - [Security](docs/security.md)
92
+ - [Testing](docs/testing.md)
93
+ - [Current limitations and roadmap](docs/roadmap.md)
295
94
 
296
- From a source checkout, the benchmark runners are TypeScript and execute
297
- directly on Node 22.18+; no transpilation step or Python environment is
298
- required. The slim npm artifact keeps only the pinned invoice data needed by
299
- runtime scoring; use the repository checkout for these developer commands:
95
+ ## Repository layout
300
96
 
301
- ```bash
302
- npm run benchmark:invoice -- --models google/gemini-3.8-flash
303
- npm run benchmark:invoice:round2 -- --verify-only
304
- npm run benchmark:text-to-sql -- --verify-only
305
- npm run benchmark:policy:validate
306
- npm run benchmark:policy -- --verify-only
307
- npm run benchmark:verify-recorded
308
- ```
97
+ - `packages/protocol` — harness-neutral types for intents, results, metrics, policies, and frontier evidence.
98
+ - `packages/core` — adapter SDK, durable coordinator, pluggable persistence, policy, and frontier conformance verification.
99
+ - `packages/pi` — Pi extension that turns OpenMerit intents into harness work.
100
+ - `docs` — user, operator, architecture, and normative documentation.
309
101
 
310
- - `benchmark/invoice_ocr/` — two real vision/OCR model slates over the same
311
- pinned 10-invoice dataset.
312
- - `benchmark/text_to_sql/` — executable SQLite evaluation over 10 analytical
313
- tasks.
314
- - `benchmark/policy_adjudication/` — 12 adversarial reimbursement-policy cases,
315
- including prompt-injection cases and frozen input hashes.
102
+ The workspace modules are private development boundaries. The published
103
+ package exposes `openmerit/core`, `openmerit/protocol`, and `openmerit/pi`.
316
104
 
317
- All three benchmark families preserve raw JSONL observations, audited summaries,
318
- cost/latency/quality rankings, and Pareto layers. `npm run typecheck` covers both
319
- the application and every benchmark runner. `npm run benchmark:verify-recorded`
320
- replays the checked-in raw results locally and compares them with their audited
321
- artifacts without making provider calls or writing new results.
105
+ ## Development from source
322
106
 
323
- ## Maintainer release
324
-
325
- OpenMerit follows pi's [package format](https://github.com/earendil-works/pi/blob/main/packages/coding-agent/docs/packages.md):
326
- the npm manifest has the `pi-package` keyword, a `pi.extensions` entry, and the
327
- pi API as a peer dependency. Public npm packages with that keyword are eligible
328
- for pi's package gallery; a separate approval from the pi team is not part of
329
- the standard listing flow.
330
-
331
- Before publishing a release:
332
-
333
- ```bash
334
- npm run check
335
- npm pack --dry-run
336
- npm publish --access public
107
+ ```sh
108
+ npm ci
109
+ npm run verify
110
+ ./node_modules/.bin/pi -e ./packages/pi/src/index.ts
337
111
  ```
338
112
 
339
- Then verify the actual public artifact with `npm view openmerit` and a clean
340
- `pi install npm:openmerit`. Gallery indexing may not be immediate. A featured
341
- or curated mention in pi-owned documentation is separate and would require the
342
- maintainers to accept a contribution or request.
113
+ Credentials must never be committed. The repository intentionally contains no provider or Jev API key.
@@ -0,0 +1,90 @@
1
+ import type { AnyIntentResult, AutomationPlan, AutomationSignal, CandidateAssessment, FrontierSnapshot, HarnessCapability, HarnessDescriptor, HarnessDispatchReceipt, HarnessWakeupRequest, TaskProfile, ImprovementCycleState, IntentLifecycleUpdate, IntentKind, OpenMeritIntent, OpenMeritProjectConfig, OpenMeritProjectState, ObservabilityCoverage, JsonValue, SupervisedGraduationPolicy } from "../../protocol/src/index.ts";
2
+ import type { OpenMeritStore } from "./store.ts";
3
+ export { OPENMERIT_DIRECTORY, ProjectStore, type AuditEventExporter, type OpenMeritStore, type ProjectStoreOptions, } from "./store.ts";
4
+ export interface HarnessAdapter {
5
+ readonly descriptor: HarnessDescriptor;
6
+ dispatch(intent: OpenMeritIntent): Promise<HarnessDispatchReceipt>;
7
+ scheduleWakeup?(request: HarnessWakeupRequest): Promise<HarnessDispatchReceipt>;
8
+ }
9
+ export interface IntentFactoryOptions {
10
+ readonly taskProfileId?: string;
11
+ readonly authorizationPolicyId?: string;
12
+ readonly policy?: SupervisedGraduationPolicy;
13
+ readonly id?: string;
14
+ readonly requestedAt?: string;
15
+ }
16
+ export declare function createOpenMeritIntent(kind: IntentKind, options?: IntentFactoryOptions): OpenMeritIntent;
17
+ export declare function missingHarnessCapabilities(descriptor: HarnessDescriptor, kind: IntentKind): readonly HarnessCapability[];
18
+ export declare function validateHarnessDescriptor(descriptor: HarnessDescriptor): VerificationResult;
19
+ export declare function dispatchToHarness(adapter: HarnessAdapter, intent: OpenMeritIntent): Promise<HarnessDispatchReceipt>;
20
+ export declare function startedStage(kind: IntentKind): ImprovementCycleState["stage"];
21
+ export declare function completedStage(kind: IntentKind, succeeded: boolean): ImprovementCycleState["stage"];
22
+ export interface VerificationResult {
23
+ readonly valid: boolean;
24
+ readonly errors: readonly string[];
25
+ }
26
+ export interface ResultInterpretation extends VerificationResult {
27
+ readonly nextStage?: ImprovementCycleState["stage"];
28
+ readonly config?: OpenMeritProjectConfig;
29
+ readonly cyclePatch?: Partial<ImprovementCycleState>;
30
+ }
31
+ export declare function validateObservabilityCoverage(profile: TaskProfile, coverage: ObservabilityCoverage, requireReady: boolean): readonly string[];
32
+ export declare function interpretIntentResult(intent: OpenMeritIntent, result: AnyIntentResult, cycle: ImprovementCycleState, config: OpenMeritProjectConfig): ResultInterpretation;
33
+ export type OrchestrationDecision = {
34
+ readonly action: "issue_intent";
35
+ readonly intentKind: IntentKind;
36
+ readonly reason: string;
37
+ } | {
38
+ readonly action: "await_user";
39
+ readonly reason: string;
40
+ } | {
41
+ readonly action: "observe";
42
+ readonly reason: string;
43
+ };
44
+ export declare function nextImprovementAction(state: ImprovementCycleState): OrchestrationDecision;
45
+ export declare function verifyFrontierSnapshot(profile: TaskProfile, assessments: readonly CandidateAssessment[], snapshot: FrontierSnapshot): VerificationResult;
46
+ export interface IssuedIntent {
47
+ readonly intent: OpenMeritIntent;
48
+ readonly state: OpenMeritProjectState;
49
+ readonly receipt: HarnessDispatchReceipt;
50
+ }
51
+ export interface AcceptedIntentResult {
52
+ readonly intentKind: IntentKind;
53
+ readonly state: OpenMeritProjectState;
54
+ readonly interpretation: ResultInterpretation;
55
+ readonly automation: {
56
+ readonly plan: AutomationPlan;
57
+ readonly receipt?: HarnessDispatchReceipt;
58
+ };
59
+ }
60
+ export interface ObservationResult {
61
+ readonly state: OpenMeritProjectState;
62
+ readonly assessmentDue: boolean;
63
+ }
64
+ export interface CatalogChangeResult {
65
+ readonly state: OpenMeritProjectState;
66
+ readonly changed: boolean;
67
+ readonly reassessmentDue: boolean;
68
+ }
69
+ export interface AutomationSignalResult {
70
+ readonly state: OpenMeritProjectState;
71
+ readonly duplicate: boolean;
72
+ readonly decision: OrchestrationDecision;
73
+ }
74
+ export declare function buildAutomationPlan(state: OpenMeritProjectState, descriptor: HarnessDescriptor): AutomationPlan;
75
+ /** Harness-neutral durable coordinator. Adapters transport jobs; this class owns lifecycle state. */
76
+ export declare class OpenMeritCoordinator {
77
+ private readonly store;
78
+ private readonly adapter;
79
+ constructor(store: OpenMeritStore, adapter: HarnessAdapter);
80
+ issue(kind: IntentKind): Promise<IssuedIntent>;
81
+ reconcileAutomation(now?: string): Promise<{
82
+ readonly plan: AutomationPlan;
83
+ readonly receipt?: HarnessDispatchReceipt;
84
+ }>;
85
+ acceptResult(result: AnyIntentResult): Promise<AcceptedIntentResult>;
86
+ recordLifecycle(update: IntentLifecycleUpdate): Promise<void>;
87
+ recordTaskObservation(details?: Readonly<Record<string, JsonValue>>): Promise<ObservationResult>;
88
+ recordAutomationSignal(signal: AutomationSignal): Promise<AutomationSignalResult>;
89
+ recordModelCatalogFingerprint(fingerprint: string): Promise<CatalogChangeResult>;
90
+ }