openmerit 0.1.4 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/CHANGELOG.md +26 -0
  2. package/README.md +83 -372
  3. package/dist/core/src/index.d.ts +90 -0
  4. package/dist/core/src/index.js +1137 -0
  5. package/dist/core/src/store.d.ts +35 -0
  6. package/dist/core/src/store.js +102 -0
  7. package/dist/pi/src/index.d.ts +15 -0
  8. package/dist/pi/src/index.js +423 -0
  9. package/dist/protocol/src/index.d.ts +402 -0
  10. package/dist/protocol/src/index.js +47 -0
  11. package/dist/protocol/src/schemas.d.ts +450 -0
  12. package/dist/protocol/src/schemas.js +224 -0
  13. package/docs/adapter-guide.md +172 -0
  14. package/docs/architecture.md +55 -0
  15. package/docs/automation.md +66 -0
  16. package/docs/getting-started.md +55 -0
  17. package/docs/lifecycle.md +30 -0
  18. package/docs/metrics-and-evidence.md +40 -0
  19. package/docs/operations.md +31 -0
  20. package/docs/pareto-spec.md +76 -0
  21. package/docs/pi-extension.md +44 -0
  22. package/docs/roadmap.md +26 -0
  23. package/docs/security.md +23 -0
  24. package/docs/testing.md +36 -0
  25. package/docs/ux-reference.md +32 -0
  26. package/package.json +45 -54
  27. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
  28. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  29. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
  30. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  31. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
  32. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  33. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
  34. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  35. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
  36. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  37. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
  38. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  39. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
  40. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  41. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
  42. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  43. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
  44. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  45. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
  46. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  47. package/benchmark/invoice_ocr/data/manifest.json +0 -97
  48. package/dist/benchmarks.js +0 -98
  49. package/dist/catalog.js +0 -61
  50. package/dist/cli.js +0 -188
  51. package/dist/daemon.js +0 -407
  52. package/dist/diagnostics.js +0 -227
  53. package/dist/frontier.js +0 -56
  54. package/dist/harness.js +0 -1
  55. package/dist/integrations.js +0 -19
  56. package/dist/invoice-eval.js +0 -33
  57. package/dist/invoice-score.js +0 -124
  58. package/dist/judge.js +0 -43
  59. package/dist/llm.js +0 -207
  60. package/dist/pi-config.js +0 -46
  61. package/dist/pi-trials.js +0 -373
  62. package/dist/policy.js +0 -185
  63. package/dist/providers.js +0 -1
  64. package/dist/recommend.js +0 -76
  65. package/dist/routes.js +0 -74
  66. package/dist/standalone.js +0 -224
  67. package/dist/store.js +0 -89
  68. package/dist/strategist.js +0 -68
  69. package/dist/task-input.js +0 -54
  70. package/dist/traces.js +0 -127
  71. package/dist/trials.js +0 -140
  72. package/dist/types.js +0 -2
  73. package/examples/invoice-prompt.txt +0 -19
  74. package/examples/task.example.json +0 -7
  75. package/extension/openmerit.ts +0 -947
  76. package/instructions/OPENMERIT.md +0 -63
  77. package/instructions/openmerit.policy.json +0 -37
  78. package/rules.md +0 -43
package/CHANGELOG.md ADDED
@@ -0,0 +1,26 @@
1
+ # Changelog
2
+
3
+ ## 0.1.5 — 2026-09-23
4
+
5
+ OpenMerit now ships as one npm package with a Pi extension, a reusable
6
+ coordinator, and a harness-neutral protocol.
7
+
8
+ ### Changed
9
+
10
+ - Pi issues typed, bounded work to the coding harness and checks structured
11
+ results and durable evidence before advancing the model-improvement lifecycle.
12
+ - Project state, audit events, and evidence manifests live under `.openmerit/`
13
+ in the product repository.
14
+ - Baseline assessment, candidate trials, frontier verification, supervised
15
+ swaps, post-swap verification, and rollback follow a confirmed project policy.
16
+ - The package exposes `openmerit/core`, `openmerit/protocol`, and `openmerit/pi`.
17
+ - Node.js 22.19 or newer and Pi 0.87 are required for the Pi integration.
18
+
19
+ ### Removed from the 0.1.4 package
20
+
21
+ - The background watcher, direct provider clients, standalone CLI, and
22
+ session-bound route comparison flow.
23
+ - Automatic use of the earlier home-directory state. The new project-local
24
+ state starts with a confirmed task profile and leaves older files untouched.
25
+
26
+ See [the README](README.md) for installation and the current command surface.
package/README.md CHANGED
@@ -1,402 +1,113 @@
1
- # openmerit
1
+ # OpenMerit
2
2
 
3
- An **experimental model-merit harness** for [pi](https://pi.dev). Start
4
- pi with model A, and the OpenMerit extension observes
5
- its completed trace, trials B and C sequentially through pi, compares quality,
6
- price, and latency, and recommends a model for that exact pi session. The
7
- shipped policy is supervised: you review and apply a recommendation with
8
- `/openmerit apply`. Automatic swaps are available as an explicit opt-in.
3
+ OpenMerit is a harness-neutral orchestration layer that helps a coding harness continuously find and maintain the Pareto frontier of models for a real product task.
9
4
 
10
- ```text
11
- pi task A → saved pi trace → extension trial job → pi trials B, C → scored frontier
12
- ↑ │
13
- └──── pi.setModel() ← approval/policy ← recommendation ┘
14
- ```
15
-
16
- ## How it maps to the design
17
-
18
- 1. **Instruction file + policy file** — [`instructions/OPENMERIT.md`](instructions/OPENMERIT.md)
19
- goes into the main harness's context (AGENTS.md-style);
20
- [`instructions/openmerit.policy.json`](instructions/openmerit.policy.json)
21
- (copied to `~/.openmerit/policy.json` by `openmerit init`) gates autonomy:
22
- supervised recommendations by default, optional auto-apply thresholds,
23
- budgets, and provider allow-lists. Public-benchmark relevance lives in
24
- [`src/benchmarks.ts`](src/benchmarks.ts) (task category → benchmarks that
25
- matter) plus a refreshable digest at `~/.openmerit/benchmarks/digest.json`.
26
- 2. **Trial workflow** — the pi extension starts a session-bound comparison
27
- after a supported task settles. It runs candidate trials sequentially under
28
- configured candidate-trial caps. `openmerit watch` remains an optional CLI
29
- mode. Pi's traces and JSON event stream supply results, cost,
30
- latency, and errors; a judge or known-ground-truth scorer measures quality.
31
- 3. **Per-task pareto frontier** — trials are keyed by task signature
32
- ([`src/frontier.ts`](src/frontier.ts), 3 objectives: quality ↑, price ↓,
33
- latency ↓). The aggregate across all of an agent's tasks is the union of
34
- its task frontiers (`openmerit frontier`).
35
- 4. **Model discovery** — the active Pi session's scoped models (or Pi's
36
- authenticated available-model registry when unscoped) are the candidate
37
- pool. OpenRouter is an optional route and enrichment source: its catalog can
38
- be snapshotted and its benchmark signals can improve the shortlist
39
- ([`src/catalog.ts`](src/catalog.ts) + [`src/daemon.ts`](src/daemon.ts)).
40
- 5. **Informing the main harness** — recommendations land in
41
- `~/.openmerit/recommendations.jsonl`. The pi extension
42
- ([`extension/openmerit.ts`](extension/openmerit.ts)) reports progress, polls it, and applies
43
- swaps with `pi.setModel()` — automatically when the policy gate passes,
44
- otherwise after user approval (`/openmerit`, `/openmerit apply`). The same
45
- recommendation carries the updated fallback, used when the active model
46
- starts failing.
47
-
48
- ## Install from npm
5
+ OpenMerit does not perform the intelligent work. It nudges a connected coding harness, such as Pi, with typed outcome requests. The harness understands the project, establishes evaluations and observability, gathers evidence, discovers and tests candidates, calculates the Pareto frontier, and performs an authorized model change. OpenMerit validates the returned contracts, preserves evidence and preferences, and advances the approved lifecycle.
49
6
 
50
- You need Node 22.18+, pi 0.85.1+, and at least two eligible model routes
51
- authenticated in Pi. OpenRouter is supported but not required. Install the
52
- package into pi, then initialize its policy and state:
53
-
54
- ```bash
55
- pi install npm:openmerit
56
- npx --yes openmerit init
57
- npx --yes openmerit verify
58
- pi list
59
- ```
60
-
61
- `pi install` makes the extension and its trial engine available to pi. The
62
- one-off `npx` command creates `~/.openmerit/policy.json`; install OpenMerit
63
- globally with `npm install --global openmerit` only if you also want persistent
64
- shell access to `openmerit status`, `doctor`, `verify`, `frontier`, or the optional watcher. Install
65
- the extension from only one source—remove any older copied `openmerit.ts` first
66
- so pi does not load it twice.
67
-
68
- For development from a local checkout instead, run:
69
-
70
- ```bash
71
- npm ci
72
- node dist/cli.js init
73
- pi install "$PWD"
74
- ```
7
+ The npm package `openmerit` contains the Pi extension, the reusable coordinator,
8
+ and the protocol in one install. The Pi adapter is the first integration; other
9
+ harnesses can use the same core and protocol exports.
75
10
 
76
- Keep that checkout available because pi loads a local package from its path.
11
+ ## Why OpenMerit
77
12
 
78
- ## Run OpenMerit with pi
13
+ The cheapest model, fastest model, and highest-quality model are often different. Public benchmarks are useful before product evidence exists, but they cannot establish which model is best for a particular workload. OpenMerit connects the initial choice to production evidence and repeated reassessment.
79
14
 
80
- 1. Initialize OpenMerit if you did not do so during installation:
15
+ It distinguishes:
81
16
 
82
- ```bash
83
- npx --yes openmerit init
84
- ```
17
+ - one-run observations from rate, percentile, consistency, and reliability claims;
18
+ - missing evidence from poor performance;
19
+ - feasible candidates from dominated, unresolved, or ineligible candidates;
20
+ - an automatic nudge from an intelligent action performed by the harness;
21
+ - a recommendation from an authorized, verified model swap.
85
22
 
86
- `init` creates `~/.openmerit/policy.json` if missing and **keeps an existing
87
- policy**. The shipped policy uses `"mode": "recommend"` and does not change
88
- the active model without approval.
23
+ ## Current capabilities
89
24
 
90
- 2. Authenticate the providers you want to compare in Pi, using `/login` or
91
- Pi's normal environment/model configuration. OpenMerit takes its eligible
92
- model routes, capabilities, prices, and credentials from Pi; judge and
93
- strategist calls use the same routes and do not require duplicate keys.
25
+ - Pi 0.87 extension with a single `/openmerit` command surface.
26
+ - Confirmed task profiles, objectives, constraints, sample requirements, and evaluation budgets.
27
+ - The complete initial quality, reliability, cost, latency, efficiency, and human-intervention metric catalogue.
28
+ - Baseline-first evidence collection followed by budgeted candidate discovery and controlled challenger trials.
29
+ - A normative, uncertainty-aware Pareto definition owned by OpenMerit.
30
+ - Harness-calculated frontier results independently checked by OpenMerit's conformance verifier.
31
+ - Versioned result schemas for all ten intents, reusable across harness adapters and exposed dynamically by Pi.
32
+ - Supervised graduation: early swaps require approval, every swap requires verification, and regressions can trigger rollback.
33
+ - Project-local snapshots, redacted append-only audit events, evidence manifests, and optional best-effort event exporters under `.openmerit/`.
34
+ - Model-catalog change detection that can nudge the harness to reassess a mature baseline.
35
+ - User-confirmed task-count, elapsed-time, regression, catalogue-change, and post-swap check policies.
36
+ - Idempotent harness signals, durable pending work, native wakeup planning, and explicit scheduler capability gaps.
94
37
 
95
- For example, verify one provider without printing its credential:
38
+ ## Responsibility boundary
96
39
 
97
- ```bash
98
- pi auth check --provider openai --json
99
- ```
40
+ | OpenMerit | Coding harness |
41
+ | --- | --- |
42
+ | Defines typed outcomes and Pareto conformance rules | Understands the product and user task |
43
+ | Tracks lifecycle, policy, budget, and evidence references | Builds evals and observability |
44
+ | Determines which approved nudge is due | Collects and calculates metrics |
45
+ | Verifies returned structure and frontier conformance | Discovers candidates and calculates the frontier |
46
+ | Enforces confirmation and rollback policy | Applies and verifies an authorized model change |
100
47
 
101
- Add an OpenRouter route the same way if you want its routed catalog:
48
+ Jev System One, when configured, belongs to the coding harness. Pi may use it aggressively for bounded ranking, triage, and fast judgment. OpenMerit contains no Jev client and accepts no Jev credential.
102
49
 
103
- ```bash
104
- pi auth check --provider openrouter --json
105
- ```
50
+ ## Install with Pi
106
51
 
107
- An `OPENROUTER_API_KEY` in the process environment or
108
- `~/.openmerit/.env` additionally enables OpenRouter catalog and public-
109
- benchmark enrichment. It is optional for native-provider comparisons. If
110
- using the file, protect it:
52
+ Requirements:
111
53
 
112
- ```bash
113
- nano ~/.openmerit/.env
114
- chmod 600 ~/.openmerit/.env
115
- ```
116
-
117
- Pi does not read OpenMerit's `.env` file for its ordinary sessions, so an
118
- OpenRouter route still needs Pi authentication. Older `model_search/.env`
119
- files are not read.
120
-
121
- 3. Verify the extension appears in `pi list`. The instruction file
122
- [`instructions/OPENMERIT.md`](instructions/OPENMERIT.md) can be added to a
123
- pi project's AGENTS.md for agent context, but the extension does
124
- not require it. Run `npx openmerit verify` for a provider-free core self-test,
125
- then `npx openmerit doctor` after starting Pi once to inspect the eligible
126
- route snapshot and configuration without printing secrets.
127
-
128
- 4. Start pi in one terminal with any configured model. For example:
129
-
130
- ```bash
131
- pi --provider openrouter --model openai/gpt-4o-mini
132
- ```
133
-
134
- For a first text task, ask: “Give the shortest valid word ladder from cat
135
- to dog. Each step changes one letter and must be a common English word.
136
- Return only the path.” Wait for A to finish and leave the pi session open.
137
- The extension automatically starts B and C **sequentially through their
138
- selected Pi routes**,
139
- reports each score in Pi, and writes a recommendation for this exact
140
- session. With the shipped supervised policy, use
141
- `/openmerit` inside pi to inspect the evidence and `/openmerit apply` to
142
- switch. Then send a second message to see which model actually handles it.
143
- You do not need a watcher terminal or the standalone `trial` command.
144
-
145
- Use `/openmerit pause` to stop the active comparison and suppress automatic
146
- comparisons, `/openmerit resume` to enable them for future completed tasks,
147
- and `/openmerit compare` to explicitly compare the latest completed task
148
- even while automatic comparisons are paused. `/openmerit doctor` runs the
149
- sanitized setup checks without leaving Pi.
150
-
151
- Candidate comparisons have no tools by default, even when the observed task
152
- used tools. See **Alpha boundaries** before explicitly enabling candidate
153
- tools.
154
-
155
- To opt in to automatic swaps, edit `~/.openmerit/policy.json`, set
156
- `"mode": "auto"` and `"auto_apply.enabled": true`, review the score-gain
157
- and price-ratio thresholds, and run the next task.
158
-
159
- ### Configure custom or local routes
160
-
161
- Pi normally supplies each route's price, context window, output limit, and
162
- modalities. Some custom and local providers omit that metadata. OpenMerit does
163
- not guess that a local model is free: add an override keyed by the exact
164
- `provider:modelId` route in `~/.openmerit/policy.json` instead:
165
-
166
- ```json
167
- {
168
- "route_overrides": {
169
- "ollama:qwen3:8b": {
170
- "cost": { "input": 0, "output": 0 },
171
- "context_window": 32768,
172
- "max_tokens": 4096,
173
- "input": ["text"]
174
- }
175
- }
176
- }
177
- ```
54
+ - Node.js 22.19 or newer
55
+ - Pi 0.87 with a supported model provider configured
178
56
 
179
- All fields are optional, but both `cost.input` and `cost.output` are required
180
- when declaring cost. Prices use Pi's dollars-per-million-token units. The
181
- override applies only to that exact route; it does not change the stable
182
- `vendor/model` identity or another provider's route to the same model.
183
-
184
- If the provider itself is registered at runtime by a Pi extension, explicitly
185
- allow that provider-registration file in the same policy. Paths must be
186
- absolute, existing files:
187
-
188
- ```json
189
- {
190
- "pi": {
191
- "provider_extensions": [
192
- "/absolute/path/to/ollama-provider.ts"
193
- ]
194
- }
195
- }
196
- ```
197
-
198
- OpenMerit still launches subprocesses with extension discovery disabled, then
199
- loads only these explicit files. Do not add OpenMerit's own extension. An
200
- allowlisted extension executes code in every candidate, judge, and strategist
201
- Pi subprocess, so list only provider extensions you trust. Run
202
- `npx openmerit doctor` or `/openmerit doctor` to validate the files and confirm
203
- that route overrides match Pi's visible routes.
204
-
205
- Existing `0.1.x` policy files remain valid: missing `route_overrides` and
206
- `pi.provider_extensions` fields normalize to empty safe defaults. `openmerit init`
207
- continues to preserve an existing policy, so add these fields manually
208
- only when you need them.
209
-
210
- ### Try an invoice-to-JSON task
211
-
212
- Use the same one-terminal setup. Start pi with a vision-capable model, for
213
- example `openai/gpt-4o-mini`. In pi, type `@` to select
214
- [`benchmark/invoice_ocr/data/invoice_01_row_2.jpg`](benchmark/invoice_ocr/data/invoice_01_row_2.jpg)
215
- and paste the **entire, unchanged** text from
216
- [`examples/invoice-prompt.txt`](examples/invoice-prompt.txt) into the same
217
- message. Pi also accepts pasted or dragged images. Wait for the JSON answer;
218
- keep pi open while the extension trials two other vision models. OpenMerit uses
219
- the existing benchmark's visibility-audited exact-field scorer when the saved
220
- task contains the exact example prompt and original bytes of a pinned JPEG.
221
- Pi may resize an attached image or add a file header to the prompt; in that
222
- case, the run uses the vision judge that sees the image and answer. Other
223
- uploaded invoices also use that judge because they have no ground truth.
224
-
225
- A real pinned-invoice run produced this trial output (models and scores will
226
- vary between runs):
227
-
228
- ```text
229
- [openmerit] task 0e7642522dc5: A=openai/gpt-4o-mini score=1.00 from pi trace
230
- [openmerit] task 0e7642522dc5: google/gemma-3-12b-it score=0.31 cost=$0.0001
231
- [openmerit] task 0e7642522dc5: mistralai/ministral-14b-2512 score=1.00 cost=$0.0007
232
- [openmerit] task 0e7642522dc5: selected mistralai/ministral-14b-2512; auto=false
233
- ```
234
-
235
- Pi candidate runs rely on the explicit invoice prompt for JSON structure.
236
- The standalone invoice benchmark sends a strict OpenRouter `response_format`
237
- schema, so its published scores are not directly comparable to these pi runs.
238
-
239
- ### Inspect or troubleshoot a run
240
-
241
- `npx openmerit doctor` checks Pi, policy, eligible routes and prices, saved
242
- session state, duplicate package sources, append-only files, and optional
243
- OpenRouter enrichment. Add `--json` for a sanitized diagnostic report suitable
244
- for a bug report; it contains no credentials, prompts, or trace contents.
245
- `npx openmerit verify` runs an offline self-test of atomic state, JSONL recovery,
246
- route preservation, policy evidence, and neutral events without contacting a
247
- provider. `npx openmerit status` shows the latest pi model and pending recommendations;
248
- `npx openmerit frontier` shows measured quality, blended price, latency, and
249
- the chosen frontier per task. `/openmerit` inside pi shows the current model,
250
- fallback, the model currently being compared, completed models with quality,
251
- cost, and latency, trial budget, pending recommendations, and any exact-task result
252
- measured in another session. When the gate declines an automatic swap, it
253
- prints reasons and `/openmerit apply` remains available. It also lists every
254
- route skipped during the current comparison with the relevant policy, pricing,
255
- modality, or per-trial budget reason.
256
-
257
- The daily trial-count and dollar limits come from `~/.openmerit/policy.json`.
258
- If a limit is reached, the extension reports why it skipped the comparison;
259
- the count resets at midnight UTC. You may raise `budgets.max_trials_per_day`
260
- for local experiments while keeping `budgets.max_usd_per_day` as a conservative
261
- candidate-spend threshold.
262
-
263
- If no comparison starts after Pi settles, confirm that pi loaded the extension
264
- (`pi list`), the task finished, and the pi session is saved (do not use
265
- `--no-session`). Image candidates must advertise image input in Pi's model
266
- registry. The extension queues
267
- completed tasks from its current session and runs one comparison at a time.
268
- Closing or switching the session cancels the active job.
269
- Candidate runs use the same text and uploaded image bytes, but they do not
270
- replay earlier answers or file changes. An exact task in another session
271
- (including identical image bytes) appears as **advice**, not a pending swap;
272
- the new session still gets its own comparison.
273
-
274
- The optional `trial` command runs controlled text-task comparisons through the
275
- same exact Pi provider routes and credential store as the automatic session
276
- path:
277
-
278
- ```bash
279
- npx openmerit trial examples/task.example.json --rounds 3
57
+ ```sh
58
+ pi install npm:openmerit
59
+ pi list
280
60
  ```
281
61
 
282
- `initial_model` is the stable `vendor/model` identity. When Pi exposes that
283
- model through more than one provider, set `initial_route` to
284
- `provider:model-id` (for example `openai:gpt-4o-mini` or
285
- `openrouter:openai/gpt-4o-mini`). Judge and strategist calls also run through
286
- Pi. An OpenRouter key only adds optional catalog and public-benchmark metadata.
287
-
288
- ## Alpha boundaries
289
-
290
- - With the extension installed and at least two eligible Pi routes, each
291
- supported settled task can start comparison calls automatically. Candidate,
292
- judge, and strategist requests send the task text, attached files or images,
293
- and candidate output to the configured providers and can incur charges. Use
294
- non-sensitive test tasks and conservative account limits while evaluating
295
- this alpha.
296
- - The extension queues tasks from its current pi session and compares one at a
297
- time. Candidate runs use isolated temporary copies of the original working
298
- directory and have no Pi tools by default. Their temporary changes are
299
- discarded and recorded as counts, and raw Pi JSON event streams are saved
300
- under `~/.openmerit/traces/trials/`.
301
- Candidate sessions replay the task text and images, not earlier conversation
302
- context or workspace changes.
303
- - Set `OPENMERIT_PI_TRIAL_TOOLS` to an explicit comma-separated allowlist such
304
- as `read,grep,find,ls` to let candidate models use tools; `none` keeps them
305
- disabled. Any tool access is an advanced opt-in: the copied working directory
306
- prevents ordinary project writes from touching the original, but it is not an
307
- OS sandbox. Tools may accept absolute paths, and shell tools may access the
308
- network. Keep the no-tools default for untrusted or sensitive projects.
309
- - File attachments that Pi records as `<file name="…">` are copied into every
310
- candidate sandbox and passed back to Pi as `@` file inputs. This covers PDFs,
311
- CSVs, spreadsheets, and other files that Pi can open; OpenMerit does not
312
- implement a separate parser for them.
313
- - `ledger.json` counts reported candidate, rubric, judge, and strategist spend
314
- for automatic session comparisons plus the daily candidate count.
315
- `max_usd_per_trial` is a conservative admission estimate based on known Pi
316
- prices and a 4K answer; it is not a provider-side hard cap. Routes without
317
- known pricing are excluded as candidates and cannot auto-apply. Use an exact
318
- `route_overrides` entry for a custom/local route whose price is known; zero
319
- cost must be stated explicitly.
320
- - Each comparison has one observed baseline plus a small candidate slate and
321
- one quality score per answer. Treat recommendations as experimental evidence,
322
- not a universal model ranking.
323
- - Candidate execution, judging, and strategy use the provider/model routes
324
- exposed by Pi. OpenRouter remains an optional route plus catalog/benchmark
325
- enrichment source. `HarnessAdapter`, `ModelProviderAdapter`,
326
- `ObservationSource`, and `EventSink` remain separate integration boundaries;
327
- Pi and local JSONL are the implementations shipped in this release.
328
- - Candidate subprocesses can use Pi built-ins, configured custom/local routes,
329
- and providers registered by explicitly allowlisted extension files. Normal
330
- extension discovery remains disabled, and OpenMerit refuses to load its own
331
- extension recursively.
332
- - JSON state snapshots are replaced atomically. Append-only readers skip and
333
- report malformed or interrupted lines while retaining later valid records.
334
- A job owned by a crashed process is reclaimable instead of remaining stuck
335
- in `running`; completed jobs remain final.
62
+ When Pi enters an unconfigured product with an interactive UI, OpenMerit automatically issues the setup nudge. Pi infers a task profile, asks the user to confirm requirements and an evaluation budget, establishes task-specific eval and observability artifacts, verifies them, and reports evidence to OpenMerit. `/openmerit setup` remains available as an explicit recovery or rerun command.
336
63
 
337
- ## State layout (`~/.openmerit/`)
64
+ Run `/openmerit doctor` inside Pi to check the installation. OpenMerit keeps
65
+ project evidence under `.openmerit/` in the product repository. Version 0.1.5
66
+ replaces the 0.1.4 background watcher and standalone CLI with this
67
+ harness-driven lifecycle; it does not migrate the earlier home-directory state.
338
68
 
339
- | file | contents |
340
- |---|---|
341
- | `policy.json` | gate thresholds, budgets, intervals, exact-route metadata overrides, and allowlisted Pi provider extensions |
342
- | `harness-state.json` | current/fallback routes, Pi's eligible route snapshot, pause state, latest session and settled task |
343
- | `recommendations.jsonl` | append-only session-bound recommendations, routes, evidence, gate reasons, and status updates |
344
- | `trials.jsonl` | every model trial point, including its provider route when known |
345
- | `traces/observations.jsonl` | task observations extracted from session traces |
346
- | `events.jsonl` | versioned provider-neutral observation, trial, and recommendation events for future sinks |
347
- | `traces/trials/*.jsonl` | raw Pi JSON event streams for candidate trials |
348
- | `catalog/snapshot.json` + `candidates.json` | catalog snapshot (including input modalities) + new-model queue |
349
- | `benchmarks/digest.json` | public-benchmark scores per model (seed + refresh) |
350
- | `ledger.json` | daily trial spend (budget enforcement) |
351
- | `watch/processed.json` | latest task handled by optional CLI watcher |
352
- | `watch/jobs/*.json` | per-session comparison status and retry marker |
69
+ ## Commands
353
70
 
354
- ## Benchmarks
71
+ - `/openmerit` or `/openmerit status` — show evidence readiness and lifecycle stage.
72
+ - `/openmerit setup` — explicitly rerun or recover automatic setup.
73
+ - `/openmerit assess` — request an evidence-sufficiency assessment now.
74
+ - `/openmerit logs` — show the canonical audit, exporter-error, evidence, and state paths.
75
+ - `/openmerit frontier` — ask the harness to calculate a frontier now; OpenMerit verifies it.
76
+ - `/openmerit approve` — approve a verified proposal during supervised graduation.
77
+ - `/openmerit pause` and `/openmerit resume` — control proactive nudges.
78
+ - `/openmerit doctor` — show sanitized adapter diagnostics.
355
79
 
356
- From a source checkout, the benchmark runners are TypeScript and execute
357
- directly on Node 22.18+; no transpilation step or Python environment is
358
- required. The slim npm artifact keeps only the pinned invoice data needed by
359
- runtime scoring; use the repository checkout for these developer commands:
80
+ ## Documentation
360
81
 
361
- ```bash
362
- npm run benchmark:invoice -- --models google/gemini-3.8-flash
363
- npm run benchmark:invoice:round2 -- --verify-only
364
- npm run benchmark:text-to-sql -- --verify-only
365
- npm run benchmark:policy:validate
366
- npm run benchmark:policy -- --verify-only
367
- npm run benchmark:verify-recorded
368
- ```
369
-
370
- - `benchmark/invoice_ocr/` — two real vision/OCR model slates over the same
371
- pinned 10-invoice dataset.
372
- - `benchmark/text_to_sql/` — executable SQLite evaluation over 10 analytical
373
- tasks.
374
- - `benchmark/policy_adjudication/` — 12 adversarial reimbursement-policy cases,
375
- including prompt-injection cases and frozen input hashes.
82
+ - [Getting started](docs/getting-started.md)
83
+ - [Architecture and responsibility boundary](docs/architecture.md)
84
+ - [Metrics and evidence](docs/metrics-and-evidence.md)
85
+ - [Improvement lifecycle](docs/lifecycle.md)
86
+ - [Harness-neutral automation](docs/automation.md)
87
+ - [Pi extension guide](docs/pi-extension.md)
88
+ - [Building another harness adapter](docs/adapter-guide.md)
89
+ - [Pareto conformance specification](docs/pareto-spec.md)
90
+ - [Project state and operations](docs/operations.md)
91
+ - [Security](docs/security.md)
92
+ - [Testing](docs/testing.md)
93
+ - [Current limitations and roadmap](docs/roadmap.md)
376
94
 
377
- All three benchmark families preserve raw JSONL observations, audited summaries,
378
- cost/latency/quality rankings, and Pareto layers. `npm run typecheck` covers both
379
- the application and every benchmark runner. `npm run benchmark:verify-recorded`
380
- replays the checked-in raw results locally and compares them with their audited
381
- artifacts without making provider calls or writing new results.
95
+ ## Repository layout
382
96
 
383
- ## Maintainer release
97
+ - `packages/protocol` — harness-neutral types for intents, results, metrics, policies, and frontier evidence.
98
+ - `packages/core` — adapter SDK, durable coordinator, pluggable persistence, policy, and frontier conformance verification.
99
+ - `packages/pi` — Pi extension that turns OpenMerit intents into harness work.
100
+ - `docs` — user, operator, architecture, and normative documentation.
384
101
 
385
- OpenMerit follows pi's [package format](https://github.com/earendil-works/pi/blob/main/packages/coding-agent/docs/packages.md):
386
- the npm manifest has the `pi-package` keyword, a `pi.extensions` entry, and the
387
- pi API as a peer dependency. Public npm packages with that keyword are eligible
388
- for pi's package gallery; a separate approval from the pi team is not part of
389
- the standard listing flow.
102
+ The workspace modules are private development boundaries. The published
103
+ package exposes `openmerit/core`, `openmerit/protocol`, and `openmerit/pi`.
390
104
 
391
- Before publishing a release:
105
+ ## Development from source
392
106
 
393
- ```bash
394
- npm run check
395
- npm pack --dry-run
396
- npm publish --access public
107
+ ```sh
108
+ npm ci
109
+ npm run verify
110
+ ./node_modules/.bin/pi -e ./packages/pi/src/index.ts
397
111
  ```
398
112
 
399
- Then verify the actual public artifact with `npm view openmerit` and a clean
400
- `pi install npm:openmerit`. Gallery indexing may not be immediate. A featured
401
- or curated mention in pi-owned documentation is separate and would require the
402
- maintainers to accept a contribution or request.
113
+ Credentials must never be committed. The repository intentionally contains no provider or Jev API key.
@@ -0,0 +1,90 @@
1
+ import type { AnyIntentResult, AutomationPlan, AutomationSignal, CandidateAssessment, FrontierSnapshot, HarnessCapability, HarnessDescriptor, HarnessDispatchReceipt, HarnessWakeupRequest, TaskProfile, ImprovementCycleState, IntentLifecycleUpdate, IntentKind, OpenMeritIntent, OpenMeritProjectConfig, OpenMeritProjectState, ObservabilityCoverage, JsonValue, SupervisedGraduationPolicy } from "../../protocol/src/index.ts";
2
+ import type { OpenMeritStore } from "./store.ts";
3
+ export { OPENMERIT_DIRECTORY, ProjectStore, type AuditEventExporter, type OpenMeritStore, type ProjectStoreOptions, } from "./store.ts";
4
+ export interface HarnessAdapter {
5
+ readonly descriptor: HarnessDescriptor;
6
+ dispatch(intent: OpenMeritIntent): Promise<HarnessDispatchReceipt>;
7
+ scheduleWakeup?(request: HarnessWakeupRequest): Promise<HarnessDispatchReceipt>;
8
+ }
9
+ export interface IntentFactoryOptions {
10
+ readonly taskProfileId?: string;
11
+ readonly authorizationPolicyId?: string;
12
+ readonly policy?: SupervisedGraduationPolicy;
13
+ readonly id?: string;
14
+ readonly requestedAt?: string;
15
+ }
16
+ export declare function createOpenMeritIntent(kind: IntentKind, options?: IntentFactoryOptions): OpenMeritIntent;
17
+ export declare function missingHarnessCapabilities(descriptor: HarnessDescriptor, kind: IntentKind): readonly HarnessCapability[];
18
+ export declare function validateHarnessDescriptor(descriptor: HarnessDescriptor): VerificationResult;
19
+ export declare function dispatchToHarness(adapter: HarnessAdapter, intent: OpenMeritIntent): Promise<HarnessDispatchReceipt>;
20
+ export declare function startedStage(kind: IntentKind): ImprovementCycleState["stage"];
21
+ export declare function completedStage(kind: IntentKind, succeeded: boolean): ImprovementCycleState["stage"];
22
+ export interface VerificationResult {
23
+ readonly valid: boolean;
24
+ readonly errors: readonly string[];
25
+ }
26
+ export interface ResultInterpretation extends VerificationResult {
27
+ readonly nextStage?: ImprovementCycleState["stage"];
28
+ readonly config?: OpenMeritProjectConfig;
29
+ readonly cyclePatch?: Partial<ImprovementCycleState>;
30
+ }
31
+ export declare function validateObservabilityCoverage(profile: TaskProfile, coverage: ObservabilityCoverage, requireReady: boolean): readonly string[];
32
+ export declare function interpretIntentResult(intent: OpenMeritIntent, result: AnyIntentResult, cycle: ImprovementCycleState, config: OpenMeritProjectConfig): ResultInterpretation;
33
+ export type OrchestrationDecision = {
34
+ readonly action: "issue_intent";
35
+ readonly intentKind: IntentKind;
36
+ readonly reason: string;
37
+ } | {
38
+ readonly action: "await_user";
39
+ readonly reason: string;
40
+ } | {
41
+ readonly action: "observe";
42
+ readonly reason: string;
43
+ };
44
+ export declare function nextImprovementAction(state: ImprovementCycleState): OrchestrationDecision;
45
+ export declare function verifyFrontierSnapshot(profile: TaskProfile, assessments: readonly CandidateAssessment[], snapshot: FrontierSnapshot): VerificationResult;
46
+ export interface IssuedIntent {
47
+ readonly intent: OpenMeritIntent;
48
+ readonly state: OpenMeritProjectState;
49
+ readonly receipt: HarnessDispatchReceipt;
50
+ }
51
+ export interface AcceptedIntentResult {
52
+ readonly intentKind: IntentKind;
53
+ readonly state: OpenMeritProjectState;
54
+ readonly interpretation: ResultInterpretation;
55
+ readonly automation: {
56
+ readonly plan: AutomationPlan;
57
+ readonly receipt?: HarnessDispatchReceipt;
58
+ };
59
+ }
60
+ export interface ObservationResult {
61
+ readonly state: OpenMeritProjectState;
62
+ readonly assessmentDue: boolean;
63
+ }
64
+ export interface CatalogChangeResult {
65
+ readonly state: OpenMeritProjectState;
66
+ readonly changed: boolean;
67
+ readonly reassessmentDue: boolean;
68
+ }
69
+ export interface AutomationSignalResult {
70
+ readonly state: OpenMeritProjectState;
71
+ readonly duplicate: boolean;
72
+ readonly decision: OrchestrationDecision;
73
+ }
74
+ export declare function buildAutomationPlan(state: OpenMeritProjectState, descriptor: HarnessDescriptor): AutomationPlan;
75
+ /** Harness-neutral durable coordinator. Adapters transport jobs; this class owns lifecycle state. */
76
+ export declare class OpenMeritCoordinator {
77
+ private readonly store;
78
+ private readonly adapter;
79
+ constructor(store: OpenMeritStore, adapter: HarnessAdapter);
80
+ issue(kind: IntentKind): Promise<IssuedIntent>;
81
+ reconcileAutomation(now?: string): Promise<{
82
+ readonly plan: AutomationPlan;
83
+ readonly receipt?: HarnessDispatchReceipt;
84
+ }>;
85
+ acceptResult(result: AnyIntentResult): Promise<AcceptedIntentResult>;
86
+ recordLifecycle(update: IntentLifecycleUpdate): Promise<void>;
87
+ recordTaskObservation(details?: Readonly<Record<string, JsonValue>>): Promise<ObservationResult>;
88
+ recordAutomationSignal(signal: AutomationSignal): Promise<AutomationSignalResult>;
89
+ recordModelCatalogFingerprint(fingerprint: string): Promise<CatalogChangeResult>;
90
+ }