openmerit 0.1.3 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/CHANGELOG.md +26 -0
  2. package/README.md +83 -312
  3. package/dist/core/src/index.d.ts +90 -0
  4. package/dist/core/src/index.js +1137 -0
  5. package/dist/core/src/store.d.ts +35 -0
  6. package/dist/core/src/store.js +102 -0
  7. package/dist/pi/src/index.d.ts +15 -0
  8. package/dist/pi/src/index.js +423 -0
  9. package/dist/protocol/src/index.d.ts +402 -0
  10. package/dist/protocol/src/index.js +47 -0
  11. package/dist/protocol/src/schemas.d.ts +450 -0
  12. package/dist/protocol/src/schemas.js +224 -0
  13. package/docs/adapter-guide.md +172 -0
  14. package/docs/architecture.md +55 -0
  15. package/docs/automation.md +66 -0
  16. package/docs/getting-started.md +55 -0
  17. package/docs/lifecycle.md +30 -0
  18. package/docs/metrics-and-evidence.md +40 -0
  19. package/docs/operations.md +31 -0
  20. package/docs/pareto-spec.md +76 -0
  21. package/docs/pi-extension.md +44 -0
  22. package/docs/roadmap.md +26 -0
  23. package/docs/security.md +23 -0
  24. package/docs/testing.md +36 -0
  25. package/docs/ux-reference.md +32 -0
  26. package/package.json +45 -54
  27. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
  28. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  29. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
  30. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  31. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
  32. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  33. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
  34. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  35. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
  36. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  37. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
  38. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  39. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
  40. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  41. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
  42. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  43. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
  44. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  45. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
  46. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  47. package/benchmark/invoice_ocr/data/manifest.json +0 -97
  48. package/dist/benchmarks.js +0 -98
  49. package/dist/catalog.js +0 -61
  50. package/dist/cli.js +0 -188
  51. package/dist/daemon.js +0 -388
  52. package/dist/diagnostics.js +0 -194
  53. package/dist/frontier.js +0 -56
  54. package/dist/harness.js +0 -1
  55. package/dist/integrations.js +0 -19
  56. package/dist/invoice-eval.js +0 -33
  57. package/dist/invoice-score.js +0 -124
  58. package/dist/judge.js +0 -43
  59. package/dist/llm.js +0 -203
  60. package/dist/pi-trials.js +0 -366
  61. package/dist/policy.js +0 -115
  62. package/dist/providers.js +0 -1
  63. package/dist/recommend.js +0 -76
  64. package/dist/routes.js +0 -59
  65. package/dist/standalone.js +0 -220
  66. package/dist/store.js +0 -89
  67. package/dist/strategist.js +0 -68
  68. package/dist/task-input.js +0 -54
  69. package/dist/traces.js +0 -127
  70. package/dist/trials.js +0 -140
  71. package/dist/types.js +0 -2
  72. package/examples/invoice-prompt.txt +0 -19
  73. package/examples/task.example.json +0 -7
  74. package/extension/openmerit.ts +0 -820
  75. package/instructions/OPENMERIT.md +0 -54
  76. package/instructions/openmerit.policy.json +0 -33
  77. package/rules.md +0 -39
@@ -0,0 +1,172 @@
1
+ # Building a harness adapter
2
+
3
+ OpenMerit adapters are transport bridges. They do not reproduce OpenMerit's lifecycle, validation, persistence, or Pareto rules, and OpenMerit does not reproduce the harness's intelligence.
4
+
5
+ ## Required components
6
+
7
+ An adapter provides a `HarnessAdapter` with:
8
+
9
+ - a stable harness descriptor;
10
+ - the OpenMerit protocol versions it supports;
11
+ - declared capabilities;
12
+ - supported execution modes;
13
+ - a `dispatch` method that accepts a canonical `OpenMeritIntent` and returns an acceptance receipt.
14
+
15
+ ```ts
16
+ import {
17
+ OpenMeritCoordinator,
18
+ ProjectStore,
19
+ type HarnessAdapter,
20
+ } from "openmerit/core";
21
+ import { PROTOCOL_VERSION } from "openmerit/protocol";
22
+
23
+ const adapter: HarnessAdapter = {
24
+ descriptor: {
25
+ id: "my-harness",
26
+ name: "My coding harness",
27
+ version: "1.0.0",
28
+ protocolVersions: [PROTOCOL_VERSION],
29
+ capabilities: [
30
+ "user_confirmation",
31
+ "evaluation_authoring",
32
+ "observability_instrumentation",
33
+ "production_observation",
34
+ "candidate_discovery",
35
+ "challenger_execution",
36
+ "frontier_calculation",
37
+ "model_mutation",
38
+ "post_swap_verification",
39
+ "rollback",
40
+ ],
41
+ executionModes: ["interactive"],
42
+ },
43
+ async dispatch(intent) {
44
+ await myHarness.enqueue(intent);
45
+ return { accepted: true, harnessJobId: intent.id };
46
+ },
47
+ };
48
+
49
+ const coordinator = new OpenMeritCoordinator(
50
+ new ProjectStore(process.cwd()),
51
+ adapter,
52
+ );
53
+ ```
54
+
55
+ `ProjectStore` is the included local-filesystem backend. A remote or hosted integration can implement the `OpenMeritStore` interface instead.
56
+
57
+ ## External telemetry export
58
+
59
+ OpenMerit always writes its canonical, redacted audit stream locally. A host integration may pass `eventExporters` to `ProjectStore` to mirror those same safe events to OpenTelemetry, Langfuse, LangSmith, or another observability backend:
60
+
61
+ ```ts
62
+ const store = new ProjectStore(projectRoot, {
63
+ eventExporters: [{
64
+ id: "my-otel-bridge",
65
+ async export(event) {
66
+ await telemetry.emit("openmerit.audit", event);
67
+ },
68
+ }],
69
+ });
70
+ ```
71
+
72
+ Exporters are not the durable source of truth. They are best effort, and failures are recorded locally without blocking OpenMerit decisions.
73
+
74
+ ## Dispatching work
75
+
76
+ ```ts
77
+ const issued = await coordinator.issue("establish_evals");
78
+ ```
79
+
80
+ The coordinator builds the canonical intent, checks the descriptor's protocol and capabilities, creates durable state, records audit events, and calls the adapter. The harness decides how to execute the job.
81
+
82
+ If a required capability is absent, dispatch is rejected before harness work starts.
83
+
84
+ ## Reporting progress
85
+
86
+ A harness may send durable lifecycle progress through `recordLifecycle`:
87
+
88
+ ```ts
89
+ await coordinator.recordLifecycle({
90
+ protocolVersion: PROTOCOL_VERSION,
91
+ intentId: issued.intent.id,
92
+ status: "running",
93
+ occurredAt: new Date().toISOString(),
94
+ message: "Generating representative evaluation cases",
95
+ });
96
+ ```
97
+
98
+ Use `waiting_for_user` when an interactive confirmation is required but cannot be completed in the current execution mode.
99
+
100
+ ## Completing work
101
+
102
+ Each intent kind has a typed output in `IntentOutputMap`. Return an `IntentResult<K>` with durable evidence references, then submit it to the coordinator:
103
+
104
+ ```ts
105
+ await coordinator.acceptResult(result);
106
+ ```
107
+
108
+ The coordinator validates the active intent, version, evidence, intent-specific output, frontier conformance, mutation authorization, and lifecycle transition. Adapters must not duplicate this logic.
109
+
110
+ ### Present the canonical result schema
111
+
112
+ Import `completionToolSchemaFor(intent.kind)` from `openmerit/protocol` and present it through the harness's strongest structured-output mechanism. Do not recreate result schemas inside an adapter. A harness that supports changing tool definitions should specialize the completion tool whenever an intent becomes active; other harnesses may expose intent-specific tools or perform adapter-side validation.
113
+
114
+ Declare the behavior in `HarnessDescriptor.structuredOutput`:
115
+
116
+ - `schemaDialect` identifies the canonical OpenMerit JSON Schema dialect;
117
+ - `presentation` states whether the adapter uses a dynamic tool, a static union, or adapter validation;
118
+ - `enforcement` states whether validation occurs at the provider and harness, harness only, or adapter only.
119
+
120
+ Schema validation catches malformed structure. The coordinator still owns semantic checks such as authorized candidate identity, evidence sufficiency, lifecycle consistency, and Pareto conformance.
121
+
122
+ ## Reporting normal work and catalogue changes
123
+
124
+ After a real product task completes:
125
+
126
+ ```ts
127
+ const observation = await coordinator.recordTaskObservation({
128
+ modelId: "vendor/model",
129
+ });
130
+
131
+ if (observation.assessmentDue) {
132
+ await coordinator.issue("run_assessment");
133
+ }
134
+ ```
135
+
136
+ When the harness's available model catalogue changes, compute a stable fingerprint and call `recordModelCatalogFingerprint`. The returned `reassessmentDue` flag tells the adapter whether the lifecycle justifies candidate discovery.
137
+
138
+ New adapters should use `recordAutomationSignal` for product tasks, metric windows, scheduled ticks, catalogue observations, and verification windows. Signal IDs must remain stable across delivery retries. If the returned decision issues an intent, dispatch it through the coordinator. `recordTaskObservation` and `recordModelCatalogFingerprint` remain compatibility helpers.
139
+
140
+ Call `reconcileAutomation` after configuration changes or when restoring a project. It returns an `AutomationPlan`. Implement `scheduleWakeup` only when the harness can genuinely arrange the requested wakeup. Declare persistent scheduling, background execution, supported signal sources, and provisioning requirements accurately in `HarnessDescriptor.automation`.
141
+
142
+ If persistent scheduling is unavailable, expose the plan's active-session or external-scheduler fallback. Do not report a seven-day schedule as configured merely because the adapter checks policy when an interactive session starts.
143
+
144
+ ## Capability mapping
145
+
146
+ | Intent | Required capability |
147
+ | --- | --- |
148
+ | `establish_evals` | confirmation, evaluation authoring, observability instrumentation |
149
+ | `instrument_observability` | observability instrumentation |
150
+ | `run_assessment` | production observation |
151
+ | `discover_candidates` | candidate discovery |
152
+ | `run_challenger_trials` | challenger execution |
153
+ | `calculate_frontier` | frontier calculation |
154
+ | `investigate_regression` | production observation |
155
+ | `apply_model_swap` | model mutation |
156
+ | `verify_model_swap` | post-swap verification |
157
+ | `rollback_model_swap` | rollback |
158
+
159
+ ## Conformance checklist
160
+
161
+ An adapter is ready when it:
162
+
163
+ 1. advertises only capabilities it can actually execute;
164
+ 2. preserves intent IDs and protocol versions;
165
+ 3. never reports success without durable evidence;
166
+ 4. returns the typed output for every supported intent;
167
+ 5. performs model mutation only when the intent explicitly authorizes it;
168
+ 6. reports ordinary product tasks without counting OpenMerit orchestration turns;
169
+ 7. passes a durable issue-and-complete cycle using `OpenMeritCoordinator`;
170
+ 8. proves the requested side effect, not merely a generated description.
171
+
172
+ The fake non-Pi adapter in `packages/core/test/adapter-conformance.test.ts` is the executable reference.
@@ -0,0 +1,55 @@
1
+ # Architecture
2
+
3
+ ## Design principle
4
+
5
+ OpenMerit is an orchestrator, not an intelligence layer.
6
+
7
+ ```text
8
+ User or lifecycle event
9
+ |
10
+ v
11
+ OpenMerit deterministic orchestration
12
+ |
13
+ | typed intent: outcome, constraints, evidence
14
+ v
15
+ Coding harness such as Pi
16
+ |
17
+ | intelligent work, tools, evals, calculation
18
+ v
19
+ Typed result and durable evidence references
20
+ |
21
+ v
22
+ OpenMerit validation, policy, persistence, next nudge
23
+ ```
24
+
25
+ An automatic OpenMerit action means automatic issuance of an approved intent. OpenMerit does not perform candidate research, evaluate quality, calculate a frontier, or mutate model configuration.
26
+
27
+ ## Packages
28
+
29
+ ### `openmerit/protocol`
30
+
31
+ Defines the JSON-safe boundary: durable intents, evidence references, metrics, task profiles, candidate assessments, frontier snapshots, budgets, policies, state, and audit events.
32
+
33
+ It also owns the versioned runtime schema registry for all intent results. Harness adapters consume these schemas rather than defining their own result shapes, report their structured-output capabilities, and may present the active schema using harness-native tools or adapter validation.
34
+
35
+ ### `openmerit/core`
36
+
37
+ Contains deterministic behavior: choosing the next lifecycle intent, persisting state, checking harness-produced frontiers against the normative definition, and enforcing confirmation or rollback policy. The verifier rejects invalid harness results; it is not a model-selection engine.
38
+
39
+ `OpenMeritCoordinator` is the reusable adapter boundary. It constructs canonical intents, negotiates capabilities, records lifecycle progress and observations, validates results, and advances durable state. `OpenMeritStore` makes persistence replaceable; `ProjectStore` is the bundled filesystem backend.
40
+
41
+ The coordinator also accepts idempotent automation signals and evaluates the user-confirmed `AutomationCheckPolicy`. `buildAutomationPlan` tells an adapter which signals and wakeups are required and reports unsupported persistent scheduling honestly. OpenMerit never installs a scheduler or creates a paid cloud worker itself.
42
+
43
+ ### `openmerit/pi`
44
+
45
+ Exposes the command surface, supplies private typed intent context, triggers Pi, accepts one structured completion for the active intent, records evidence, advances the lifecycle, and detects model-catalog changes.
46
+
47
+ The Pi package implements `HarnessAdapter`. It does not define canonical intents, per-intent result contracts, stage transitions, or validation rules.
48
+
49
+ ## Intelligence resources
50
+
51
+ Models, web research, tools, evaluators, and fast decision systems such as Jev System One are harness resources. They are configured and invoked by Pi. This keeps OpenMerit provider-neutral and lets future adapters choose their own implementation.
52
+
53
+ ## Trust boundary
54
+
55
+ The harness is trusted to perform the work and return honest evidence. OpenMerit reduces unsupported claims through structured outputs, evidence references, compatible profile revisions, complete comparisons, and conservative uncertainty handling. It cannot prove that an arbitrary external evidence URI contains truthful data.
@@ -0,0 +1,66 @@
1
+ # Harness-neutral automation
2
+
3
+ The user defines cadence, thresholds, budget, scheduling mode, and automation permission. OpenMerit stores that policy, evaluates durable signals, and decides whether a bounded intent is due. A harness adapter arranges native wakeups and reports events. The coding harness performs the resulting intelligent work.
4
+
5
+ ```text
6
+ User-confirmed policy
7
+ ↓
8
+ OpenMerit due-check engine
9
+ ↓
10
+ Harness-native wakeup or lifecycle hook
11
+ ↓
12
+ Standard automation signal
13
+ ↓
14
+ No work due, or one typed OpenMerit intent
15
+ ```
16
+
17
+ ## Check policy
18
+
19
+ `AutomationCheckPolicy` independently configures:
20
+
21
+ - baseline assessment cadence;
22
+ - frontier reassessment cadence and catalogue-change behavior;
23
+ - regression-check cadence and per-metric relative deterioration thresholds;
24
+ - post-swap verification windows;
25
+ - a cooldown between mature-system checks;
26
+ - active-session or persistent execution;
27
+ - whether wakeup provisioning may make infrastructure changes.
28
+
29
+ Task-count and elapsed-time triggers use OR semantics. A policy with `afterCompletedTasks: 100` and `afterElapsedSeconds: 604800` becomes due at the first threshold reached. A scheduled wakeup does not directly launch expensive work: it submits `scheduled_tick`, and OpenMerit evaluates policy and durable state before issuing an intent.
30
+
31
+ ## Signals
32
+
33
+ Adapters may report:
34
+
35
+ - `product_task_completed`;
36
+ - `metric_window_available`;
37
+ - `scheduled_tick`;
38
+ - `model_catalog_changed`;
39
+ - `verification_window_completed`.
40
+
41
+ Signal IDs are idempotency keys. OpenMerit persists a bounded history, does not count duplicate task completions twice, and keeps due work pending until an adapter successfully dispatches it.
42
+
43
+ ## Capability negotiation
44
+
45
+ Every `HarnessDescriptor` declares supported signals, persistent scheduling, background execution, and whether wakeup provisioning is native, unavailable, or requires an infrastructure change. OpenMerit produces an `AutomationPlan` containing required signals, the next wakeup, scheduling strategy, and explicit gaps.
46
+
47
+ If persistent scheduling is requested but unsupported, the plan says `external_scheduler_required`. The adapter must not claim that a schedule exists. Active-session adapters can submit `scheduled_tick` when they start. An external scheduler may also wake a non-interactive harness and submit the same signal.
48
+
49
+ OpenMerit calls `scheduleWakeup` only when the confirmed policy requests persistent execution, the adapter declares persistent scheduling and background execution, and any infrastructure-changing provisioning has been authorized. Paid agents, system scheduler installation, and cloud infrastructure remain harness implementation details and authorization boundaries.
50
+
51
+ ## Intent mapping
52
+
53
+ OpenMerit owns deterministic mapping from due conditions to intents:
54
+
55
+ | Due condition | Intent |
56
+ | --- | --- |
57
+ | Baseline cadence | `run_assessment` |
58
+ | Mature frontier cadence or changed catalogue | `discover_candidates` |
59
+ | Regression cadence or crossed metric threshold | `investigate_regression` |
60
+ | Completed post-swap window | `verify_model_swap` |
61
+
62
+ Subsequent lifecycle transitions continue through candidate trials, frontier calculation, supervised swap, verification, or rollback. The harness executes those intents; OpenMerit validates their structured result and evidence.
63
+
64
+ ## Pi behavior
65
+
66
+ Pi reports catalogue and scheduled-tick signals on session start and product-task signals after ordinary settled turns. Its typed `openmerit_report_signal` tool accepts verified metric and post-swap windows from eval or observability work. Pi cannot currently provision persistent background wakeups, so its descriptor says so. A persistent policy therefore produces an external-scheduler gap instead of a false success claim. A Pi-compatible scheduler can start Pi non-interactively; Pi then submits the tick and OpenMerit decides whether work is due.
@@ -0,0 +1,55 @@
1
+ # Getting started
2
+
3
+ ## Prerequisites
4
+
5
+ - Node.js 22.19 or newer.
6
+ - npm.
7
+ - Pi 0.87-compatible provider authentication.
8
+ - A product repository in which Pi can create and verify evaluation and observability artifacts.
9
+
10
+ OpenMerit does not require or accept an LLM provider credential. Provider access belongs to the coding harness.
11
+
12
+ ## Install from npm
13
+
14
+ ```sh
15
+ pi install npm:openmerit
16
+ pi list
17
+ ```
18
+
19
+ Run `/openmerit doctor` inside Pi to confirm that the extension is available.
20
+
21
+ ## Install from source
22
+
23
+ ```sh
24
+ git clone https://github.com/laz-aslam/openmerit.git
25
+ cd openmerit
26
+ npm ci
27
+ ```
28
+
29
+ Start Pi with the extension:
30
+
31
+ ```sh
32
+ ./node_modules/.bin/pi -e ./packages/pi/src/index.ts
33
+ ```
34
+
35
+ Run `/openmerit doctor` inside Pi to confirm that the adapter and intent executor are available.
36
+
37
+ ## Establish a product profile
38
+
39
+ Open the product repository in interactive Pi with the extension loaded. If the project has no OpenMerit state, setup begins automatically. `/openmerit setup` is only needed to explicitly rerun or recover setup.
40
+
41
+ OpenMerit sends Pi a private, typed `establish_evals` intent. Pi must understand the product task, infer objectives, ask the user to confirm requirements, an evaluation budget, check cadence, thresholds, scheduling mode, and automation permissions, create and verify the necessary infrastructure, and complete the intent with durable evidence.
42
+
43
+ OpenMerit rejects successful setup without a confirmed task profile, at least one valid objective, a supervised-graduation policy, a budget, and evidence.
44
+
45
+ ## Collect the baseline
46
+
47
+ After setup, use Pi normally. OpenMerit records completed-task checkpoints without treating its own orchestration turn as product evidence. At the required sample threshold, OpenMerit nudges Pi to assess the baseline. Missing evidence is not treated as zero or failure.
48
+
49
+ ## Find the frontier
50
+
51
+ When Pi reports a sufficient baseline, OpenMerit issues the next approved nudges: discover candidates, run controlled challenger trials, and calculate the Pareto frontier. Pi performs the work. OpenMerit checks the returned result against the [Pareto specification](pareto-spec.md).
52
+
53
+ ## Approve an early swap
54
+
55
+ During supervised graduation, a verified model proposal pauses for `/openmerit approve`. That authorizes Pi to apply only the proposed change. OpenMerit then requires post-swap verification; the approved policy determines rollback behavior.
@@ -0,0 +1,30 @@
1
+ # Improvement lifecycle
2
+
3
+ ## Stages
4
+
5
+ 1. **Unconfigured** — no confirmed profile exists.
6
+ 2. **Establishing evidence** — Pi creates and verifies evals and observability.
7
+ 3. **Collecting baseline** — normal product work accumulates evidence.
8
+ 4. **Baseline ready** — Pi establishes sufficiency for every required objective.
9
+ 5. **Discovering candidates** — Pi researches credible candidates within budget.
10
+ 6. **Running challenger trials** — Pi runs comparable controlled trials.
11
+ 7. **Calculating frontier** — Pi calculates and explains the Pareto frontier.
12
+ 8. **Frontier ready** — OpenMerit verifies the returned snapshot.
13
+ 9. **Approval or swap** — policy decides whether confirmation is required.
14
+ 10. **Verifying swap** — every applied change receives post-swap evaluation.
15
+ 11. **Monitoring** — production evidence continues to accumulate.
16
+ 12. **Rollback** — a covered regression restores the prior verified route.
17
+
18
+ ## Event-driven nudges
19
+
20
+ OpenMerit remains quiet when no action is justified. On first interactive entry to an unconfigured product, it automatically nudges Pi to establish the user-confirmed profile, evals, observability, check policy, and automation permission. Later nudges become due when durable harness signals satisfy the configured task-count or elapsed-time cadence, a metric crosses a regression threshold, the model catalogue changes under an enabled policy, a post-swap window completes, an intent enables the next stage, rollback policy covers a regression, or the user invokes a command.
21
+
22
+ Every signal has an idempotency key. A due intent is persisted before dispatch so duplicate webhooks, restarts, and retrying adapters do not double-count tasks or silently lose pending work.
23
+
24
+ ## Evaluation budget
25
+
26
+ Candidate work can cost money but is bounded by the confirmed policy. The contract can constrain currency, spend, candidate count, runs per candidate, and expiration. Pi decides how to work within those limits and must return evidence.
27
+
28
+ ## Supervised graduation
29
+
30
+ Early proposals wait for `/openmerit approve`. A policy may enable bounded automatic swaps only after the configured number of verified swaps. Post-swap verification remains mandatory.
@@ -0,0 +1,40 @@
1
+ # Metrics and evidence
2
+
3
+ ## Task-specific profiles
4
+
5
+ The full catalogue is available to every product, but a confirmed task profile selects required metrics, direction, minimum sample count, tolerance, and optional hard constraints.
6
+
7
+ ## Initial metric catalogue
8
+
9
+ Quality and behavior: task quality, task success, consistency, instruction following, tool-use performance, structured-output reliability, hallucination rate, reasoning efficiency, recovery ability, context handling, retrieval-use quality, long-horizon performance, and preference fit.
10
+
11
+ Cost and resources: input tokens, output tokens, total task cost, cost per successful task, and agent steps.
12
+
13
+ Latency and operations: time to first useful output, end-to-end latency, throughput, retry rate, and human-intervention rate.
14
+
15
+ ## Metric states
16
+
17
+ - `measured` — numeric value and evidence exist.
18
+ - `not_applicable` — the metric does not apply.
19
+ - `not_configured` — collection or evaluation is missing.
20
+ - `insufficient_evidence` — relevant, but without enough representative samples.
21
+
22
+ Missing measurements are never converted to zero.
23
+
24
+ ## Single and repeated runs
25
+
26
+ A single execution can report its cost, latency, token use, steps, success, and evaluator score. Distribution and reliability claims require repeated runs: rates, consistency, variance, averages, percentiles, throughput, prompt sensitivity, and regression stability.
27
+
28
+ Two forms of repetition may be needed: repeat the same input to measure stochastic variance, and run representative task instances to measure general task performance.
29
+
30
+ ## Evidence references
31
+
32
+ Results point to durable evidence rather than embedding arbitrary traces. Sources include harness traces, evaluation artifacts, observability records, external sources, and user feedback. References may contain a URI, media type, and digest. Successful intents require evidence; frontier certification also requires calculation evidence and exact assessment IDs.
33
+
34
+ ## Readiness contract
35
+
36
+ Setup must return an explicit observability coverage report containing every required metric, which metrics are covered, which remain missing, and when the check occurred. OpenMerit derives readiness from those sets; a harness cannot label coverage `ready` while a required metric is absent.
37
+
38
+ If setup reports a required gap, OpenMerit issues `instrument_observability`. A successful instrumentation result is accepted only when every required metric is covered. This prevents a configured task profile from being mistaken for working telemetry.
39
+
40
+ Product instrumentation and OpenMerit auditing are separate. The product emits measurements such as quality, success, tokens, cost, latency, retries, tool calls, and intervention. OpenMerit records the signals, policy decision, intent lifecycle, evidence identifiers, and readiness or frontier decision that followed.
@@ -0,0 +1,31 @@
1
+ # Project state and operations
2
+
3
+ ## State location
4
+
5
+ Operational data lives under `.openmerit/` in the target product repository. It should be ignored by version control because it can contain local lifecycle state and evidence locations.
6
+
7
+ - `config.json` — task profiles and policy.
8
+ - `state.json` — current cycle, active or pending intent, last result, explicit observability coverage, processed signal IDs, task counters, metric baselines, catalogue fingerprint, and assessment/swap checkpoints.
9
+ - `events.jsonl` — append-only audit events.
10
+ - `audit-export-errors.jsonl` — failures from optional external event exporters; these failures never replace or erase the canonical local event.
11
+ - `evidence/*.json` — evidence-reference manifests.
12
+
13
+ Snapshot writes are atomic. Files use owner-only permissions where POSIX modes are supported.
14
+
15
+ ## Diagnostics
16
+
17
+ Start with `/openmerit status`, which reports profile, metric and evaluation readiness, active work, next assessment, lifecycle, and pause state. `/openmerit doctor` confirms adapter loading without printing secrets.
18
+
19
+ `/openmerit pause` stops proactive nudges for the current process; explicit commands still work. `/openmerit resume` re-enables them.
20
+
21
+ `/openmerit logs` prints the exact local paths to the audit log, exporter failures, evidence manifests, and current state. During an end-to-end run, `tail -f .openmerit/events.jsonl | jq .` shows signals and decisions as they occur.
22
+
23
+ ## Audit and recovery
24
+
25
+ OpenMerit records initialization, intent requests and completions, automation signals and duplicates, due checks, task checkpoints, catalogue changes, frontier validation, confirmation boundaries, and lifecycle advancement.
26
+
27
+ Automation-signal records include safe measurement summaries: model ID and token context for completed tasks; or metric ID, state, scope, direction, method, value, unit, sample count, interval/window, and evidence IDs for metric windows. Intent completions record their kind, status, next stage, harness, summary, and evidence IDs/sources. Secret-shaped keys and common API-key patterns are redacted before local persistence or export.
28
+
29
+ The local JSONL file is the canonical source of truth. `ProjectStore` can also receive optional event exporters for OpenTelemetry or vendor-specific adapters. Export is best effort: a failed exporter is written to `audit-export-errors.jsonl` and does not interrupt lifecycle state or canonical logging.
30
+
31
+ If Pi stops during an intent, durable state retains it. Inspect `state.json` and `events.jsonl` before manual changes. Deleting `.openmerit/` discards the audit trail, budget, profile, and evidence references.
@@ -0,0 +1,76 @@
1
+ # OpenMerit Pareto conformance specification
2
+
3
+ Status: initial normative specification
4
+
5
+ OpenMerit owns this definition. A connected harness performs the calculation, returns a `FrontierSnapshot`, and supplies calculation evidence. OpenMerit verifies the returned result against these rules.
6
+
7
+ ## Comparable assessment set
8
+
9
+ Every assessment in one frontier calculation must share:
10
+
11
+ - task-profile ID and revision;
12
+ - product revision;
13
+ - harness identity and materially equivalent tool configuration;
14
+ - objective definitions, units, and directions;
15
+ - a representative evaluation workload and compatible evidence window.
16
+
17
+ The harness must not silently compare measurements from incompatible conditions.
18
+
19
+ ## Metric readiness
20
+
21
+ A required objective is ready only when:
22
+
23
+ - a metric record exists;
24
+ - its state is `measured`;
25
+ - it has a numeric value;
26
+ - its sample count meets the objective's minimum;
27
+ - its unit and direction are compatible across candidates.
28
+
29
+ Missing or immature evidence is `unresolved`. It is never converted to zero and never treated as success.
30
+
31
+ ## Feasibility
32
+
33
+ Hard constraints are evaluated conservatively against uncertainty intervals. If an interval is absent, the point value is both bounds.
34
+
35
+ - `at_least`: eligible only when the lower bound meets the threshold; ineligible when the upper bound is below it; otherwise unresolved.
36
+ - `at_most`: eligible only when the upper bound meets the threshold; ineligible when the lower bound exceeds it; otherwise unresolved.
37
+
38
+ Only eligible candidates may be certified as members of the frontier.
39
+
40
+ ## Dominance
41
+
42
+ For candidate A to dominate candidate B, A must be proven no worse than B on every objective and proven materially better on at least one objective.
43
+
44
+ For a maximize objective with tolerance `t`:
45
+
46
+ - A is proven no worse when `A.lower + t >= B.upper`.
47
+ - A is proven materially better when `A.lower > B.upper + t`.
48
+
49
+ For a minimize objective with tolerance `t`:
50
+
51
+ - A is proven no worse when `A.upper <= B.lower + t`.
52
+ - A is proven materially better when `A.upper + t < B.lower`.
53
+
54
+ If complete evidence proves A worse on any objective, A does not dominate B. If the evidence cannot establish no-worse or worse because intervals overlap, the comparison is unresolved.
55
+
56
+ ## Frontier
57
+
58
+ An eligible candidate is on the certified frontier when:
59
+
60
+ - no eligible candidate is proven to dominate it; and
61
+ - none of its required pairwise comparisons is unresolved.
62
+
63
+ Candidates with incomplete evidence, unresolved constraints, or unresolved comparisons are reported separately. A dominated, ineligible, or unresolved candidate must not appear on the certified frontier.
64
+
65
+ ## Required result evidence
66
+
67
+ Every frontier result must include:
68
+
69
+ - the assessment IDs used;
70
+ - all directed pairwise comparisons between eligible candidates;
71
+ - reasons for every comparison;
72
+ - separate frontier, dominated, ineligible, and unresolved sets;
73
+ - at least one durable calculation-evidence reference.
74
+
75
+ The calculation evidence should identify the harness execution or artifact that produced the result.
76
+
@@ -0,0 +1,44 @@
1
+ # Pi extension
2
+
3
+ ## Supported version
4
+
5
+ The adapter targets Pi 0.87 and declares `>=0.87.0 <0.88.0`. Development uses 0.87.0 without modifying a globally installed Pi.
6
+
7
+ ## Loading
8
+
9
+ Install the published package with `pi install npm:openmerit`. For development
10
+ from a checkout, load the source extension with:
11
+
12
+ ```sh
13
+ ./node_modules/.bin/pi -e ./packages/pi/src/index.ts
14
+ ```
15
+
16
+ Use `/openmerit doctor` to confirm that the adapter and intent executor are available.
17
+
18
+ Use `/openmerit logs` to locate the project-local audit stream, exporter failures, evidence manifests, and state while testing an end-to-end run.
19
+
20
+ On `session_start`, an interactive unconfigured project automatically receives one `establish_evals` intent. Durable active-intent state prevents the setup nudge from being reissued on every session. Non-interactive runs do not attempt a confirmation flow; `/openmerit setup` remains an explicit recovery command.
21
+
22
+ For configured projects, session start reports the current model catalogue and a scheduled tick. Ordinary settled product work reports an idempotent completed-task signal. OpenMerit evaluates the confirmed policy before issuing work. Pi currently declares no persistent scheduling or background execution, so persistent policies require an external scheduler to start Pi; the adapter reports that gap explicitly.
23
+
24
+ Pi also exposes `openmerit_report_signal` for verified `metric_window_available` and `verification_window_completed` events. Evals or observability work performed by Pi can submit measured metric records through this typed ingress using stable signal IDs. This does not create a background scheduler; it gives the active harness a validated way to connect the infrastructure it built to OpenMerit's due-check engine.
25
+
26
+ ## Intent delivery
27
+
28
+ The extension persists each intent as a custom Pi session entry, sends private extension-authored context with `display: false`, and triggers a follow-up Pi turn. The intent describes the outcome, constraints, evidence, task profile, and authorization boundary without prescribing Pi's method.
29
+
30
+ Pi must finish by calling `openmerit_complete_intent` exactly once. Success without evidence is rejected. Setup, assessment, and frontier intents receive additional structured-output checks.
31
+
32
+ The extension replaces the completion tool definition whenever an intent becomes active. Its `outputs` property therefore exposes the canonical schema for that exact intent instead of untyped JSON. Pi validates the arguments before execution and requests provider-side JSON Schema constrained sampling when supported. OpenMerit validates the canonical schema and semantic rules again when accepting the result.
33
+
34
+ ## Model changes
35
+
36
+ The extension does not use Pi's model setter as a substitute for harness work. After policy authorization, it sends a bounded `apply_model_swap` intent. Pi performs the change and returns configuration proof.
37
+
38
+ ## Jev System One
39
+
40
+ Jev is not integrated into OpenMerit. If configured in Pi, the intent asks Pi to use harness-configured fast decision resources aggressively for bounded ranking, triage, and judgment. Credentials remain outside OpenMerit.
41
+
42
+ ## Non-interactive modes
43
+
44
+ A harness must not attempt interactive confirmation in a mode without dialog-capable UI. It should return a waiting-for-user or failed result with an actionable reason.
@@ -0,0 +1,26 @@
1
+ # Current limitations and roadmap
2
+
3
+ ## Current status
4
+
5
+ OpenMerit is ready for an initial supervised product trial with Pi. Its protocol, lifecycle, persistence, metric catalogue, frontier verifier, and Pi adapter are implemented and tested. That is not evidence that it optimizes every product; each target workload requires its own profile and representative evaluations.
6
+
7
+ ## Known limitations
8
+
9
+ - No second harness adapter exists.
10
+ - Jev must be configured and invoked by the harness; OpenMerit intentionally provides no client.
11
+ - Generic Pi task checkpoints do not replace product telemetry. Setup must create the infrastructure required by the profile.
12
+ - Pi reports active-session signals and typed metric windows but cannot provision persistent background wakeups; persistent policies require an external scheduler until a capable adapter is added.
13
+ - Pause state is process-local rather than persisted.
14
+ - External evidence authenticity depends on the referenced system.
15
+ - The core and protocol workspaces are private; the single `openmerit` npm
16
+ package exposes their public entry points.
17
+
18
+ ## Next validation milestones
19
+
20
+ 1. Run setup in one representative product.
21
+ 2. Verify generated eval and observability artifacts and their side effects.
22
+ 3. Collect repeated evidence for the confirmed profile.
23
+ 4. Complete a budgeted A/B/C challenger cycle.
24
+ 5. Validate the harness-calculated frontier and recommendation.
25
+ 6. Perform a supervised swap, post-swap verification, and rollback drill.
26
+ 7. Repeat with a materially different product task before broad readiness claims.
@@ -0,0 +1,23 @@
1
+ # Security
2
+
3
+ ## Credential ownership
4
+
5
+ OpenMerit accepts no model-provider or Jev credential. Credentials belong to the coding harness, its environment, or an external secret store.
6
+
7
+ Never place secrets in profiles, policies, intent constraints, summaries, evidence URIs, audit events, fixtures, traces, or committed environment files.
8
+
9
+ ## Mutation authorization
10
+
11
+ Most intents carry `modelMutationAllowed: false`. Only apply-swap and rollback intents can authorize a mutation, limited to the active proposal. Early swaps require confirmation; automatic swaps require explicit policy opt-in and sufficient verified history. Every swap requires verification.
12
+
13
+ Persistent wakeup provisioning is a separate authorization boundary. An adapter that needs to install a system scheduler, create a paid cloud agent, or change infrastructure must declare `wakeupProvisioning: "infrastructure_change"`; OpenMerit will not request that provisioning unless the confirmed check policy sets `infrastructureChangesAllowed: true`.
14
+
15
+ ## Local state
16
+
17
+ `.openmerit/` may reveal task names, candidate identifiers, evidence locations, budgets, or model choices. Treat it as sensitive project metadata even though it must not contain credentials.
18
+
19
+ ## Evidence integrity
20
+
21
+ OpenMerit validates structure and deterministic Pareto conformance, not the truth of arbitrary external evidence. Use authenticated observability, immutable artifacts, and digests where stronger provenance is required.
22
+
23
+ Do not include credentials, private traces, or proprietary evaluation data in public issues.