openmerit 0.1.5 → 0.1.6-preview.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +33 -9
  3. package/dist/core/src/index.d.ts +12 -1
  4. package/dist/core/src/index.js +552 -40
  5. package/dist/pi/src/index.d.ts +17 -0
  6. package/dist/pi/src/index.js +448 -77
  7. package/dist/pi/src/scheduler.d.ts +11 -0
  8. package/dist/pi/src/scheduler.js +137 -0
  9. package/dist/pi/src/wakeup.d.ts +2 -0
  10. package/dist/pi/src/wakeup.js +108 -0
  11. package/dist/protocol/src/index.d.ts +86 -4
  12. package/dist/protocol/src/index.js +1 -1
  13. package/dist/protocol/src/schemas.d.ts +128 -2
  14. package/dist/protocol/src/schemas.js +57 -1
  15. package/dist/terminal/public/app.js +297 -0
  16. package/dist/terminal/public/brands/anthropic.png +0 -0
  17. package/dist/terminal/public/brands/baai.png +0 -0
  18. package/dist/terminal/public/brands/baseten.png +0 -0
  19. package/dist/terminal/public/brands/cerebras.png +0 -0
  20. package/dist/terminal/public/brands/cohere.png +0 -0
  21. package/dist/terminal/public/brands/deepseek.ico +0 -0
  22. package/dist/terminal/public/brands/google.png +0 -0
  23. package/dist/terminal/public/brands/groq.ico +0 -0
  24. package/dist/terminal/public/brands/lm-studio.png +0 -0
  25. package/dist/terminal/public/brands/meta.ico +0 -0
  26. package/dist/terminal/public/brands/mistral.png +0 -0
  27. package/dist/terminal/public/brands/nomic.png +0 -0
  28. package/dist/terminal/public/brands/ollama.png +0 -0
  29. package/dist/terminal/public/brands/openai.png +0 -0
  30. package/dist/terminal/public/brands/openrouter.png +0 -0
  31. package/dist/terminal/public/brands/qwen.png +0 -0
  32. package/dist/terminal/public/brands/vllm.ico +0 -0
  33. package/dist/terminal/public/brands/vllm.png +0 -0
  34. package/dist/terminal/public/favicon.svg +1 -0
  35. package/dist/terminal/public/flow.css +1 -0
  36. package/dist/terminal/public/flow.js +770 -0
  37. package/dist/terminal/public/index.html +21 -0
  38. package/dist/terminal/public/styles.css +779 -0
  39. package/dist/terminal/src/activity-merge.mjs +64 -0
  40. package/dist/terminal/src/browser.mjs +29 -0
  41. package/dist/terminal/src/cli.mjs +60 -0
  42. package/dist/terminal/src/collect.mjs +311 -0
  43. package/dist/terminal/src/discovery.mjs +93 -0
  44. package/dist/terminal/src/hardware.mjs +57 -0
  45. package/dist/terminal/src/project-activity.mjs +156 -0
  46. package/dist/terminal/src/sample.mjs +171 -0
  47. package/dist/terminal/src/server.mjs +56 -0
  48. package/dist/terminal/src/services.mjs +62 -0
  49. package/dist/terminal/src/topology.mjs +30 -0
  50. package/docs/adapter-guide.md +24 -7
  51. package/docs/architecture.md +8 -4
  52. package/docs/automation.md +10 -2
  53. package/docs/budgets.md +37 -0
  54. package/docs/commands.md +85 -0
  55. package/docs/demo-backfill.md +29 -0
  56. package/docs/demo-fieldkit.md +47 -0
  57. package/docs/demo-placement.md +30 -0
  58. package/docs/demo-spam.md +15 -0
  59. package/docs/demo-support.md +42 -0
  60. package/docs/demo.md +57 -0
  61. package/docs/first-trial.md +60 -0
  62. package/docs/getting-started.md +18 -8
  63. package/docs/index.md +40 -0
  64. package/docs/inference-terminal.md +439 -0
  65. package/docs/lifecycle.md +9 -9
  66. package/docs/memo.md +126 -0
  67. package/docs/metrics-and-evidence.md +11 -3
  68. package/docs/operations.md +12 -3
  69. package/docs/pi-extension.md +17 -7
  70. package/docs/roadmap.md +4 -2
  71. package/docs/security.md +15 -1
  72. package/docs/site-artwork-linocut.md +23 -0
  73. package/docs/site-artwork-miniature-diverse.md +28 -0
  74. package/docs/site-artwork-miniature.md +26 -0
  75. package/docs/site-demo.md +177 -0
  76. package/docs/site-design.md +94 -0
  77. package/docs/site-documentation.md +83 -0
  78. package/docs/site-dynamic-og.md +35 -0
  79. package/docs/site-faq-maintenance.md +115 -0
  80. package/docs/site-hero-resolution.md +60 -0
  81. package/docs/site-illustration-sequences.md +227 -0
  82. package/docs/site-inference-terminal.md +203 -0
  83. package/docs/site-memo.md +39 -0
  84. package/docs/site-og-image.md +38 -0
  85. package/docs/site-og-workshop.md +21 -0
  86. package/docs/site-section-artwork.md +56 -0
  87. package/docs/site-skill-review.md +57 -0
  88. package/docs/site-terminal-preview.md +85 -0
  89. package/docs/testing.md +85 -3
  90. package/docs/troubleshooting.md +55 -0
  91. package/package.json +48 -7
@@ -1,8 +1,12 @@
1
+ import { readdirSync, readFileSync, statSync } from "node:fs";
1
2
  import { createHash } from "node:crypto";
3
+ import { basename, join, relative } from "node:path";
4
+ import { fileURLToPath } from "node:url";
2
5
  import { StringEnum } from "@earendil-works/pi-ai";
3
6
  import { nextImprovementAction, OpenMeritCoordinator, ProjectStore, } from "../../core/src/index.js";
4
7
  import { completionToolSchemaFor, OPENMERIT_SCHEMA_DIALECT, PROTOCOL_VERSION, MetricRecordSchema, } from "../../protocol/src/index.js";
5
8
  import { Type } from "typebox";
9
+ import { piWakeupPlistPath, provisionPiWakeup, removePiWakeup } from "./scheduler.js";
6
10
  const INTENT_ENTRY_TYPE = "openmerit.intent";
7
11
  const RESULT_ENTRY_TYPE = "openmerit.result";
8
12
  const INTENT_MESSAGE_TYPE = "openmerit.intent-context";
@@ -34,34 +38,146 @@ export const PI_HARNESS_DESCRIPTOR = {
34
38
  "product_task_completed", "metric_window_available", "scheduled_tick",
35
39
  "model_catalog_changed", "verification_window_completed",
36
40
  ],
37
- persistentScheduling: false,
38
- backgroundExecution: false,
39
- wakeupProvisioning: "none",
41
+ persistentScheduling: process.platform === "darwin",
42
+ backgroundExecution: process.platform === "darwin",
43
+ wakeupProvisioning: process.platform === "darwin" ? "infrastructure_change" : "none",
40
44
  },
41
45
  };
42
46
  export function createInitialUiState() {
43
47
  return {
44
48
  proactiveNudgesPaused: false,
49
+ applicationTarget: "not_detected",
45
50
  taskProfile: "not_configured",
46
51
  metricReadiness: "not_assessed",
47
52
  evaluationReadiness: "not_assessed",
48
53
  activeIntent: null,
49
54
  nextAssessment: null,
50
55
  lifecycleStage: "not_initialized",
56
+ incumbentCandidate: null,
57
+ candidateCount: 0,
58
+ assessmentCount: 0,
59
+ evaluationSpend: 0,
60
+ proposedCandidate: null,
51
61
  };
52
62
  }
53
63
  export function renderStatus(state) {
54
64
  return [
55
65
  "OpenMerit evidence readiness",
66
+ `application target : ${state.applicationTarget.replaceAll("_", " ")}`,
56
67
  `task profile : ${state.taskProfile.replaceAll("_", " ")}`,
57
68
  `metric readiness : ${state.metricReadiness.replaceAll("_", " ")}`,
58
69
  `evaluation status : ${state.evaluationReadiness.replaceAll("_", " ")}`,
59
70
  `active work : ${state.activeIntent ?? "none"}`,
60
71
  `next assessment : ${state.nextAssessment ?? "not scheduled"}`,
61
72
  `frontier lifecycle : ${state.lifecycleStage.replaceAll("_", " ")}`,
73
+ `incumbent candidate: ${state.incumbentCandidate ?? "not confirmed"}`,
74
+ `candidate set : ${state.candidateCount}`,
75
+ `assessments : ${state.assessmentCount}`,
76
+ `evaluation spend : $${state.evaluationSpend.toFixed(6)}`,
77
+ `proposed candidate : ${state.proposedCandidate ?? "none"}`,
62
78
  `proactive nudges : ${state.proactiveNudgesPaused ? "paused" : "enabled"}`,
63
79
  ].join("\n");
64
80
  }
81
+ const DETECTION_IGNORES = new Set([
82
+ ".git", ".openmerit", "node_modules", "dist", "build", "coverage", ".next", "vendor", "tests", "test", "__tests__",
83
+ ]);
84
+ const DETECTION_EXTENSIONS = new Set([".js", ".jsx", ".mjs", ".cjs", ".ts", ".tsx", ".py", ".go", ".rs", ".java", ".kt"]);
85
+ const APPLICATION_CALL_PATTERNS = [
86
+ { provider: "openai-compatible", pattern: /\b(?:chat\.completions|responses)\.create\s*\(/ },
87
+ { provider: "anthropic", pattern: /\bmessages\.create\s*\(/ },
88
+ { provider: "google", pattern: /\bgenerateContent(?:Stream)?\s*\(/ },
89
+ { provider: "vercel-ai", pattern: /\b(?:generateText|streamText|generateObject|streamObject)\s*\(/ },
90
+ { provider: "langchain", pattern: /\b(?:ChatOpenAI|ChatAnthropic|ChatGoogleGenerativeAI)\s*\(/ },
91
+ { provider: "openrouter", pattern: /openrouter\.ai\/api\/v1[\s\S]{0,5000}\/(?:chat\/completions|responses)/ },
92
+ ];
93
+ function extensionOf(path) {
94
+ const match = path.match(/(\.[^.\/]+)$/);
95
+ return match?.[1]?.toLowerCase() ?? "";
96
+ }
97
+ /** Bounded, adapter-local detection. It requires an actual application call pattern, not only an SDK dependency. */
98
+ export function detectApplicationLlmTargets(projectRoot) {
99
+ const files = [];
100
+ const pending = [projectRoot];
101
+ while (pending.length && files.length < 500) {
102
+ const directory = pending.shift();
103
+ let entries;
104
+ try {
105
+ entries = readdirSync(directory, { withFileTypes: true });
106
+ }
107
+ catch {
108
+ continue;
109
+ }
110
+ for (const entry of entries) {
111
+ if (DETECTION_IGNORES.has(entry.name))
112
+ continue;
113
+ const path = join(directory, entry.name);
114
+ if (entry.isDirectory())
115
+ pending.push(path);
116
+ else if (entry.isFile() && DETECTION_EXTENSIONS.has(extensionOf(path)))
117
+ files.push(path);
118
+ if (files.length >= 500)
119
+ break;
120
+ }
121
+ }
122
+ const applicationId = basename(projectRoot);
123
+ const targets = [];
124
+ for (const path of files) {
125
+ let source;
126
+ try {
127
+ if (statSync(path).size > 1_000_000)
128
+ continue;
129
+ source = readFileSync(path, "utf8");
130
+ }
131
+ catch {
132
+ continue;
133
+ }
134
+ for (const { provider, pattern } of APPLICATION_CALL_PATTERNS) {
135
+ const match = pattern.exec(source);
136
+ if (!match)
137
+ continue;
138
+ const projectPath = relative(projectRoot, path).replaceAll("\\", "/");
139
+ const line = source.slice(0, match.index).split("\n").length;
140
+ const identity = `${applicationId}:${projectPath}:${provider}`;
141
+ targets.push({
142
+ id: `app-llm-${createHash("sha256").update(identity).digest("hex").slice(0, 12)}`,
143
+ kind: "application_llm_call",
144
+ name: `${provider} application call in ${projectPath}`,
145
+ applicationId,
146
+ callSites: [`${projectPath}:${line}`],
147
+ routeKey: `${projectPath}#${provider}`,
148
+ ...((source.match(/(?:DEFAULT_MODEL|defaultModel|model)\s*=\s*["'`]([^"'`]+)["'`]/)?.[1])
149
+ ? { configuredModelId: source.match(/(?:DEFAULT_MODEL|defaultModel|model)\s*=\s*["'`]([^"'`]+)["'`]/)[1] }
150
+ : {}),
151
+ });
152
+ }
153
+ }
154
+ return targets;
155
+ }
156
+ function detectedTargetsConstraint(targets) {
157
+ return targets.map((target) => ({
158
+ id: target.id,
159
+ kind: target.kind,
160
+ name: target.name,
161
+ applicationId: target.applicationId,
162
+ callSites: [...target.callSites],
163
+ routeKey: target.routeKey,
164
+ ...(target.configuredModelId ? { configuredModelId: target.configuredModelId } : {}),
165
+ }));
166
+ }
167
+ function validateReportedMetrics(metrics, targetId) {
168
+ for (const metric of metrics) {
169
+ if (metric.targetId !== targetId) {
170
+ throw new Error(`${metric.metricId} target does not match the reported application LLM target.`);
171
+ }
172
+ if (metric.state === "measured" && (!metric.evidence.length || metric.value === undefined)) {
173
+ throw new Error(`Measured metric ${metric.metricId} requires a numeric value and evidence.`);
174
+ }
175
+ if (metric.metricId === "tool_use_performance" && metric.state === "measured" &&
176
+ !metric.evidence.some((item) => item.source === "harness_trace")) {
177
+ throw new Error("tool_use_performance requires harness-trace evidence; test coverage is not tool-use evidence.");
178
+ }
179
+ }
180
+ }
65
181
  function modelCatalogFingerprint(ctx) {
66
182
  const catalogue = ctx.modelRegistry.getAvailable().map((model) => ({
67
183
  provider: model.provider,
@@ -80,56 +196,69 @@ function privateIntentMessage(intent) {
80
196
  "Perform the intelligent work using the current project, session context, and available tools.",
81
197
  "Use harness-configured fast decision resources such as Jev System One aggressively for bounded ranking, triage, and judgment when available; OpenMerit does not call those services itself.",
82
198
  "You, the coding harness, own all research, evaluation design, instrumentation, measurement, Pareto calculation, and authorized configuration work. OpenMerit only validates the returned contract and evidence.",
83
- "Instrument every required metric selected by the user. Report explicit covered and missing metric IDs; never describe observability as ready while a required metric is missing.",
199
+ "The system under test is the user-confirmed application LLM target. Pi's own model, tokens, cost, latency, and session are orchestration data and must never be submitted as application evidence.",
200
+ intent.targetId
201
+ ? `Every result and metric must reference application target ${intent.targetId}.`
202
+ : "Before setup succeeds, identify an actual application LLM call or shared application route and ask the user to confirm its stable target ID, call sites, and route key.",
203
+ "For setup, infer task-relevant metrics from the application rather than requiring the user to invent them. Before any setup run, show the user the proposed metrics, rationale, sampling strategy, representative case count, repetitions per case, aggregation, true performance thresholds if any, and estimated cost, then obtain confirmation or edits. Rate, reliability, distribution, percentile, variance, and average claims require multiple runs. Distinguish repeating the same input for stochastic variance from running representative task instances for category performance; use both when the claim needs both. minimumSamples is evidence quantity and must never be copied into constraint.value. A constraint is optional and represents only a real metric threshold in that metric's unit. For an active-session demo, default baselineAssessment and postSwapVerification to afterCompletedTasks: 1 unless the user changes the cadence; never return an empty cadence. Instrument every confirmed required metric. For setup or instrumentation, actually run each task-specific grader and collector through the application path with working provider authentication, inspect the resulting metric values, and cite the run artifacts. Report explicit covered and missing metric IDs; never describe observability as ready when a required metric cannot be produced. A setup smoke run is verification, not a production baseline sample.",
204
+ "Capture cost, token use, and task latency from the application LLM provider or application observability. Do not substitute Pi provider telemetry or zero for unavailable application telemetry.",
205
+ intent.kind === "instrument_observability"
206
+ ? "For instrumentation, execute the confirmed application target itself and return one measured MetricRecord for every covered required metric. Map provider/application usage fields to input_token_consumption and cost_per_successful_task, use the exact metric IDs and objective aggregation, include sampleCount and valid startedAt/completedAt windows, and attach durable observability_record or harness_trace evidence. Do not merely describe telemetry or add arbitrary fields to the application response; those do not count until converted into MetricRecords in the completion result."
207
+ : "",
208
+ "Tool-use performance must be derived from tool-call traces and outcomes; code coverage is not tool-use evidence.",
84
209
  intent.constraints.modelMutationAllowed
85
- ? "This intent authorizes only the model mutation described by the active OpenMerit cycle."
210
+ ? `This intent authorizes only the application model mutation for target ${intent.targetId ?? "(missing target)"}. Never change Pi's model.`
86
211
  : "Do not change Pi's model or application model configuration.",
87
212
  "Before finishing, call openmerit_complete_intent exactly once with the result and durable evidence references.",
88
213
  JSON.stringify(intent),
89
214
  ].join("\n\n");
90
215
  }
91
- function piHarnessAdapter(pi) {
216
+ function piHarnessAdapter(pi, ctx) {
92
217
  return {
93
218
  descriptor: PI_HARNESS_DESCRIPTOR,
94
219
  async dispatch(intent) {
95
220
  pi.appendEntry(INTENT_ENTRY_TYPE, { intent });
96
- pi.sendMessage({
97
- customType: INTENT_MESSAGE_TYPE,
98
- content: privateIntentMessage(intent),
99
- display: false,
100
- details: { intentId: intent.id, kind: intent.kind },
101
- }, { deliverAs: "followUp", triggerTurn: true });
221
+ if (ctx.mode === "tui" || ctx.mode === "rpc") {
222
+ pi.sendMessage({
223
+ customType: INTENT_MESSAGE_TYPE,
224
+ content: privateIntentMessage(intent),
225
+ display: false,
226
+ details: { intentId: intent.id, kind: intent.kind },
227
+ }, { deliverAs: "followUp", triggerTurn: true });
228
+ }
102
229
  return { accepted: true, harnessJobId: intent.id };
103
230
  },
231
+ async scheduleWakeup(request) {
232
+ if (request.signal !== "scheduled_tick")
233
+ return { accepted: false, reason: "Pi only schedules OpenMerit ticks." };
234
+ if (!ctx.model)
235
+ return { accepted: false, reason: "Select an authenticated Pi model before enabling background checks." };
236
+ try {
237
+ const path = await provisionPiWakeup({
238
+ projectRoot: ctx.cwd,
239
+ extensionPath: fileURLToPath(import.meta.url),
240
+ provider: ctx.model.provider,
241
+ modelId: ctx.model.id,
242
+ });
243
+ return { accepted: true, harnessJobId: path };
244
+ }
245
+ catch (error) {
246
+ return { accepted: false, reason: error instanceof Error ? error.message : String(error) };
247
+ }
248
+ },
104
249
  };
105
250
  }
106
- function restoreState(ctx, state) {
107
- state.activeIntent = null;
108
- for (const entry of ctx.sessionManager.getEntries()) {
109
- if (entry.type !== "custom")
110
- continue;
111
- if (entry.customType === INTENT_ENTRY_TYPE) {
112
- const stored = entry.data;
113
- if (stored?.intent?.id)
114
- state.activeIntent = stored.intent.id;
115
- }
116
- if (entry.customType === RESULT_ENTRY_TYPE) {
117
- const stored = entry.data;
118
- if (stored?.result?.intentId === state.activeIntent)
119
- state.activeIntent = null;
120
- }
121
- }
122
- }
123
- async function dispatchIntent(pi, ctx, state, store, kind, prepareCompletionTool) {
251
+ async function dispatchIntent(pi, ctx, state, store, kind, prepareCompletionTool, additionalConstraints) {
124
252
  if (state.activeIntent) {
125
253
  ctx.ui.notify(`openmerit: work is already active (${state.activeIntent})`, "warning");
126
254
  return;
127
255
  }
128
256
  prepareCompletionTool(kind);
129
- const issued = await new OpenMeritCoordinator(store, piHarnessAdapter(pi)).issue(kind);
257
+ const issued = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).issue(kind, { additionalConstraints });
130
258
  const { intent } = issued;
131
259
  state.activeIntent = intent.id;
132
260
  if (kind === "establish_evals") {
261
+ state.applicationTarget = "detected";
133
262
  state.taskProfile = "pending";
134
263
  state.metricReadiness = "pending";
135
264
  state.evaluationReadiness = "pending";
@@ -142,7 +271,38 @@ async function dispatchIntent(pi, ctx, state, store, kind, prepareCompletionTool
142
271
  ctx.ui.notify(`openmerit: ${label} started (${intent.id})`, "info");
143
272
  }
144
273
  async function reportAutomationSignal(pi, ctx, state, store, signal, prepareCompletionTool) {
145
- const result = await new OpenMeritCoordinator(store, piHarnessAdapter(pi)).recordAutomationSignal(signal);
274
+ const result = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).recordAutomationSignal(signal);
275
+ const persistedState = state.activeIntent ? await store.readState() : undefined;
276
+ if (state.activeIntent && persistedState?.activeIntent?.kind === "instrument_observability" && signal.type === "metric_window_available") {
277
+ const projectConfig = await store.readConfig();
278
+ const profile = projectConfig?.taskProfiles.find((item) => item.id === projectConfig.activeTaskProfileId);
279
+ const requiredObjectives = profile?.objectives.filter((item) => item.required) ?? [];
280
+ const measuredMetricIds = new Set(signal.metrics
281
+ .filter((metric) => metric.targetId === signal.targetId && metric.state === "measured" && metric.value !== undefined && metric.sampleCount > 0 && metric.evidence.length > 0)
282
+ .map((metric) => metric.metricId));
283
+ const coveredMetricIds = requiredObjectives.filter((objective) => measuredMetricIds.has(objective.metricId)).map((objective) => objective.metricId);
284
+ const missingMetricIds = requiredObjectives.filter((objective) => !measuredMetricIds.has(objective.metricId)).map((objective) => objective.metricId);
285
+ const evidence = [...new Map(signal.metrics.flatMap((metric) => metric.evidence).map((item) => [item.id, item])).values()];
286
+ const completion = {
287
+ protocolVersion: PROTOCOL_VERSION,
288
+ intentId: state.activeIntent,
289
+ status: "succeeded",
290
+ completedAt: new Date().toISOString(),
291
+ summary: `Instrumentation verified ${coveredMetricIds.length} required metric(s); ${missingMetricIds.length} remain missing.`,
292
+ evidence,
293
+ outputs: {
294
+ targetId: signal.targetId,
295
+ coveredMetricIds,
296
+ missingMetricIds,
297
+ },
298
+ };
299
+ const accepted = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).acceptResult(completion);
300
+ state.activeIntent = null;
301
+ state.lifecycleStage = accepted.state.cycle.stage;
302
+ state.metricReadiness = accepted.state.observability?.status === "ready" ? "assessed" : "pending";
303
+ prepareCompletionTool("instrument_observability");
304
+ return;
305
+ }
146
306
  if (result.decision.action === "issue_intent" && !state.activeIntent) {
147
307
  await dispatchIntent(pi, ctx, state, store, result.decision.intentKind, prepareCompletionTool);
148
308
  }
@@ -150,7 +310,15 @@ async function reportAutomationSignal(pi, ctx, state, store, signal, prepareComp
150
310
  export default function openMeritExtension(pi) {
151
311
  const state = createInitialUiState();
152
312
  let store;
153
- let suppressNextSettledObservation = false;
313
+ const queuedUserInputs = [];
314
+ const releaseQueuedUserInput = () => {
315
+ if (state.activeIntent || !queuedUserInputs.length)
316
+ return;
317
+ const queued = queuedUserInputs.splice(0);
318
+ const text = queued.map((item) => item.text).join("\n\n");
319
+ const images = queued.flatMap((item) => item.images ?? []);
320
+ pi.sendUserMessage(images.length ? [{ type: "text", text }, ...images] : text, { deliverAs: "followUp" });
321
+ };
154
322
  const registerCompletionTool = (kind) => {
155
323
  const parameters = kind
156
324
  ? completionToolSchemaFor(kind)
@@ -214,13 +382,27 @@ export default function openMeritExtension(pi) {
214
382
  }
215
383
  : {}),
216
384
  };
217
- const accepted = await new OpenMeritCoordinator(store, piHarnessAdapter(pi)).acceptResult(result);
385
+ const accepted = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).acceptResult(result);
386
+ if (accepted.intentKind === "establish_evals" &&
387
+ (accepted.state.cycle.policy?.checkPolicy.execution.mode !== "persistent" ||
388
+ !accepted.state.cycle.policy?.checkPolicy.execution.infrastructureChangesAllowed)) {
389
+ try {
390
+ await removePiWakeup(ctx.cwd);
391
+ }
392
+ catch (error) {
393
+ ctx.ui.notify(`openmerit: could not remove prior wakeup: ${error instanceof Error ? error.message : String(error)}`, "warning");
394
+ }
395
+ }
218
396
  pi.appendEntry(RESULT_ENTRY_TYPE, { result });
219
- suppressNextSettledObservation = true;
220
397
  const { intentKind, state: nextState } = accepted;
221
398
  const stage = nextState.cycle.stage;
222
399
  state.activeIntent = null;
223
400
  state.lifecycleStage = stage;
401
+ state.incumbentCandidate = nextState.cycle.activeCandidateId ?? null;
402
+ state.candidateCount = nextState.candidates?.length ?? 0;
403
+ state.assessmentCount = nextState.assessments?.length ?? 0;
404
+ state.evaluationSpend = nextState.automation.evaluationSpend ?? 0;
405
+ state.proposedCandidate = nextState.cycle.proposedCandidateId ?? null;
224
406
  registerCompletionTool();
225
407
  state.nextAssessment = accepted.automation.plan.nextWakeupAt ??
226
408
  (accepted.automation.plan.scheduling === "external_scheduler_required" ? "external scheduler required" : "signal driven");
@@ -228,6 +410,7 @@ export default function openMeritExtension(pi) {
228
410
  ctx.ui.notify(`openmerit automation: ${accepted.automation.plan.gaps.join(" ")}`, "warning");
229
411
  }
230
412
  if (input.status === "succeeded" && intentKind === "establish_evals") {
413
+ state.applicationTarget = "confirmed";
231
414
  state.taskProfile = "configured";
232
415
  state.metricReadiness = nextState.observability?.status === "ready" ? "assessed" : "pending";
233
416
  state.evaluationReadiness = "assessed";
@@ -244,6 +427,10 @@ export default function openMeritExtension(pi) {
244
427
  ctx.ui.notify(`openmerit: ${decision.reason}`, "warning");
245
428
  }
246
429
  }
430
+ // A successful result can immediately issue another OpenMerit intent
431
+ // (for example, setup -> instrumentation). Release product work only
432
+ // after that chain has no active follow-up.
433
+ releaseQueuedUserInput();
247
434
  return {
248
435
  content: [{ type: "text", text: `Recorded OpenMerit intent ${input.intentId} as ${input.status}.` }],
249
436
  details: { result },
@@ -255,46 +442,159 @@ export default function openMeritExtension(pi) {
255
442
  pi.registerTool({
256
443
  name: "openmerit_report_signal",
257
444
  label: "Report OpenMerit Signal",
258
- description: "Report a verified observability metric window or completed post-swap verification window to OpenMerit.",
259
- promptSnippet: "Report verified OpenMerit observability signals",
445
+ description: "Report a verified application LLM task, observability metric window, or completed post-swap verification window to OpenMerit.",
446
+ promptSnippet: "Report verified application-target OpenMerit signals",
260
447
  promptGuidelines: [
261
448
  "Use stable signalId values so retried deliveries remain idempotent.",
262
- "Report only measured metrics backed by the referenced observability or evaluation artifacts.",
449
+ "Every signal and metric must reference the confirmed application LLM target.",
450
+ "Report only application-provider telemetry and measured metrics backed by referenced application observability or evaluation artifacts. Never report Pi's own model usage.",
263
451
  ],
264
452
  parameters: Type.Object({
265
453
  signalId: Type.String({ minLength: 1 }),
266
- type: StringEnum(["metric_window_available", "verification_window_completed"]),
454
+ type: StringEnum(["product_task_completed", "metric_window_available", "verification_window_completed"]),
267
455
  occurredAt: Type.Optional(Type.String({ minLength: 1 })),
268
- metrics: Type.Array(MetricRecordSchema),
456
+ targetId: Type.String({ minLength: 1 }),
457
+ metrics: Type.Optional(Type.Array(MetricRecordSchema)),
458
+ modelId: Type.Optional(Type.String({ minLength: 1 })),
459
+ evidence: Type.Optional(Type.Array(Type.Object({
460
+ id: Type.String({ minLength: 1 }),
461
+ source: StringEnum(["harness_trace", "evaluation_artifact", "observability_record", "external_source", "user_feedback"]),
462
+ uri: Type.String({ minLength: 1 }),
463
+ mediaType: Type.Optional(Type.String({ minLength: 1 })),
464
+ digest: Type.Optional(Type.String({ minLength: 1 })),
465
+ }, { additionalProperties: false }))),
466
+ telemetry: Type.Optional(Type.Object({
467
+ startedAt: Type.String({ minLength: 1 }),
468
+ completedAt: Type.String({ minLength: 1 }),
469
+ durationMs: Type.Number({ minimum: 0 }),
470
+ inputTokens: Type.Number({ minimum: 0 }),
471
+ outputTokens: Type.Number({ minimum: 0 }),
472
+ cacheReadTokens: Type.Number({ minimum: 0 }),
473
+ cacheWriteTokens: Type.Number({ minimum: 0 }),
474
+ cost: Type.Number({ minimum: 0 }),
475
+ currency: Type.Literal("USD"),
476
+ }, { additionalProperties: false })),
269
477
  }, { additionalProperties: false }),
270
478
  constrainedSampling: { type: "json_schema", strict: "prefer" },
271
479
  async execute(_toolCallId, input, _signal, _onUpdate, ctx) {
272
480
  store ??= new ProjectStore(ctx.cwd);
273
- await reportAutomationSignal(pi, ctx, state, store, {
274
- protocolVersion: PROTOCOL_VERSION,
275
- id: input.signalId,
276
- type: input.type,
277
- occurredAt: input.occurredAt ?? new Date().toISOString(),
278
- metrics: input.metrics,
279
- }, registerCompletionTool);
481
+ if (input.type === "product_task_completed") {
482
+ if (!input.modelId || !input.evidence?.length) {
483
+ throw new Error("A completed application task requires the application modelId and durable evidence.");
484
+ }
485
+ await reportAutomationSignal(pi, ctx, state, store, {
486
+ protocolVersion: PROTOCOL_VERSION,
487
+ id: input.signalId,
488
+ type: input.type,
489
+ occurredAt: input.occurredAt ?? new Date().toISOString(),
490
+ targetId: input.targetId,
491
+ modelId: input.modelId,
492
+ evidence: input.evidence,
493
+ ...(input.telemetry ? { telemetry: input.telemetry } : {}),
494
+ }, registerCompletionTool);
495
+ }
496
+ else {
497
+ if (!input.metrics)
498
+ throw new Error(`${input.type} requires application-target metrics.`);
499
+ validateReportedMetrics(input.metrics, input.targetId);
500
+ await reportAutomationSignal(pi, ctx, state, store, {
501
+ protocolVersion: PROTOCOL_VERSION,
502
+ id: input.signalId,
503
+ type: input.type,
504
+ occurredAt: input.occurredAt ?? new Date().toISOString(),
505
+ targetId: input.targetId,
506
+ metrics: input.metrics,
507
+ }, registerCompletionTool);
508
+ }
280
509
  return {
281
510
  content: [{ type: "text", text: `Recorded OpenMerit automation signal ${input.signalId}.` }],
282
511
  details: { signalId: input.signalId, type: input.type },
283
512
  };
284
513
  },
285
514
  });
515
+ pi.on("input", (event, ctx) => {
516
+ if (ctx.mode !== "tui" || !state.activeIntent || event.source !== "interactive" || ctx.isIdle())
517
+ return;
518
+ queuedUserInputs.push({ text: event.text, ...(event.images ? { images: event.images } : {}) });
519
+ ctx.ui.notify(`openmerit: queued your request until active setup work finishes (${state.activeIntent})`, "info");
520
+ return { action: "handled" };
521
+ });
522
+ // A fresh demo can start in an empty directory. Re-scan only after Pi has
523
+ // settled the user's build turn, so setup sees a complete application call
524
+ // instead of racing partially written files.
525
+ pi.on("turn_end", async (_event, ctx) => {
526
+ if (!ctx.hasUI || state.proactiveNudgesPaused || state.activeIntent)
527
+ return;
528
+ store ??= new ProjectStore(ctx.cwd);
529
+ const projectState = await store.readState();
530
+ if (projectState?.cycle.targetId || (projectState && projectState.cycle.stage !== "unconfigured"))
531
+ return;
532
+ const detectedApplicationTargets = detectApplicationLlmTargets(ctx.cwd);
533
+ if (!detectedApplicationTargets.length)
534
+ return;
535
+ ctx.ui.notify(`openmerit: detected ${detectedApplicationTargets.length} application LLM target candidate(s) after the build; starting evidence setup`, "info");
536
+ await dispatchIntent(pi, ctx, state, store, "establish_evals", registerCompletionTool, {
537
+ applicationTargetRequired: true,
538
+ detectedApplicationTargets: detectedTargetsConstraint(detectedApplicationTargets),
539
+ ...(detectedApplicationTargets.length === 1 && detectedApplicationTargets[0].configuredModelId ? { expectedIncumbentModelId: detectedApplicationTargets[0].configuredModelId } : {}),
540
+ });
541
+ });
286
542
  pi.on("session_start", async (_event, ctx) => {
287
543
  store = new ProjectStore(ctx.cwd);
288
544
  let projectState = await store.readState();
289
545
  if (projectState) {
290
546
  state.lifecycleStage = projectState.cycle.stage;
547
+ state.proactiveNudgesPaused = projectState.automation.paused === true;
291
548
  state.activeIntent = projectState.activeIntent?.id ?? null;
292
- if (projectState.activeIntent)
549
+ if (projectState.activeIntent) {
293
550
  registerCompletionTool(projectState.activeIntent.kind);
551
+ if (ctx.hasUI || ctx.mode === "rpc") {
552
+ pi.sendMessage({
553
+ customType: INTENT_MESSAGE_TYPE,
554
+ content: privateIntentMessage(projectState.activeIntent),
555
+ display: false,
556
+ details: { intentId: projectState.activeIntent.id, kind: projectState.activeIntent.kind, resumed: true },
557
+ }, { deliverAs: "followUp", triggerTurn: true });
558
+ }
559
+ }
560
+ state.applicationTarget = projectState.cycle.targetId ? "confirmed" : "not_detected";
294
561
  state.taskProfile = projectState.cycle.taskProfileId ? "configured" : "not_configured";
295
562
  state.metricReadiness = projectState.observability?.status === "ready" ? "assessed" : "pending";
296
563
  state.evaluationReadiness = projectState.cycle.taskProfileId ? "assessed" : "not_assessed";
297
- const automation = await new OpenMeritCoordinator(store, piHarnessAdapter(pi)).reconcileAutomation();
564
+ state.incumbentCandidate = projectState.cycle.activeCandidateId ?? null;
565
+ state.candidateCount = projectState.candidates?.length ?? 0;
566
+ state.assessmentCount = projectState.assessments?.length ?? 0;
567
+ state.evaluationSpend = projectState.automation.evaluationSpend ?? 0;
568
+ state.proposedCandidate = projectState.cycle.proposedCandidateId ?? null;
569
+ if (!projectState.cycle.targetId) {
570
+ if (ctx.hasUI && !state.proactiveNudgesPaused && projectState.cycle.stage === "unconfigured") {
571
+ const detectedApplicationTargets = detectApplicationLlmTargets(ctx.cwd);
572
+ if (detectedApplicationTargets.length) {
573
+ await dispatchIntent(pi, ctx, state, store, "establish_evals", registerCompletionTool, {
574
+ applicationTargetRequired: true,
575
+ detectedApplicationTargets: detectedTargetsConstraint(detectedApplicationTargets),
576
+ ...(detectedApplicationTargets.length === 1 && detectedApplicationTargets[0].configuredModelId ? { expectedIncumbentModelId: detectedApplicationTargets[0].configuredModelId } : {}),
577
+ });
578
+ }
579
+ }
580
+ else if (state.proactiveNudgesPaused) {
581
+ ctx.ui.notify("openmerit: proactive checks remain paused; use /openmerit resume to re-enable them", "info");
582
+ }
583
+ else {
584
+ ctx.ui.notify("openmerit: existing state has no confirmed application LLM target; use /openmerit setup", "warning");
585
+ }
586
+ return;
587
+ }
588
+ if (state.proactiveNudgesPaused || projectState.cycle.policy?.checkPolicy.execution.mode !== "persistent" ||
589
+ !projectState.cycle.policy?.checkPolicy.execution.infrastructureChangesAllowed) {
590
+ try {
591
+ await removePiWakeup(ctx.cwd);
592
+ }
593
+ catch (error) {
594
+ ctx.ui.notify(`openmerit: could not remove prior wakeup: ${error instanceof Error ? error.message : String(error)}`, "warning");
595
+ }
596
+ }
597
+ const automation = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).reconcileAutomation();
298
598
  state.nextAssessment = automation.plan.nextWakeupAt ??
299
599
  (automation.plan.scheduling === "external_scheduler_required" ? "external scheduler required" : "signal driven");
300
600
  if (automation.plan.gaps.length) {
@@ -306,6 +606,7 @@ export default function openMeritExtension(pi) {
306
606
  id: `signal-${crypto.randomUUID()}`,
307
607
  type: "model_catalog_changed",
308
608
  occurredAt: new Date().toISOString(),
609
+ targetId: projectState.cycle.targetId,
309
610
  fingerprint,
310
611
  }, registerCompletionTool);
311
612
  if (!state.activeIntent) {
@@ -314,35 +615,30 @@ export default function openMeritExtension(pi) {
314
615
  id: `signal-${crypto.randomUUID()}`,
315
616
  type: "scheduled_tick",
316
617
  occurredAt: new Date().toISOString(),
618
+ targetId: projectState.cycle.targetId,
317
619
  }, registerCompletionTool);
318
620
  }
319
621
  }
320
622
  else if (ctx.hasUI && !state.proactiveNudgesPaused) {
321
- ctx.ui.notify("openmerit: no frontier profile found; asking Pi to establish evals and observability", "info");
322
- await dispatchIntent(pi, ctx, state, store, "establish_evals", registerCompletionTool);
623
+ const detectedApplicationTargets = detectApplicationLlmTargets(ctx.cwd);
624
+ if (detectedApplicationTargets.length) {
625
+ ctx.ui.notify(`openmerit: detected ${detectedApplicationTargets.length} application LLM target candidate(s); asking Pi to confirm the target before setup`, "info");
626
+ await dispatchIntent(pi, ctx, state, store, "establish_evals", registerCompletionTool, {
627
+ applicationTargetRequired: true,
628
+ detectedApplicationTargets: detectedTargetsConstraint(detectedApplicationTargets),
629
+ ...(detectedApplicationTargets.length === 1 && detectedApplicationTargets[0].configuredModelId ? { expectedIncumbentModelId: detectedApplicationTargets[0].configuredModelId } : {}),
630
+ });
631
+ }
632
+ else {
633
+ ctx.ui.notify("openmerit: no application LLM call detected yet; automatic setup will check again after Pi finishes a build turn (/openmerit setup is available for recovery)", "info");
634
+ }
323
635
  }
324
- restoreState(ctx, state);
325
- });
326
- pi.on("agent_settled", async (_event, ctx) => {
327
- if (suppressNextSettledObservation) {
328
- suppressNextSettledObservation = false;
329
- return;
636
+ if (projectState?.activeIntent) {
637
+ ctx.ui.notify(`openmerit: recovered interrupted work (${projectState.activeIntent.id}); use /openmerit cancel to abandon it`, "warning");
330
638
  }
331
- store ??= new ProjectStore(ctx.cwd);
332
- const current = await store.readState();
333
- if (!current || current.activeIntent)
334
- return;
335
- await reportAutomationSignal(pi, ctx, state, store, {
336
- protocolVersion: PROTOCOL_VERSION,
337
- id: `signal-${crypto.randomUUID()}`,
338
- type: "product_task_completed",
339
- occurredAt: new Date().toISOString(),
340
- modelId: ctx.model ? `${ctx.model.provider}/${ctx.model.id}` : "unknown",
341
- contextTokens: ctx.getContextUsage()?.tokens ?? null,
342
- }, registerCompletionTool);
343
639
  });
344
640
  pi.registerCommand("openmerit", {
345
- description: "OpenMerit: status, setup, assess, frontier, approve, pause/resume, logs, doctor",
641
+ description: "OpenMerit: status, setup, assess, frontier, approve, cancel, pause/resume, logs, doctor",
346
642
  handler: async (args, ctx) => {
347
643
  const subcommand = args.trim().split(/\s+/)[0]?.toLowerCase() || "status";
348
644
  switch (subcommand) {
@@ -351,11 +647,27 @@ export default function openMeritExtension(pi) {
351
647
  return;
352
648
  case "setup":
353
649
  store ??= new ProjectStore(ctx.cwd);
354
- await dispatchIntent(pi, ctx, state, store, "establish_evals", registerCompletionTool);
650
+ {
651
+ const detectedApplicationTargets = detectApplicationLlmTargets(ctx.cwd);
652
+ await dispatchIntent(pi, ctx, state, store, "establish_evals", registerCompletionTool, {
653
+ applicationTargetRequired: true,
654
+ detectedApplicationTargets: detectedTargetsConstraint(detectedApplicationTargets),
655
+ ...(detectedApplicationTargets.length === 1 && detectedApplicationTargets[0].configuredModelId ? { expectedIncumbentModelId: detectedApplicationTargets[0].configuredModelId } : {}),
656
+ });
657
+ }
355
658
  return;
356
659
  case "assess":
357
660
  store ??= new ProjectStore(ctx.cwd);
358
- await dispatchIntent(pi, ctx, state, store, "run_assessment", registerCompletionTool);
661
+ {
662
+ const assessment = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).assessBaseline();
663
+ state.lifecycleStage = assessment.state.cycle.stage;
664
+ ctx.ui.notify(assessment.errors.length
665
+ ? `openmerit: baseline is not ready: ${assessment.errors.join("; ")}`
666
+ : "openmerit: baseline meets the required metric thresholds; starting candidate discovery", assessment.errors.length ? "warning" : "info");
667
+ if (assessment.decision.action === "issue_intent") {
668
+ await dispatchIntent(pi, ctx, state, store, assessment.decision.intentKind, registerCompletionTool);
669
+ }
670
+ }
359
671
  return;
360
672
  case "frontier":
361
673
  store ??= new ProjectStore(ctx.cwd);
@@ -386,22 +698,81 @@ export default function openMeritExtension(pi) {
386
698
  await dispatchIntent(pi, ctx, state, store, "apply_model_swap", registerCompletionTool);
387
699
  return;
388
700
  }
389
- case "pause":
701
+ case "cancel": {
702
+ store ??= new ProjectStore(ctx.cwd);
703
+ const projectState = await store.readState();
704
+ if (!projectState?.activeIntent) {
705
+ ctx.ui.notify("openmerit: there is no active work to cancel", "warning");
706
+ return;
707
+ }
708
+ const cancelled = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).acceptResult({
709
+ protocolVersion: PROTOCOL_VERSION,
710
+ intentId: projectState.activeIntent.id,
711
+ status: "cancelled",
712
+ completedAt: new Date().toISOString(),
713
+ summary: "Cancelled explicitly by the user after an interrupted harness run.",
714
+ evidence: [],
715
+ });
716
+ state.activeIntent = null;
717
+ state.lifecycleStage = cancelled.state.cycle.stage;
718
+ registerCompletionTool();
719
+ releaseQueuedUserInput();
720
+ ctx.ui.notify(`openmerit: cancelled ${projectState.activeIntent.id}`, "info");
721
+ return;
722
+ }
723
+ case "pause": {
724
+ store ??= new ProjectStore(ctx.cwd);
725
+ const existing = await store.readState();
726
+ const now = new Date().toISOString();
727
+ const paused = existing
728
+ ? { ...existing, automation: { ...existing.automation, paused: true }, updatedAt: now }
729
+ : {
730
+ schemaVersion: 1,
731
+ cycle: { id: `cycle-${crypto.randomUUID()}`, stage: "unconfigured", verifiedSwapCount: 0, baselineEvidenceSufficient: false },
732
+ automation: { setupNudgeShown: false, observedCompletedTasks: 0, lastAssessmentTaskCount: 0, paused: true },
733
+ updatedAt: now,
734
+ };
735
+ await store.writeState(paused);
390
736
  state.proactiveNudgesPaused = true;
737
+ state.nextAssessment = "paused";
738
+ try {
739
+ await removePiWakeup(ctx.cwd);
740
+ }
741
+ catch (error) {
742
+ ctx.ui.notify(`openmerit: could not unload wakeup: ${error instanceof Error ? error.message : String(error)}`, "warning");
743
+ }
391
744
  ctx.ui.notify("openmerit: proactive nudges paused", "info");
392
745
  return;
393
- case "resume":
746
+ }
747
+ case "resume": {
748
+ store ??= new ProjectStore(ctx.cwd);
749
+ const existing = await store.readState();
750
+ if (existing) {
751
+ await store.writeState({ ...existing, automation: { ...existing.automation, paused: false }, updatedAt: new Date().toISOString() });
752
+ }
394
753
  state.proactiveNudgesPaused = false;
754
+ if (existing?.cycle.targetId) {
755
+ const automation = await new OpenMeritCoordinator(store, piHarnessAdapter(pi, ctx)).reconcileAutomation();
756
+ state.nextAssessment = automation.plan.nextWakeupAt ??
757
+ (automation.plan.scheduling === "external_scheduler_required" ? "external scheduler required" : "signal driven");
758
+ if (automation.plan.gaps.length) {
759
+ ctx.ui.notify(`openmerit automation: ${automation.plan.gaps.join(" ")}`, "warning");
760
+ }
761
+ }
395
762
  ctx.ui.notify("openmerit: proactive nudges resumed", "info");
396
763
  return;
764
+ }
397
765
  case "doctor":
398
766
  ctx.ui.notify([
399
767
  "OpenMerit doctor",
400
768
  "Pi adapter : loaded",
401
769
  "intent executor : connected",
402
770
  `audit log : ${store?.eventsPath ?? `${ctx.cwd}/.openmerit/events.jsonl`}`,
403
- "model mutation : disabled",
404
- "persistent schedule: unavailable (active-session ticks only)",
771
+ "Pi model mutation : prohibited",
772
+ "application swap : target-bound after approval",
773
+ `persistent schedule: ${process.platform === "darwin" ? "opt-in LaunchAgent when policy permits" : "unavailable on this platform"}`,
774
+ `schedule file : ${piWakeupPlistPath(ctx.cwd)}`,
775
+ `schedule status : ${ctx.cwd}/.openmerit/scheduler-status.json`,
405
776
  `interactive UI : ${ctx.hasUI ? "available" : "unavailable"}`,
406
777
  ].join("\n"), "info");
407
778
  return;
@@ -416,7 +787,7 @@ export default function openMeritExtension(pi) {
416
787
  ].join("\n"), "info");
417
788
  return;
418
789
  default:
419
- ctx.ui.notify("openmerit: usage: /openmerit [status|setup|assess|frontier|approve|pause|resume|logs|doctor]", "warning");
790
+ ctx.ui.notify("openmerit: usage: /openmerit [status|setup|assess|frontier|approve|cancel|pause|resume|logs|doctor]", "warning");
420
791
  }
421
792
  },
422
793
  });