openmerit 0.1.5 → 0.1.6-preview.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (91) hide show
  1. package/CHANGELOG.md +14 -0
  2. package/README.md +33 -9
  3. package/dist/core/src/index.d.ts +12 -1
  4. package/dist/core/src/index.js +552 -40
  5. package/dist/pi/src/index.d.ts +17 -0
  6. package/dist/pi/src/index.js +448 -77
  7. package/dist/pi/src/scheduler.d.ts +11 -0
  8. package/dist/pi/src/scheduler.js +137 -0
  9. package/dist/pi/src/wakeup.d.ts +2 -0
  10. package/dist/pi/src/wakeup.js +108 -0
  11. package/dist/protocol/src/index.d.ts +86 -4
  12. package/dist/protocol/src/index.js +1 -1
  13. package/dist/protocol/src/schemas.d.ts +128 -2
  14. package/dist/protocol/src/schemas.js +57 -1
  15. package/dist/terminal/public/app.js +297 -0
  16. package/dist/terminal/public/brands/anthropic.png +0 -0
  17. package/dist/terminal/public/brands/baai.png +0 -0
  18. package/dist/terminal/public/brands/baseten.png +0 -0
  19. package/dist/terminal/public/brands/cerebras.png +0 -0
  20. package/dist/terminal/public/brands/cohere.png +0 -0
  21. package/dist/terminal/public/brands/deepseek.ico +0 -0
  22. package/dist/terminal/public/brands/google.png +0 -0
  23. package/dist/terminal/public/brands/groq.ico +0 -0
  24. package/dist/terminal/public/brands/lm-studio.png +0 -0
  25. package/dist/terminal/public/brands/meta.ico +0 -0
  26. package/dist/terminal/public/brands/mistral.png +0 -0
  27. package/dist/terminal/public/brands/nomic.png +0 -0
  28. package/dist/terminal/public/brands/ollama.png +0 -0
  29. package/dist/terminal/public/brands/openai.png +0 -0
  30. package/dist/terminal/public/brands/openrouter.png +0 -0
  31. package/dist/terminal/public/brands/qwen.png +0 -0
  32. package/dist/terminal/public/brands/vllm.ico +0 -0
  33. package/dist/terminal/public/brands/vllm.png +0 -0
  34. package/dist/terminal/public/favicon.svg +1 -0
  35. package/dist/terminal/public/flow.css +1 -0
  36. package/dist/terminal/public/flow.js +770 -0
  37. package/dist/terminal/public/index.html +21 -0
  38. package/dist/terminal/public/styles.css +779 -0
  39. package/dist/terminal/src/activity-merge.mjs +64 -0
  40. package/dist/terminal/src/browser.mjs +29 -0
  41. package/dist/terminal/src/cli.mjs +60 -0
  42. package/dist/terminal/src/collect.mjs +311 -0
  43. package/dist/terminal/src/discovery.mjs +93 -0
  44. package/dist/terminal/src/hardware.mjs +57 -0
  45. package/dist/terminal/src/project-activity.mjs +156 -0
  46. package/dist/terminal/src/sample.mjs +171 -0
  47. package/dist/terminal/src/server.mjs +56 -0
  48. package/dist/terminal/src/services.mjs +62 -0
  49. package/dist/terminal/src/topology.mjs +30 -0
  50. package/docs/adapter-guide.md +24 -7
  51. package/docs/architecture.md +8 -4
  52. package/docs/automation.md +10 -2
  53. package/docs/budgets.md +37 -0
  54. package/docs/commands.md +85 -0
  55. package/docs/demo-backfill.md +29 -0
  56. package/docs/demo-fieldkit.md +47 -0
  57. package/docs/demo-placement.md +30 -0
  58. package/docs/demo-spam.md +15 -0
  59. package/docs/demo-support.md +42 -0
  60. package/docs/demo.md +57 -0
  61. package/docs/first-trial.md +60 -0
  62. package/docs/getting-started.md +18 -8
  63. package/docs/index.md +40 -0
  64. package/docs/inference-terminal.md +439 -0
  65. package/docs/lifecycle.md +9 -9
  66. package/docs/memo.md +126 -0
  67. package/docs/metrics-and-evidence.md +11 -3
  68. package/docs/operations.md +12 -3
  69. package/docs/pi-extension.md +17 -7
  70. package/docs/roadmap.md +4 -2
  71. package/docs/security.md +15 -1
  72. package/docs/site-artwork-linocut.md +23 -0
  73. package/docs/site-artwork-miniature-diverse.md +28 -0
  74. package/docs/site-artwork-miniature.md +26 -0
  75. package/docs/site-demo.md +177 -0
  76. package/docs/site-design.md +94 -0
  77. package/docs/site-documentation.md +83 -0
  78. package/docs/site-dynamic-og.md +35 -0
  79. package/docs/site-faq-maintenance.md +115 -0
  80. package/docs/site-hero-resolution.md +60 -0
  81. package/docs/site-illustration-sequences.md +227 -0
  82. package/docs/site-inference-terminal.md +203 -0
  83. package/docs/site-memo.md +39 -0
  84. package/docs/site-og-image.md +38 -0
  85. package/docs/site-og-workshop.md +21 -0
  86. package/docs/site-section-artwork.md +56 -0
  87. package/docs/site-skill-review.md +57 -0
  88. package/docs/site-terminal-preview.md +85 -0
  89. package/docs/testing.md +85 -3
  90. package/docs/troubleshooting.md +55 -0
  91. package/package.json +48 -7
@@ -1,14 +1,14 @@
1
1
  import { METRIC_CATALOG, OPENMERIT_SCHEMA_DIALECT, PROTOCOL_VERSION, REQUIRED_CAPABILITIES, validateIntentOutputSchema, } from "../../protocol/src/index.js";
2
2
  export { OPENMERIT_DIRECTORY, ProjectStore, } from "./store.js";
3
3
  const outcomes = {
4
- establish_evals: "Infer the current task profile and metric objectives, ask the user to confirm requirements, evaluation budget, check cadence, thresholds, scheduling mode, and automation permissions, establish applicable evals and observability, verify them, and return the confirmed taskProfile and policy. Do not change model configuration.",
5
- instrument_observability: "Establish and verify the missing observability required by the confirmed task profile.",
6
- run_assessment: "Assess whether current baseline evidence satisfies every required metric and sample threshold. Return measured, missing, and insufficient-evidence metrics explicitly.",
7
- discover_candidates: "Discover credible model candidates for the confirmed task profile using current capability, reputation, price, availability, and benchmark signals without exceeding the approved evaluation budget.",
8
- run_challenger_trials: "Run comparable controlled challenger trials under the approved budget and return candidate assessments with durable evidence.",
4
+ establish_evals: "Identify the real application LLM call or shared application route under evaluation and its exact incumbent application model. Infer the metrics required for that application's task from its outputs, tools, failure modes, and representative work. Before running setup, tell the user which metrics will be collected, why each matters, whether it uses repeated identical inputs, representative task instances, both, or load testing, the case and repetition counts, aggregation, any true performance threshold, and estimated evaluation cost; ask the user to confirm or edit that plan together with the target, budget, check cadence, scheduling mode, and automation permissions. Never turn a sample-count requirement into a metric-value constraint. Rate, distribution, reliability, percentile, variance, and average claims require multiple runs. Build runnable task-specific evaluation cases and application-call collectors for every confirmed metric. Execute a representative smoke run through the same application authentication and runtime path, run the grader, and inspect actual outputs. Preserve the commands and resulting artifacts as setup evidence. Mark any metric missing if its collector or grader cannot produce a value; do not declare coverage ready based only on code existing. Do not count a setup smoke run as a production baseline sample. Return the confirmed taskProfile, incumbent, and policy. During setup, research a preliminary model shortlist with current sourced price estimates and reputation; report an empty list if no credible leads are available. These are research priors only: do not treat them as measured task results or change model configuration. The coding harness model is not an evaluation target.",
5
+ instrument_observability: "Establish the missing observability required by the confirmed task profile. Exercise each collector and task-specific grader through a representative application run with the same authentication and runtime path, inspect actual metric outputs, and preserve verification artifacts. Report a metric covered only if that path produced its value; otherwise keep it missing. Do not count a setup smoke run as a production baseline sample.",
6
+ run_assessment: "Assess existing application-target baseline metric windows against every required metric and sample threshold. Do not synthesize samples, count setup smoke runs as production data, or wait for new data during this check. If a required window is absent or immature, promptly return baselineEvidenceSufficient false with measured, missing, and insufficient-evidence metrics explicitly and cite the records inspected.",
7
+ discover_candidates: "Refresh the preliminary model leads collected at setup and discover other credible candidates for the confirmed task profile using current capability, reputation, price, availability, and benchmark signals without exceeding the approved evaluation budget. Preliminary leads are research priors, not measured task evidence.",
8
+ run_challenger_trials: "Run a balanced, reproducible comparison of the frozen incumbent and challenger set under the approved budget. Use the same evaluation cases, application revision, prompt, tools, grader, and run count for every usable model; use a recorded seed to shuffle execution order. Attribute every run to the exact application model, preserve per-run quality, cost, token, and latency evidence, and return candidate assessments plus one durable experiment manifest. Never substitute catalogue prices for measured application-call cost.",
9
9
  calculate_frontier: "Calculate the Pareto frontier using the OpenMerit conformance definition, explain every candidate classification, and return a FrontierSnapshot with calculation evidence.",
10
10
  investigate_regression: "Investigate the observed regression and return its likely cause, affected metrics, and supporting evidence.",
11
- apply_model_swap: "Apply the approved model change in the harness or product target and return proof of the resulting configuration. Do not broaden the authorized change.",
11
+ apply_model_swap: "Apply the approved model change only to the bound application LLM target and return proof of the resulting application configuration. Never change the coding harness model and do not broaden the authorized change.",
12
12
  verify_model_swap: "Verify the applied model against the approved post-swap evaluation window and report any material regression.",
13
13
  rollback_model_swap: "Restore the previously verified model configuration and validate the rollback.",
14
14
  };
@@ -17,31 +17,47 @@ export function createOpenMeritIntent(kind, options = {}) {
17
17
  protocolVersion: PROTOCOL_VERSION,
18
18
  id: options.id ?? `intent-${crypto.randomUUID()}`,
19
19
  kind,
20
+ ...(options.targetId ? { targetId: options.targetId } : {}),
20
21
  taskProfileId: options.taskProfileId ?? "current-project",
21
22
  authorizationPolicyId: options.authorizationPolicyId ?? "observe-and-evaluate-only",
22
23
  requestedOutcome: outcomes[kind],
23
24
  constraints: {
24
25
  modelMutationAllowed: kind === "apply_model_swap" || kind === "rollback_model_swap",
26
+ harnessModelMutationAllowed: false,
27
+ mutationTarget: "application_llm_call",
25
28
  requireVerifiedArtifacts: true,
26
29
  distinguishSingleRunFromMultiRunClaims: true,
27
30
  ...(options.policy ? { evaluationBudget: options.policy.evaluationBudget } : {}),
31
+ ...(options.policy ? {
32
+ evaluationSpend: options.evaluationSpend ?? 0,
33
+ remainingEvaluationBudget: Math.max(0, options.policy.evaluationBudget.maximumSpend - (options.evaluationSpend ?? 0)),
34
+ } : {}),
28
35
  ...(options.policy ? { checkPolicy: options.policy.checkPolicy } : {}),
36
+ ...(options.additionalConstraints ?? {}),
29
37
  },
30
38
  requiredEvidence: kind === "establish_evals"
31
39
  ? [
32
40
  { id: "task-profile", description: "The inferred task profile and the user constraints it represents." },
41
+ { id: "application-target", description: "The user-confirmed application LLM call or shared route, including stable call-site and route identifiers." },
42
+ { id: "incumbent-model", description: "The exact application model configured before challenger trials." },
33
43
  { id: "evaluation-artifacts", description: "Runnable evaluation artifacts and verification results." },
34
44
  { id: "observability-coverage", description: "The observable metrics and explicit coverage gaps." },
45
+ { id: "preliminary-model-research", description: "Sources for preliminary model pricing and reputation, or an explanation that no credible leads were found." },
35
46
  ]
36
- : kind === "calculate_frontier"
47
+ : kind === "run_challenger_trials"
37
48
  ? [
38
- { id: "frontier-snapshot", description: "A complete harness-calculated FrontierSnapshot." },
39
- { id: "frontier-calculation", description: "Durable evidence of the frontier calculation." },
49
+ { id: "experiment-manifest", description: "A durable manifest containing the seed, frozen cases, application revision, exact model IDs, and per-run evidence." },
50
+ { id: "assessment-results", description: "Comparable candidate assessments tied to the active task profile." },
40
51
  ]
41
- : [
42
- { id: "assessment-results", description: "Assessment results tied to the active task profile." },
43
- { id: "metric-coverage", description: "Observed, derived, and insufficient-evidence metric states." },
44
- ],
52
+ : kind === "calculate_frontier"
53
+ ? [
54
+ { id: "frontier-snapshot", description: "A complete harness-calculated FrontierSnapshot." },
55
+ { id: "frontier-calculation", description: "Durable evidence of the frontier calculation." },
56
+ ]
57
+ : [
58
+ { id: "assessment-results", description: "Assessment results tied to the active task profile." },
59
+ { id: "metric-coverage", description: "Observed, derived, and insufficient-evidence metric states." },
60
+ ],
45
61
  requestedAt: options.requestedAt ?? new Date().toISOString(),
46
62
  };
47
63
  }
@@ -126,13 +142,16 @@ const knownMetricIds = new Set(METRIC_CATALOG.map((metric) => metric.id));
126
142
  function isRecord(value) {
127
143
  return typeof value === "object" && value !== null && !Array.isArray(value);
128
144
  }
129
- function validateSetupOutput(output) {
130
- if (!isRecord(output) || !isRecord(output.taskProfile) || !isRecord(output.policy)) {
131
- return ["setup output requires taskProfile and policy"];
145
+ function validateSetupOutput(output, expectedIncumbentModelId) {
146
+ if (!isRecord(output) || !isRecord(output.taskProfile) || !isRecord(output.incumbent) || !isRecord(output.policy)) {
147
+ return ["setup output requires taskProfile, incumbent, and policy"];
132
148
  }
133
149
  const profile = output.taskProfile;
134
150
  const policy = output.policy;
135
151
  const errors = [];
152
+ if (expectedIncumbentModelId && output.incumbent.modelId !== expectedIncumbentModelId) {
153
+ errors.push(`incumbent model ${String(output.incumbent.modelId)} does not match the application source model ${expectedIncumbentModelId}`);
154
+ }
136
155
  if (typeof profile.id !== "string" || typeof profile.name !== "string" ||
137
156
  typeof profile.goal !== "string" || !Number.isInteger(profile.revision) ||
138
157
  typeof profile.confirmedAt !== "string" || !Array.isArray(profile.objectives) ||
@@ -148,6 +167,49 @@ function validateSetupOutput(output) {
148
167
  errors.push("taskProfile contains an invalid metric objective");
149
168
  break;
150
169
  }
170
+ const samplingPlan = objective.samplingPlan;
171
+ if (!isRecord(samplingPlan) || typeof samplingPlan.strategy !== "string" ||
172
+ !Number.isInteger(samplingPlan.representativeCaseCount) || samplingPlan.representativeCaseCount < 1 ||
173
+ !Number.isInteger(samplingPlan.repetitionsPerCase) || samplingPlan.repetitionsPerCase < 1 ||
174
+ typeof samplingPlan.aggregation !== "string" || typeof samplingPlan.rationale !== "string" ||
175
+ !samplingPlan.rationale.trim()) {
176
+ errors.push(`${objective.metricId} requires a complete sampling plan`);
177
+ continue;
178
+ }
179
+ const plannedSamples = samplingPlan.representativeCaseCount *
180
+ samplingPlan.repetitionsPerCase;
181
+ if (objective.minimumSamples < plannedSamples) {
182
+ errors.push(`${objective.metricId} minimumSamples must cover its sampling plan`);
183
+ }
184
+ const aggregateClaim = ["mean", "rate", "distribution", "percentile"].includes(samplingPlan.aggregation);
185
+ const metricDefinition = METRIC_CATALOG.find((metric) => metric.id === objective.metricId);
186
+ if ((aggregateClaim || metricDefinition?.requiresRepeatedRuns) && plannedSamples < 2) {
187
+ errors.push(`${objective.metricId} requires multiple planned runs for its ${String(samplingPlan.aggregation)} claim`);
188
+ }
189
+ if ((aggregateClaim || metricDefinition?.requiresRepeatedRuns) && samplingPlan.strategy === "single_run") {
190
+ errors.push(`${objective.metricId} cannot use single_run for a repeated-run claim`);
191
+ }
192
+ if (isRecord(objective.constraint) &&
193
+ ["task_success", "structured_output_reliability", "hallucination_rate", "retry_rate", "recovery_ability", "human_intervention_rate"].includes(objective.metricId) &&
194
+ (typeof objective.constraint.value !== "number" || objective.constraint.value < 0 || objective.constraint.value > 1)) {
195
+ errors.push(`${objective.metricId} ratio constraint must be between 0 and 1`);
196
+ }
197
+ }
198
+ }
199
+ if (!isRecord(profile.applicationTarget) || profile.applicationTarget.kind !== "application_llm_call" ||
200
+ typeof profile.applicationTarget.id !== "string" || !profile.applicationTarget.id.trim() ||
201
+ typeof profile.applicationTarget.applicationId !== "string" || !profile.applicationTarget.applicationId.trim() ||
202
+ typeof profile.applicationTarget.routeKey !== "string" || !profile.applicationTarget.routeKey.trim() ||
203
+ typeof profile.applicationTarget.confirmedAt !== "string" ||
204
+ !Array.isArray(profile.applicationTarget.callSites) || profile.applicationTarget.callSites.length === 0 ||
205
+ profile.applicationTarget.callSites.some((callSite) => typeof callSite !== "string" || !callSite.trim())) {
206
+ errors.push("taskProfile requires a user-confirmed application LLM target with stable call sites and route key");
207
+ }
208
+ if (isRecord(profile.applicationTarget)) {
209
+ errors.push(...validateTargetId(output.incumbent.targetId, profile.applicationTarget.id, "incumbent"));
210
+ if (typeof output.incumbent.candidateId !== "string" || !output.incumbent.candidateId.trim() ||
211
+ typeof output.incumbent.modelId !== "string" || !output.incumbent.modelId.trim()) {
212
+ errors.push("incumbent requires stable candidateId and exact modelId");
151
213
  }
152
214
  }
153
215
  const budget = policy.evaluationBudget;
@@ -158,20 +220,71 @@ function validateSetupOutput(output) {
158
220
  policy.requirePostSwapVerification !== true || typeof policy.rollbackOnRegression !== "boolean") {
159
221
  errors.push("policy must be a valid supervised-graduation policy with an evaluation budget");
160
222
  }
223
+ const checkPolicy = policy.checkPolicy;
224
+ if (!isRecord(checkPolicy) || !isRecord(checkPolicy.baselineAssessment) ||
225
+ (typeof checkPolicy.baselineAssessment.afterCompletedTasks !== "number" &&
226
+ typeof checkPolicy.baselineAssessment.afterElapsedSeconds !== "number")) {
227
+ errors.push("policy requires a baseline assessment cadence");
228
+ }
229
+ if (!isRecord(checkPolicy) || !isRecord(checkPolicy.postSwapVerification) ||
230
+ (typeof checkPolicy.postSwapVerification.afterCompletedTasks !== "number" &&
231
+ typeof checkPolicy.postSwapVerification.afterElapsedSeconds !== "number")) {
232
+ errors.push("policy requires a post-swap verification cadence");
233
+ }
161
234
  if (!isRecord(output.observability)) {
162
235
  errors.push("setup output requires an observability coverage report");
163
236
  }
164
237
  else if (Array.isArray(profile.objectives)) {
165
238
  errors.push(...validateObservabilityCoverage(profile, output.observability, false));
166
239
  }
240
+ if (Array.isArray(output.preliminaryModelLeads) && isRecord(profile.applicationTarget)) {
241
+ const seen = new Set();
242
+ for (const lead of output.preliminaryModelLeads) {
243
+ if (!isRecord(lead))
244
+ continue;
245
+ errors.push(...validateTargetId(lead.targetId, profile.applicationTarget.id, `preliminary lead ${String(lead.modelId)}`));
246
+ const identity = `${String(lead.modelId)}@${String(lead.modelVersion ?? "")}`;
247
+ if (seen.has(identity))
248
+ errors.push(`duplicate preliminary model lead ${identity}`);
249
+ seen.add(identity);
250
+ }
251
+ }
252
+ return errors;
253
+ }
254
+ function validateMetricAggregation(metric, objective) {
255
+ const errors = [];
256
+ if (metric.metricId === "total_task_cost" && metric.state === "measured" &&
257
+ metric.scope === "sample_window" && metric.aggregation !== "sum") {
258
+ errors.push("total_task_cost sample windows must use sum aggregation because the value is used for evaluation-budget accounting");
259
+ }
260
+ if (objective?.samplingPlan && metric.state === "measured" &&
261
+ metric.aggregation !== objective.samplingPlan.aggregation &&
262
+ !(objective.metricId === "total_task_cost" && metric.aggregation === "sum")) {
263
+ errors.push(`${objective.metricId} aggregation ${metric.aggregation} does not match its sampling plan ${objective.samplingPlan.aggregation}`);
264
+ }
167
265
  return errors;
168
266
  }
267
+ function validateMetricWindow(metric, startedAt, completedAt) {
268
+ if (!metric.window)
269
+ return [];
270
+ const start = Date.parse(metric.window.startedAt);
271
+ const end = Date.parse(metric.window.endedAt);
272
+ const intentStart = Date.parse(startedAt);
273
+ const intentEnd = Date.parse(completedAt);
274
+ if (![start, end, intentStart, intentEnd].every(Number.isFinite) || end < start || start < intentStart || end > intentEnd) {
275
+ return [`${metric.metricId} window must be valid and contained within the intent execution window`];
276
+ }
277
+ return [];
278
+ }
169
279
  export function validateObservabilityCoverage(profile, coverage, requireReady) {
170
280
  const required = new Set(profile.objectives.filter((item) => item.required).map((item) => item.metricId));
171
281
  const declaredRequired = new Set(coverage.requiredMetricIds);
172
282
  const covered = new Set(coverage.coveredMetricIds);
173
283
  const missing = new Set(coverage.missingMetricIds);
174
284
  const errors = [];
285
+ if (coverage.targetId !== profile.applicationTarget.id) {
286
+ errors.push("observability target does not match the confirmed application LLM target");
287
+ }
175
288
  for (const metricId of required) {
176
289
  if (!declaredRequired.has(metricId))
177
290
  errors.push(`observability omits required metric ${metricId}`);
@@ -191,6 +304,15 @@ export function validateObservabilityCoverage(profile, coverage, requireReady) {
191
304
  }
192
305
  return errors;
193
306
  }
307
+ function activeTarget(config) {
308
+ return config.taskProfiles.find((item) => item.id === config.activeTaskProfileId)?.applicationTarget;
309
+ }
310
+ function validateTargetId(actual, expected, subject) {
311
+ return actual === expected ? [] : [`${subject} target does not match the confirmed application LLM target`];
312
+ }
313
+ function validateMetricTargets(metrics, targetId, subject) {
314
+ return metrics.flatMap((metric) => validateTargetId(metric.targetId, targetId, `${subject} metric ${metric.metricId}`));
315
+ }
194
316
  export function interpretIntentResult(intent, result, cycle, config) {
195
317
  const errors = [];
196
318
  if (result.protocolVersion !== PROTOCOL_VERSION)
@@ -208,6 +330,35 @@ export function interpretIntentResult(intent, result, cycle, config) {
208
330
  }
209
331
  const output = result.outputs;
210
332
  errors.push(...validateIntentOutputSchema(intent.kind, output).map((error) => `result outputs ${error}`));
333
+ if (isRecord(output) && (intent.kind === "run_assessment" || intent.kind === "run_challenger_trials")) {
334
+ const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
335
+ const objectives = new Map((profile?.objectives ?? []).map((objective) => [objective.metricId, objective]));
336
+ const metricGroups = intent.kind === "run_assessment"
337
+ ? [{ metrics: output.metrics, startedAt: undefined, completedAt: undefined }]
338
+ : (Array.isArray(output.assessments) ? output.assessments.map((assessment) => isRecord(assessment)
339
+ ? { metrics: assessment.metrics, startedAt: assessment.startedAt, completedAt: assessment.completedAt }
340
+ : { metrics: undefined, startedAt: undefined, completedAt: undefined }) : []);
341
+ for (const group of metricGroups) {
342
+ const metrics = group.metrics;
343
+ if (!Array.isArray(metrics))
344
+ continue;
345
+ for (const metric of metrics) {
346
+ if (!isRecord(metric))
347
+ continue;
348
+ const typedMetric = metric;
349
+ errors.push(...validateMetricAggregation(typedMetric, objectives.get(typedMetric.metricId)));
350
+ if (group.startedAt && group.completedAt)
351
+ errors.push(...validateMetricWindow(typedMetric, String(group.startedAt), String(group.completedAt)));
352
+ }
353
+ }
354
+ }
355
+ const target = activeTarget(config);
356
+ if (intent.kind !== "establish_evals") {
357
+ if (!target)
358
+ errors.push("no confirmed application LLM target exists");
359
+ else
360
+ errors.push(...validateTargetId(intent.targetId, target.id, "intent"));
361
+ }
211
362
  if (errors.length) {
212
363
  return { valid: false, errors, nextStage: completedStage(intent.kind, false) };
213
364
  }
@@ -216,20 +367,27 @@ export function interpretIntentResult(intent, result, cycle, config) {
216
367
  let cyclePatch;
217
368
  switch (intent.kind) {
218
369
  case "establish_evals": {
219
- errors.push(...validateSetupOutput(output));
370
+ errors.push(...validateSetupOutput(output, typeof intent.constraints.expectedIncumbentModelId === "string"
371
+ ? intent.constraints.expectedIncumbentModelId : undefined));
220
372
  if (!errors.length && isRecord(output)) {
221
373
  const taskProfile = output.taskProfile;
222
374
  const policy = output.policy;
223
375
  const observability = output.observability;
376
+ const incumbent = output.incumbent;
377
+ const preliminaryModelLeads = (output.preliminaryModelLeads ?? []);
224
378
  nextConfig = {
225
379
  schemaVersion: 1,
226
380
  activeTaskProfileId: taskProfile.id,
227
381
  taskProfiles: [...config.taskProfiles.filter((item) => item.id !== taskProfile.id), taskProfile],
228
382
  policy,
383
+ preliminaryModelLeads,
229
384
  };
230
385
  cyclePatch = {
386
+ targetId: taskProfile.applicationTarget.id,
231
387
  taskProfileId: taskProfile.id,
388
+ activeCandidateId: incumbent.candidateId,
232
389
  policy,
390
+ baselineEvidenceSufficient: false,
233
391
  observabilityReady: observability.status === "ready",
234
392
  };
235
393
  if (observability.status !== "ready")
@@ -246,7 +404,9 @@ export function interpretIntentResult(intent, result, cycle, config) {
246
404
  if (!profile)
247
405
  errors.push("no confirmed task profile exists for observability verification");
248
406
  else {
407
+ errors.push(...validateTargetId(output.targetId, profile.applicationTarget.id, "observability result"));
249
408
  const coverage = {
409
+ targetId: profile.applicationTarget.id,
250
410
  status: output.missingMetricIds.length ? "incomplete" : "ready",
251
411
  requiredMetricIds: profile.objectives.filter((item) => item.required).map((item) => item.metricId),
252
412
  coveredMetricIds: output.coveredMetricIds,
@@ -259,10 +419,18 @@ export function interpretIntentResult(intent, result, cycle, config) {
259
419
  }
260
420
  break;
261
421
  case "run_assessment":
262
- if (!isRecord(output) || typeof output.baselineEvidenceSufficient !== "boolean") {
422
+ if (!isRecord(output) || typeof output.baselineEvidenceSufficient !== "boolean" || !Array.isArray(output.metrics)) {
263
423
  errors.push("assessment result requires baselineEvidenceSufficient");
264
424
  }
265
425
  else {
426
+ if (target) {
427
+ errors.push(...validateTargetId(output.targetId, target.id, "baseline assessment"));
428
+ errors.push(...validateMetricTargets(output.metrics, target.id, "baseline assessment"));
429
+ const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
430
+ if (output.baselineEvidenceSufficient) {
431
+ errors.push(...baselineReadinessErrors(profile, output.metrics));
432
+ }
433
+ }
266
434
  cyclePatch = { baselineEvidenceSufficient: output.baselineEvidenceSufficient };
267
435
  if (!output.baselineEvidenceSufficient)
268
436
  nextStage = "collecting_baseline";
@@ -271,10 +439,61 @@ export function interpretIntentResult(intent, result, cycle, config) {
271
439
  case "discover_candidates":
272
440
  if (!isRecord(output) || !Array.isArray(output.candidates))
273
441
  errors.push("candidate discovery requires candidates");
442
+ else if (target) {
443
+ errors.push(...validateTargetId(output.targetId, target.id, "candidate discovery"));
444
+ const candidateIds = new Set();
445
+ for (const candidate of output.candidates) {
446
+ if (isRecord(candidate)) {
447
+ errors.push(...validateTargetId(candidate.targetId, target.id, `candidate ${String(candidate.candidateId)}`));
448
+ if (typeof candidate.candidateId === "string") {
449
+ if (candidateIds.has(candidate.candidateId))
450
+ errors.push(`duplicate candidate ${candidate.candidateId}`);
451
+ candidateIds.add(candidate.candidateId);
452
+ }
453
+ }
454
+ }
455
+ if (cycle.activeCandidateId && !candidateIds.has(cycle.activeCandidateId)) {
456
+ errors.push("candidate discovery must include the incumbent application model");
457
+ }
458
+ const maximum = config.policy?.evaluationBudget.maximumCandidateCount;
459
+ if (maximum !== undefined && output.candidates.length > maximum) {
460
+ errors.push(`candidate discovery exceeds maximumCandidateCount ${maximum}`);
461
+ }
462
+ }
274
463
  break;
275
464
  case "run_challenger_trials":
276
- if (!isRecord(output) || !Array.isArray(output.assessments))
277
- errors.push("challenger trials require assessments");
465
+ if (!isRecord(output) || !Array.isArray(output.assessments) || !isRecord(output.experimentManifest)) {
466
+ errors.push("challenger trials require assessments and an experimentManifest");
467
+ }
468
+ else {
469
+ if (target)
470
+ errors.push(...validateTargetId(output.targetId, target.id, "challenger trials"));
471
+ const expectedCandidateIds = Array.isArray(intent.constraints.candidates)
472
+ ? new Set(intent.constraints.candidates.flatMap((candidate) => isRecord(candidate) && typeof candidate.candidateId === "string" ? [candidate.candidateId] : []))
473
+ : undefined;
474
+ const assessedCandidateIds = new Set();
475
+ const productRevisions = new Set();
476
+ for (const assessment of output.assessments) {
477
+ assessedCandidateIds.add(assessment.candidateId);
478
+ productRevisions.add(assessment.productRevision);
479
+ if (target) {
480
+ errors.push(...validateTargetId(assessment.targetId, target.id, `challenger assessment ${assessment.id}`));
481
+ errors.push(...validateMetricTargets(assessment.metrics, target.id, `challenger assessment ${assessment.id}`));
482
+ }
483
+ const costs = assessment.metrics.filter((metric) => metric.metricId === "total_task_cost" && metric.state === "measured" &&
484
+ metric.value !== undefined && metric.value >= 0 && metric.evidence.length > 0);
485
+ if (costs.length !== 1) {
486
+ errors.push(`challenger assessment ${assessment.id} requires exactly one measured total_task_cost with evidence`);
487
+ }
488
+ }
489
+ if (expectedCandidateIds &&
490
+ (expectedCandidateIds.size !== assessedCandidateIds.size ||
491
+ [...expectedCandidateIds].some((candidateId) => !assessedCandidateIds.has(candidateId)))) {
492
+ errors.push("challenger assessments must exactly cover the frozen candidate set");
493
+ }
494
+ if (productRevisions.size !== 1)
495
+ errors.push("challenger assessments must use one application product revision");
496
+ }
278
497
  break;
279
498
  case "calculate_frontier": {
280
499
  if (!isRecord(output) || !Array.isArray(output.assessments) || !isRecord(output.frontierSnapshot)) {
@@ -286,6 +505,16 @@ export function interpretIntentResult(intent, result, cycle, config) {
286
505
  errors.push("no confirmed task profile exists for frontier verification");
287
506
  break;
288
507
  }
508
+ errors.push(...validateTargetId(output.targetId, profile.applicationTarget.id, "frontier result"));
509
+ const expectedAssessmentIds = Array.isArray(intent.constraints.assessments)
510
+ ? new Set(intent.constraints.assessments.flatMap((assessment) => isRecord(assessment) && typeof assessment.id === "string" ? [assessment.id] : []))
511
+ : undefined;
512
+ const suppliedAssessmentIds = new Set(output.assessments.map((assessment) => assessment.id));
513
+ if (expectedAssessmentIds &&
514
+ (expectedAssessmentIds.size !== suppliedAssessmentIds.size ||
515
+ [...expectedAssessmentIds].some((assessmentId) => !suppliedAssessmentIds.has(assessmentId)))) {
516
+ errors.push("frontier calculation must use exactly the accepted challenger assessments");
517
+ }
289
518
  const verification = verifyFrontierSnapshot(profile, output.assessments, output.frontierSnapshot);
290
519
  errors.push(...verification.errors);
291
520
  if (output.recommendation !== undefined) {
@@ -295,6 +524,12 @@ export function interpretIntentResult(intent, result, cycle, config) {
295
524
  else if (!output.frontierSnapshot.frontierCandidateIds.includes(output.recommendation.selectedCandidateId)) {
296
525
  errors.push("recommended candidate is not on the verified frontier");
297
526
  }
527
+ else if (output.recommendation.targetId !== profile.applicationTarget.id) {
528
+ errors.push("frontier recommendation target does not match the confirmed application LLM target");
529
+ }
530
+ else if (cycle.activeCandidateId && output.recommendation.currentCandidateId !== cycle.activeCandidateId) {
531
+ errors.push("frontier recommendation current candidate does not match the incumbent application model");
532
+ }
298
533
  else {
299
534
  cyclePatch = { proposedCandidateId: output.recommendation.selectedCandidateId };
300
535
  }
@@ -305,11 +540,16 @@ export function interpretIntentResult(intent, result, cycle, config) {
305
540
  if (!isRecord(output) || typeof output.regressionDetected !== "boolean" || !Array.isArray(output.affectedMetricIds)) {
306
541
  errors.push("regression result requires regressionDetected and affectedMetricIds");
307
542
  }
543
+ else if (target)
544
+ errors.push(...validateTargetId(output.targetId, target.id, "regression investigation"));
308
545
  break;
309
546
  case "apply_model_swap":
310
547
  if (!isRecord(output) || typeof output.appliedCandidateId !== "string") {
311
548
  errors.push("swap result requires appliedCandidateId");
312
549
  }
550
+ else if (target && output.targetId !== target.id) {
551
+ errors.push("swap target does not match the confirmed application LLM target");
552
+ }
313
553
  else if (cycle.proposedCandidateId && output.appliedCandidateId !== cycle.proposedCandidateId) {
314
554
  errors.push("applied candidate does not match the authorized proposal");
315
555
  }
@@ -324,6 +564,9 @@ export function interpretIntentResult(intent, result, cycle, config) {
324
564
  if (!isRecord(output) || typeof output.verified !== "boolean" || typeof output.regressionDetected !== "boolean") {
325
565
  errors.push("swap verification requires verified and regressionDetected");
326
566
  }
567
+ else if (target && output.targetId !== target.id) {
568
+ errors.push("swap verification target does not match the confirmed application LLM target");
569
+ }
327
570
  else if (!output.verified || output.regressionDetected) {
328
571
  nextStage = "verification_failed";
329
572
  }
@@ -335,6 +578,9 @@ export function interpretIntentResult(intent, result, cycle, config) {
335
578
  if (!isRecord(output) || typeof output.restoredCandidateId !== "string") {
336
579
  errors.push("rollback result requires restoredCandidateId");
337
580
  }
581
+ else if (target && output.targetId !== target.id) {
582
+ errors.push("rollback target does not match the confirmed application LLM target");
583
+ }
338
584
  else if (cycle.previousCandidateId && output.restoredCandidateId !== cycle.previousCandidateId) {
339
585
  errors.push("rollback did not restore the previous verified candidate");
340
586
  }
@@ -359,7 +605,7 @@ export function nextImprovementAction(state) {
359
605
  : { action: "issue_intent", intentKind: "instrument_observability", reason: "Required metric instrumentation is incomplete." };
360
606
  case "collecting_baseline":
361
607
  return state.baselineEvidenceSufficient
362
- ? { action: "issue_intent", intentKind: "run_assessment", reason: "Baseline evidence has reached the approved sufficiency threshold." }
608
+ ? { action: "issue_intent", intentKind: "discover_candidates", reason: "The baseline is ready for candidate discovery." }
363
609
  : { action: "observe", reason: "Continue collecting baseline production evidence." };
364
610
  case "baseline_ready":
365
611
  return { action: "issue_intent", intentKind: "discover_candidates", reason: "The baseline is ready for budgeted candidate discovery." };
@@ -411,22 +657,71 @@ function bounds(metric) {
411
657
  }
412
658
  function readinessErrors(profile, assessment) {
413
659
  const errors = [];
660
+ if (assessment.targetId !== profile.applicationTarget.id) {
661
+ errors.push("assessment target does not match the confirmed application LLM target");
662
+ }
414
663
  for (const objective of profile.objectives) {
415
664
  if (!objective.required)
416
665
  continue;
417
666
  const metric = metricFor(assessment, objective);
418
667
  if (!metric)
419
668
  errors.push(`${objective.metricId}: missing`);
420
- else if (metric.state !== "measured")
421
- errors.push(`${objective.metricId}: ${metric.state}`);
422
- else if (metric.value === undefined)
423
- errors.push(`${objective.metricId}: no numeric value`);
424
- else if (metric.sampleCount < objective.minimumSamples) {
425
- errors.push(`${objective.metricId}: ${metric.sampleCount}/${objective.minimumSamples} samples`);
669
+ else {
670
+ if (metric.state !== "measured")
671
+ errors.push(`${objective.metricId}: ${metric.state}`);
672
+ else if (metric.value === undefined)
673
+ errors.push(`${objective.metricId}: no numeric value`);
674
+ errors.push(...validateMetricAggregation(metric, objective));
675
+ if (metric.sampleCount < objective.minimumSamples) {
676
+ errors.push(`${objective.metricId}: ${metric.sampleCount}/${objective.minimumSamples} samples`);
677
+ }
426
678
  }
427
679
  }
428
680
  return errors;
429
681
  }
682
+ function baselineReadinessErrors(profile, metrics) {
683
+ if (!profile)
684
+ return ["confirmed task profile is missing"];
685
+ const errors = [];
686
+ if (!profile.objectives.some((objective) => objective.required)) {
687
+ errors.push("task profile has no required metric objectives");
688
+ }
689
+ for (const objective of profile.objectives) {
690
+ if (!objective.required)
691
+ continue;
692
+ const metric = metrics.find((item) => item.metricId === objective.metricId &&
693
+ item.targetId === profile.applicationTarget.id);
694
+ if (!metric)
695
+ errors.push(`${objective.metricId}: missing`);
696
+ else {
697
+ if (metric.state !== "measured")
698
+ errors.push(`${objective.metricId}: ${metric.state}`);
699
+ else if (metric.value === undefined || !Number.isFinite(metric.value))
700
+ errors.push(`${objective.metricId}: no finite numeric value`);
701
+ errors.push(...validateMetricAggregation(metric, objective));
702
+ if (metric.sampleCount < objective.minimumSamples) {
703
+ errors.push(`${objective.metricId}: ${metric.sampleCount}/${objective.minimumSamples} samples`);
704
+ }
705
+ else if (!metric.evidence.length)
706
+ errors.push(`${objective.metricId}: no evidence reference`);
707
+ }
708
+ }
709
+ return errors;
710
+ }
711
+ function baselineAssessmentErrors(state, profile, metrics) {
712
+ return [
713
+ ...(state.cycle.observabilityReady ? [] : ["required evaluation and observability setup is not ready"]),
714
+ ...baselineReadinessErrors(profile, metrics),
715
+ ];
716
+ }
717
+ function updatedBaselineWindows(previous, signal) {
718
+ if (signal.type !== "metric_window_available")
719
+ return previous ?? [];
720
+ const windows = new Map((previous ?? []).map((metric) => [metric.metricId, metric]));
721
+ for (const metric of signal.metrics)
722
+ windows.set(metric.metricId, metric);
723
+ return [...windows.values()];
724
+ }
430
725
  function feasibility(profile, assessment) {
431
726
  if (readinessErrors(profile, assessment).length)
432
727
  return "unresolved";
@@ -525,6 +820,9 @@ export function verifyFrontierSnapshot(profile, assessments, snapshot) {
525
820
  if (snapshot.taskProfileId !== profile.id || snapshot.taskProfileRevision !== profile.revision) {
526
821
  errors.push("snapshot task profile does not match");
527
822
  }
823
+ if (snapshot.targetId !== profile.applicationTarget.id) {
824
+ errors.push("snapshot target does not match the confirmed application LLM target");
825
+ }
528
826
  if (snapshot.calculationEvidence.length === 0)
529
827
  errors.push("calculation evidence is required");
530
828
  if (!sameMembers(snapshot.assessmentIds, assessments.map((assessment) => assessment.id))) {
@@ -534,6 +832,11 @@ export function verifyFrontierSnapshot(profile, assessments, snapshot) {
534
832
  const expectedIneligible = [];
535
833
  const expectedUnresolved = new Set();
536
834
  for (const assessment of assessments) {
835
+ if (assessment.targetId !== profile.applicationTarget.id) {
836
+ errors.push(`${assessment.candidateId}: application target mismatch`);
837
+ continue;
838
+ }
839
+ errors.push(...validateMetricTargets(assessment.metrics, profile.applicationTarget.id, assessment.candidateId));
537
840
  if (assessment.taskProfileId !== profile.id || assessment.taskProfileRevision !== profile.revision) {
538
841
  errors.push(`${assessment.candidateId}: task profile revision mismatch`);
539
842
  continue;
@@ -611,6 +914,14 @@ function earliest(values) {
611
914
  .sort((left, right) => Date.parse(left) - Date.parse(right))[0];
612
915
  }
613
916
  export function buildAutomationPlan(state, descriptor) {
917
+ if (state.automation.paused) {
918
+ return {
919
+ mode: state.cycle.policy?.checkPolicy.execution.mode ?? "active_session_only",
920
+ requiredSignals: [],
921
+ scheduling: "active_session_fallback",
922
+ gaps: ["Proactive OpenMerit checks are paused."],
923
+ };
924
+ }
614
925
  const policy = state.cycle.policy?.checkPolicy;
615
926
  if (!policy) {
616
927
  return {
@@ -640,7 +951,7 @@ export function buildAutomationPlan(state, descriptor) {
640
951
  required.add("verification_window_completed");
641
952
  }
642
953
  const nextWakeupAt = state.cycle.stage === "collecting_baseline"
643
- ? nextAt(state.automation.lastAssessmentAt ?? state.updatedAt, policy.baselineAssessment.afterElapsedSeconds)
954
+ ? nextAt(state.automation.lastAssessmentAt ?? state.automation.baselineStartedAt ?? state.lastResult?.completedAt ?? state.updatedAt, policy.baselineAssessment.afterElapsedSeconds)
644
955
  : ["monitoring", "swap_verified"].includes(state.cycle.stage)
645
956
  ? earliest([
646
957
  nextAt(state.automation.lastReassessmentAt ?? state.updatedAt, policy.frontierReassessment.afterElapsedSeconds),
@@ -701,6 +1012,8 @@ function regressionMetric(signal, previous, policy) {
701
1012
  return undefined;
702
1013
  }
703
1014
  function dueDecision(state, signal, now) {
1015
+ if (state.automation.paused)
1016
+ return { action: "observe", reason: "Proactive OpenMerit checks are paused." };
704
1017
  if (state.activeIntent)
705
1018
  return { action: "observe", reason: "Harness work is already active." };
706
1019
  if (state.automation.pendingIntentKind) {
@@ -719,7 +1032,7 @@ function dueDecision(state, signal, now) {
719
1032
  .sort((left, right) => Date.parse(right) - Date.parse(left))[0];
720
1033
  const cooldownSatisfied = !cooldownAnchor ||
721
1034
  Date.parse(now) - Date.parse(cooldownAnchor) >= policy.checkPolicy.cooldownSeconds * 1000;
722
- if (state.cycle.stage === "collecting_baseline" && cadenceDue(policy.checkPolicy.baselineAssessment, completed, state.automation.lastAssessmentTaskCount, state.automation.lastAssessmentAt ?? state.updatedAt, now)) {
1035
+ if (state.cycle.stage === "collecting_baseline" && cadenceDue(policy.checkPolicy.baselineAssessment, completed, state.automation.lastAssessmentTaskCount, state.automation.lastAssessmentAt ?? state.automation.baselineStartedAt ?? state.lastResult?.completedAt ?? state.updatedAt, now)) {
723
1036
  return { action: "issue_intent", intentKind: "run_assessment", reason: "The user-confirmed baseline assessment cadence is due." };
724
1037
  }
725
1038
  if (["monitoring", "swap_verified"].includes(state.cycle.stage) && cooldownSatisfied) {
@@ -749,14 +1062,17 @@ function auditSignalData(signal) {
749
1062
  switch (signal.type) {
750
1063
  case "product_task_completed":
751
1064
  return {
1065
+ targetId: signal.targetId,
752
1066
  modelId: signal.modelId,
753
- contextTokens: signal.contextTokens ?? null,
1067
+ ...(signal.telemetry ? { telemetry: signal.telemetry } : {}),
754
1068
  evidenceIds: (signal.evidence ?? []).map((item) => item.id),
755
1069
  };
756
1070
  case "metric_window_available":
757
1071
  case "verification_window_completed":
758
1072
  return {
1073
+ targetId: signal.targetId,
759
1074
  metrics: (signal.metrics ?? []).map((metric) => ({
1075
+ targetId: metric.targetId,
760
1076
  metricId: metric.metricId,
761
1077
  state: metric.state,
762
1078
  scope: metric.scope,
@@ -771,9 +1087,9 @@ function auditSignalData(signal) {
771
1087
  })),
772
1088
  };
773
1089
  case "model_catalog_changed":
774
- return { fingerprint: signal.fingerprint };
1090
+ return { targetId: signal.targetId, fingerprint: signal.fingerprint };
775
1091
  case "scheduled_tick":
776
- return {};
1092
+ return { targetId: signal.targetId };
777
1093
  }
778
1094
  }
779
1095
  /** Harness-neutral durable coordinator. Adapters transport jobs; this class owns lifecycle state. */
@@ -784,7 +1100,7 @@ export class OpenMeritCoordinator {
784
1100
  this.store = store;
785
1101
  this.adapter = adapter;
786
1102
  }
787
- async issue(kind) {
1103
+ async issue(kind, options = {}) {
788
1104
  const missing = missingHarnessCapabilities(this.adapter.descriptor, kind);
789
1105
  if (missing.length)
790
1106
  throw new Error(`Harness lacks required capabilities: ${missing.join(", ")}`);
@@ -792,16 +1108,80 @@ export class OpenMeritCoordinator {
792
1108
  const existing = await this.store.readState();
793
1109
  if (existing?.activeIntent)
794
1110
  throw new Error(`OpenMerit intent ${existing.activeIntent.id} is already active.`);
1111
+ if (kind === "run_challenger_trials" && config.policy) {
1112
+ const spent = existing?.automation.evaluationSpend ?? 0;
1113
+ const limit = config.policy.evaluationBudget.maximumSpend;
1114
+ if (spent >= limit) {
1115
+ throw new Error(`Evaluation budget exhausted: ${spent.toFixed(6)} ${config.policy.evaluationBudget.currency} spent of ${limit.toFixed(6)}.`);
1116
+ }
1117
+ }
1118
+ const targetId = activeTarget(config)?.id;
1119
+ const confirmedProfile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
1120
+ if (kind !== "establish_evals" && !targetId) {
1121
+ throw new Error(`OpenMerit cannot issue ${kind} without a confirmed application LLM target.`);
1122
+ }
795
1123
  const intent = createOpenMeritIntent(kind, {
796
1124
  taskProfileId: config.activeTaskProfileId,
797
1125
  policy: config.policy,
1126
+ evaluationSpend: existing?.automation.evaluationSpend,
1127
+ targetId,
1128
+ additionalConstraints: {
1129
+ ...options.additionalConstraints,
1130
+ ...(confirmedProfile ? { taskProfile: confirmedProfile } : {}),
1131
+ ...(kind === "discover_candidates"
1132
+ ? {
1133
+ preliminaryModelLeads: (config.preliminaryModelLeads ?? []),
1134
+ incumbentCandidate: (existing?.candidates?.find((candidate) => candidate.candidateId === existing.cycle.activeCandidateId) ?? existing?.candidates?.[0]),
1135
+ candidateSelection: {
1136
+ totalCandidateCount: Math.min(3, config.policy?.evaluationBudget.maximumCandidateCount ?? 3),
1137
+ includeIncumbent: true,
1138
+ requireAvailabilityProbe: true,
1139
+ requireExactModelIds: true,
1140
+ requirePriceAndCapabilitySources: true,
1141
+ },
1142
+ }
1143
+ : {}),
1144
+ ...(kind === "run_challenger_trials"
1145
+ ? { candidates: (existing?.candidates ?? []) }
1146
+ : {}),
1147
+ ...(kind === "calculate_frontier"
1148
+ ? {
1149
+ assessments: (existing?.assessments ?? []),
1150
+ ...(existing?.experimentManifest
1151
+ ? { experimentManifest: existing.experimentManifest }
1152
+ : {}),
1153
+ }
1154
+ : {}),
1155
+ },
798
1156
  });
799
1157
  const now = new Date().toISOString();
800
1158
  const state = existing
801
1159
  ? {
802
1160
  ...existing,
803
- cycle: { ...existing.cycle, stage: startedStage(kind) },
1161
+ cycle: kind === "establish_evals"
1162
+ ? {
1163
+ ...existing.cycle,
1164
+ id: `cycle-${crypto.randomUUID()}`,
1165
+ targetId: undefined,
1166
+ taskProfileId: undefined,
1167
+ activeCandidateId: undefined,
1168
+ previousCandidateId: undefined,
1169
+ proposedCandidateId: undefined,
1170
+ stage: startedStage(kind),
1171
+ verifiedSwapCount: 0,
1172
+ baselineEvidenceSufficient: false,
1173
+ observabilityReady: false,
1174
+ policy: undefined,
1175
+ }
1176
+ : { ...existing.cycle, stage: startedStage(kind) },
804
1177
  activeIntent: intent,
1178
+ ...(kind === "establish_evals" ? {
1179
+ candidates: undefined,
1180
+ assessments: undefined,
1181
+ experimentManifest: undefined,
1182
+ observability: undefined,
1183
+ lastResult: undefined,
1184
+ } : {}),
805
1185
  automation: {
806
1186
  ...existing.automation,
807
1187
  ...(existing.automation.pendingIntentKind === kind
@@ -889,6 +1269,16 @@ export class OpenMeritCoordinator {
889
1269
  signal: "scheduled_tick",
890
1270
  reason: "Wake the harness so OpenMerit can evaluate the user-confirmed check policy.",
891
1271
  });
1272
+ if (!receipt.accepted) {
1273
+ return {
1274
+ plan: {
1275
+ ...plan,
1276
+ scheduling: "external_scheduler_required",
1277
+ gaps: [...plan.gaps, `Harness wakeup provisioning failed: ${receipt.reason ?? "unknown reason"}`],
1278
+ },
1279
+ receipt,
1280
+ };
1281
+ }
892
1282
  return { plan, receipt };
893
1283
  }
894
1284
  async acceptResult(result) {
@@ -903,6 +1293,17 @@ export class OpenMeritCoordinator {
903
1293
  if (interpretation.config)
904
1294
  await this.store.writeConfig(interpretation.config);
905
1295
  const now = new Date().toISOString();
1296
+ const evaluationSpend = result.status === "succeeded" && state.activeIntent.kind === "run_challenger_trials"
1297
+ ? result.outputs.assessments.reduce((sum, assessment) => sum + assessment.metrics
1298
+ .filter((metric) => metric.metricId === "total_task_cost" && metric.state === "measured")
1299
+ .reduce((metricSum, metric) => metricSum + (metric.value ?? 0), 0), state.automation.evaluationSpend ?? 0)
1300
+ : state.automation.evaluationSpend;
1301
+ if (config.policy && evaluationSpend !== undefined &&
1302
+ evaluationSpend > config.policy.evaluationBudget.maximumSpend) {
1303
+ throw new Error(`OpenMerit rejected challenger results: evaluation spend ${evaluationSpend.toFixed(6)} ` +
1304
+ `${config.policy.evaluationBudget.currency} exceeds the approved limit ` +
1305
+ `${config.policy.evaluationBudget.maximumSpend.toFixed(6)}.`);
1306
+ }
906
1307
  let observability = state.observability;
907
1308
  if (result.status === "succeeded" && state.activeIntent.kind === "establish_evals") {
908
1309
  observability = result.outputs.observability;
@@ -913,6 +1314,7 @@ export class OpenMeritCoordinator {
913
1314
  const output = result.outputs;
914
1315
  const requiredMetricIds = profile.objectives.filter((item) => item.required).map((item) => item.metricId);
915
1316
  observability = {
1317
+ targetId: profile.applicationTarget.id,
916
1318
  status: "ready",
917
1319
  requiredMetricIds,
918
1320
  coveredMetricIds: output.coveredMetricIds,
@@ -930,9 +1332,31 @@ export class OpenMeritCoordinator {
930
1332
  },
931
1333
  activeIntent: undefined,
932
1334
  lastResult: result,
1335
+ ...(result.status === "succeeded" && state.activeIntent.kind === "establish_evals"
1336
+ ? { candidates: [result.outputs.incumbent] }
1337
+ : {}),
1338
+ ...(result.status === "succeeded" && state.activeIntent.kind === "discover_candidates"
1339
+ ? { candidates: result.outputs.candidates }
1340
+ : {}),
1341
+ ...(result.status === "succeeded" && state.activeIntent.kind === "run_challenger_trials"
1342
+ ? {
1343
+ assessments: result.outputs.assessments,
1344
+ experimentManifest: result.outputs.experimentManifest,
1345
+ }
1346
+ : {}),
933
1347
  ...(observability ? { observability } : {}),
934
1348
  automation: {
935
1349
  ...state.automation,
1350
+ ...(result.status === "succeeded" && ["establish_evals", "instrument_observability"].includes(state.activeIntent.kind) &&
1351
+ interpretation.nextStage === "collecting_baseline"
1352
+ ? {
1353
+ baselineStartedAt: now,
1354
+ lastAssessmentAt: undefined,
1355
+ lastAssessmentTaskCount: state.automation.observedCompletedTasks,
1356
+ baselineMetricWindows: [],
1357
+ }
1358
+ : {}),
1359
+ ...(evaluationSpend !== undefined ? { evaluationSpend } : {}),
936
1360
  ...(state.activeIntent.kind === "run_assessment"
937
1361
  ? { lastAssessmentTaskCount: state.automation.observedCompletedTasks, lastAssessmentAt: now }
938
1362
  : {}),
@@ -984,6 +1408,7 @@ export class OpenMeritCoordinator {
984
1408
  cycleId: nextState.cycle.id,
985
1409
  intentId: result.intentId,
986
1410
  data: {
1411
+ targetId: observability.targetId,
987
1412
  status: observability.status,
988
1413
  requiredMetricIds: observability.requiredMetricIds,
989
1414
  coveredMetricIds: observability.coveredMetricIds,
@@ -1016,21 +1441,54 @@ export class OpenMeritCoordinator {
1016
1441
  }
1017
1442
  async recordTaskObservation(details = {}) {
1018
1443
  const now = new Date().toISOString();
1444
+ const state = await this.store.readState();
1445
+ if (!state?.cycle.targetId)
1446
+ throw new Error("A confirmed application LLM target is required.");
1019
1447
  const result = await this.recordAutomationSignal({
1020
1448
  protocolVersion: PROTOCOL_VERSION,
1021
1449
  id: `legacy-task-${crypto.randomUUID()}`,
1022
1450
  type: "product_task_completed",
1023
1451
  occurredAt: now,
1452
+ targetId: state.cycle.targetId,
1024
1453
  modelId: typeof details.modelId === "string" ? details.modelId : "unknown",
1025
- contextTokens: typeof details.contextTokens === "number" || details.contextTokens === null
1026
- ? details.contextTokens
1027
- : undefined,
1028
1454
  });
1029
1455
  return {
1030
1456
  state: result.state,
1031
- assessmentDue: result.decision.action === "issue_intent" && result.decision.intentKind === "run_assessment",
1457
+ assessmentDue: result.state.automation.lastAssessmentAt === now,
1032
1458
  };
1033
1459
  }
1460
+ async assessBaseline(now = new Date().toISOString()) {
1461
+ const state = await this.store.readState();
1462
+ if (!state || state.cycle.stage !== "collecting_baseline" || state.activeIntent) {
1463
+ throw new Error("Baseline assessment requires an idle collecting-baseline cycle.");
1464
+ }
1465
+ const config = await this.store.readConfig();
1466
+ const profile = config?.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
1467
+ const errors = baselineAssessmentErrors(state, profile, state.automation.baselineMetricWindows ?? []);
1468
+ const sufficient = errors.length === 0;
1469
+ const decision = sufficient
1470
+ ? { action: "issue_intent", intentKind: "discover_candidates", reason: "Required baseline metric windows meet the confirmed sample thresholds; discover candidates." }
1471
+ : { action: "observe", reason: `Baseline evidence remains insufficient: ${errors.join("; ")}.` };
1472
+ const nextState = {
1473
+ ...state,
1474
+ cycle: { ...state.cycle, baselineEvidenceSufficient: sufficient, stage: sufficient ? "baseline_ready" : "collecting_baseline" },
1475
+ automation: {
1476
+ ...state.automation,
1477
+ lastAssessmentAt: now,
1478
+ lastAssessmentTaskCount: state.automation.observedCompletedTasks,
1479
+ pendingIntentKind: decision.action === "issue_intent" ? decision.intentKind : undefined,
1480
+ pendingIntentReason: decision.action === "issue_intent" ? decision.reason : undefined,
1481
+ },
1482
+ updatedAt: now,
1483
+ };
1484
+ await this.store.writeState(nextState);
1485
+ await this.store.appendEvent({
1486
+ schemaVersion: 1, id: `event-${crypto.randomUUID()}`, type: "baseline_sufficiency_checked",
1487
+ occurredAt: now, cycleId: state.cycle.id,
1488
+ data: { sufficient, errors, metricIds: (state.automation.baselineMetricWindows ?? []).map((metric) => metric.metricId), source: "manual" },
1489
+ });
1490
+ return { state: nextState, errors, decision };
1491
+ }
1034
1492
  async recordAutomationSignal(signal) {
1035
1493
  const state = await this.store.readState();
1036
1494
  if (!state)
@@ -1039,6 +1497,14 @@ export class OpenMeritCoordinator {
1039
1497
  throw new Error("Automation signal protocol version does not match.");
1040
1498
  if (!signal.id.trim())
1041
1499
  throw new Error("Automation signal ID is required.");
1500
+ if (!state.cycle.targetId || signal.targetId !== state.cycle.targetId) {
1501
+ throw new Error("Automation signal target does not match the confirmed application LLM target.");
1502
+ }
1503
+ if (signal.type === "metric_window_available" || signal.type === "verification_window_completed") {
1504
+ const mismatched = (signal.metrics ?? []).find((metric) => metric.targetId !== state.cycle.targetId);
1505
+ if (mismatched)
1506
+ throw new Error(`Metric ${mismatched.metricId} target does not match the confirmed application LLM target.`);
1507
+ }
1042
1508
  const processed = state.automation.processedSignalIds ?? [];
1043
1509
  if (processed.includes(signal.id)) {
1044
1510
  await this.store.appendEvent({
@@ -1053,12 +1519,31 @@ export class OpenMeritCoordinator {
1053
1519
  }
1054
1520
  const completedTasks = state.automation.observedCompletedTasks +
1055
1521
  (signal.type === "product_task_completed" ? 1 : 0);
1522
+ const baselineMetricWindows = state.cycle.stage === "collecting_baseline"
1523
+ ? updatedBaselineWindows(state.automation.baselineMetricWindows, signal)
1524
+ : state.automation.baselineMetricWindows ?? [];
1056
1525
  const decisionState = {
1057
1526
  ...state,
1058
- automation: { ...state.automation, observedCompletedTasks: completedTasks },
1527
+ automation: { ...state.automation, observedCompletedTasks: completedTasks, baselineMetricWindows },
1059
1528
  };
1060
- const decision = dueDecision(decisionState, signal, signal.occurredAt);
1529
+ const due = dueDecision(decisionState, signal, signal.occurredAt);
1530
+ const baselineCheckDue = due.action === "issue_intent" && due.intentKind === "run_assessment";
1531
+ const config = baselineCheckDue ? await this.store.readConfig() : undefined;
1532
+ const profile = config?.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
1533
+ const baselineErrors = baselineCheckDue ? baselineAssessmentErrors(state, profile, baselineMetricWindows) : [];
1534
+ const baselineReady = baselineCheckDue && baselineErrors.length === 0;
1535
+ const decision = baselineCheckDue
1536
+ ? baselineReady
1537
+ ? { action: "issue_intent", intentKind: "discover_candidates", reason: "Required baseline metric windows meet the confirmed sample thresholds; discover candidates." }
1538
+ : { action: "observe", reason: `Baseline evidence remains insufficient: ${baselineErrors.join("; ")}.` }
1539
+ : due;
1061
1540
  const latestMetricValues = { ...(state.automation.latestMetricValues ?? {}) };
1541
+ if (signal.type === "product_task_completed" && signal.telemetry) {
1542
+ latestMetricValues.input_token_consumption = signal.telemetry.inputTokens;
1543
+ latestMetricValues.output_token_consumption = signal.telemetry.outputTokens;
1544
+ latestMetricValues.total_task_cost = signal.telemetry.cost;
1545
+ latestMetricValues.end_to_end_task_latency = signal.telemetry.durationMs;
1546
+ }
1062
1547
  if (signal.type === "metric_window_available" || signal.type === "verification_window_completed") {
1063
1548
  for (const metric of signal.metrics ?? []) {
1064
1549
  if (metric.state === "measured" && metric.value !== undefined)
@@ -1067,10 +1552,19 @@ export class OpenMeritCoordinator {
1067
1552
  }
1068
1553
  const nextState = {
1069
1554
  ...decisionState,
1555
+ cycle: baselineCheckDue
1556
+ ? { ...decisionState.cycle, baselineEvidenceSufficient: baselineReady, stage: baselineReady ? "baseline_ready" : "collecting_baseline" }
1557
+ : decisionState.cycle,
1070
1558
  automation: {
1071
1559
  ...decisionState.automation,
1072
1560
  processedSignalIds: [...processed, signal.id].slice(-MAX_PROCESSED_SIGNAL_IDS),
1073
1561
  latestMetricValues,
1562
+ ...(baselineCheckDue ? {
1563
+ lastAssessmentAt: signal.occurredAt,
1564
+ lastAssessmentTaskCount: completedTasks,
1565
+ pendingIntentKind: decision.action === "issue_intent" ? decision.intentKind : undefined,
1566
+ pendingIntentReason: decision.action === "issue_intent" ? decision.reason : undefined,
1567
+ } : {}),
1074
1568
  ...(signal.type === "model_catalog_changed" ? {
1075
1569
  modelCatalogFingerprint: signal.fingerprint,
1076
1570
  ...(state.automation.modelCatalogFingerprint && state.automation.modelCatalogFingerprint !== signal.fingerprint
@@ -1099,6 +1593,21 @@ export class OpenMeritCoordinator {
1099
1593
  ...auditSignalData(signal),
1100
1594
  },
1101
1595
  });
1596
+ if (baselineCheckDue) {
1597
+ await this.store.appendEvent({
1598
+ schemaVersion: 1,
1599
+ id: `event-${crypto.randomUUID()}`,
1600
+ type: "baseline_sufficiency_checked",
1601
+ occurredAt: signal.occurredAt,
1602
+ cycleId: state.cycle.id,
1603
+ data: {
1604
+ sufficient: baselineReady,
1605
+ errors: baselineErrors,
1606
+ metricIds: baselineMetricWindows.map((metric) => metric.metricId),
1607
+ sourceSignalId: signal.id,
1608
+ },
1609
+ });
1610
+ }
1102
1611
  if (decision.action === "issue_intent") {
1103
1612
  await this.store.appendEvent({
1104
1613
  schemaVersion: 1,
@@ -1115,6 +1624,8 @@ export class OpenMeritCoordinator {
1115
1624
  const state = await this.store.readState();
1116
1625
  if (!state)
1117
1626
  throw new Error("OpenMerit project is not initialized.");
1627
+ if (!state.cycle.targetId)
1628
+ throw new Error("A confirmed application LLM target is required.");
1118
1629
  const previousFingerprint = state.automation.modelCatalogFingerprint;
1119
1630
  const changed = previousFingerprint !== undefined && previousFingerprint !== fingerprint;
1120
1631
  if (previousFingerprint === fingerprint) {
@@ -1126,6 +1637,7 @@ export class OpenMeritCoordinator {
1126
1637
  id: `legacy-catalog-${crypto.randomUUID()}`,
1127
1638
  type: "model_catalog_changed",
1128
1639
  occurredAt: now,
1640
+ targetId: state.cycle.targetId,
1129
1641
  fingerprint,
1130
1642
  });
1131
1643
  return {