openmerit 0.1.5 → 0.1.6-preview.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +33 -9
- package/dist/core/src/index.d.ts +12 -1
- package/dist/core/src/index.js +552 -40
- package/dist/pi/src/index.d.ts +17 -0
- package/dist/pi/src/index.js +448 -77
- package/dist/pi/src/scheduler.d.ts +11 -0
- package/dist/pi/src/scheduler.js +137 -0
- package/dist/pi/src/wakeup.d.ts +2 -0
- package/dist/pi/src/wakeup.js +108 -0
- package/dist/protocol/src/index.d.ts +86 -4
- package/dist/protocol/src/index.js +1 -1
- package/dist/protocol/src/schemas.d.ts +128 -2
- package/dist/protocol/src/schemas.js +57 -1
- package/dist/terminal/public/app.js +297 -0
- package/dist/terminal/public/brands/anthropic.png +0 -0
- package/dist/terminal/public/brands/baai.png +0 -0
- package/dist/terminal/public/brands/baseten.png +0 -0
- package/dist/terminal/public/brands/cerebras.png +0 -0
- package/dist/terminal/public/brands/cohere.png +0 -0
- package/dist/terminal/public/brands/deepseek.ico +0 -0
- package/dist/terminal/public/brands/google.png +0 -0
- package/dist/terminal/public/brands/groq.ico +0 -0
- package/dist/terminal/public/brands/lm-studio.png +0 -0
- package/dist/terminal/public/brands/meta.ico +0 -0
- package/dist/terminal/public/brands/mistral.png +0 -0
- package/dist/terminal/public/brands/nomic.png +0 -0
- package/dist/terminal/public/brands/ollama.png +0 -0
- package/dist/terminal/public/brands/openai.png +0 -0
- package/dist/terminal/public/brands/openrouter.png +0 -0
- package/dist/terminal/public/brands/qwen.png +0 -0
- package/dist/terminal/public/brands/vllm.ico +0 -0
- package/dist/terminal/public/brands/vllm.png +0 -0
- package/dist/terminal/public/favicon.svg +1 -0
- package/dist/terminal/public/flow.css +1 -0
- package/dist/terminal/public/flow.js +770 -0
- package/dist/terminal/public/index.html +21 -0
- package/dist/terminal/public/styles.css +779 -0
- package/dist/terminal/src/activity-merge.mjs +64 -0
- package/dist/terminal/src/browser.mjs +29 -0
- package/dist/terminal/src/cli.mjs +60 -0
- package/dist/terminal/src/collect.mjs +311 -0
- package/dist/terminal/src/discovery.mjs +93 -0
- package/dist/terminal/src/hardware.mjs +57 -0
- package/dist/terminal/src/project-activity.mjs +156 -0
- package/dist/terminal/src/sample.mjs +171 -0
- package/dist/terminal/src/server.mjs +56 -0
- package/dist/terminal/src/services.mjs +62 -0
- package/dist/terminal/src/topology.mjs +30 -0
- package/docs/adapter-guide.md +24 -7
- package/docs/architecture.md +8 -4
- package/docs/automation.md +10 -2
- package/docs/budgets.md +37 -0
- package/docs/commands.md +85 -0
- package/docs/demo-backfill.md +29 -0
- package/docs/demo-fieldkit.md +47 -0
- package/docs/demo-placement.md +30 -0
- package/docs/demo-spam.md +15 -0
- package/docs/demo-support.md +42 -0
- package/docs/demo.md +57 -0
- package/docs/first-trial.md +60 -0
- package/docs/getting-started.md +18 -8
- package/docs/index.md +40 -0
- package/docs/inference-terminal.md +439 -0
- package/docs/lifecycle.md +9 -9
- package/docs/memo.md +126 -0
- package/docs/metrics-and-evidence.md +11 -3
- package/docs/operations.md +12 -3
- package/docs/pi-extension.md +17 -7
- package/docs/roadmap.md +4 -2
- package/docs/security.md +15 -1
- package/docs/site-artwork-linocut.md +23 -0
- package/docs/site-artwork-miniature-diverse.md +28 -0
- package/docs/site-artwork-miniature.md +26 -0
- package/docs/site-demo.md +177 -0
- package/docs/site-design.md +94 -0
- package/docs/site-documentation.md +83 -0
- package/docs/site-dynamic-og.md +35 -0
- package/docs/site-faq-maintenance.md +115 -0
- package/docs/site-hero-resolution.md +60 -0
- package/docs/site-illustration-sequences.md +227 -0
- package/docs/site-inference-terminal.md +203 -0
- package/docs/site-memo.md +39 -0
- package/docs/site-og-image.md +38 -0
- package/docs/site-og-workshop.md +21 -0
- package/docs/site-section-artwork.md +56 -0
- package/docs/site-skill-review.md +57 -0
- package/docs/site-terminal-preview.md +85 -0
- package/docs/testing.md +85 -3
- package/docs/troubleshooting.md +55 -0
- package/package.json +48 -7
package/dist/core/src/index.js
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
import { METRIC_CATALOG, OPENMERIT_SCHEMA_DIALECT, PROTOCOL_VERSION, REQUIRED_CAPABILITIES, validateIntentOutputSchema, } from "../../protocol/src/index.js";
|
|
2
2
|
export { OPENMERIT_DIRECTORY, ProjectStore, } from "./store.js";
|
|
3
3
|
const outcomes = {
|
|
4
|
-
establish_evals: "Infer the
|
|
5
|
-
instrument_observability: "Establish
|
|
6
|
-
run_assessment: "Assess
|
|
7
|
-
discover_candidates: "
|
|
8
|
-
run_challenger_trials: "Run
|
|
4
|
+
establish_evals: "Identify the real application LLM call or shared application route under evaluation and its exact incumbent application model. Infer the metrics required for that application's task from its outputs, tools, failure modes, and representative work. Before running setup, tell the user which metrics will be collected, why each matters, whether it uses repeated identical inputs, representative task instances, both, or load testing, the case and repetition counts, aggregation, any true performance threshold, and estimated evaluation cost; ask the user to confirm or edit that plan together with the target, budget, check cadence, scheduling mode, and automation permissions. Never turn a sample-count requirement into a metric-value constraint. Rate, distribution, reliability, percentile, variance, and average claims require multiple runs. Build runnable task-specific evaluation cases and application-call collectors for every confirmed metric. Execute a representative smoke run through the same application authentication and runtime path, run the grader, and inspect actual outputs. Preserve the commands and resulting artifacts as setup evidence. Mark any metric missing if its collector or grader cannot produce a value; do not declare coverage ready based only on code existing. Do not count a setup smoke run as a production baseline sample. Return the confirmed taskProfile, incumbent, and policy. During setup, research a preliminary model shortlist with current sourced price estimates and reputation; report an empty list if no credible leads are available. These are research priors only: do not treat them as measured task results or change model configuration. The coding harness model is not an evaluation target.",
|
|
5
|
+
instrument_observability: "Establish the missing observability required by the confirmed task profile. Exercise each collector and task-specific grader through a representative application run with the same authentication and runtime path, inspect actual metric outputs, and preserve verification artifacts. Report a metric covered only if that path produced its value; otherwise keep it missing. Do not count a setup smoke run as a production baseline sample.",
|
|
6
|
+
run_assessment: "Assess existing application-target baseline metric windows against every required metric and sample threshold. Do not synthesize samples, count setup smoke runs as production data, or wait for new data during this check. If a required window is absent or immature, promptly return baselineEvidenceSufficient false with measured, missing, and insufficient-evidence metrics explicitly and cite the records inspected.",
|
|
7
|
+
discover_candidates: "Refresh the preliminary model leads collected at setup and discover other credible candidates for the confirmed task profile using current capability, reputation, price, availability, and benchmark signals without exceeding the approved evaluation budget. Preliminary leads are research priors, not measured task evidence.",
|
|
8
|
+
run_challenger_trials: "Run a balanced, reproducible comparison of the frozen incumbent and challenger set under the approved budget. Use the same evaluation cases, application revision, prompt, tools, grader, and run count for every usable model; use a recorded seed to shuffle execution order. Attribute every run to the exact application model, preserve per-run quality, cost, token, and latency evidence, and return candidate assessments plus one durable experiment manifest. Never substitute catalogue prices for measured application-call cost.",
|
|
9
9
|
calculate_frontier: "Calculate the Pareto frontier using the OpenMerit conformance definition, explain every candidate classification, and return a FrontierSnapshot with calculation evidence.",
|
|
10
10
|
investigate_regression: "Investigate the observed regression and return its likely cause, affected metrics, and supporting evidence.",
|
|
11
|
-
apply_model_swap: "Apply the approved model change
|
|
11
|
+
apply_model_swap: "Apply the approved model change only to the bound application LLM target and return proof of the resulting application configuration. Never change the coding harness model and do not broaden the authorized change.",
|
|
12
12
|
verify_model_swap: "Verify the applied model against the approved post-swap evaluation window and report any material regression.",
|
|
13
13
|
rollback_model_swap: "Restore the previously verified model configuration and validate the rollback.",
|
|
14
14
|
};
|
|
@@ -17,31 +17,47 @@ export function createOpenMeritIntent(kind, options = {}) {
|
|
|
17
17
|
protocolVersion: PROTOCOL_VERSION,
|
|
18
18
|
id: options.id ?? `intent-${crypto.randomUUID()}`,
|
|
19
19
|
kind,
|
|
20
|
+
...(options.targetId ? { targetId: options.targetId } : {}),
|
|
20
21
|
taskProfileId: options.taskProfileId ?? "current-project",
|
|
21
22
|
authorizationPolicyId: options.authorizationPolicyId ?? "observe-and-evaluate-only",
|
|
22
23
|
requestedOutcome: outcomes[kind],
|
|
23
24
|
constraints: {
|
|
24
25
|
modelMutationAllowed: kind === "apply_model_swap" || kind === "rollback_model_swap",
|
|
26
|
+
harnessModelMutationAllowed: false,
|
|
27
|
+
mutationTarget: "application_llm_call",
|
|
25
28
|
requireVerifiedArtifacts: true,
|
|
26
29
|
distinguishSingleRunFromMultiRunClaims: true,
|
|
27
30
|
...(options.policy ? { evaluationBudget: options.policy.evaluationBudget } : {}),
|
|
31
|
+
...(options.policy ? {
|
|
32
|
+
evaluationSpend: options.evaluationSpend ?? 0,
|
|
33
|
+
remainingEvaluationBudget: Math.max(0, options.policy.evaluationBudget.maximumSpend - (options.evaluationSpend ?? 0)),
|
|
34
|
+
} : {}),
|
|
28
35
|
...(options.policy ? { checkPolicy: options.policy.checkPolicy } : {}),
|
|
36
|
+
...(options.additionalConstraints ?? {}),
|
|
29
37
|
},
|
|
30
38
|
requiredEvidence: kind === "establish_evals"
|
|
31
39
|
? [
|
|
32
40
|
{ id: "task-profile", description: "The inferred task profile and the user constraints it represents." },
|
|
41
|
+
{ id: "application-target", description: "The user-confirmed application LLM call or shared route, including stable call-site and route identifiers." },
|
|
42
|
+
{ id: "incumbent-model", description: "The exact application model configured before challenger trials." },
|
|
33
43
|
{ id: "evaluation-artifacts", description: "Runnable evaluation artifacts and verification results." },
|
|
34
44
|
{ id: "observability-coverage", description: "The observable metrics and explicit coverage gaps." },
|
|
45
|
+
{ id: "preliminary-model-research", description: "Sources for preliminary model pricing and reputation, or an explanation that no credible leads were found." },
|
|
35
46
|
]
|
|
36
|
-
: kind === "
|
|
47
|
+
: kind === "run_challenger_trials"
|
|
37
48
|
? [
|
|
38
|
-
{ id: "
|
|
39
|
-
{ id: "
|
|
49
|
+
{ id: "experiment-manifest", description: "A durable manifest containing the seed, frozen cases, application revision, exact model IDs, and per-run evidence." },
|
|
50
|
+
{ id: "assessment-results", description: "Comparable candidate assessments tied to the active task profile." },
|
|
40
51
|
]
|
|
41
|
-
:
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
52
|
+
: kind === "calculate_frontier"
|
|
53
|
+
? [
|
|
54
|
+
{ id: "frontier-snapshot", description: "A complete harness-calculated FrontierSnapshot." },
|
|
55
|
+
{ id: "frontier-calculation", description: "Durable evidence of the frontier calculation." },
|
|
56
|
+
]
|
|
57
|
+
: [
|
|
58
|
+
{ id: "assessment-results", description: "Assessment results tied to the active task profile." },
|
|
59
|
+
{ id: "metric-coverage", description: "Observed, derived, and insufficient-evidence metric states." },
|
|
60
|
+
],
|
|
45
61
|
requestedAt: options.requestedAt ?? new Date().toISOString(),
|
|
46
62
|
};
|
|
47
63
|
}
|
|
@@ -126,13 +142,16 @@ const knownMetricIds = new Set(METRIC_CATALOG.map((metric) => metric.id));
|
|
|
126
142
|
function isRecord(value) {
|
|
127
143
|
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
128
144
|
}
|
|
129
|
-
function validateSetupOutput(output) {
|
|
130
|
-
if (!isRecord(output) || !isRecord(output.taskProfile) || !isRecord(output.policy)) {
|
|
131
|
-
return ["setup output requires taskProfile and policy"];
|
|
145
|
+
function validateSetupOutput(output, expectedIncumbentModelId) {
|
|
146
|
+
if (!isRecord(output) || !isRecord(output.taskProfile) || !isRecord(output.incumbent) || !isRecord(output.policy)) {
|
|
147
|
+
return ["setup output requires taskProfile, incumbent, and policy"];
|
|
132
148
|
}
|
|
133
149
|
const profile = output.taskProfile;
|
|
134
150
|
const policy = output.policy;
|
|
135
151
|
const errors = [];
|
|
152
|
+
if (expectedIncumbentModelId && output.incumbent.modelId !== expectedIncumbentModelId) {
|
|
153
|
+
errors.push(`incumbent model ${String(output.incumbent.modelId)} does not match the application source model ${expectedIncumbentModelId}`);
|
|
154
|
+
}
|
|
136
155
|
if (typeof profile.id !== "string" || typeof profile.name !== "string" ||
|
|
137
156
|
typeof profile.goal !== "string" || !Number.isInteger(profile.revision) ||
|
|
138
157
|
typeof profile.confirmedAt !== "string" || !Array.isArray(profile.objectives) ||
|
|
@@ -148,6 +167,49 @@ function validateSetupOutput(output) {
|
|
|
148
167
|
errors.push("taskProfile contains an invalid metric objective");
|
|
149
168
|
break;
|
|
150
169
|
}
|
|
170
|
+
const samplingPlan = objective.samplingPlan;
|
|
171
|
+
if (!isRecord(samplingPlan) || typeof samplingPlan.strategy !== "string" ||
|
|
172
|
+
!Number.isInteger(samplingPlan.representativeCaseCount) || samplingPlan.representativeCaseCount < 1 ||
|
|
173
|
+
!Number.isInteger(samplingPlan.repetitionsPerCase) || samplingPlan.repetitionsPerCase < 1 ||
|
|
174
|
+
typeof samplingPlan.aggregation !== "string" || typeof samplingPlan.rationale !== "string" ||
|
|
175
|
+
!samplingPlan.rationale.trim()) {
|
|
176
|
+
errors.push(`${objective.metricId} requires a complete sampling plan`);
|
|
177
|
+
continue;
|
|
178
|
+
}
|
|
179
|
+
const plannedSamples = samplingPlan.representativeCaseCount *
|
|
180
|
+
samplingPlan.repetitionsPerCase;
|
|
181
|
+
if (objective.minimumSamples < plannedSamples) {
|
|
182
|
+
errors.push(`${objective.metricId} minimumSamples must cover its sampling plan`);
|
|
183
|
+
}
|
|
184
|
+
const aggregateClaim = ["mean", "rate", "distribution", "percentile"].includes(samplingPlan.aggregation);
|
|
185
|
+
const metricDefinition = METRIC_CATALOG.find((metric) => metric.id === objective.metricId);
|
|
186
|
+
if ((aggregateClaim || metricDefinition?.requiresRepeatedRuns) && plannedSamples < 2) {
|
|
187
|
+
errors.push(`${objective.metricId} requires multiple planned runs for its ${String(samplingPlan.aggregation)} claim`);
|
|
188
|
+
}
|
|
189
|
+
if ((aggregateClaim || metricDefinition?.requiresRepeatedRuns) && samplingPlan.strategy === "single_run") {
|
|
190
|
+
errors.push(`${objective.metricId} cannot use single_run for a repeated-run claim`);
|
|
191
|
+
}
|
|
192
|
+
if (isRecord(objective.constraint) &&
|
|
193
|
+
["task_success", "structured_output_reliability", "hallucination_rate", "retry_rate", "recovery_ability", "human_intervention_rate"].includes(objective.metricId) &&
|
|
194
|
+
(typeof objective.constraint.value !== "number" || objective.constraint.value < 0 || objective.constraint.value > 1)) {
|
|
195
|
+
errors.push(`${objective.metricId} ratio constraint must be between 0 and 1`);
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
if (!isRecord(profile.applicationTarget) || profile.applicationTarget.kind !== "application_llm_call" ||
|
|
200
|
+
typeof profile.applicationTarget.id !== "string" || !profile.applicationTarget.id.trim() ||
|
|
201
|
+
typeof profile.applicationTarget.applicationId !== "string" || !profile.applicationTarget.applicationId.trim() ||
|
|
202
|
+
typeof profile.applicationTarget.routeKey !== "string" || !profile.applicationTarget.routeKey.trim() ||
|
|
203
|
+
typeof profile.applicationTarget.confirmedAt !== "string" ||
|
|
204
|
+
!Array.isArray(profile.applicationTarget.callSites) || profile.applicationTarget.callSites.length === 0 ||
|
|
205
|
+
profile.applicationTarget.callSites.some((callSite) => typeof callSite !== "string" || !callSite.trim())) {
|
|
206
|
+
errors.push("taskProfile requires a user-confirmed application LLM target with stable call sites and route key");
|
|
207
|
+
}
|
|
208
|
+
if (isRecord(profile.applicationTarget)) {
|
|
209
|
+
errors.push(...validateTargetId(output.incumbent.targetId, profile.applicationTarget.id, "incumbent"));
|
|
210
|
+
if (typeof output.incumbent.candidateId !== "string" || !output.incumbent.candidateId.trim() ||
|
|
211
|
+
typeof output.incumbent.modelId !== "string" || !output.incumbent.modelId.trim()) {
|
|
212
|
+
errors.push("incumbent requires stable candidateId and exact modelId");
|
|
151
213
|
}
|
|
152
214
|
}
|
|
153
215
|
const budget = policy.evaluationBudget;
|
|
@@ -158,20 +220,71 @@ function validateSetupOutput(output) {
|
|
|
158
220
|
policy.requirePostSwapVerification !== true || typeof policy.rollbackOnRegression !== "boolean") {
|
|
159
221
|
errors.push("policy must be a valid supervised-graduation policy with an evaluation budget");
|
|
160
222
|
}
|
|
223
|
+
const checkPolicy = policy.checkPolicy;
|
|
224
|
+
if (!isRecord(checkPolicy) || !isRecord(checkPolicy.baselineAssessment) ||
|
|
225
|
+
(typeof checkPolicy.baselineAssessment.afterCompletedTasks !== "number" &&
|
|
226
|
+
typeof checkPolicy.baselineAssessment.afterElapsedSeconds !== "number")) {
|
|
227
|
+
errors.push("policy requires a baseline assessment cadence");
|
|
228
|
+
}
|
|
229
|
+
if (!isRecord(checkPolicy) || !isRecord(checkPolicy.postSwapVerification) ||
|
|
230
|
+
(typeof checkPolicy.postSwapVerification.afterCompletedTasks !== "number" &&
|
|
231
|
+
typeof checkPolicy.postSwapVerification.afterElapsedSeconds !== "number")) {
|
|
232
|
+
errors.push("policy requires a post-swap verification cadence");
|
|
233
|
+
}
|
|
161
234
|
if (!isRecord(output.observability)) {
|
|
162
235
|
errors.push("setup output requires an observability coverage report");
|
|
163
236
|
}
|
|
164
237
|
else if (Array.isArray(profile.objectives)) {
|
|
165
238
|
errors.push(...validateObservabilityCoverage(profile, output.observability, false));
|
|
166
239
|
}
|
|
240
|
+
if (Array.isArray(output.preliminaryModelLeads) && isRecord(profile.applicationTarget)) {
|
|
241
|
+
const seen = new Set();
|
|
242
|
+
for (const lead of output.preliminaryModelLeads) {
|
|
243
|
+
if (!isRecord(lead))
|
|
244
|
+
continue;
|
|
245
|
+
errors.push(...validateTargetId(lead.targetId, profile.applicationTarget.id, `preliminary lead ${String(lead.modelId)}`));
|
|
246
|
+
const identity = `${String(lead.modelId)}@${String(lead.modelVersion ?? "")}`;
|
|
247
|
+
if (seen.has(identity))
|
|
248
|
+
errors.push(`duplicate preliminary model lead ${identity}`);
|
|
249
|
+
seen.add(identity);
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
return errors;
|
|
253
|
+
}
|
|
254
|
+
function validateMetricAggregation(metric, objective) {
|
|
255
|
+
const errors = [];
|
|
256
|
+
if (metric.metricId === "total_task_cost" && metric.state === "measured" &&
|
|
257
|
+
metric.scope === "sample_window" && metric.aggregation !== "sum") {
|
|
258
|
+
errors.push("total_task_cost sample windows must use sum aggregation because the value is used for evaluation-budget accounting");
|
|
259
|
+
}
|
|
260
|
+
if (objective?.samplingPlan && metric.state === "measured" &&
|
|
261
|
+
metric.aggregation !== objective.samplingPlan.aggregation &&
|
|
262
|
+
!(objective.metricId === "total_task_cost" && metric.aggregation === "sum")) {
|
|
263
|
+
errors.push(`${objective.metricId} aggregation ${metric.aggregation} does not match its sampling plan ${objective.samplingPlan.aggregation}`);
|
|
264
|
+
}
|
|
167
265
|
return errors;
|
|
168
266
|
}
|
|
267
|
+
function validateMetricWindow(metric, startedAt, completedAt) {
|
|
268
|
+
if (!metric.window)
|
|
269
|
+
return [];
|
|
270
|
+
const start = Date.parse(metric.window.startedAt);
|
|
271
|
+
const end = Date.parse(metric.window.endedAt);
|
|
272
|
+
const intentStart = Date.parse(startedAt);
|
|
273
|
+
const intentEnd = Date.parse(completedAt);
|
|
274
|
+
if (![start, end, intentStart, intentEnd].every(Number.isFinite) || end < start || start < intentStart || end > intentEnd) {
|
|
275
|
+
return [`${metric.metricId} window must be valid and contained within the intent execution window`];
|
|
276
|
+
}
|
|
277
|
+
return [];
|
|
278
|
+
}
|
|
169
279
|
export function validateObservabilityCoverage(profile, coverage, requireReady) {
|
|
170
280
|
const required = new Set(profile.objectives.filter((item) => item.required).map((item) => item.metricId));
|
|
171
281
|
const declaredRequired = new Set(coverage.requiredMetricIds);
|
|
172
282
|
const covered = new Set(coverage.coveredMetricIds);
|
|
173
283
|
const missing = new Set(coverage.missingMetricIds);
|
|
174
284
|
const errors = [];
|
|
285
|
+
if (coverage.targetId !== profile.applicationTarget.id) {
|
|
286
|
+
errors.push("observability target does not match the confirmed application LLM target");
|
|
287
|
+
}
|
|
175
288
|
for (const metricId of required) {
|
|
176
289
|
if (!declaredRequired.has(metricId))
|
|
177
290
|
errors.push(`observability omits required metric ${metricId}`);
|
|
@@ -191,6 +304,15 @@ export function validateObservabilityCoverage(profile, coverage, requireReady) {
|
|
|
191
304
|
}
|
|
192
305
|
return errors;
|
|
193
306
|
}
|
|
307
|
+
function activeTarget(config) {
|
|
308
|
+
return config.taskProfiles.find((item) => item.id === config.activeTaskProfileId)?.applicationTarget;
|
|
309
|
+
}
|
|
310
|
+
function validateTargetId(actual, expected, subject) {
|
|
311
|
+
return actual === expected ? [] : [`${subject} target does not match the confirmed application LLM target`];
|
|
312
|
+
}
|
|
313
|
+
function validateMetricTargets(metrics, targetId, subject) {
|
|
314
|
+
return metrics.flatMap((metric) => validateTargetId(metric.targetId, targetId, `${subject} metric ${metric.metricId}`));
|
|
315
|
+
}
|
|
194
316
|
export function interpretIntentResult(intent, result, cycle, config) {
|
|
195
317
|
const errors = [];
|
|
196
318
|
if (result.protocolVersion !== PROTOCOL_VERSION)
|
|
@@ -208,6 +330,35 @@ export function interpretIntentResult(intent, result, cycle, config) {
|
|
|
208
330
|
}
|
|
209
331
|
const output = result.outputs;
|
|
210
332
|
errors.push(...validateIntentOutputSchema(intent.kind, output).map((error) => `result outputs ${error}`));
|
|
333
|
+
if (isRecord(output) && (intent.kind === "run_assessment" || intent.kind === "run_challenger_trials")) {
|
|
334
|
+
const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
335
|
+
const objectives = new Map((profile?.objectives ?? []).map((objective) => [objective.metricId, objective]));
|
|
336
|
+
const metricGroups = intent.kind === "run_assessment"
|
|
337
|
+
? [{ metrics: output.metrics, startedAt: undefined, completedAt: undefined }]
|
|
338
|
+
: (Array.isArray(output.assessments) ? output.assessments.map((assessment) => isRecord(assessment)
|
|
339
|
+
? { metrics: assessment.metrics, startedAt: assessment.startedAt, completedAt: assessment.completedAt }
|
|
340
|
+
: { metrics: undefined, startedAt: undefined, completedAt: undefined }) : []);
|
|
341
|
+
for (const group of metricGroups) {
|
|
342
|
+
const metrics = group.metrics;
|
|
343
|
+
if (!Array.isArray(metrics))
|
|
344
|
+
continue;
|
|
345
|
+
for (const metric of metrics) {
|
|
346
|
+
if (!isRecord(metric))
|
|
347
|
+
continue;
|
|
348
|
+
const typedMetric = metric;
|
|
349
|
+
errors.push(...validateMetricAggregation(typedMetric, objectives.get(typedMetric.metricId)));
|
|
350
|
+
if (group.startedAt && group.completedAt)
|
|
351
|
+
errors.push(...validateMetricWindow(typedMetric, String(group.startedAt), String(group.completedAt)));
|
|
352
|
+
}
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
const target = activeTarget(config);
|
|
356
|
+
if (intent.kind !== "establish_evals") {
|
|
357
|
+
if (!target)
|
|
358
|
+
errors.push("no confirmed application LLM target exists");
|
|
359
|
+
else
|
|
360
|
+
errors.push(...validateTargetId(intent.targetId, target.id, "intent"));
|
|
361
|
+
}
|
|
211
362
|
if (errors.length) {
|
|
212
363
|
return { valid: false, errors, nextStage: completedStage(intent.kind, false) };
|
|
213
364
|
}
|
|
@@ -216,20 +367,27 @@ export function interpretIntentResult(intent, result, cycle, config) {
|
|
|
216
367
|
let cyclePatch;
|
|
217
368
|
switch (intent.kind) {
|
|
218
369
|
case "establish_evals": {
|
|
219
|
-
errors.push(...validateSetupOutput(output
|
|
370
|
+
errors.push(...validateSetupOutput(output, typeof intent.constraints.expectedIncumbentModelId === "string"
|
|
371
|
+
? intent.constraints.expectedIncumbentModelId : undefined));
|
|
220
372
|
if (!errors.length && isRecord(output)) {
|
|
221
373
|
const taskProfile = output.taskProfile;
|
|
222
374
|
const policy = output.policy;
|
|
223
375
|
const observability = output.observability;
|
|
376
|
+
const incumbent = output.incumbent;
|
|
377
|
+
const preliminaryModelLeads = (output.preliminaryModelLeads ?? []);
|
|
224
378
|
nextConfig = {
|
|
225
379
|
schemaVersion: 1,
|
|
226
380
|
activeTaskProfileId: taskProfile.id,
|
|
227
381
|
taskProfiles: [...config.taskProfiles.filter((item) => item.id !== taskProfile.id), taskProfile],
|
|
228
382
|
policy,
|
|
383
|
+
preliminaryModelLeads,
|
|
229
384
|
};
|
|
230
385
|
cyclePatch = {
|
|
386
|
+
targetId: taskProfile.applicationTarget.id,
|
|
231
387
|
taskProfileId: taskProfile.id,
|
|
388
|
+
activeCandidateId: incumbent.candidateId,
|
|
232
389
|
policy,
|
|
390
|
+
baselineEvidenceSufficient: false,
|
|
233
391
|
observabilityReady: observability.status === "ready",
|
|
234
392
|
};
|
|
235
393
|
if (observability.status !== "ready")
|
|
@@ -246,7 +404,9 @@ export function interpretIntentResult(intent, result, cycle, config) {
|
|
|
246
404
|
if (!profile)
|
|
247
405
|
errors.push("no confirmed task profile exists for observability verification");
|
|
248
406
|
else {
|
|
407
|
+
errors.push(...validateTargetId(output.targetId, profile.applicationTarget.id, "observability result"));
|
|
249
408
|
const coverage = {
|
|
409
|
+
targetId: profile.applicationTarget.id,
|
|
250
410
|
status: output.missingMetricIds.length ? "incomplete" : "ready",
|
|
251
411
|
requiredMetricIds: profile.objectives.filter((item) => item.required).map((item) => item.metricId),
|
|
252
412
|
coveredMetricIds: output.coveredMetricIds,
|
|
@@ -259,10 +419,18 @@ export function interpretIntentResult(intent, result, cycle, config) {
|
|
|
259
419
|
}
|
|
260
420
|
break;
|
|
261
421
|
case "run_assessment":
|
|
262
|
-
if (!isRecord(output) || typeof output.baselineEvidenceSufficient !== "boolean") {
|
|
422
|
+
if (!isRecord(output) || typeof output.baselineEvidenceSufficient !== "boolean" || !Array.isArray(output.metrics)) {
|
|
263
423
|
errors.push("assessment result requires baselineEvidenceSufficient");
|
|
264
424
|
}
|
|
265
425
|
else {
|
|
426
|
+
if (target) {
|
|
427
|
+
errors.push(...validateTargetId(output.targetId, target.id, "baseline assessment"));
|
|
428
|
+
errors.push(...validateMetricTargets(output.metrics, target.id, "baseline assessment"));
|
|
429
|
+
const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
430
|
+
if (output.baselineEvidenceSufficient) {
|
|
431
|
+
errors.push(...baselineReadinessErrors(profile, output.metrics));
|
|
432
|
+
}
|
|
433
|
+
}
|
|
266
434
|
cyclePatch = { baselineEvidenceSufficient: output.baselineEvidenceSufficient };
|
|
267
435
|
if (!output.baselineEvidenceSufficient)
|
|
268
436
|
nextStage = "collecting_baseline";
|
|
@@ -271,10 +439,61 @@ export function interpretIntentResult(intent, result, cycle, config) {
|
|
|
271
439
|
case "discover_candidates":
|
|
272
440
|
if (!isRecord(output) || !Array.isArray(output.candidates))
|
|
273
441
|
errors.push("candidate discovery requires candidates");
|
|
442
|
+
else if (target) {
|
|
443
|
+
errors.push(...validateTargetId(output.targetId, target.id, "candidate discovery"));
|
|
444
|
+
const candidateIds = new Set();
|
|
445
|
+
for (const candidate of output.candidates) {
|
|
446
|
+
if (isRecord(candidate)) {
|
|
447
|
+
errors.push(...validateTargetId(candidate.targetId, target.id, `candidate ${String(candidate.candidateId)}`));
|
|
448
|
+
if (typeof candidate.candidateId === "string") {
|
|
449
|
+
if (candidateIds.has(candidate.candidateId))
|
|
450
|
+
errors.push(`duplicate candidate ${candidate.candidateId}`);
|
|
451
|
+
candidateIds.add(candidate.candidateId);
|
|
452
|
+
}
|
|
453
|
+
}
|
|
454
|
+
}
|
|
455
|
+
if (cycle.activeCandidateId && !candidateIds.has(cycle.activeCandidateId)) {
|
|
456
|
+
errors.push("candidate discovery must include the incumbent application model");
|
|
457
|
+
}
|
|
458
|
+
const maximum = config.policy?.evaluationBudget.maximumCandidateCount;
|
|
459
|
+
if (maximum !== undefined && output.candidates.length > maximum) {
|
|
460
|
+
errors.push(`candidate discovery exceeds maximumCandidateCount ${maximum}`);
|
|
461
|
+
}
|
|
462
|
+
}
|
|
274
463
|
break;
|
|
275
464
|
case "run_challenger_trials":
|
|
276
|
-
if (!isRecord(output) || !Array.isArray(output.assessments))
|
|
277
|
-
errors.push("challenger trials require assessments");
|
|
465
|
+
if (!isRecord(output) || !Array.isArray(output.assessments) || !isRecord(output.experimentManifest)) {
|
|
466
|
+
errors.push("challenger trials require assessments and an experimentManifest");
|
|
467
|
+
}
|
|
468
|
+
else {
|
|
469
|
+
if (target)
|
|
470
|
+
errors.push(...validateTargetId(output.targetId, target.id, "challenger trials"));
|
|
471
|
+
const expectedCandidateIds = Array.isArray(intent.constraints.candidates)
|
|
472
|
+
? new Set(intent.constraints.candidates.flatMap((candidate) => isRecord(candidate) && typeof candidate.candidateId === "string" ? [candidate.candidateId] : []))
|
|
473
|
+
: undefined;
|
|
474
|
+
const assessedCandidateIds = new Set();
|
|
475
|
+
const productRevisions = new Set();
|
|
476
|
+
for (const assessment of output.assessments) {
|
|
477
|
+
assessedCandidateIds.add(assessment.candidateId);
|
|
478
|
+
productRevisions.add(assessment.productRevision);
|
|
479
|
+
if (target) {
|
|
480
|
+
errors.push(...validateTargetId(assessment.targetId, target.id, `challenger assessment ${assessment.id}`));
|
|
481
|
+
errors.push(...validateMetricTargets(assessment.metrics, target.id, `challenger assessment ${assessment.id}`));
|
|
482
|
+
}
|
|
483
|
+
const costs = assessment.metrics.filter((metric) => metric.metricId === "total_task_cost" && metric.state === "measured" &&
|
|
484
|
+
metric.value !== undefined && metric.value >= 0 && metric.evidence.length > 0);
|
|
485
|
+
if (costs.length !== 1) {
|
|
486
|
+
errors.push(`challenger assessment ${assessment.id} requires exactly one measured total_task_cost with evidence`);
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
if (expectedCandidateIds &&
|
|
490
|
+
(expectedCandidateIds.size !== assessedCandidateIds.size ||
|
|
491
|
+
[...expectedCandidateIds].some((candidateId) => !assessedCandidateIds.has(candidateId)))) {
|
|
492
|
+
errors.push("challenger assessments must exactly cover the frozen candidate set");
|
|
493
|
+
}
|
|
494
|
+
if (productRevisions.size !== 1)
|
|
495
|
+
errors.push("challenger assessments must use one application product revision");
|
|
496
|
+
}
|
|
278
497
|
break;
|
|
279
498
|
case "calculate_frontier": {
|
|
280
499
|
if (!isRecord(output) || !Array.isArray(output.assessments) || !isRecord(output.frontierSnapshot)) {
|
|
@@ -286,6 +505,16 @@ export function interpretIntentResult(intent, result, cycle, config) {
|
|
|
286
505
|
errors.push("no confirmed task profile exists for frontier verification");
|
|
287
506
|
break;
|
|
288
507
|
}
|
|
508
|
+
errors.push(...validateTargetId(output.targetId, profile.applicationTarget.id, "frontier result"));
|
|
509
|
+
const expectedAssessmentIds = Array.isArray(intent.constraints.assessments)
|
|
510
|
+
? new Set(intent.constraints.assessments.flatMap((assessment) => isRecord(assessment) && typeof assessment.id === "string" ? [assessment.id] : []))
|
|
511
|
+
: undefined;
|
|
512
|
+
const suppliedAssessmentIds = new Set(output.assessments.map((assessment) => assessment.id));
|
|
513
|
+
if (expectedAssessmentIds &&
|
|
514
|
+
(expectedAssessmentIds.size !== suppliedAssessmentIds.size ||
|
|
515
|
+
[...expectedAssessmentIds].some((assessmentId) => !suppliedAssessmentIds.has(assessmentId)))) {
|
|
516
|
+
errors.push("frontier calculation must use exactly the accepted challenger assessments");
|
|
517
|
+
}
|
|
289
518
|
const verification = verifyFrontierSnapshot(profile, output.assessments, output.frontierSnapshot);
|
|
290
519
|
errors.push(...verification.errors);
|
|
291
520
|
if (output.recommendation !== undefined) {
|
|
@@ -295,6 +524,12 @@ export function interpretIntentResult(intent, result, cycle, config) {
|
|
|
295
524
|
else if (!output.frontierSnapshot.frontierCandidateIds.includes(output.recommendation.selectedCandidateId)) {
|
|
296
525
|
errors.push("recommended candidate is not on the verified frontier");
|
|
297
526
|
}
|
|
527
|
+
else if (output.recommendation.targetId !== profile.applicationTarget.id) {
|
|
528
|
+
errors.push("frontier recommendation target does not match the confirmed application LLM target");
|
|
529
|
+
}
|
|
530
|
+
else if (cycle.activeCandidateId && output.recommendation.currentCandidateId !== cycle.activeCandidateId) {
|
|
531
|
+
errors.push("frontier recommendation current candidate does not match the incumbent application model");
|
|
532
|
+
}
|
|
298
533
|
else {
|
|
299
534
|
cyclePatch = { proposedCandidateId: output.recommendation.selectedCandidateId };
|
|
300
535
|
}
|
|
@@ -305,11 +540,16 @@ export function interpretIntentResult(intent, result, cycle, config) {
|
|
|
305
540
|
if (!isRecord(output) || typeof output.regressionDetected !== "boolean" || !Array.isArray(output.affectedMetricIds)) {
|
|
306
541
|
errors.push("regression result requires regressionDetected and affectedMetricIds");
|
|
307
542
|
}
|
|
543
|
+
else if (target)
|
|
544
|
+
errors.push(...validateTargetId(output.targetId, target.id, "regression investigation"));
|
|
308
545
|
break;
|
|
309
546
|
case "apply_model_swap":
|
|
310
547
|
if (!isRecord(output) || typeof output.appliedCandidateId !== "string") {
|
|
311
548
|
errors.push("swap result requires appliedCandidateId");
|
|
312
549
|
}
|
|
550
|
+
else if (target && output.targetId !== target.id) {
|
|
551
|
+
errors.push("swap target does not match the confirmed application LLM target");
|
|
552
|
+
}
|
|
313
553
|
else if (cycle.proposedCandidateId && output.appliedCandidateId !== cycle.proposedCandidateId) {
|
|
314
554
|
errors.push("applied candidate does not match the authorized proposal");
|
|
315
555
|
}
|
|
@@ -324,6 +564,9 @@ export function interpretIntentResult(intent, result, cycle, config) {
|
|
|
324
564
|
if (!isRecord(output) || typeof output.verified !== "boolean" || typeof output.regressionDetected !== "boolean") {
|
|
325
565
|
errors.push("swap verification requires verified and regressionDetected");
|
|
326
566
|
}
|
|
567
|
+
else if (target && output.targetId !== target.id) {
|
|
568
|
+
errors.push("swap verification target does not match the confirmed application LLM target");
|
|
569
|
+
}
|
|
327
570
|
else if (!output.verified || output.regressionDetected) {
|
|
328
571
|
nextStage = "verification_failed";
|
|
329
572
|
}
|
|
@@ -335,6 +578,9 @@ export function interpretIntentResult(intent, result, cycle, config) {
|
|
|
335
578
|
if (!isRecord(output) || typeof output.restoredCandidateId !== "string") {
|
|
336
579
|
errors.push("rollback result requires restoredCandidateId");
|
|
337
580
|
}
|
|
581
|
+
else if (target && output.targetId !== target.id) {
|
|
582
|
+
errors.push("rollback target does not match the confirmed application LLM target");
|
|
583
|
+
}
|
|
338
584
|
else if (cycle.previousCandidateId && output.restoredCandidateId !== cycle.previousCandidateId) {
|
|
339
585
|
errors.push("rollback did not restore the previous verified candidate");
|
|
340
586
|
}
|
|
@@ -359,7 +605,7 @@ export function nextImprovementAction(state) {
|
|
|
359
605
|
: { action: "issue_intent", intentKind: "instrument_observability", reason: "Required metric instrumentation is incomplete." };
|
|
360
606
|
case "collecting_baseline":
|
|
361
607
|
return state.baselineEvidenceSufficient
|
|
362
|
-
? { action: "issue_intent", intentKind: "
|
|
608
|
+
? { action: "issue_intent", intentKind: "discover_candidates", reason: "The baseline is ready for candidate discovery." }
|
|
363
609
|
: { action: "observe", reason: "Continue collecting baseline production evidence." };
|
|
364
610
|
case "baseline_ready":
|
|
365
611
|
return { action: "issue_intent", intentKind: "discover_candidates", reason: "The baseline is ready for budgeted candidate discovery." };
|
|
@@ -411,22 +657,71 @@ function bounds(metric) {
|
|
|
411
657
|
}
|
|
412
658
|
function readinessErrors(profile, assessment) {
|
|
413
659
|
const errors = [];
|
|
660
|
+
if (assessment.targetId !== profile.applicationTarget.id) {
|
|
661
|
+
errors.push("assessment target does not match the confirmed application LLM target");
|
|
662
|
+
}
|
|
414
663
|
for (const objective of profile.objectives) {
|
|
415
664
|
if (!objective.required)
|
|
416
665
|
continue;
|
|
417
666
|
const metric = metricFor(assessment, objective);
|
|
418
667
|
if (!metric)
|
|
419
668
|
errors.push(`${objective.metricId}: missing`);
|
|
420
|
-
else
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
errors.push(
|
|
669
|
+
else {
|
|
670
|
+
if (metric.state !== "measured")
|
|
671
|
+
errors.push(`${objective.metricId}: ${metric.state}`);
|
|
672
|
+
else if (metric.value === undefined)
|
|
673
|
+
errors.push(`${objective.metricId}: no numeric value`);
|
|
674
|
+
errors.push(...validateMetricAggregation(metric, objective));
|
|
675
|
+
if (metric.sampleCount < objective.minimumSamples) {
|
|
676
|
+
errors.push(`${objective.metricId}: ${metric.sampleCount}/${objective.minimumSamples} samples`);
|
|
677
|
+
}
|
|
426
678
|
}
|
|
427
679
|
}
|
|
428
680
|
return errors;
|
|
429
681
|
}
|
|
682
|
+
function baselineReadinessErrors(profile, metrics) {
|
|
683
|
+
if (!profile)
|
|
684
|
+
return ["confirmed task profile is missing"];
|
|
685
|
+
const errors = [];
|
|
686
|
+
if (!profile.objectives.some((objective) => objective.required)) {
|
|
687
|
+
errors.push("task profile has no required metric objectives");
|
|
688
|
+
}
|
|
689
|
+
for (const objective of profile.objectives) {
|
|
690
|
+
if (!objective.required)
|
|
691
|
+
continue;
|
|
692
|
+
const metric = metrics.find((item) => item.metricId === objective.metricId &&
|
|
693
|
+
item.targetId === profile.applicationTarget.id);
|
|
694
|
+
if (!metric)
|
|
695
|
+
errors.push(`${objective.metricId}: missing`);
|
|
696
|
+
else {
|
|
697
|
+
if (metric.state !== "measured")
|
|
698
|
+
errors.push(`${objective.metricId}: ${metric.state}`);
|
|
699
|
+
else if (metric.value === undefined || !Number.isFinite(metric.value))
|
|
700
|
+
errors.push(`${objective.metricId}: no finite numeric value`);
|
|
701
|
+
errors.push(...validateMetricAggregation(metric, objective));
|
|
702
|
+
if (metric.sampleCount < objective.minimumSamples) {
|
|
703
|
+
errors.push(`${objective.metricId}: ${metric.sampleCount}/${objective.minimumSamples} samples`);
|
|
704
|
+
}
|
|
705
|
+
else if (!metric.evidence.length)
|
|
706
|
+
errors.push(`${objective.metricId}: no evidence reference`);
|
|
707
|
+
}
|
|
708
|
+
}
|
|
709
|
+
return errors;
|
|
710
|
+
}
|
|
711
|
+
function baselineAssessmentErrors(state, profile, metrics) {
|
|
712
|
+
return [
|
|
713
|
+
...(state.cycle.observabilityReady ? [] : ["required evaluation and observability setup is not ready"]),
|
|
714
|
+
...baselineReadinessErrors(profile, metrics),
|
|
715
|
+
];
|
|
716
|
+
}
|
|
717
|
+
function updatedBaselineWindows(previous, signal) {
|
|
718
|
+
if (signal.type !== "metric_window_available")
|
|
719
|
+
return previous ?? [];
|
|
720
|
+
const windows = new Map((previous ?? []).map((metric) => [metric.metricId, metric]));
|
|
721
|
+
for (const metric of signal.metrics)
|
|
722
|
+
windows.set(metric.metricId, metric);
|
|
723
|
+
return [...windows.values()];
|
|
724
|
+
}
|
|
430
725
|
function feasibility(profile, assessment) {
|
|
431
726
|
if (readinessErrors(profile, assessment).length)
|
|
432
727
|
return "unresolved";
|
|
@@ -525,6 +820,9 @@ export function verifyFrontierSnapshot(profile, assessments, snapshot) {
|
|
|
525
820
|
if (snapshot.taskProfileId !== profile.id || snapshot.taskProfileRevision !== profile.revision) {
|
|
526
821
|
errors.push("snapshot task profile does not match");
|
|
527
822
|
}
|
|
823
|
+
if (snapshot.targetId !== profile.applicationTarget.id) {
|
|
824
|
+
errors.push("snapshot target does not match the confirmed application LLM target");
|
|
825
|
+
}
|
|
528
826
|
if (snapshot.calculationEvidence.length === 0)
|
|
529
827
|
errors.push("calculation evidence is required");
|
|
530
828
|
if (!sameMembers(snapshot.assessmentIds, assessments.map((assessment) => assessment.id))) {
|
|
@@ -534,6 +832,11 @@ export function verifyFrontierSnapshot(profile, assessments, snapshot) {
|
|
|
534
832
|
const expectedIneligible = [];
|
|
535
833
|
const expectedUnresolved = new Set();
|
|
536
834
|
for (const assessment of assessments) {
|
|
835
|
+
if (assessment.targetId !== profile.applicationTarget.id) {
|
|
836
|
+
errors.push(`${assessment.candidateId}: application target mismatch`);
|
|
837
|
+
continue;
|
|
838
|
+
}
|
|
839
|
+
errors.push(...validateMetricTargets(assessment.metrics, profile.applicationTarget.id, assessment.candidateId));
|
|
537
840
|
if (assessment.taskProfileId !== profile.id || assessment.taskProfileRevision !== profile.revision) {
|
|
538
841
|
errors.push(`${assessment.candidateId}: task profile revision mismatch`);
|
|
539
842
|
continue;
|
|
@@ -611,6 +914,14 @@ function earliest(values) {
|
|
|
611
914
|
.sort((left, right) => Date.parse(left) - Date.parse(right))[0];
|
|
612
915
|
}
|
|
613
916
|
export function buildAutomationPlan(state, descriptor) {
|
|
917
|
+
if (state.automation.paused) {
|
|
918
|
+
return {
|
|
919
|
+
mode: state.cycle.policy?.checkPolicy.execution.mode ?? "active_session_only",
|
|
920
|
+
requiredSignals: [],
|
|
921
|
+
scheduling: "active_session_fallback",
|
|
922
|
+
gaps: ["Proactive OpenMerit checks are paused."],
|
|
923
|
+
};
|
|
924
|
+
}
|
|
614
925
|
const policy = state.cycle.policy?.checkPolicy;
|
|
615
926
|
if (!policy) {
|
|
616
927
|
return {
|
|
@@ -640,7 +951,7 @@ export function buildAutomationPlan(state, descriptor) {
|
|
|
640
951
|
required.add("verification_window_completed");
|
|
641
952
|
}
|
|
642
953
|
const nextWakeupAt = state.cycle.stage === "collecting_baseline"
|
|
643
|
-
? nextAt(state.automation.lastAssessmentAt ?? state.updatedAt, policy.baselineAssessment.afterElapsedSeconds)
|
|
954
|
+
? nextAt(state.automation.lastAssessmentAt ?? state.automation.baselineStartedAt ?? state.lastResult?.completedAt ?? state.updatedAt, policy.baselineAssessment.afterElapsedSeconds)
|
|
644
955
|
: ["monitoring", "swap_verified"].includes(state.cycle.stage)
|
|
645
956
|
? earliest([
|
|
646
957
|
nextAt(state.automation.lastReassessmentAt ?? state.updatedAt, policy.frontierReassessment.afterElapsedSeconds),
|
|
@@ -701,6 +1012,8 @@ function regressionMetric(signal, previous, policy) {
|
|
|
701
1012
|
return undefined;
|
|
702
1013
|
}
|
|
703
1014
|
function dueDecision(state, signal, now) {
|
|
1015
|
+
if (state.automation.paused)
|
|
1016
|
+
return { action: "observe", reason: "Proactive OpenMerit checks are paused." };
|
|
704
1017
|
if (state.activeIntent)
|
|
705
1018
|
return { action: "observe", reason: "Harness work is already active." };
|
|
706
1019
|
if (state.automation.pendingIntentKind) {
|
|
@@ -719,7 +1032,7 @@ function dueDecision(state, signal, now) {
|
|
|
719
1032
|
.sort((left, right) => Date.parse(right) - Date.parse(left))[0];
|
|
720
1033
|
const cooldownSatisfied = !cooldownAnchor ||
|
|
721
1034
|
Date.parse(now) - Date.parse(cooldownAnchor) >= policy.checkPolicy.cooldownSeconds * 1000;
|
|
722
|
-
if (state.cycle.stage === "collecting_baseline" && cadenceDue(policy.checkPolicy.baselineAssessment, completed, state.automation.lastAssessmentTaskCount, state.automation.lastAssessmentAt ?? state.updatedAt, now)) {
|
|
1035
|
+
if (state.cycle.stage === "collecting_baseline" && cadenceDue(policy.checkPolicy.baselineAssessment, completed, state.automation.lastAssessmentTaskCount, state.automation.lastAssessmentAt ?? state.automation.baselineStartedAt ?? state.lastResult?.completedAt ?? state.updatedAt, now)) {
|
|
723
1036
|
return { action: "issue_intent", intentKind: "run_assessment", reason: "The user-confirmed baseline assessment cadence is due." };
|
|
724
1037
|
}
|
|
725
1038
|
if (["monitoring", "swap_verified"].includes(state.cycle.stage) && cooldownSatisfied) {
|
|
@@ -749,14 +1062,17 @@ function auditSignalData(signal) {
|
|
|
749
1062
|
switch (signal.type) {
|
|
750
1063
|
case "product_task_completed":
|
|
751
1064
|
return {
|
|
1065
|
+
targetId: signal.targetId,
|
|
752
1066
|
modelId: signal.modelId,
|
|
753
|
-
|
|
1067
|
+
...(signal.telemetry ? { telemetry: signal.telemetry } : {}),
|
|
754
1068
|
evidenceIds: (signal.evidence ?? []).map((item) => item.id),
|
|
755
1069
|
};
|
|
756
1070
|
case "metric_window_available":
|
|
757
1071
|
case "verification_window_completed":
|
|
758
1072
|
return {
|
|
1073
|
+
targetId: signal.targetId,
|
|
759
1074
|
metrics: (signal.metrics ?? []).map((metric) => ({
|
|
1075
|
+
targetId: metric.targetId,
|
|
760
1076
|
metricId: metric.metricId,
|
|
761
1077
|
state: metric.state,
|
|
762
1078
|
scope: metric.scope,
|
|
@@ -771,9 +1087,9 @@ function auditSignalData(signal) {
|
|
|
771
1087
|
})),
|
|
772
1088
|
};
|
|
773
1089
|
case "model_catalog_changed":
|
|
774
|
-
return { fingerprint: signal.fingerprint };
|
|
1090
|
+
return { targetId: signal.targetId, fingerprint: signal.fingerprint };
|
|
775
1091
|
case "scheduled_tick":
|
|
776
|
-
return {};
|
|
1092
|
+
return { targetId: signal.targetId };
|
|
777
1093
|
}
|
|
778
1094
|
}
|
|
779
1095
|
/** Harness-neutral durable coordinator. Adapters transport jobs; this class owns lifecycle state. */
|
|
@@ -784,7 +1100,7 @@ export class OpenMeritCoordinator {
|
|
|
784
1100
|
this.store = store;
|
|
785
1101
|
this.adapter = adapter;
|
|
786
1102
|
}
|
|
787
|
-
async issue(kind) {
|
|
1103
|
+
async issue(kind, options = {}) {
|
|
788
1104
|
const missing = missingHarnessCapabilities(this.adapter.descriptor, kind);
|
|
789
1105
|
if (missing.length)
|
|
790
1106
|
throw new Error(`Harness lacks required capabilities: ${missing.join(", ")}`);
|
|
@@ -792,16 +1108,80 @@ export class OpenMeritCoordinator {
|
|
|
792
1108
|
const existing = await this.store.readState();
|
|
793
1109
|
if (existing?.activeIntent)
|
|
794
1110
|
throw new Error(`OpenMerit intent ${existing.activeIntent.id} is already active.`);
|
|
1111
|
+
if (kind === "run_challenger_trials" && config.policy) {
|
|
1112
|
+
const spent = existing?.automation.evaluationSpend ?? 0;
|
|
1113
|
+
const limit = config.policy.evaluationBudget.maximumSpend;
|
|
1114
|
+
if (spent >= limit) {
|
|
1115
|
+
throw new Error(`Evaluation budget exhausted: ${spent.toFixed(6)} ${config.policy.evaluationBudget.currency} spent of ${limit.toFixed(6)}.`);
|
|
1116
|
+
}
|
|
1117
|
+
}
|
|
1118
|
+
const targetId = activeTarget(config)?.id;
|
|
1119
|
+
const confirmedProfile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
1120
|
+
if (kind !== "establish_evals" && !targetId) {
|
|
1121
|
+
throw new Error(`OpenMerit cannot issue ${kind} without a confirmed application LLM target.`);
|
|
1122
|
+
}
|
|
795
1123
|
const intent = createOpenMeritIntent(kind, {
|
|
796
1124
|
taskProfileId: config.activeTaskProfileId,
|
|
797
1125
|
policy: config.policy,
|
|
1126
|
+
evaluationSpend: existing?.automation.evaluationSpend,
|
|
1127
|
+
targetId,
|
|
1128
|
+
additionalConstraints: {
|
|
1129
|
+
...options.additionalConstraints,
|
|
1130
|
+
...(confirmedProfile ? { taskProfile: confirmedProfile } : {}),
|
|
1131
|
+
...(kind === "discover_candidates"
|
|
1132
|
+
? {
|
|
1133
|
+
preliminaryModelLeads: (config.preliminaryModelLeads ?? []),
|
|
1134
|
+
incumbentCandidate: (existing?.candidates?.find((candidate) => candidate.candidateId === existing.cycle.activeCandidateId) ?? existing?.candidates?.[0]),
|
|
1135
|
+
candidateSelection: {
|
|
1136
|
+
totalCandidateCount: Math.min(3, config.policy?.evaluationBudget.maximumCandidateCount ?? 3),
|
|
1137
|
+
includeIncumbent: true,
|
|
1138
|
+
requireAvailabilityProbe: true,
|
|
1139
|
+
requireExactModelIds: true,
|
|
1140
|
+
requirePriceAndCapabilitySources: true,
|
|
1141
|
+
},
|
|
1142
|
+
}
|
|
1143
|
+
: {}),
|
|
1144
|
+
...(kind === "run_challenger_trials"
|
|
1145
|
+
? { candidates: (existing?.candidates ?? []) }
|
|
1146
|
+
: {}),
|
|
1147
|
+
...(kind === "calculate_frontier"
|
|
1148
|
+
? {
|
|
1149
|
+
assessments: (existing?.assessments ?? []),
|
|
1150
|
+
...(existing?.experimentManifest
|
|
1151
|
+
? { experimentManifest: existing.experimentManifest }
|
|
1152
|
+
: {}),
|
|
1153
|
+
}
|
|
1154
|
+
: {}),
|
|
1155
|
+
},
|
|
798
1156
|
});
|
|
799
1157
|
const now = new Date().toISOString();
|
|
800
1158
|
const state = existing
|
|
801
1159
|
? {
|
|
802
1160
|
...existing,
|
|
803
|
-
cycle:
|
|
1161
|
+
cycle: kind === "establish_evals"
|
|
1162
|
+
? {
|
|
1163
|
+
...existing.cycle,
|
|
1164
|
+
id: `cycle-${crypto.randomUUID()}`,
|
|
1165
|
+
targetId: undefined,
|
|
1166
|
+
taskProfileId: undefined,
|
|
1167
|
+
activeCandidateId: undefined,
|
|
1168
|
+
previousCandidateId: undefined,
|
|
1169
|
+
proposedCandidateId: undefined,
|
|
1170
|
+
stage: startedStage(kind),
|
|
1171
|
+
verifiedSwapCount: 0,
|
|
1172
|
+
baselineEvidenceSufficient: false,
|
|
1173
|
+
observabilityReady: false,
|
|
1174
|
+
policy: undefined,
|
|
1175
|
+
}
|
|
1176
|
+
: { ...existing.cycle, stage: startedStage(kind) },
|
|
804
1177
|
activeIntent: intent,
|
|
1178
|
+
...(kind === "establish_evals" ? {
|
|
1179
|
+
candidates: undefined,
|
|
1180
|
+
assessments: undefined,
|
|
1181
|
+
experimentManifest: undefined,
|
|
1182
|
+
observability: undefined,
|
|
1183
|
+
lastResult: undefined,
|
|
1184
|
+
} : {}),
|
|
805
1185
|
automation: {
|
|
806
1186
|
...existing.automation,
|
|
807
1187
|
...(existing.automation.pendingIntentKind === kind
|
|
@@ -889,6 +1269,16 @@ export class OpenMeritCoordinator {
|
|
|
889
1269
|
signal: "scheduled_tick",
|
|
890
1270
|
reason: "Wake the harness so OpenMerit can evaluate the user-confirmed check policy.",
|
|
891
1271
|
});
|
|
1272
|
+
if (!receipt.accepted) {
|
|
1273
|
+
return {
|
|
1274
|
+
plan: {
|
|
1275
|
+
...plan,
|
|
1276
|
+
scheduling: "external_scheduler_required",
|
|
1277
|
+
gaps: [...plan.gaps, `Harness wakeup provisioning failed: ${receipt.reason ?? "unknown reason"}`],
|
|
1278
|
+
},
|
|
1279
|
+
receipt,
|
|
1280
|
+
};
|
|
1281
|
+
}
|
|
892
1282
|
return { plan, receipt };
|
|
893
1283
|
}
|
|
894
1284
|
async acceptResult(result) {
|
|
@@ -903,6 +1293,17 @@ export class OpenMeritCoordinator {
|
|
|
903
1293
|
if (interpretation.config)
|
|
904
1294
|
await this.store.writeConfig(interpretation.config);
|
|
905
1295
|
const now = new Date().toISOString();
|
|
1296
|
+
const evaluationSpend = result.status === "succeeded" && state.activeIntent.kind === "run_challenger_trials"
|
|
1297
|
+
? result.outputs.assessments.reduce((sum, assessment) => sum + assessment.metrics
|
|
1298
|
+
.filter((metric) => metric.metricId === "total_task_cost" && metric.state === "measured")
|
|
1299
|
+
.reduce((metricSum, metric) => metricSum + (metric.value ?? 0), 0), state.automation.evaluationSpend ?? 0)
|
|
1300
|
+
: state.automation.evaluationSpend;
|
|
1301
|
+
if (config.policy && evaluationSpend !== undefined &&
|
|
1302
|
+
evaluationSpend > config.policy.evaluationBudget.maximumSpend) {
|
|
1303
|
+
throw new Error(`OpenMerit rejected challenger results: evaluation spend ${evaluationSpend.toFixed(6)} ` +
|
|
1304
|
+
`${config.policy.evaluationBudget.currency} exceeds the approved limit ` +
|
|
1305
|
+
`${config.policy.evaluationBudget.maximumSpend.toFixed(6)}.`);
|
|
1306
|
+
}
|
|
906
1307
|
let observability = state.observability;
|
|
907
1308
|
if (result.status === "succeeded" && state.activeIntent.kind === "establish_evals") {
|
|
908
1309
|
observability = result.outputs.observability;
|
|
@@ -913,6 +1314,7 @@ export class OpenMeritCoordinator {
|
|
|
913
1314
|
const output = result.outputs;
|
|
914
1315
|
const requiredMetricIds = profile.objectives.filter((item) => item.required).map((item) => item.metricId);
|
|
915
1316
|
observability = {
|
|
1317
|
+
targetId: profile.applicationTarget.id,
|
|
916
1318
|
status: "ready",
|
|
917
1319
|
requiredMetricIds,
|
|
918
1320
|
coveredMetricIds: output.coveredMetricIds,
|
|
@@ -930,9 +1332,31 @@ export class OpenMeritCoordinator {
|
|
|
930
1332
|
},
|
|
931
1333
|
activeIntent: undefined,
|
|
932
1334
|
lastResult: result,
|
|
1335
|
+
...(result.status === "succeeded" && state.activeIntent.kind === "establish_evals"
|
|
1336
|
+
? { candidates: [result.outputs.incumbent] }
|
|
1337
|
+
: {}),
|
|
1338
|
+
...(result.status === "succeeded" && state.activeIntent.kind === "discover_candidates"
|
|
1339
|
+
? { candidates: result.outputs.candidates }
|
|
1340
|
+
: {}),
|
|
1341
|
+
...(result.status === "succeeded" && state.activeIntent.kind === "run_challenger_trials"
|
|
1342
|
+
? {
|
|
1343
|
+
assessments: result.outputs.assessments,
|
|
1344
|
+
experimentManifest: result.outputs.experimentManifest,
|
|
1345
|
+
}
|
|
1346
|
+
: {}),
|
|
933
1347
|
...(observability ? { observability } : {}),
|
|
934
1348
|
automation: {
|
|
935
1349
|
...state.automation,
|
|
1350
|
+
...(result.status === "succeeded" && ["establish_evals", "instrument_observability"].includes(state.activeIntent.kind) &&
|
|
1351
|
+
interpretation.nextStage === "collecting_baseline"
|
|
1352
|
+
? {
|
|
1353
|
+
baselineStartedAt: now,
|
|
1354
|
+
lastAssessmentAt: undefined,
|
|
1355
|
+
lastAssessmentTaskCount: state.automation.observedCompletedTasks,
|
|
1356
|
+
baselineMetricWindows: [],
|
|
1357
|
+
}
|
|
1358
|
+
: {}),
|
|
1359
|
+
...(evaluationSpend !== undefined ? { evaluationSpend } : {}),
|
|
936
1360
|
...(state.activeIntent.kind === "run_assessment"
|
|
937
1361
|
? { lastAssessmentTaskCount: state.automation.observedCompletedTasks, lastAssessmentAt: now }
|
|
938
1362
|
: {}),
|
|
@@ -984,6 +1408,7 @@ export class OpenMeritCoordinator {
|
|
|
984
1408
|
cycleId: nextState.cycle.id,
|
|
985
1409
|
intentId: result.intentId,
|
|
986
1410
|
data: {
|
|
1411
|
+
targetId: observability.targetId,
|
|
987
1412
|
status: observability.status,
|
|
988
1413
|
requiredMetricIds: observability.requiredMetricIds,
|
|
989
1414
|
coveredMetricIds: observability.coveredMetricIds,
|
|
@@ -1016,21 +1441,54 @@ export class OpenMeritCoordinator {
|
|
|
1016
1441
|
}
|
|
1017
1442
|
async recordTaskObservation(details = {}) {
|
|
1018
1443
|
const now = new Date().toISOString();
|
|
1444
|
+
const state = await this.store.readState();
|
|
1445
|
+
if (!state?.cycle.targetId)
|
|
1446
|
+
throw new Error("A confirmed application LLM target is required.");
|
|
1019
1447
|
const result = await this.recordAutomationSignal({
|
|
1020
1448
|
protocolVersion: PROTOCOL_VERSION,
|
|
1021
1449
|
id: `legacy-task-${crypto.randomUUID()}`,
|
|
1022
1450
|
type: "product_task_completed",
|
|
1023
1451
|
occurredAt: now,
|
|
1452
|
+
targetId: state.cycle.targetId,
|
|
1024
1453
|
modelId: typeof details.modelId === "string" ? details.modelId : "unknown",
|
|
1025
|
-
contextTokens: typeof details.contextTokens === "number" || details.contextTokens === null
|
|
1026
|
-
? details.contextTokens
|
|
1027
|
-
: undefined,
|
|
1028
1454
|
});
|
|
1029
1455
|
return {
|
|
1030
1456
|
state: result.state,
|
|
1031
|
-
assessmentDue: result.
|
|
1457
|
+
assessmentDue: result.state.automation.lastAssessmentAt === now,
|
|
1032
1458
|
};
|
|
1033
1459
|
}
|
|
1460
|
+
async assessBaseline(now = new Date().toISOString()) {
|
|
1461
|
+
const state = await this.store.readState();
|
|
1462
|
+
if (!state || state.cycle.stage !== "collecting_baseline" || state.activeIntent) {
|
|
1463
|
+
throw new Error("Baseline assessment requires an idle collecting-baseline cycle.");
|
|
1464
|
+
}
|
|
1465
|
+
const config = await this.store.readConfig();
|
|
1466
|
+
const profile = config?.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
1467
|
+
const errors = baselineAssessmentErrors(state, profile, state.automation.baselineMetricWindows ?? []);
|
|
1468
|
+
const sufficient = errors.length === 0;
|
|
1469
|
+
const decision = sufficient
|
|
1470
|
+
? { action: "issue_intent", intentKind: "discover_candidates", reason: "Required baseline metric windows meet the confirmed sample thresholds; discover candidates." }
|
|
1471
|
+
: { action: "observe", reason: `Baseline evidence remains insufficient: ${errors.join("; ")}.` };
|
|
1472
|
+
const nextState = {
|
|
1473
|
+
...state,
|
|
1474
|
+
cycle: { ...state.cycle, baselineEvidenceSufficient: sufficient, stage: sufficient ? "baseline_ready" : "collecting_baseline" },
|
|
1475
|
+
automation: {
|
|
1476
|
+
...state.automation,
|
|
1477
|
+
lastAssessmentAt: now,
|
|
1478
|
+
lastAssessmentTaskCount: state.automation.observedCompletedTasks,
|
|
1479
|
+
pendingIntentKind: decision.action === "issue_intent" ? decision.intentKind : undefined,
|
|
1480
|
+
pendingIntentReason: decision.action === "issue_intent" ? decision.reason : undefined,
|
|
1481
|
+
},
|
|
1482
|
+
updatedAt: now,
|
|
1483
|
+
};
|
|
1484
|
+
await this.store.writeState(nextState);
|
|
1485
|
+
await this.store.appendEvent({
|
|
1486
|
+
schemaVersion: 1, id: `event-${crypto.randomUUID()}`, type: "baseline_sufficiency_checked",
|
|
1487
|
+
occurredAt: now, cycleId: state.cycle.id,
|
|
1488
|
+
data: { sufficient, errors, metricIds: (state.automation.baselineMetricWindows ?? []).map((metric) => metric.metricId), source: "manual" },
|
|
1489
|
+
});
|
|
1490
|
+
return { state: nextState, errors, decision };
|
|
1491
|
+
}
|
|
1034
1492
|
async recordAutomationSignal(signal) {
|
|
1035
1493
|
const state = await this.store.readState();
|
|
1036
1494
|
if (!state)
|
|
@@ -1039,6 +1497,14 @@ export class OpenMeritCoordinator {
|
|
|
1039
1497
|
throw new Error("Automation signal protocol version does not match.");
|
|
1040
1498
|
if (!signal.id.trim())
|
|
1041
1499
|
throw new Error("Automation signal ID is required.");
|
|
1500
|
+
if (!state.cycle.targetId || signal.targetId !== state.cycle.targetId) {
|
|
1501
|
+
throw new Error("Automation signal target does not match the confirmed application LLM target.");
|
|
1502
|
+
}
|
|
1503
|
+
if (signal.type === "metric_window_available" || signal.type === "verification_window_completed") {
|
|
1504
|
+
const mismatched = (signal.metrics ?? []).find((metric) => metric.targetId !== state.cycle.targetId);
|
|
1505
|
+
if (mismatched)
|
|
1506
|
+
throw new Error(`Metric ${mismatched.metricId} target does not match the confirmed application LLM target.`);
|
|
1507
|
+
}
|
|
1042
1508
|
const processed = state.automation.processedSignalIds ?? [];
|
|
1043
1509
|
if (processed.includes(signal.id)) {
|
|
1044
1510
|
await this.store.appendEvent({
|
|
@@ -1053,12 +1519,31 @@ export class OpenMeritCoordinator {
|
|
|
1053
1519
|
}
|
|
1054
1520
|
const completedTasks = state.automation.observedCompletedTasks +
|
|
1055
1521
|
(signal.type === "product_task_completed" ? 1 : 0);
|
|
1522
|
+
const baselineMetricWindows = state.cycle.stage === "collecting_baseline"
|
|
1523
|
+
? updatedBaselineWindows(state.automation.baselineMetricWindows, signal)
|
|
1524
|
+
: state.automation.baselineMetricWindows ?? [];
|
|
1056
1525
|
const decisionState = {
|
|
1057
1526
|
...state,
|
|
1058
|
-
automation: { ...state.automation, observedCompletedTasks: completedTasks },
|
|
1527
|
+
automation: { ...state.automation, observedCompletedTasks: completedTasks, baselineMetricWindows },
|
|
1059
1528
|
};
|
|
1060
|
-
const
|
|
1529
|
+
const due = dueDecision(decisionState, signal, signal.occurredAt);
|
|
1530
|
+
const baselineCheckDue = due.action === "issue_intent" && due.intentKind === "run_assessment";
|
|
1531
|
+
const config = baselineCheckDue ? await this.store.readConfig() : undefined;
|
|
1532
|
+
const profile = config?.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
1533
|
+
const baselineErrors = baselineCheckDue ? baselineAssessmentErrors(state, profile, baselineMetricWindows) : [];
|
|
1534
|
+
const baselineReady = baselineCheckDue && baselineErrors.length === 0;
|
|
1535
|
+
const decision = baselineCheckDue
|
|
1536
|
+
? baselineReady
|
|
1537
|
+
? { action: "issue_intent", intentKind: "discover_candidates", reason: "Required baseline metric windows meet the confirmed sample thresholds; discover candidates." }
|
|
1538
|
+
: { action: "observe", reason: `Baseline evidence remains insufficient: ${baselineErrors.join("; ")}.` }
|
|
1539
|
+
: due;
|
|
1061
1540
|
const latestMetricValues = { ...(state.automation.latestMetricValues ?? {}) };
|
|
1541
|
+
if (signal.type === "product_task_completed" && signal.telemetry) {
|
|
1542
|
+
latestMetricValues.input_token_consumption = signal.telemetry.inputTokens;
|
|
1543
|
+
latestMetricValues.output_token_consumption = signal.telemetry.outputTokens;
|
|
1544
|
+
latestMetricValues.total_task_cost = signal.telemetry.cost;
|
|
1545
|
+
latestMetricValues.end_to_end_task_latency = signal.telemetry.durationMs;
|
|
1546
|
+
}
|
|
1062
1547
|
if (signal.type === "metric_window_available" || signal.type === "verification_window_completed") {
|
|
1063
1548
|
for (const metric of signal.metrics ?? []) {
|
|
1064
1549
|
if (metric.state === "measured" && metric.value !== undefined)
|
|
@@ -1067,10 +1552,19 @@ export class OpenMeritCoordinator {
|
|
|
1067
1552
|
}
|
|
1068
1553
|
const nextState = {
|
|
1069
1554
|
...decisionState,
|
|
1555
|
+
cycle: baselineCheckDue
|
|
1556
|
+
? { ...decisionState.cycle, baselineEvidenceSufficient: baselineReady, stage: baselineReady ? "baseline_ready" : "collecting_baseline" }
|
|
1557
|
+
: decisionState.cycle,
|
|
1070
1558
|
automation: {
|
|
1071
1559
|
...decisionState.automation,
|
|
1072
1560
|
processedSignalIds: [...processed, signal.id].slice(-MAX_PROCESSED_SIGNAL_IDS),
|
|
1073
1561
|
latestMetricValues,
|
|
1562
|
+
...(baselineCheckDue ? {
|
|
1563
|
+
lastAssessmentAt: signal.occurredAt,
|
|
1564
|
+
lastAssessmentTaskCount: completedTasks,
|
|
1565
|
+
pendingIntentKind: decision.action === "issue_intent" ? decision.intentKind : undefined,
|
|
1566
|
+
pendingIntentReason: decision.action === "issue_intent" ? decision.reason : undefined,
|
|
1567
|
+
} : {}),
|
|
1074
1568
|
...(signal.type === "model_catalog_changed" ? {
|
|
1075
1569
|
modelCatalogFingerprint: signal.fingerprint,
|
|
1076
1570
|
...(state.automation.modelCatalogFingerprint && state.automation.modelCatalogFingerprint !== signal.fingerprint
|
|
@@ -1099,6 +1593,21 @@ export class OpenMeritCoordinator {
|
|
|
1099
1593
|
...auditSignalData(signal),
|
|
1100
1594
|
},
|
|
1101
1595
|
});
|
|
1596
|
+
if (baselineCheckDue) {
|
|
1597
|
+
await this.store.appendEvent({
|
|
1598
|
+
schemaVersion: 1,
|
|
1599
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1600
|
+
type: "baseline_sufficiency_checked",
|
|
1601
|
+
occurredAt: signal.occurredAt,
|
|
1602
|
+
cycleId: state.cycle.id,
|
|
1603
|
+
data: {
|
|
1604
|
+
sufficient: baselineReady,
|
|
1605
|
+
errors: baselineErrors,
|
|
1606
|
+
metricIds: baselineMetricWindows.map((metric) => metric.metricId),
|
|
1607
|
+
sourceSignalId: signal.id,
|
|
1608
|
+
},
|
|
1609
|
+
});
|
|
1610
|
+
}
|
|
1102
1611
|
if (decision.action === "issue_intent") {
|
|
1103
1612
|
await this.store.appendEvent({
|
|
1104
1613
|
schemaVersion: 1,
|
|
@@ -1115,6 +1624,8 @@ export class OpenMeritCoordinator {
|
|
|
1115
1624
|
const state = await this.store.readState();
|
|
1116
1625
|
if (!state)
|
|
1117
1626
|
throw new Error("OpenMerit project is not initialized.");
|
|
1627
|
+
if (!state.cycle.targetId)
|
|
1628
|
+
throw new Error("A confirmed application LLM target is required.");
|
|
1118
1629
|
const previousFingerprint = state.automation.modelCatalogFingerprint;
|
|
1119
1630
|
const changed = previousFingerprint !== undefined && previousFingerprint !== fingerprint;
|
|
1120
1631
|
if (previousFingerprint === fingerprint) {
|
|
@@ -1126,6 +1637,7 @@ export class OpenMeritCoordinator {
|
|
|
1126
1637
|
id: `legacy-catalog-${crypto.randomUUID()}`,
|
|
1127
1638
|
type: "model_catalog_changed",
|
|
1128
1639
|
occurredAt: now,
|
|
1640
|
+
targetId: state.cycle.targetId,
|
|
1129
1641
|
fingerprint,
|
|
1130
1642
|
});
|
|
1131
1643
|
return {
|