openmerit 0.1.4 → 0.1.6-preview.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/README.md +121 -386
- package/dist/core/src/index.d.ts +101 -0
- package/dist/core/src/index.js +1649 -0
- package/dist/core/src/store.d.ts +35 -0
- package/dist/core/src/store.js +102 -0
- package/dist/pi/src/index.d.ts +32 -0
- package/dist/pi/src/index.js +794 -0
- package/dist/pi/src/scheduler.d.ts +11 -0
- package/dist/pi/src/scheduler.js +137 -0
- package/dist/pi/src/wakeup.d.ts +2 -0
- package/dist/pi/src/wakeup.js +108 -0
- package/dist/protocol/src/index.d.ts +484 -0
- package/dist/protocol/src/index.js +47 -0
- package/dist/protocol/src/schemas.d.ts +576 -0
- package/dist/protocol/src/schemas.js +280 -0
- package/dist/terminal/public/app.js +297 -0
- package/dist/terminal/public/brands/anthropic.png +0 -0
- package/dist/terminal/public/brands/baai.png +0 -0
- package/dist/terminal/public/brands/baseten.png +0 -0
- package/dist/terminal/public/brands/cerebras.png +0 -0
- package/dist/terminal/public/brands/cohere.png +0 -0
- package/dist/terminal/public/brands/deepseek.ico +0 -0
- package/dist/terminal/public/brands/google.png +0 -0
- package/dist/terminal/public/brands/groq.ico +0 -0
- package/dist/terminal/public/brands/lm-studio.png +0 -0
- package/dist/terminal/public/brands/meta.ico +0 -0
- package/dist/terminal/public/brands/mistral.png +0 -0
- package/dist/terminal/public/brands/nomic.png +0 -0
- package/dist/terminal/public/brands/ollama.png +0 -0
- package/dist/terminal/public/brands/openai.png +0 -0
- package/dist/terminal/public/brands/openrouter.png +0 -0
- package/dist/terminal/public/brands/qwen.png +0 -0
- package/dist/terminal/public/brands/vllm.ico +0 -0
- package/dist/terminal/public/brands/vllm.png +0 -0
- package/dist/terminal/public/favicon.svg +1 -0
- package/dist/terminal/public/flow.css +1 -0
- package/dist/terminal/public/flow.js +770 -0
- package/dist/terminal/public/index.html +21 -0
- package/dist/terminal/public/styles.css +779 -0
- package/dist/terminal/src/activity-merge.mjs +64 -0
- package/dist/terminal/src/browser.mjs +29 -0
- package/dist/terminal/src/cli.mjs +60 -0
- package/dist/terminal/src/collect.mjs +311 -0
- package/dist/terminal/src/discovery.mjs +93 -0
- package/dist/terminal/src/hardware.mjs +57 -0
- package/dist/terminal/src/project-activity.mjs +156 -0
- package/dist/terminal/src/sample.mjs +171 -0
- package/dist/terminal/src/server.mjs +56 -0
- package/dist/terminal/src/services.mjs +62 -0
- package/dist/terminal/src/topology.mjs +30 -0
- package/docs/adapter-guide.md +189 -0
- package/docs/architecture.md +59 -0
- package/docs/automation.md +74 -0
- package/docs/budgets.md +37 -0
- package/docs/commands.md +85 -0
- package/docs/demo-backfill.md +29 -0
- package/docs/demo-fieldkit.md +47 -0
- package/docs/demo-placement.md +30 -0
- package/docs/demo-spam.md +15 -0
- package/docs/demo-support.md +42 -0
- package/docs/demo.md +57 -0
- package/docs/first-trial.md +60 -0
- package/docs/getting-started.md +65 -0
- package/docs/index.md +40 -0
- package/docs/inference-terminal.md +439 -0
- package/docs/lifecycle.md +30 -0
- package/docs/memo.md +126 -0
- package/docs/metrics-and-evidence.md +48 -0
- package/docs/operations.md +40 -0
- package/docs/pareto-spec.md +76 -0
- package/docs/pi-extension.md +54 -0
- package/docs/roadmap.md +28 -0
- package/docs/security.md +37 -0
- package/docs/site-artwork-linocut.md +23 -0
- package/docs/site-artwork-miniature-diverse.md +28 -0
- package/docs/site-artwork-miniature.md +26 -0
- package/docs/site-demo.md +177 -0
- package/docs/site-design.md +94 -0
- package/docs/site-documentation.md +83 -0
- package/docs/site-dynamic-og.md +35 -0
- package/docs/site-faq-maintenance.md +115 -0
- package/docs/site-hero-resolution.md +60 -0
- package/docs/site-illustration-sequences.md +227 -0
- package/docs/site-inference-terminal.md +203 -0
- package/docs/site-memo.md +39 -0
- package/docs/site-og-image.md +38 -0
- package/docs/site-og-workshop.md +21 -0
- package/docs/site-section-artwork.md +56 -0
- package/docs/site-skill-review.md +57 -0
- package/docs/site-terminal-preview.md +85 -0
- package/docs/testing.md +118 -0
- package/docs/troubleshooting.md +55 -0
- package/docs/ux-reference.md +32 -0
- package/package.json +74 -42
- package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
- package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
- package/benchmark/invoice_ocr/data/manifest.json +0 -97
- package/dist/benchmarks.js +0 -98
- package/dist/catalog.js +0 -61
- package/dist/cli.js +0 -188
- package/dist/daemon.js +0 -407
- package/dist/diagnostics.js +0 -227
- package/dist/frontier.js +0 -56
- package/dist/harness.js +0 -1
- package/dist/integrations.js +0 -19
- package/dist/invoice-eval.js +0 -33
- package/dist/invoice-score.js +0 -124
- package/dist/judge.js +0 -43
- package/dist/llm.js +0 -207
- package/dist/pi-config.js +0 -46
- package/dist/pi-trials.js +0 -373
- package/dist/policy.js +0 -185
- package/dist/providers.js +0 -1
- package/dist/recommend.js +0 -76
- package/dist/routes.js +0 -74
- package/dist/standalone.js +0 -224
- package/dist/store.js +0 -89
- package/dist/strategist.js +0 -68
- package/dist/task-input.js +0 -54
- package/dist/traces.js +0 -127
- package/dist/trials.js +0 -140
- package/dist/types.js +0 -2
- package/examples/invoice-prompt.txt +0 -19
- package/examples/task.example.json +0 -7
- package/extension/openmerit.ts +0 -947
- package/instructions/OPENMERIT.md +0 -63
- package/instructions/openmerit.policy.json +0 -37
- package/rules.md +0 -43
|
@@ -0,0 +1,1649 @@
|
|
|
1
|
+
import { METRIC_CATALOG, OPENMERIT_SCHEMA_DIALECT, PROTOCOL_VERSION, REQUIRED_CAPABILITIES, validateIntentOutputSchema, } from "../../protocol/src/index.js";
|
|
2
|
+
export { OPENMERIT_DIRECTORY, ProjectStore, } from "./store.js";
|
|
3
|
+
const outcomes = {
|
|
4
|
+
establish_evals: "Identify the real application LLM call or shared application route under evaluation and its exact incumbent application model. Infer the metrics required for that application's task from its outputs, tools, failure modes, and representative work. Before running setup, tell the user which metrics will be collected, why each matters, whether it uses repeated identical inputs, representative task instances, both, or load testing, the case and repetition counts, aggregation, any true performance threshold, and estimated evaluation cost; ask the user to confirm or edit that plan together with the target, budget, check cadence, scheduling mode, and automation permissions. Never turn a sample-count requirement into a metric-value constraint. Rate, distribution, reliability, percentile, variance, and average claims require multiple runs. Build runnable task-specific evaluation cases and application-call collectors for every confirmed metric. Execute a representative smoke run through the same application authentication and runtime path, run the grader, and inspect actual outputs. Preserve the commands and resulting artifacts as setup evidence. Mark any metric missing if its collector or grader cannot produce a value; do not declare coverage ready based only on code existing. Do not count a setup smoke run as a production baseline sample. Return the confirmed taskProfile, incumbent, and policy. During setup, research a preliminary model shortlist with current sourced price estimates and reputation; report an empty list if no credible leads are available. These are research priors only: do not treat them as measured task results or change model configuration. The coding harness model is not an evaluation target.",
|
|
5
|
+
instrument_observability: "Establish the missing observability required by the confirmed task profile. Exercise each collector and task-specific grader through a representative application run with the same authentication and runtime path, inspect actual metric outputs, and preserve verification artifacts. Report a metric covered only if that path produced its value; otherwise keep it missing. Do not count a setup smoke run as a production baseline sample.",
|
|
6
|
+
run_assessment: "Assess existing application-target baseline metric windows against every required metric and sample threshold. Do not synthesize samples, count setup smoke runs as production data, or wait for new data during this check. If a required window is absent or immature, promptly return baselineEvidenceSufficient false with measured, missing, and insufficient-evidence metrics explicitly and cite the records inspected.",
|
|
7
|
+
discover_candidates: "Refresh the preliminary model leads collected at setup and discover other credible candidates for the confirmed task profile using current capability, reputation, price, availability, and benchmark signals without exceeding the approved evaluation budget. Preliminary leads are research priors, not measured task evidence.",
|
|
8
|
+
run_challenger_trials: "Run a balanced, reproducible comparison of the frozen incumbent and challenger set under the approved budget. Use the same evaluation cases, application revision, prompt, tools, grader, and run count for every usable model; use a recorded seed to shuffle execution order. Attribute every run to the exact application model, preserve per-run quality, cost, token, and latency evidence, and return candidate assessments plus one durable experiment manifest. Never substitute catalogue prices for measured application-call cost.",
|
|
9
|
+
calculate_frontier: "Calculate the Pareto frontier using the OpenMerit conformance definition, explain every candidate classification, and return a FrontierSnapshot with calculation evidence.",
|
|
10
|
+
investigate_regression: "Investigate the observed regression and return its likely cause, affected metrics, and supporting evidence.",
|
|
11
|
+
apply_model_swap: "Apply the approved model change only to the bound application LLM target and return proof of the resulting application configuration. Never change the coding harness model and do not broaden the authorized change.",
|
|
12
|
+
verify_model_swap: "Verify the applied model against the approved post-swap evaluation window and report any material regression.",
|
|
13
|
+
rollback_model_swap: "Restore the previously verified model configuration and validate the rollback.",
|
|
14
|
+
};
|
|
15
|
+
export function createOpenMeritIntent(kind, options = {}) {
|
|
16
|
+
return {
|
|
17
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
18
|
+
id: options.id ?? `intent-${crypto.randomUUID()}`,
|
|
19
|
+
kind,
|
|
20
|
+
...(options.targetId ? { targetId: options.targetId } : {}),
|
|
21
|
+
taskProfileId: options.taskProfileId ?? "current-project",
|
|
22
|
+
authorizationPolicyId: options.authorizationPolicyId ?? "observe-and-evaluate-only",
|
|
23
|
+
requestedOutcome: outcomes[kind],
|
|
24
|
+
constraints: {
|
|
25
|
+
modelMutationAllowed: kind === "apply_model_swap" || kind === "rollback_model_swap",
|
|
26
|
+
harnessModelMutationAllowed: false,
|
|
27
|
+
mutationTarget: "application_llm_call",
|
|
28
|
+
requireVerifiedArtifacts: true,
|
|
29
|
+
distinguishSingleRunFromMultiRunClaims: true,
|
|
30
|
+
...(options.policy ? { evaluationBudget: options.policy.evaluationBudget } : {}),
|
|
31
|
+
...(options.policy ? {
|
|
32
|
+
evaluationSpend: options.evaluationSpend ?? 0,
|
|
33
|
+
remainingEvaluationBudget: Math.max(0, options.policy.evaluationBudget.maximumSpend - (options.evaluationSpend ?? 0)),
|
|
34
|
+
} : {}),
|
|
35
|
+
...(options.policy ? { checkPolicy: options.policy.checkPolicy } : {}),
|
|
36
|
+
...(options.additionalConstraints ?? {}),
|
|
37
|
+
},
|
|
38
|
+
requiredEvidence: kind === "establish_evals"
|
|
39
|
+
? [
|
|
40
|
+
{ id: "task-profile", description: "The inferred task profile and the user constraints it represents." },
|
|
41
|
+
{ id: "application-target", description: "The user-confirmed application LLM call or shared route, including stable call-site and route identifiers." },
|
|
42
|
+
{ id: "incumbent-model", description: "The exact application model configured before challenger trials." },
|
|
43
|
+
{ id: "evaluation-artifacts", description: "Runnable evaluation artifacts and verification results." },
|
|
44
|
+
{ id: "observability-coverage", description: "The observable metrics and explicit coverage gaps." },
|
|
45
|
+
{ id: "preliminary-model-research", description: "Sources for preliminary model pricing and reputation, or an explanation that no credible leads were found." },
|
|
46
|
+
]
|
|
47
|
+
: kind === "run_challenger_trials"
|
|
48
|
+
? [
|
|
49
|
+
{ id: "experiment-manifest", description: "A durable manifest containing the seed, frozen cases, application revision, exact model IDs, and per-run evidence." },
|
|
50
|
+
{ id: "assessment-results", description: "Comparable candidate assessments tied to the active task profile." },
|
|
51
|
+
]
|
|
52
|
+
: kind === "calculate_frontier"
|
|
53
|
+
? [
|
|
54
|
+
{ id: "frontier-snapshot", description: "A complete harness-calculated FrontierSnapshot." },
|
|
55
|
+
{ id: "frontier-calculation", description: "Durable evidence of the frontier calculation." },
|
|
56
|
+
]
|
|
57
|
+
: [
|
|
58
|
+
{ id: "assessment-results", description: "Assessment results tied to the active task profile." },
|
|
59
|
+
{ id: "metric-coverage", description: "Observed, derived, and insufficient-evidence metric states." },
|
|
60
|
+
],
|
|
61
|
+
requestedAt: options.requestedAt ?? new Date().toISOString(),
|
|
62
|
+
};
|
|
63
|
+
}
|
|
64
|
+
export function missingHarnessCapabilities(descriptor, kind) {
|
|
65
|
+
if (!descriptor.protocolVersions.includes(PROTOCOL_VERSION))
|
|
66
|
+
return REQUIRED_CAPABILITIES[kind];
|
|
67
|
+
const available = new Set(descriptor.capabilities);
|
|
68
|
+
return REQUIRED_CAPABILITIES[kind].filter((capability) => !available.has(capability));
|
|
69
|
+
}
|
|
70
|
+
export function validateHarnessDescriptor(descriptor) {
|
|
71
|
+
const errors = [];
|
|
72
|
+
if (!descriptor.id.trim())
|
|
73
|
+
errors.push("harness ID is required");
|
|
74
|
+
if (!descriptor.name.trim())
|
|
75
|
+
errors.push("harness name is required");
|
|
76
|
+
if (!descriptor.version.trim())
|
|
77
|
+
errors.push("harness version is required");
|
|
78
|
+
if (!descriptor.protocolVersions.includes(PROTOCOL_VERSION)) {
|
|
79
|
+
errors.push(`harness does not support OpenMerit protocol ${PROTOCOL_VERSION}`);
|
|
80
|
+
}
|
|
81
|
+
if (new Set(descriptor.capabilities).size !== descriptor.capabilities.length) {
|
|
82
|
+
errors.push("harness capabilities contain duplicates");
|
|
83
|
+
}
|
|
84
|
+
if (descriptor.executionModes.length === 0)
|
|
85
|
+
errors.push("at least one execution mode is required");
|
|
86
|
+
if (!descriptor.structuredOutput) {
|
|
87
|
+
errors.push("harness must declare structured-output behavior");
|
|
88
|
+
}
|
|
89
|
+
else if (descriptor.structuredOutput.schemaDialect !== OPENMERIT_SCHEMA_DIALECT) {
|
|
90
|
+
errors.push(`harness does not support OpenMerit schema dialect ${OPENMERIT_SCHEMA_DIALECT}`);
|
|
91
|
+
}
|
|
92
|
+
if (!descriptor.automation || descriptor.automation.supportedSignals.length === 0) {
|
|
93
|
+
errors.push("harness must declare at least one supported automation signal");
|
|
94
|
+
}
|
|
95
|
+
else if (descriptor.automation.persistentScheduling && descriptor.automation.wakeupProvisioning === "none") {
|
|
96
|
+
errors.push("persistent scheduling requires a wakeup provisioning mechanism");
|
|
97
|
+
}
|
|
98
|
+
return { valid: errors.length === 0, errors };
|
|
99
|
+
}
|
|
100
|
+
export async function dispatchToHarness(adapter, intent) {
|
|
101
|
+
const descriptorValidation = validateHarnessDescriptor(adapter.descriptor);
|
|
102
|
+
if (!descriptorValidation.valid) {
|
|
103
|
+
return { accepted: false, reason: descriptorValidation.errors.join("; ") };
|
|
104
|
+
}
|
|
105
|
+
const missing = missingHarnessCapabilities(adapter.descriptor, intent.kind);
|
|
106
|
+
if (missing.length) {
|
|
107
|
+
return { accepted: false, reason: `Harness lacks required capabilities: ${missing.join(", ")}` };
|
|
108
|
+
}
|
|
109
|
+
return adapter.dispatch(intent);
|
|
110
|
+
}
|
|
111
|
+
export function startedStage(kind) {
|
|
112
|
+
switch (kind) {
|
|
113
|
+
case "establish_evals":
|
|
114
|
+
case "instrument_observability": return "establishing_evidence";
|
|
115
|
+
case "run_assessment": return "collecting_baseline";
|
|
116
|
+
case "discover_candidates": return "discovering_candidates";
|
|
117
|
+
case "run_challenger_trials": return "running_challenger_trials";
|
|
118
|
+
case "calculate_frontier": return "calculating_frontier";
|
|
119
|
+
case "apply_model_swap": return "applying_swap";
|
|
120
|
+
case "verify_model_swap": return "verifying_swap";
|
|
121
|
+
case "rollback_model_swap": return "rolling_back";
|
|
122
|
+
case "investigate_regression": return "monitoring";
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
export function completedStage(kind, succeeded) {
|
|
126
|
+
if (!succeeded)
|
|
127
|
+
return kind === "verify_model_swap" ? "verification_failed" : "monitoring";
|
|
128
|
+
switch (kind) {
|
|
129
|
+
case "establish_evals":
|
|
130
|
+
case "instrument_observability": return "collecting_baseline";
|
|
131
|
+
case "run_assessment": return "baseline_ready";
|
|
132
|
+
case "discover_candidates": return "candidates_ready";
|
|
133
|
+
case "run_challenger_trials": return "challengers_ready";
|
|
134
|
+
case "calculate_frontier": return "frontier_ready";
|
|
135
|
+
case "apply_model_swap": return "swap_applied";
|
|
136
|
+
case "verify_model_swap": return "swap_verified";
|
|
137
|
+
case "rollback_model_swap":
|
|
138
|
+
case "investigate_regression": return "monitoring";
|
|
139
|
+
}
|
|
140
|
+
}
|
|
141
|
+
const knownMetricIds = new Set(METRIC_CATALOG.map((metric) => metric.id));
|
|
142
|
+
function isRecord(value) {
|
|
143
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
144
|
+
}
|
|
145
|
+
function validateSetupOutput(output, expectedIncumbentModelId) {
|
|
146
|
+
if (!isRecord(output) || !isRecord(output.taskProfile) || !isRecord(output.incumbent) || !isRecord(output.policy)) {
|
|
147
|
+
return ["setup output requires taskProfile, incumbent, and policy"];
|
|
148
|
+
}
|
|
149
|
+
const profile = output.taskProfile;
|
|
150
|
+
const policy = output.policy;
|
|
151
|
+
const errors = [];
|
|
152
|
+
if (expectedIncumbentModelId && output.incumbent.modelId !== expectedIncumbentModelId) {
|
|
153
|
+
errors.push(`incumbent model ${String(output.incumbent.modelId)} does not match the application source model ${expectedIncumbentModelId}`);
|
|
154
|
+
}
|
|
155
|
+
if (typeof profile.id !== "string" || typeof profile.name !== "string" ||
|
|
156
|
+
typeof profile.goal !== "string" || !Number.isInteger(profile.revision) ||
|
|
157
|
+
typeof profile.confirmedAt !== "string" || !Array.isArray(profile.objectives) ||
|
|
158
|
+
profile.objectives.length === 0) {
|
|
159
|
+
errors.push("taskProfile must be confirmed and contain at least one objective");
|
|
160
|
+
}
|
|
161
|
+
else {
|
|
162
|
+
for (const objective of profile.objectives) {
|
|
163
|
+
if (!isRecord(objective) || typeof objective.metricId !== "string" ||
|
|
164
|
+
!knownMetricIds.has(objective.metricId) || typeof objective.required !== "boolean" ||
|
|
165
|
+
!Number.isInteger(objective.minimumSamples) || objective.minimumSamples < 1 ||
|
|
166
|
+
typeof objective.tolerance !== "number") {
|
|
167
|
+
errors.push("taskProfile contains an invalid metric objective");
|
|
168
|
+
break;
|
|
169
|
+
}
|
|
170
|
+
const samplingPlan = objective.samplingPlan;
|
|
171
|
+
if (!isRecord(samplingPlan) || typeof samplingPlan.strategy !== "string" ||
|
|
172
|
+
!Number.isInteger(samplingPlan.representativeCaseCount) || samplingPlan.representativeCaseCount < 1 ||
|
|
173
|
+
!Number.isInteger(samplingPlan.repetitionsPerCase) || samplingPlan.repetitionsPerCase < 1 ||
|
|
174
|
+
typeof samplingPlan.aggregation !== "string" || typeof samplingPlan.rationale !== "string" ||
|
|
175
|
+
!samplingPlan.rationale.trim()) {
|
|
176
|
+
errors.push(`${objective.metricId} requires a complete sampling plan`);
|
|
177
|
+
continue;
|
|
178
|
+
}
|
|
179
|
+
const plannedSamples = samplingPlan.representativeCaseCount *
|
|
180
|
+
samplingPlan.repetitionsPerCase;
|
|
181
|
+
if (objective.minimumSamples < plannedSamples) {
|
|
182
|
+
errors.push(`${objective.metricId} minimumSamples must cover its sampling plan`);
|
|
183
|
+
}
|
|
184
|
+
const aggregateClaim = ["mean", "rate", "distribution", "percentile"].includes(samplingPlan.aggregation);
|
|
185
|
+
const metricDefinition = METRIC_CATALOG.find((metric) => metric.id === objective.metricId);
|
|
186
|
+
if ((aggregateClaim || metricDefinition?.requiresRepeatedRuns) && plannedSamples < 2) {
|
|
187
|
+
errors.push(`${objective.metricId} requires multiple planned runs for its ${String(samplingPlan.aggregation)} claim`);
|
|
188
|
+
}
|
|
189
|
+
if ((aggregateClaim || metricDefinition?.requiresRepeatedRuns) && samplingPlan.strategy === "single_run") {
|
|
190
|
+
errors.push(`${objective.metricId} cannot use single_run for a repeated-run claim`);
|
|
191
|
+
}
|
|
192
|
+
if (isRecord(objective.constraint) &&
|
|
193
|
+
["task_success", "structured_output_reliability", "hallucination_rate", "retry_rate", "recovery_ability", "human_intervention_rate"].includes(objective.metricId) &&
|
|
194
|
+
(typeof objective.constraint.value !== "number" || objective.constraint.value < 0 || objective.constraint.value > 1)) {
|
|
195
|
+
errors.push(`${objective.metricId} ratio constraint must be between 0 and 1`);
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
}
|
|
199
|
+
if (!isRecord(profile.applicationTarget) || profile.applicationTarget.kind !== "application_llm_call" ||
|
|
200
|
+
typeof profile.applicationTarget.id !== "string" || !profile.applicationTarget.id.trim() ||
|
|
201
|
+
typeof profile.applicationTarget.applicationId !== "string" || !profile.applicationTarget.applicationId.trim() ||
|
|
202
|
+
typeof profile.applicationTarget.routeKey !== "string" || !profile.applicationTarget.routeKey.trim() ||
|
|
203
|
+
typeof profile.applicationTarget.confirmedAt !== "string" ||
|
|
204
|
+
!Array.isArray(profile.applicationTarget.callSites) || profile.applicationTarget.callSites.length === 0 ||
|
|
205
|
+
profile.applicationTarget.callSites.some((callSite) => typeof callSite !== "string" || !callSite.trim())) {
|
|
206
|
+
errors.push("taskProfile requires a user-confirmed application LLM target with stable call sites and route key");
|
|
207
|
+
}
|
|
208
|
+
if (isRecord(profile.applicationTarget)) {
|
|
209
|
+
errors.push(...validateTargetId(output.incumbent.targetId, profile.applicationTarget.id, "incumbent"));
|
|
210
|
+
if (typeof output.incumbent.candidateId !== "string" || !output.incumbent.candidateId.trim() ||
|
|
211
|
+
typeof output.incumbent.modelId !== "string" || !output.incumbent.modelId.trim()) {
|
|
212
|
+
errors.push("incumbent requires stable candidateId and exact modelId");
|
|
213
|
+
}
|
|
214
|
+
}
|
|
215
|
+
const budget = policy.evaluationBudget;
|
|
216
|
+
if (policy.mode !== "supervised_graduation" || !isRecord(budget) ||
|
|
217
|
+
typeof budget.currency !== "string" || typeof budget.maximumSpend !== "number" ||
|
|
218
|
+
budget.maximumSpend < 0 || typeof policy.automaticSwapsEnabled !== "boolean" ||
|
|
219
|
+
!Number.isInteger(policy.confirmationRequiredUntilVerifiedSwaps) ||
|
|
220
|
+
policy.requirePostSwapVerification !== true || typeof policy.rollbackOnRegression !== "boolean") {
|
|
221
|
+
errors.push("policy must be a valid supervised-graduation policy with an evaluation budget");
|
|
222
|
+
}
|
|
223
|
+
const checkPolicy = policy.checkPolicy;
|
|
224
|
+
if (!isRecord(checkPolicy) || !isRecord(checkPolicy.baselineAssessment) ||
|
|
225
|
+
(typeof checkPolicy.baselineAssessment.afterCompletedTasks !== "number" &&
|
|
226
|
+
typeof checkPolicy.baselineAssessment.afterElapsedSeconds !== "number")) {
|
|
227
|
+
errors.push("policy requires a baseline assessment cadence");
|
|
228
|
+
}
|
|
229
|
+
if (!isRecord(checkPolicy) || !isRecord(checkPolicy.postSwapVerification) ||
|
|
230
|
+
(typeof checkPolicy.postSwapVerification.afterCompletedTasks !== "number" &&
|
|
231
|
+
typeof checkPolicy.postSwapVerification.afterElapsedSeconds !== "number")) {
|
|
232
|
+
errors.push("policy requires a post-swap verification cadence");
|
|
233
|
+
}
|
|
234
|
+
if (!isRecord(output.observability)) {
|
|
235
|
+
errors.push("setup output requires an observability coverage report");
|
|
236
|
+
}
|
|
237
|
+
else if (Array.isArray(profile.objectives)) {
|
|
238
|
+
errors.push(...validateObservabilityCoverage(profile, output.observability, false));
|
|
239
|
+
}
|
|
240
|
+
if (Array.isArray(output.preliminaryModelLeads) && isRecord(profile.applicationTarget)) {
|
|
241
|
+
const seen = new Set();
|
|
242
|
+
for (const lead of output.preliminaryModelLeads) {
|
|
243
|
+
if (!isRecord(lead))
|
|
244
|
+
continue;
|
|
245
|
+
errors.push(...validateTargetId(lead.targetId, profile.applicationTarget.id, `preliminary lead ${String(lead.modelId)}`));
|
|
246
|
+
const identity = `${String(lead.modelId)}@${String(lead.modelVersion ?? "")}`;
|
|
247
|
+
if (seen.has(identity))
|
|
248
|
+
errors.push(`duplicate preliminary model lead ${identity}`);
|
|
249
|
+
seen.add(identity);
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
return errors;
|
|
253
|
+
}
|
|
254
|
+
function validateMetricAggregation(metric, objective) {
|
|
255
|
+
const errors = [];
|
|
256
|
+
if (metric.metricId === "total_task_cost" && metric.state === "measured" &&
|
|
257
|
+
metric.scope === "sample_window" && metric.aggregation !== "sum") {
|
|
258
|
+
errors.push("total_task_cost sample windows must use sum aggregation because the value is used for evaluation-budget accounting");
|
|
259
|
+
}
|
|
260
|
+
if (objective?.samplingPlan && metric.state === "measured" &&
|
|
261
|
+
metric.aggregation !== objective.samplingPlan.aggregation &&
|
|
262
|
+
!(objective.metricId === "total_task_cost" && metric.aggregation === "sum")) {
|
|
263
|
+
errors.push(`${objective.metricId} aggregation ${metric.aggregation} does not match its sampling plan ${objective.samplingPlan.aggregation}`);
|
|
264
|
+
}
|
|
265
|
+
return errors;
|
|
266
|
+
}
|
|
267
|
+
function validateMetricWindow(metric, startedAt, completedAt) {
|
|
268
|
+
if (!metric.window)
|
|
269
|
+
return [];
|
|
270
|
+
const start = Date.parse(metric.window.startedAt);
|
|
271
|
+
const end = Date.parse(metric.window.endedAt);
|
|
272
|
+
const intentStart = Date.parse(startedAt);
|
|
273
|
+
const intentEnd = Date.parse(completedAt);
|
|
274
|
+
if (![start, end, intentStart, intentEnd].every(Number.isFinite) || end < start || start < intentStart || end > intentEnd) {
|
|
275
|
+
return [`${metric.metricId} window must be valid and contained within the intent execution window`];
|
|
276
|
+
}
|
|
277
|
+
return [];
|
|
278
|
+
}
|
|
279
|
+
export function validateObservabilityCoverage(profile, coverage, requireReady) {
|
|
280
|
+
const required = new Set(profile.objectives.filter((item) => item.required).map((item) => item.metricId));
|
|
281
|
+
const declaredRequired = new Set(coverage.requiredMetricIds);
|
|
282
|
+
const covered = new Set(coverage.coveredMetricIds);
|
|
283
|
+
const missing = new Set(coverage.missingMetricIds);
|
|
284
|
+
const errors = [];
|
|
285
|
+
if (coverage.targetId !== profile.applicationTarget.id) {
|
|
286
|
+
errors.push("observability target does not match the confirmed application LLM target");
|
|
287
|
+
}
|
|
288
|
+
for (const metricId of required) {
|
|
289
|
+
if (!declaredRequired.has(metricId))
|
|
290
|
+
errors.push(`observability omits required metric ${metricId}`);
|
|
291
|
+
if (!covered.has(metricId) && !missing.has(metricId))
|
|
292
|
+
errors.push(`observability does not classify required metric ${metricId}`);
|
|
293
|
+
}
|
|
294
|
+
for (const metricId of covered) {
|
|
295
|
+
if (missing.has(metricId))
|
|
296
|
+
errors.push(`observability marks ${metricId} as both covered and missing`);
|
|
297
|
+
}
|
|
298
|
+
const requiredMissing = [...required].filter((metricId) => !covered.has(metricId));
|
|
299
|
+
const expectedStatus = requiredMissing.length === 0 ? "ready" : "incomplete";
|
|
300
|
+
if (coverage.status !== expectedStatus)
|
|
301
|
+
errors.push(`observability status must be ${expectedStatus}`);
|
|
302
|
+
if (requireReady && requiredMissing.length) {
|
|
303
|
+
errors.push(`required observability remains missing: ${requiredMissing.join(", ")}`);
|
|
304
|
+
}
|
|
305
|
+
return errors;
|
|
306
|
+
}
|
|
307
|
+
function activeTarget(config) {
|
|
308
|
+
return config.taskProfiles.find((item) => item.id === config.activeTaskProfileId)?.applicationTarget;
|
|
309
|
+
}
|
|
310
|
+
function validateTargetId(actual, expected, subject) {
|
|
311
|
+
return actual === expected ? [] : [`${subject} target does not match the confirmed application LLM target`];
|
|
312
|
+
}
|
|
313
|
+
function validateMetricTargets(metrics, targetId, subject) {
|
|
314
|
+
return metrics.flatMap((metric) => validateTargetId(metric.targetId, targetId, `${subject} metric ${metric.metricId}`));
|
|
315
|
+
}
|
|
316
|
+
export function interpretIntentResult(intent, result, cycle, config) {
|
|
317
|
+
const errors = [];
|
|
318
|
+
if (result.protocolVersion !== PROTOCOL_VERSION)
|
|
319
|
+
errors.push("result protocol version does not match");
|
|
320
|
+
if (result.intentId !== intent.id)
|
|
321
|
+
errors.push("result intent ID does not match");
|
|
322
|
+
if (result.status === "succeeded" && result.evidence.length === 0)
|
|
323
|
+
errors.push("successful result requires evidence");
|
|
324
|
+
if (result.status !== "succeeded") {
|
|
325
|
+
return {
|
|
326
|
+
valid: errors.length === 0,
|
|
327
|
+
errors,
|
|
328
|
+
nextStage: completedStage(intent.kind, false),
|
|
329
|
+
};
|
|
330
|
+
}
|
|
331
|
+
const output = result.outputs;
|
|
332
|
+
errors.push(...validateIntentOutputSchema(intent.kind, output).map((error) => `result outputs ${error}`));
|
|
333
|
+
if (isRecord(output) && (intent.kind === "run_assessment" || intent.kind === "run_challenger_trials")) {
|
|
334
|
+
const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
335
|
+
const objectives = new Map((profile?.objectives ?? []).map((objective) => [objective.metricId, objective]));
|
|
336
|
+
const metricGroups = intent.kind === "run_assessment"
|
|
337
|
+
? [{ metrics: output.metrics, startedAt: undefined, completedAt: undefined }]
|
|
338
|
+
: (Array.isArray(output.assessments) ? output.assessments.map((assessment) => isRecord(assessment)
|
|
339
|
+
? { metrics: assessment.metrics, startedAt: assessment.startedAt, completedAt: assessment.completedAt }
|
|
340
|
+
: { metrics: undefined, startedAt: undefined, completedAt: undefined }) : []);
|
|
341
|
+
for (const group of metricGroups) {
|
|
342
|
+
const metrics = group.metrics;
|
|
343
|
+
if (!Array.isArray(metrics))
|
|
344
|
+
continue;
|
|
345
|
+
for (const metric of metrics) {
|
|
346
|
+
if (!isRecord(metric))
|
|
347
|
+
continue;
|
|
348
|
+
const typedMetric = metric;
|
|
349
|
+
errors.push(...validateMetricAggregation(typedMetric, objectives.get(typedMetric.metricId)));
|
|
350
|
+
if (group.startedAt && group.completedAt)
|
|
351
|
+
errors.push(...validateMetricWindow(typedMetric, String(group.startedAt), String(group.completedAt)));
|
|
352
|
+
}
|
|
353
|
+
}
|
|
354
|
+
}
|
|
355
|
+
const target = activeTarget(config);
|
|
356
|
+
if (intent.kind !== "establish_evals") {
|
|
357
|
+
if (!target)
|
|
358
|
+
errors.push("no confirmed application LLM target exists");
|
|
359
|
+
else
|
|
360
|
+
errors.push(...validateTargetId(intent.targetId, target.id, "intent"));
|
|
361
|
+
}
|
|
362
|
+
if (errors.length) {
|
|
363
|
+
return { valid: false, errors, nextStage: completedStage(intent.kind, false) };
|
|
364
|
+
}
|
|
365
|
+
let nextStage = completedStage(intent.kind, true);
|
|
366
|
+
let nextConfig;
|
|
367
|
+
let cyclePatch;
|
|
368
|
+
switch (intent.kind) {
|
|
369
|
+
case "establish_evals": {
|
|
370
|
+
errors.push(...validateSetupOutput(output, typeof intent.constraints.expectedIncumbentModelId === "string"
|
|
371
|
+
? intent.constraints.expectedIncumbentModelId : undefined));
|
|
372
|
+
if (!errors.length && isRecord(output)) {
|
|
373
|
+
const taskProfile = output.taskProfile;
|
|
374
|
+
const policy = output.policy;
|
|
375
|
+
const observability = output.observability;
|
|
376
|
+
const incumbent = output.incumbent;
|
|
377
|
+
const preliminaryModelLeads = (output.preliminaryModelLeads ?? []);
|
|
378
|
+
nextConfig = {
|
|
379
|
+
schemaVersion: 1,
|
|
380
|
+
activeTaskProfileId: taskProfile.id,
|
|
381
|
+
taskProfiles: [...config.taskProfiles.filter((item) => item.id !== taskProfile.id), taskProfile],
|
|
382
|
+
policy,
|
|
383
|
+
preliminaryModelLeads,
|
|
384
|
+
};
|
|
385
|
+
cyclePatch = {
|
|
386
|
+
targetId: taskProfile.applicationTarget.id,
|
|
387
|
+
taskProfileId: taskProfile.id,
|
|
388
|
+
activeCandidateId: incumbent.candidateId,
|
|
389
|
+
policy,
|
|
390
|
+
baselineEvidenceSufficient: false,
|
|
391
|
+
observabilityReady: observability.status === "ready",
|
|
392
|
+
};
|
|
393
|
+
if (observability.status !== "ready")
|
|
394
|
+
nextStage = "establishing_evidence";
|
|
395
|
+
}
|
|
396
|
+
break;
|
|
397
|
+
}
|
|
398
|
+
case "instrument_observability":
|
|
399
|
+
if (!isRecord(output) || !Array.isArray(output.coveredMetricIds) || !Array.isArray(output.missingMetricIds)) {
|
|
400
|
+
errors.push("observability result requires coveredMetricIds and missingMetricIds");
|
|
401
|
+
}
|
|
402
|
+
else {
|
|
403
|
+
const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
404
|
+
if (!profile)
|
|
405
|
+
errors.push("no confirmed task profile exists for observability verification");
|
|
406
|
+
else {
|
|
407
|
+
errors.push(...validateTargetId(output.targetId, profile.applicationTarget.id, "observability result"));
|
|
408
|
+
const coverage = {
|
|
409
|
+
targetId: profile.applicationTarget.id,
|
|
410
|
+
status: output.missingMetricIds.length ? "incomplete" : "ready",
|
|
411
|
+
requiredMetricIds: profile.objectives.filter((item) => item.required).map((item) => item.metricId),
|
|
412
|
+
coveredMetricIds: output.coveredMetricIds,
|
|
413
|
+
missingMetricIds: output.missingMetricIds,
|
|
414
|
+
checkedAt: result.completedAt,
|
|
415
|
+
};
|
|
416
|
+
errors.push(...validateObservabilityCoverage(profile, coverage, true));
|
|
417
|
+
cyclePatch = { observabilityReady: errors.length === 0 };
|
|
418
|
+
}
|
|
419
|
+
}
|
|
420
|
+
break;
|
|
421
|
+
case "run_assessment":
|
|
422
|
+
if (!isRecord(output) || typeof output.baselineEvidenceSufficient !== "boolean" || !Array.isArray(output.metrics)) {
|
|
423
|
+
errors.push("assessment result requires baselineEvidenceSufficient");
|
|
424
|
+
}
|
|
425
|
+
else {
|
|
426
|
+
if (target) {
|
|
427
|
+
errors.push(...validateTargetId(output.targetId, target.id, "baseline assessment"));
|
|
428
|
+
errors.push(...validateMetricTargets(output.metrics, target.id, "baseline assessment"));
|
|
429
|
+
const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
430
|
+
if (output.baselineEvidenceSufficient) {
|
|
431
|
+
errors.push(...baselineReadinessErrors(profile, output.metrics));
|
|
432
|
+
}
|
|
433
|
+
}
|
|
434
|
+
cyclePatch = { baselineEvidenceSufficient: output.baselineEvidenceSufficient };
|
|
435
|
+
if (!output.baselineEvidenceSufficient)
|
|
436
|
+
nextStage = "collecting_baseline";
|
|
437
|
+
}
|
|
438
|
+
break;
|
|
439
|
+
case "discover_candidates":
|
|
440
|
+
if (!isRecord(output) || !Array.isArray(output.candidates))
|
|
441
|
+
errors.push("candidate discovery requires candidates");
|
|
442
|
+
else if (target) {
|
|
443
|
+
errors.push(...validateTargetId(output.targetId, target.id, "candidate discovery"));
|
|
444
|
+
const candidateIds = new Set();
|
|
445
|
+
for (const candidate of output.candidates) {
|
|
446
|
+
if (isRecord(candidate)) {
|
|
447
|
+
errors.push(...validateTargetId(candidate.targetId, target.id, `candidate ${String(candidate.candidateId)}`));
|
|
448
|
+
if (typeof candidate.candidateId === "string") {
|
|
449
|
+
if (candidateIds.has(candidate.candidateId))
|
|
450
|
+
errors.push(`duplicate candidate ${candidate.candidateId}`);
|
|
451
|
+
candidateIds.add(candidate.candidateId);
|
|
452
|
+
}
|
|
453
|
+
}
|
|
454
|
+
}
|
|
455
|
+
if (cycle.activeCandidateId && !candidateIds.has(cycle.activeCandidateId)) {
|
|
456
|
+
errors.push("candidate discovery must include the incumbent application model");
|
|
457
|
+
}
|
|
458
|
+
const maximum = config.policy?.evaluationBudget.maximumCandidateCount;
|
|
459
|
+
if (maximum !== undefined && output.candidates.length > maximum) {
|
|
460
|
+
errors.push(`candidate discovery exceeds maximumCandidateCount ${maximum}`);
|
|
461
|
+
}
|
|
462
|
+
}
|
|
463
|
+
break;
|
|
464
|
+
case "run_challenger_trials":
|
|
465
|
+
if (!isRecord(output) || !Array.isArray(output.assessments) || !isRecord(output.experimentManifest)) {
|
|
466
|
+
errors.push("challenger trials require assessments and an experimentManifest");
|
|
467
|
+
}
|
|
468
|
+
else {
|
|
469
|
+
if (target)
|
|
470
|
+
errors.push(...validateTargetId(output.targetId, target.id, "challenger trials"));
|
|
471
|
+
const expectedCandidateIds = Array.isArray(intent.constraints.candidates)
|
|
472
|
+
? new Set(intent.constraints.candidates.flatMap((candidate) => isRecord(candidate) && typeof candidate.candidateId === "string" ? [candidate.candidateId] : []))
|
|
473
|
+
: undefined;
|
|
474
|
+
const assessedCandidateIds = new Set();
|
|
475
|
+
const productRevisions = new Set();
|
|
476
|
+
for (const assessment of output.assessments) {
|
|
477
|
+
assessedCandidateIds.add(assessment.candidateId);
|
|
478
|
+
productRevisions.add(assessment.productRevision);
|
|
479
|
+
if (target) {
|
|
480
|
+
errors.push(...validateTargetId(assessment.targetId, target.id, `challenger assessment ${assessment.id}`));
|
|
481
|
+
errors.push(...validateMetricTargets(assessment.metrics, target.id, `challenger assessment ${assessment.id}`));
|
|
482
|
+
}
|
|
483
|
+
const costs = assessment.metrics.filter((metric) => metric.metricId === "total_task_cost" && metric.state === "measured" &&
|
|
484
|
+
metric.value !== undefined && metric.value >= 0 && metric.evidence.length > 0);
|
|
485
|
+
if (costs.length !== 1) {
|
|
486
|
+
errors.push(`challenger assessment ${assessment.id} requires exactly one measured total_task_cost with evidence`);
|
|
487
|
+
}
|
|
488
|
+
}
|
|
489
|
+
if (expectedCandidateIds &&
|
|
490
|
+
(expectedCandidateIds.size !== assessedCandidateIds.size ||
|
|
491
|
+
[...expectedCandidateIds].some((candidateId) => !assessedCandidateIds.has(candidateId)))) {
|
|
492
|
+
errors.push("challenger assessments must exactly cover the frozen candidate set");
|
|
493
|
+
}
|
|
494
|
+
if (productRevisions.size !== 1)
|
|
495
|
+
errors.push("challenger assessments must use one application product revision");
|
|
496
|
+
}
|
|
497
|
+
break;
|
|
498
|
+
case "calculate_frontier": {
|
|
499
|
+
if (!isRecord(output) || !Array.isArray(output.assessments) || !isRecord(output.frontierSnapshot)) {
|
|
500
|
+
errors.push("frontier result requires assessments and frontierSnapshot");
|
|
501
|
+
break;
|
|
502
|
+
}
|
|
503
|
+
const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
504
|
+
if (!profile) {
|
|
505
|
+
errors.push("no confirmed task profile exists for frontier verification");
|
|
506
|
+
break;
|
|
507
|
+
}
|
|
508
|
+
errors.push(...validateTargetId(output.targetId, profile.applicationTarget.id, "frontier result"));
|
|
509
|
+
const expectedAssessmentIds = Array.isArray(intent.constraints.assessments)
|
|
510
|
+
? new Set(intent.constraints.assessments.flatMap((assessment) => isRecord(assessment) && typeof assessment.id === "string" ? [assessment.id] : []))
|
|
511
|
+
: undefined;
|
|
512
|
+
const suppliedAssessmentIds = new Set(output.assessments.map((assessment) => assessment.id));
|
|
513
|
+
if (expectedAssessmentIds &&
|
|
514
|
+
(expectedAssessmentIds.size !== suppliedAssessmentIds.size ||
|
|
515
|
+
[...expectedAssessmentIds].some((assessmentId) => !suppliedAssessmentIds.has(assessmentId)))) {
|
|
516
|
+
errors.push("frontier calculation must use exactly the accepted challenger assessments");
|
|
517
|
+
}
|
|
518
|
+
const verification = verifyFrontierSnapshot(profile, output.assessments, output.frontierSnapshot);
|
|
519
|
+
errors.push(...verification.errors);
|
|
520
|
+
if (output.recommendation !== undefined) {
|
|
521
|
+
if (!isRecord(output.recommendation) || typeof output.recommendation.selectedCandidateId !== "string") {
|
|
522
|
+
errors.push("frontier recommendation is invalid");
|
|
523
|
+
}
|
|
524
|
+
else if (!output.frontierSnapshot.frontierCandidateIds.includes(output.recommendation.selectedCandidateId)) {
|
|
525
|
+
errors.push("recommended candidate is not on the verified frontier");
|
|
526
|
+
}
|
|
527
|
+
else if (output.recommendation.targetId !== profile.applicationTarget.id) {
|
|
528
|
+
errors.push("frontier recommendation target does not match the confirmed application LLM target");
|
|
529
|
+
}
|
|
530
|
+
else if (cycle.activeCandidateId && output.recommendation.currentCandidateId !== cycle.activeCandidateId) {
|
|
531
|
+
errors.push("frontier recommendation current candidate does not match the incumbent application model");
|
|
532
|
+
}
|
|
533
|
+
else {
|
|
534
|
+
cyclePatch = { proposedCandidateId: output.recommendation.selectedCandidateId };
|
|
535
|
+
}
|
|
536
|
+
}
|
|
537
|
+
break;
|
|
538
|
+
}
|
|
539
|
+
case "investigate_regression":
|
|
540
|
+
if (!isRecord(output) || typeof output.regressionDetected !== "boolean" || !Array.isArray(output.affectedMetricIds)) {
|
|
541
|
+
errors.push("regression result requires regressionDetected and affectedMetricIds");
|
|
542
|
+
}
|
|
543
|
+
else if (target)
|
|
544
|
+
errors.push(...validateTargetId(output.targetId, target.id, "regression investigation"));
|
|
545
|
+
break;
|
|
546
|
+
case "apply_model_swap":
|
|
547
|
+
if (!isRecord(output) || typeof output.appliedCandidateId !== "string") {
|
|
548
|
+
errors.push("swap result requires appliedCandidateId");
|
|
549
|
+
}
|
|
550
|
+
else if (target && output.targetId !== target.id) {
|
|
551
|
+
errors.push("swap target does not match the confirmed application LLM target");
|
|
552
|
+
}
|
|
553
|
+
else if (cycle.proposedCandidateId && output.appliedCandidateId !== cycle.proposedCandidateId) {
|
|
554
|
+
errors.push("applied candidate does not match the authorized proposal");
|
|
555
|
+
}
|
|
556
|
+
else {
|
|
557
|
+
cyclePatch = {
|
|
558
|
+
previousCandidateId: cycle.activeCandidateId,
|
|
559
|
+
activeCandidateId: output.appliedCandidateId,
|
|
560
|
+
};
|
|
561
|
+
}
|
|
562
|
+
break;
|
|
563
|
+
case "verify_model_swap":
|
|
564
|
+
if (!isRecord(output) || typeof output.verified !== "boolean" || typeof output.regressionDetected !== "boolean") {
|
|
565
|
+
errors.push("swap verification requires verified and regressionDetected");
|
|
566
|
+
}
|
|
567
|
+
else if (target && output.targetId !== target.id) {
|
|
568
|
+
errors.push("swap verification target does not match the confirmed application LLM target");
|
|
569
|
+
}
|
|
570
|
+
else if (!output.verified || output.regressionDetected) {
|
|
571
|
+
nextStage = "verification_failed";
|
|
572
|
+
}
|
|
573
|
+
else {
|
|
574
|
+
cyclePatch = { verifiedSwapCount: cycle.verifiedSwapCount + 1 };
|
|
575
|
+
}
|
|
576
|
+
break;
|
|
577
|
+
case "rollback_model_swap":
|
|
578
|
+
if (!isRecord(output) || typeof output.restoredCandidateId !== "string") {
|
|
579
|
+
errors.push("rollback result requires restoredCandidateId");
|
|
580
|
+
}
|
|
581
|
+
else if (target && output.targetId !== target.id) {
|
|
582
|
+
errors.push("rollback target does not match the confirmed application LLM target");
|
|
583
|
+
}
|
|
584
|
+
else if (cycle.previousCandidateId && output.restoredCandidateId !== cycle.previousCandidateId) {
|
|
585
|
+
errors.push("rollback did not restore the previous verified candidate");
|
|
586
|
+
}
|
|
587
|
+
else {
|
|
588
|
+
cyclePatch = { activeCandidateId: output.restoredCandidateId };
|
|
589
|
+
}
|
|
590
|
+
break;
|
|
591
|
+
}
|
|
592
|
+
return {
|
|
593
|
+
valid: errors.length === 0,
|
|
594
|
+
errors,
|
|
595
|
+
...(errors.length ? {} : { nextStage, config: nextConfig, cyclePatch }),
|
|
596
|
+
};
|
|
597
|
+
}
|
|
598
|
+
export function nextImprovementAction(state) {
|
|
599
|
+
switch (state.stage) {
|
|
600
|
+
case "unconfigured":
|
|
601
|
+
return { action: "issue_intent", intentKind: "establish_evals", reason: "Task profile, evaluations, and observability are not configured." };
|
|
602
|
+
case "establishing_evidence":
|
|
603
|
+
return state.observabilityReady
|
|
604
|
+
? { action: "observe", reason: "The harness is establishing evaluations and observability." }
|
|
605
|
+
: { action: "issue_intent", intentKind: "instrument_observability", reason: "Required metric instrumentation is incomplete." };
|
|
606
|
+
case "collecting_baseline":
|
|
607
|
+
return state.baselineEvidenceSufficient
|
|
608
|
+
? { action: "issue_intent", intentKind: "discover_candidates", reason: "The baseline is ready for candidate discovery." }
|
|
609
|
+
: { action: "observe", reason: "Continue collecting baseline production evidence." };
|
|
610
|
+
case "baseline_ready":
|
|
611
|
+
return { action: "issue_intent", intentKind: "discover_candidates", reason: "The baseline is ready for budgeted candidate discovery." };
|
|
612
|
+
case "discovering_candidates":
|
|
613
|
+
case "running_challenger_trials":
|
|
614
|
+
case "calculating_frontier":
|
|
615
|
+
case "applying_swap":
|
|
616
|
+
case "verifying_swap":
|
|
617
|
+
case "rolling_back":
|
|
618
|
+
return { action: "observe", reason: "Harness work is in progress." };
|
|
619
|
+
case "candidates_ready":
|
|
620
|
+
return { action: "issue_intent", intentKind: "run_challenger_trials", reason: "Candidate models are ready for controlled comparison." };
|
|
621
|
+
case "challengers_ready":
|
|
622
|
+
return { action: "issue_intent", intentKind: "calculate_frontier", reason: "Comparable challenger evidence is ready." };
|
|
623
|
+
case "frontier_ready": {
|
|
624
|
+
if (!state.proposedCandidateId || state.proposedCandidateId === state.activeCandidateId) {
|
|
625
|
+
return { action: "observe", reason: "The harness did not identify a materially better frontier candidate." };
|
|
626
|
+
}
|
|
627
|
+
if (!state.policy)
|
|
628
|
+
return { action: "await_user", reason: "A swap policy must be approved before changing the model." };
|
|
629
|
+
const graduated = state.policy.automaticSwapsEnabled &&
|
|
630
|
+
state.verifiedSwapCount >= state.policy.confirmationRequiredUntilVerifiedSwaps;
|
|
631
|
+
return graduated
|
|
632
|
+
? { action: "issue_intent", intentKind: "apply_model_swap", reason: "Bounded automatic swapping is enabled and the graduation threshold is satisfied." }
|
|
633
|
+
: { action: "await_user", reason: "User confirmation is required for this model swap." };
|
|
634
|
+
}
|
|
635
|
+
case "awaiting_swap_approval":
|
|
636
|
+
return { action: "await_user", reason: "Waiting for the user to approve or reject the proposed model swap." };
|
|
637
|
+
case "swap_approved":
|
|
638
|
+
return { action: "issue_intent", intentKind: "apply_model_swap", reason: "The user approved the proposed model swap." };
|
|
639
|
+
case "swap_applied":
|
|
640
|
+
return { action: "observe", reason: "Collect the user-confirmed post-swap verification window before verification." };
|
|
641
|
+
case "swap_verified":
|
|
642
|
+
case "monitoring":
|
|
643
|
+
return { action: "observe", reason: "Monitor production evidence until reassessment is due." };
|
|
644
|
+
case "verification_failed":
|
|
645
|
+
return state.policy?.rollbackOnRegression
|
|
646
|
+
? { action: "issue_intent", intentKind: "rollback_model_swap", reason: "Post-swap verification detected a regression and rollback is required." }
|
|
647
|
+
: { action: "await_user", reason: "Post-swap verification failed and automatic rollback is not authorized." };
|
|
648
|
+
}
|
|
649
|
+
}
|
|
650
|
+
function metricFor(assessment, objective) {
|
|
651
|
+
return assessment.metrics.find((metric) => metric.metricId === objective.metricId);
|
|
652
|
+
}
|
|
653
|
+
function bounds(metric) {
|
|
654
|
+
if (metric.state !== "measured" || metric.value === undefined)
|
|
655
|
+
return undefined;
|
|
656
|
+
return metric.interval ?? { lower: metric.value, upper: metric.value };
|
|
657
|
+
}
|
|
658
|
+
function readinessErrors(profile, assessment) {
|
|
659
|
+
const errors = [];
|
|
660
|
+
if (assessment.targetId !== profile.applicationTarget.id) {
|
|
661
|
+
errors.push("assessment target does not match the confirmed application LLM target");
|
|
662
|
+
}
|
|
663
|
+
for (const objective of profile.objectives) {
|
|
664
|
+
if (!objective.required)
|
|
665
|
+
continue;
|
|
666
|
+
const metric = metricFor(assessment, objective);
|
|
667
|
+
if (!metric)
|
|
668
|
+
errors.push(`${objective.metricId}: missing`);
|
|
669
|
+
else {
|
|
670
|
+
if (metric.state !== "measured")
|
|
671
|
+
errors.push(`${objective.metricId}: ${metric.state}`);
|
|
672
|
+
else if (metric.value === undefined)
|
|
673
|
+
errors.push(`${objective.metricId}: no numeric value`);
|
|
674
|
+
errors.push(...validateMetricAggregation(metric, objective));
|
|
675
|
+
if (metric.sampleCount < objective.minimumSamples) {
|
|
676
|
+
errors.push(`${objective.metricId}: ${metric.sampleCount}/${objective.minimumSamples} samples`);
|
|
677
|
+
}
|
|
678
|
+
}
|
|
679
|
+
}
|
|
680
|
+
return errors;
|
|
681
|
+
}
|
|
682
|
+
function baselineReadinessErrors(profile, metrics) {
|
|
683
|
+
if (!profile)
|
|
684
|
+
return ["confirmed task profile is missing"];
|
|
685
|
+
const errors = [];
|
|
686
|
+
if (!profile.objectives.some((objective) => objective.required)) {
|
|
687
|
+
errors.push("task profile has no required metric objectives");
|
|
688
|
+
}
|
|
689
|
+
for (const objective of profile.objectives) {
|
|
690
|
+
if (!objective.required)
|
|
691
|
+
continue;
|
|
692
|
+
const metric = metrics.find((item) => item.metricId === objective.metricId &&
|
|
693
|
+
item.targetId === profile.applicationTarget.id);
|
|
694
|
+
if (!metric)
|
|
695
|
+
errors.push(`${objective.metricId}: missing`);
|
|
696
|
+
else {
|
|
697
|
+
if (metric.state !== "measured")
|
|
698
|
+
errors.push(`${objective.metricId}: ${metric.state}`);
|
|
699
|
+
else if (metric.value === undefined || !Number.isFinite(metric.value))
|
|
700
|
+
errors.push(`${objective.metricId}: no finite numeric value`);
|
|
701
|
+
errors.push(...validateMetricAggregation(metric, objective));
|
|
702
|
+
if (metric.sampleCount < objective.minimumSamples) {
|
|
703
|
+
errors.push(`${objective.metricId}: ${metric.sampleCount}/${objective.minimumSamples} samples`);
|
|
704
|
+
}
|
|
705
|
+
else if (!metric.evidence.length)
|
|
706
|
+
errors.push(`${objective.metricId}: no evidence reference`);
|
|
707
|
+
}
|
|
708
|
+
}
|
|
709
|
+
return errors;
|
|
710
|
+
}
|
|
711
|
+
function baselineAssessmentErrors(state, profile, metrics) {
|
|
712
|
+
return [
|
|
713
|
+
...(state.cycle.observabilityReady ? [] : ["required evaluation and observability setup is not ready"]),
|
|
714
|
+
...baselineReadinessErrors(profile, metrics),
|
|
715
|
+
];
|
|
716
|
+
}
|
|
717
|
+
function updatedBaselineWindows(previous, signal) {
|
|
718
|
+
if (signal.type !== "metric_window_available")
|
|
719
|
+
return previous ?? [];
|
|
720
|
+
const windows = new Map((previous ?? []).map((metric) => [metric.metricId, metric]));
|
|
721
|
+
for (const metric of signal.metrics)
|
|
722
|
+
windows.set(metric.metricId, metric);
|
|
723
|
+
return [...windows.values()];
|
|
724
|
+
}
|
|
725
|
+
function feasibility(profile, assessment) {
|
|
726
|
+
if (readinessErrors(profile, assessment).length)
|
|
727
|
+
return "unresolved";
|
|
728
|
+
let unresolved = false;
|
|
729
|
+
for (const objective of profile.objectives) {
|
|
730
|
+
if (!objective.constraint)
|
|
731
|
+
continue;
|
|
732
|
+
const range = bounds(metricFor(assessment, objective));
|
|
733
|
+
if (objective.constraint.operator === "at_least") {
|
|
734
|
+
if (range.upper < objective.constraint.value)
|
|
735
|
+
return "ineligible";
|
|
736
|
+
if (range.lower < objective.constraint.value)
|
|
737
|
+
unresolved = true;
|
|
738
|
+
}
|
|
739
|
+
else {
|
|
740
|
+
if (range.lower > objective.constraint.value)
|
|
741
|
+
return "ineligible";
|
|
742
|
+
if (range.upper > objective.constraint.value)
|
|
743
|
+
unresolved = true;
|
|
744
|
+
}
|
|
745
|
+
}
|
|
746
|
+
return unresolved ? "unresolved" : "eligible";
|
|
747
|
+
}
|
|
748
|
+
function expectedComparison(profile, candidate, other) {
|
|
749
|
+
const reasons = [];
|
|
750
|
+
let strictlyBetter = false;
|
|
751
|
+
let uncertain = false;
|
|
752
|
+
for (const objective of profile.objectives) {
|
|
753
|
+
const aMetric = metricFor(candidate, objective);
|
|
754
|
+
const bMetric = metricFor(other, objective);
|
|
755
|
+
const a = aMetric && bounds(aMetric);
|
|
756
|
+
const b = bMetric && bounds(bMetric);
|
|
757
|
+
if (!aMetric || !bMetric || !a || !b || aMetric.unit !== bMetric.unit || aMetric.direction !== bMetric.direction) {
|
|
758
|
+
return {
|
|
759
|
+
candidateId: candidate.candidateId,
|
|
760
|
+
otherCandidateId: other.candidateId,
|
|
761
|
+
state: "unresolved",
|
|
762
|
+
reasons: [`${objective.metricId}: incompatible or insufficient evidence`],
|
|
763
|
+
};
|
|
764
|
+
}
|
|
765
|
+
const tolerance = objective.tolerance;
|
|
766
|
+
if (aMetric.direction === "maximize") {
|
|
767
|
+
if (a.lower + tolerance >= b.upper) {
|
|
768
|
+
if (a.lower > b.upper + tolerance)
|
|
769
|
+
strictlyBetter = true;
|
|
770
|
+
}
|
|
771
|
+
else if (a.upper + tolerance < b.lower) {
|
|
772
|
+
reasons.push(`${objective.metricId}: proven worse`);
|
|
773
|
+
return { candidateId: candidate.candidateId, otherCandidateId: other.candidateId, state: "does_not_dominate", reasons };
|
|
774
|
+
}
|
|
775
|
+
else
|
|
776
|
+
uncertain = true;
|
|
777
|
+
}
|
|
778
|
+
else {
|
|
779
|
+
if (a.upper <= b.lower + tolerance) {
|
|
780
|
+
if (a.upper + tolerance < b.lower)
|
|
781
|
+
strictlyBetter = true;
|
|
782
|
+
}
|
|
783
|
+
else if (a.lower > b.upper + tolerance) {
|
|
784
|
+
reasons.push(`${objective.metricId}: proven worse`);
|
|
785
|
+
return { candidateId: candidate.candidateId, otherCandidateId: other.candidateId, state: "does_not_dominate", reasons };
|
|
786
|
+
}
|
|
787
|
+
else
|
|
788
|
+
uncertain = true;
|
|
789
|
+
}
|
|
790
|
+
}
|
|
791
|
+
if (uncertain) {
|
|
792
|
+
return {
|
|
793
|
+
candidateId: candidate.candidateId,
|
|
794
|
+
otherCandidateId: other.candidateId,
|
|
795
|
+
state: "unresolved",
|
|
796
|
+
reasons: ["one or more objective intervals do not establish ordering"],
|
|
797
|
+
};
|
|
798
|
+
}
|
|
799
|
+
if (!strictlyBetter) {
|
|
800
|
+
return {
|
|
801
|
+
candidateId: candidate.candidateId,
|
|
802
|
+
otherCandidateId: other.candidateId,
|
|
803
|
+
state: "does_not_dominate",
|
|
804
|
+
reasons: ["no objective is materially better"],
|
|
805
|
+
};
|
|
806
|
+
}
|
|
807
|
+
return {
|
|
808
|
+
candidateId: candidate.candidateId,
|
|
809
|
+
otherCandidateId: other.candidateId,
|
|
810
|
+
state: "dominates",
|
|
811
|
+
reasons: ["proven no worse on every objective and materially better on at least one"],
|
|
812
|
+
};
|
|
813
|
+
}
|
|
814
|
+
function sameMembers(actual, expected) {
|
|
815
|
+
return actual.length === expected.length && [...actual].sort().every((item, index) => item === [...expected].sort()[index]);
|
|
816
|
+
}
|
|
817
|
+
export function verifyFrontierSnapshot(profile, assessments, snapshot) {
|
|
818
|
+
const errors = [];
|
|
819
|
+
const byCandidate = new Map(assessments.map((assessment) => [assessment.candidateId, assessment]));
|
|
820
|
+
if (snapshot.taskProfileId !== profile.id || snapshot.taskProfileRevision !== profile.revision) {
|
|
821
|
+
errors.push("snapshot task profile does not match");
|
|
822
|
+
}
|
|
823
|
+
if (snapshot.targetId !== profile.applicationTarget.id) {
|
|
824
|
+
errors.push("snapshot target does not match the confirmed application LLM target");
|
|
825
|
+
}
|
|
826
|
+
if (snapshot.calculationEvidence.length === 0)
|
|
827
|
+
errors.push("calculation evidence is required");
|
|
828
|
+
if (!sameMembers(snapshot.assessmentIds, assessments.map((assessment) => assessment.id))) {
|
|
829
|
+
errors.push("snapshot assessment IDs do not match the supplied assessments");
|
|
830
|
+
}
|
|
831
|
+
const eligible = [];
|
|
832
|
+
const expectedIneligible = [];
|
|
833
|
+
const expectedUnresolved = new Set();
|
|
834
|
+
for (const assessment of assessments) {
|
|
835
|
+
if (assessment.targetId !== profile.applicationTarget.id) {
|
|
836
|
+
errors.push(`${assessment.candidateId}: application target mismatch`);
|
|
837
|
+
continue;
|
|
838
|
+
}
|
|
839
|
+
errors.push(...validateMetricTargets(assessment.metrics, profile.applicationTarget.id, assessment.candidateId));
|
|
840
|
+
if (assessment.taskProfileId !== profile.id || assessment.taskProfileRevision !== profile.revision) {
|
|
841
|
+
errors.push(`${assessment.candidateId}: task profile revision mismatch`);
|
|
842
|
+
continue;
|
|
843
|
+
}
|
|
844
|
+
if (assessment.productRevision !== snapshot.productRevision) {
|
|
845
|
+
errors.push(`${assessment.candidateId}: product revision mismatch`);
|
|
846
|
+
continue;
|
|
847
|
+
}
|
|
848
|
+
const state = feasibility(profile, assessment);
|
|
849
|
+
if (state === "eligible")
|
|
850
|
+
eligible.push(assessment);
|
|
851
|
+
else if (state === "ineligible")
|
|
852
|
+
expectedIneligible.push(assessment.candidateId);
|
|
853
|
+
else
|
|
854
|
+
expectedUnresolved.add(assessment.candidateId);
|
|
855
|
+
}
|
|
856
|
+
const expectedDominated = new Set();
|
|
857
|
+
const expectedComparisons = [];
|
|
858
|
+
for (const candidate of eligible) {
|
|
859
|
+
for (const other of eligible) {
|
|
860
|
+
if (candidate.candidateId === other.candidateId)
|
|
861
|
+
continue;
|
|
862
|
+
const comparison = expectedComparison(profile, candidate, other);
|
|
863
|
+
expectedComparisons.push(comparison);
|
|
864
|
+
if (comparison.state === "dominates")
|
|
865
|
+
expectedDominated.add(other.candidateId);
|
|
866
|
+
if (comparison.state === "unresolved") {
|
|
867
|
+
expectedUnresolved.add(candidate.candidateId);
|
|
868
|
+
expectedUnresolved.add(other.candidateId);
|
|
869
|
+
}
|
|
870
|
+
}
|
|
871
|
+
}
|
|
872
|
+
for (const expected of expectedComparisons) {
|
|
873
|
+
const actual = snapshot.comparisons.find((comparison) => comparison.candidateId === expected.candidateId && comparison.otherCandidateId === expected.otherCandidateId);
|
|
874
|
+
if (!actual)
|
|
875
|
+
errors.push(`missing comparison ${expected.candidateId} -> ${expected.otherCandidateId}`);
|
|
876
|
+
else if (actual.state !== expected.state) {
|
|
877
|
+
errors.push(`comparison ${expected.candidateId} -> ${expected.otherCandidateId} should be ${expected.state}, got ${actual.state}`);
|
|
878
|
+
}
|
|
879
|
+
else if (actual.reasons.length === 0) {
|
|
880
|
+
errors.push(`comparison ${expected.candidateId} -> ${expected.otherCandidateId} requires reasons`);
|
|
881
|
+
}
|
|
882
|
+
}
|
|
883
|
+
const expectedFrontier = eligible
|
|
884
|
+
.map((assessment) => assessment.candidateId)
|
|
885
|
+
.filter((candidateId) => !expectedDominated.has(candidateId) && !expectedUnresolved.has(candidateId));
|
|
886
|
+
if (!sameMembers(snapshot.frontierCandidateIds, expectedFrontier))
|
|
887
|
+
errors.push("frontier candidate set is incorrect");
|
|
888
|
+
if (!sameMembers(snapshot.dominatedCandidateIds, [...expectedDominated]))
|
|
889
|
+
errors.push("dominated candidate set is incorrect");
|
|
890
|
+
if (!sameMembers(snapshot.ineligibleCandidateIds, expectedIneligible))
|
|
891
|
+
errors.push("ineligible candidate set is incorrect");
|
|
892
|
+
if (!sameMembers(snapshot.unresolvedCandidateIds, [...expectedUnresolved]))
|
|
893
|
+
errors.push("unresolved candidate set is incorrect");
|
|
894
|
+
const allKnown = new Set(byCandidate.keys());
|
|
895
|
+
for (const id of [
|
|
896
|
+
...snapshot.frontierCandidateIds,
|
|
897
|
+
...snapshot.dominatedCandidateIds,
|
|
898
|
+
...snapshot.ineligibleCandidateIds,
|
|
899
|
+
...snapshot.unresolvedCandidateIds,
|
|
900
|
+
]) {
|
|
901
|
+
if (!allKnown.has(id))
|
|
902
|
+
errors.push(`unknown candidate ${id}`);
|
|
903
|
+
}
|
|
904
|
+
return { valid: errors.length === 0, errors };
|
|
905
|
+
}
|
|
906
|
+
const MAX_PROCESSED_SIGNAL_IDS = 256;
|
|
907
|
+
function nextAt(anchor, seconds) {
|
|
908
|
+
if (seconds === undefined)
|
|
909
|
+
return undefined;
|
|
910
|
+
return new Date(Date.parse(anchor) + seconds * 1000).toISOString();
|
|
911
|
+
}
|
|
912
|
+
function earliest(values) {
|
|
913
|
+
return values.filter((value) => value !== undefined)
|
|
914
|
+
.sort((left, right) => Date.parse(left) - Date.parse(right))[0];
|
|
915
|
+
}
|
|
916
|
+
export function buildAutomationPlan(state, descriptor) {
|
|
917
|
+
if (state.automation.paused) {
|
|
918
|
+
return {
|
|
919
|
+
mode: state.cycle.policy?.checkPolicy.execution.mode ?? "active_session_only",
|
|
920
|
+
requiredSignals: [],
|
|
921
|
+
scheduling: "active_session_fallback",
|
|
922
|
+
gaps: ["Proactive OpenMerit checks are paused."],
|
|
923
|
+
};
|
|
924
|
+
}
|
|
925
|
+
const policy = state.cycle.policy?.checkPolicy;
|
|
926
|
+
if (!policy) {
|
|
927
|
+
return {
|
|
928
|
+
mode: "active_session_only",
|
|
929
|
+
requiredSignals: [],
|
|
930
|
+
scheduling: "active_session_fallback",
|
|
931
|
+
gaps: ["No user-confirmed automation policy exists."],
|
|
932
|
+
};
|
|
933
|
+
}
|
|
934
|
+
const required = new Set();
|
|
935
|
+
const cadences = [
|
|
936
|
+
policy.baselineAssessment,
|
|
937
|
+
policy.frontierReassessment,
|
|
938
|
+
policy.regressionCheck,
|
|
939
|
+
policy.postSwapVerification,
|
|
940
|
+
];
|
|
941
|
+
if (cadences.some((cadence) => cadence.afterCompletedTasks !== undefined))
|
|
942
|
+
required.add("product_task_completed");
|
|
943
|
+
if (cadences.some((cadence) => cadence.afterElapsedSeconds !== undefined))
|
|
944
|
+
required.add("scheduled_tick");
|
|
945
|
+
if (policy.frontierReassessment.onModelCatalogChange)
|
|
946
|
+
required.add("model_catalog_changed");
|
|
947
|
+
if (policy.regressionCheck.thresholds.length)
|
|
948
|
+
required.add("metric_window_available");
|
|
949
|
+
if (policy.postSwapVerification.afterCompletedTasks === undefined &&
|
|
950
|
+
policy.postSwapVerification.afterElapsedSeconds === undefined) {
|
|
951
|
+
required.add("verification_window_completed");
|
|
952
|
+
}
|
|
953
|
+
const nextWakeupAt = state.cycle.stage === "collecting_baseline"
|
|
954
|
+
? nextAt(state.automation.lastAssessmentAt ?? state.automation.baselineStartedAt ?? state.lastResult?.completedAt ?? state.updatedAt, policy.baselineAssessment.afterElapsedSeconds)
|
|
955
|
+
: ["monitoring", "swap_verified"].includes(state.cycle.stage)
|
|
956
|
+
? earliest([
|
|
957
|
+
nextAt(state.automation.lastReassessmentAt ?? state.updatedAt, policy.frontierReassessment.afterElapsedSeconds),
|
|
958
|
+
nextAt(state.automation.lastRegressionCheckAt ?? state.updatedAt, policy.regressionCheck.afterElapsedSeconds),
|
|
959
|
+
])
|
|
960
|
+
: state.cycle.stage === "swap_applied"
|
|
961
|
+
? nextAt(state.automation.swapAppliedAt ?? state.updatedAt, policy.postSwapVerification.afterElapsedSeconds)
|
|
962
|
+
: undefined;
|
|
963
|
+
const supported = new Set(descriptor.automation.supportedSignals);
|
|
964
|
+
const gaps = [...required]
|
|
965
|
+
.filter((signal) => !supported.has(signal))
|
|
966
|
+
.map((signal) => `Harness does not report ${signal}.`);
|
|
967
|
+
let scheduling = "active_session_fallback";
|
|
968
|
+
if (policy.execution.mode === "persistent") {
|
|
969
|
+
scheduling = descriptor.automation.persistentScheduling && descriptor.automation.backgroundExecution
|
|
970
|
+
? "native"
|
|
971
|
+
: "external_scheduler_required";
|
|
972
|
+
if (scheduling === "external_scheduler_required") {
|
|
973
|
+
gaps.push("Persistent scheduling or background execution is unavailable; configure an external scheduler.");
|
|
974
|
+
}
|
|
975
|
+
if (descriptor.automation.wakeupProvisioning === "infrastructure_change" &&
|
|
976
|
+
!policy.execution.infrastructureChangesAllowed) {
|
|
977
|
+
scheduling = "external_scheduler_required";
|
|
978
|
+
gaps.push("Wakeup provisioning requires an infrastructure change that the user has not authorized.");
|
|
979
|
+
}
|
|
980
|
+
}
|
|
981
|
+
return {
|
|
982
|
+
mode: policy.execution.mode,
|
|
983
|
+
requiredSignals: [...required],
|
|
984
|
+
...(nextWakeupAt ? { nextWakeupAt } : {}),
|
|
985
|
+
scheduling,
|
|
986
|
+
gaps,
|
|
987
|
+
};
|
|
988
|
+
}
|
|
989
|
+
function cadenceDue(cadence, completedTasks, lastTaskCount, lastAt, now) {
|
|
990
|
+
const taskDue = cadence.afterCompletedTasks !== undefined &&
|
|
991
|
+
completedTasks - lastTaskCount >= cadence.afterCompletedTasks;
|
|
992
|
+
const elapsedDue = cadence.afterElapsedSeconds !== undefined &&
|
|
993
|
+
Date.parse(now) - Date.parse(lastAt) >= cadence.afterElapsedSeconds * 1000;
|
|
994
|
+
return taskDue || elapsedDue;
|
|
995
|
+
}
|
|
996
|
+
function regressionMetric(signal, previous, policy) {
|
|
997
|
+
if (signal.type !== "metric_window_available")
|
|
998
|
+
return undefined;
|
|
999
|
+
const directions = new Map(METRIC_CATALOG.map((metric) => [metric.id, metric.direction]));
|
|
1000
|
+
for (const threshold of policy.checkPolicy.regressionCheck.thresholds) {
|
|
1001
|
+
const current = signal.metrics.find((metric) => metric.metricId === threshold.metricId && metric.state === "measured" && metric.value !== undefined);
|
|
1002
|
+
const prior = previous[threshold.metricId];
|
|
1003
|
+
if (!current || current.value === undefined || prior === undefined)
|
|
1004
|
+
continue;
|
|
1005
|
+
const denominator = Math.max(Math.abs(prior), Number.EPSILON);
|
|
1006
|
+
const deterioration = directions.get(threshold.metricId) === "minimize"
|
|
1007
|
+
? (current.value - prior) / denominator
|
|
1008
|
+
: (prior - current.value) / denominator;
|
|
1009
|
+
if (deterioration >= threshold.relativeChangeAtLeast)
|
|
1010
|
+
return threshold.metricId;
|
|
1011
|
+
}
|
|
1012
|
+
return undefined;
|
|
1013
|
+
}
|
|
1014
|
+
function dueDecision(state, signal, now) {
|
|
1015
|
+
if (state.automation.paused)
|
|
1016
|
+
return { action: "observe", reason: "Proactive OpenMerit checks are paused." };
|
|
1017
|
+
if (state.activeIntent)
|
|
1018
|
+
return { action: "observe", reason: "Harness work is already active." };
|
|
1019
|
+
if (state.automation.pendingIntentKind) {
|
|
1020
|
+
return {
|
|
1021
|
+
action: "issue_intent",
|
|
1022
|
+
intentKind: state.automation.pendingIntentKind,
|
|
1023
|
+
reason: state.automation.pendingIntentReason ?? "A previously detected automation check remains due.",
|
|
1024
|
+
};
|
|
1025
|
+
}
|
|
1026
|
+
const policy = state.cycle.policy;
|
|
1027
|
+
if (!policy)
|
|
1028
|
+
return { action: "observe", reason: "No user-confirmed automation policy exists." };
|
|
1029
|
+
const completed = state.automation.observedCompletedTasks;
|
|
1030
|
+
const cooldownAnchor = [state.automation.lastReassessmentAt, state.automation.lastRegressionCheckAt]
|
|
1031
|
+
.filter((value) => value !== undefined)
|
|
1032
|
+
.sort((left, right) => Date.parse(right) - Date.parse(left))[0];
|
|
1033
|
+
const cooldownSatisfied = !cooldownAnchor ||
|
|
1034
|
+
Date.parse(now) - Date.parse(cooldownAnchor) >= policy.checkPolicy.cooldownSeconds * 1000;
|
|
1035
|
+
if (state.cycle.stage === "collecting_baseline" && cadenceDue(policy.checkPolicy.baselineAssessment, completed, state.automation.lastAssessmentTaskCount, state.automation.lastAssessmentAt ?? state.automation.baselineStartedAt ?? state.lastResult?.completedAt ?? state.updatedAt, now)) {
|
|
1036
|
+
return { action: "issue_intent", intentKind: "run_assessment", reason: "The user-confirmed baseline assessment cadence is due." };
|
|
1037
|
+
}
|
|
1038
|
+
if (["monitoring", "swap_verified"].includes(state.cycle.stage) && cooldownSatisfied) {
|
|
1039
|
+
const regressedMetric = regressionMetric(signal, state.automation.latestMetricValues ?? {}, policy);
|
|
1040
|
+
if (regressedMetric) {
|
|
1041
|
+
return { action: "issue_intent", intentKind: "investigate_regression", reason: `${regressedMetric} crossed its user-confirmed regression threshold.` };
|
|
1042
|
+
}
|
|
1043
|
+
if (cadenceDue(policy.checkPolicy.regressionCheck, completed, state.automation.lastRegressionCheckTaskCount ?? 0, state.automation.lastRegressionCheckAt ?? state.updatedAt, now)) {
|
|
1044
|
+
return { action: "issue_intent", intentKind: "investigate_regression", reason: "The user-confirmed regression check cadence is due." };
|
|
1045
|
+
}
|
|
1046
|
+
if (signal.type === "model_catalog_changed" && policy.checkPolicy.frontierReassessment.onModelCatalogChange &&
|
|
1047
|
+
state.automation.modelCatalogFingerprint !== undefined &&
|
|
1048
|
+
state.automation.modelCatalogFingerprint !== signal.fingerprint) {
|
|
1049
|
+
return { action: "issue_intent", intentKind: "discover_candidates", reason: "The model catalogue changed and the user enabled catalogue-triggered reassessment." };
|
|
1050
|
+
}
|
|
1051
|
+
if (cadenceDue(policy.checkPolicy.frontierReassessment, completed, state.automation.lastReassessmentTaskCount ?? 0, state.automation.lastReassessmentAt ?? state.updatedAt, now)) {
|
|
1052
|
+
return { action: "issue_intent", intentKind: "discover_candidates", reason: "The user-confirmed frontier reassessment cadence is due." };
|
|
1053
|
+
}
|
|
1054
|
+
}
|
|
1055
|
+
if (state.cycle.stage === "swap_applied" &&
|
|
1056
|
+
(signal.type === "verification_window_completed" || cadenceDue(policy.checkPolicy.postSwapVerification, completed, state.automation.swapAppliedTaskCount ?? completed, state.automation.swapAppliedAt ?? state.updatedAt, now))) {
|
|
1057
|
+
return { action: "issue_intent", intentKind: "verify_model_swap", reason: "The user-confirmed post-swap verification window is complete." };
|
|
1058
|
+
}
|
|
1059
|
+
return { action: "observe", reason: "No user-confirmed automation check is due." };
|
|
1060
|
+
}
|
|
1061
|
+
function auditSignalData(signal) {
|
|
1062
|
+
switch (signal.type) {
|
|
1063
|
+
case "product_task_completed":
|
|
1064
|
+
return {
|
|
1065
|
+
targetId: signal.targetId,
|
|
1066
|
+
modelId: signal.modelId,
|
|
1067
|
+
...(signal.telemetry ? { telemetry: signal.telemetry } : {}),
|
|
1068
|
+
evidenceIds: (signal.evidence ?? []).map((item) => item.id),
|
|
1069
|
+
};
|
|
1070
|
+
case "metric_window_available":
|
|
1071
|
+
case "verification_window_completed":
|
|
1072
|
+
return {
|
|
1073
|
+
targetId: signal.targetId,
|
|
1074
|
+
metrics: (signal.metrics ?? []).map((metric) => ({
|
|
1075
|
+
targetId: metric.targetId,
|
|
1076
|
+
metricId: metric.metricId,
|
|
1077
|
+
state: metric.state,
|
|
1078
|
+
scope: metric.scope,
|
|
1079
|
+
direction: metric.direction,
|
|
1080
|
+
method: metric.method,
|
|
1081
|
+
value: metric.value ?? null,
|
|
1082
|
+
unit: metric.unit,
|
|
1083
|
+
sampleCount: metric.sampleCount,
|
|
1084
|
+
...(metric.interval ? { interval: metric.interval } : {}),
|
|
1085
|
+
...(metric.window ? { window: metric.window } : {}),
|
|
1086
|
+
evidenceIds: metric.evidence.map((item) => item.id),
|
|
1087
|
+
})),
|
|
1088
|
+
};
|
|
1089
|
+
case "model_catalog_changed":
|
|
1090
|
+
return { targetId: signal.targetId, fingerprint: signal.fingerprint };
|
|
1091
|
+
case "scheduled_tick":
|
|
1092
|
+
return { targetId: signal.targetId };
|
|
1093
|
+
}
|
|
1094
|
+
}
|
|
1095
|
+
/** Harness-neutral durable coordinator. Adapters transport jobs; this class owns lifecycle state. */
|
|
1096
|
+
export class OpenMeritCoordinator {
|
|
1097
|
+
store;
|
|
1098
|
+
adapter;
|
|
1099
|
+
constructor(store, adapter) {
|
|
1100
|
+
this.store = store;
|
|
1101
|
+
this.adapter = adapter;
|
|
1102
|
+
}
|
|
1103
|
+
async issue(kind, options = {}) {
|
|
1104
|
+
const missing = missingHarnessCapabilities(this.adapter.descriptor, kind);
|
|
1105
|
+
if (missing.length)
|
|
1106
|
+
throw new Error(`Harness lacks required capabilities: ${missing.join(", ")}`);
|
|
1107
|
+
const config = await this.store.readConfig() ?? { schemaVersion: 1, taskProfiles: [] };
|
|
1108
|
+
const existing = await this.store.readState();
|
|
1109
|
+
if (existing?.activeIntent)
|
|
1110
|
+
throw new Error(`OpenMerit intent ${existing.activeIntent.id} is already active.`);
|
|
1111
|
+
if (kind === "run_challenger_trials" && config.policy) {
|
|
1112
|
+
const spent = existing?.automation.evaluationSpend ?? 0;
|
|
1113
|
+
const limit = config.policy.evaluationBudget.maximumSpend;
|
|
1114
|
+
if (spent >= limit) {
|
|
1115
|
+
throw new Error(`Evaluation budget exhausted: ${spent.toFixed(6)} ${config.policy.evaluationBudget.currency} spent of ${limit.toFixed(6)}.`);
|
|
1116
|
+
}
|
|
1117
|
+
}
|
|
1118
|
+
const targetId = activeTarget(config)?.id;
|
|
1119
|
+
const confirmedProfile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
1120
|
+
if (kind !== "establish_evals" && !targetId) {
|
|
1121
|
+
throw new Error(`OpenMerit cannot issue ${kind} without a confirmed application LLM target.`);
|
|
1122
|
+
}
|
|
1123
|
+
const intent = createOpenMeritIntent(kind, {
|
|
1124
|
+
taskProfileId: config.activeTaskProfileId,
|
|
1125
|
+
policy: config.policy,
|
|
1126
|
+
evaluationSpend: existing?.automation.evaluationSpend,
|
|
1127
|
+
targetId,
|
|
1128
|
+
additionalConstraints: {
|
|
1129
|
+
...options.additionalConstraints,
|
|
1130
|
+
...(confirmedProfile ? { taskProfile: confirmedProfile } : {}),
|
|
1131
|
+
...(kind === "discover_candidates"
|
|
1132
|
+
? {
|
|
1133
|
+
preliminaryModelLeads: (config.preliminaryModelLeads ?? []),
|
|
1134
|
+
incumbentCandidate: (existing?.candidates?.find((candidate) => candidate.candidateId === existing.cycle.activeCandidateId) ?? existing?.candidates?.[0]),
|
|
1135
|
+
candidateSelection: {
|
|
1136
|
+
totalCandidateCount: Math.min(3, config.policy?.evaluationBudget.maximumCandidateCount ?? 3),
|
|
1137
|
+
includeIncumbent: true,
|
|
1138
|
+
requireAvailabilityProbe: true,
|
|
1139
|
+
requireExactModelIds: true,
|
|
1140
|
+
requirePriceAndCapabilitySources: true,
|
|
1141
|
+
},
|
|
1142
|
+
}
|
|
1143
|
+
: {}),
|
|
1144
|
+
...(kind === "run_challenger_trials"
|
|
1145
|
+
? { candidates: (existing?.candidates ?? []) }
|
|
1146
|
+
: {}),
|
|
1147
|
+
...(kind === "calculate_frontier"
|
|
1148
|
+
? {
|
|
1149
|
+
assessments: (existing?.assessments ?? []),
|
|
1150
|
+
...(existing?.experimentManifest
|
|
1151
|
+
? { experimentManifest: existing.experimentManifest }
|
|
1152
|
+
: {}),
|
|
1153
|
+
}
|
|
1154
|
+
: {}),
|
|
1155
|
+
},
|
|
1156
|
+
});
|
|
1157
|
+
const now = new Date().toISOString();
|
|
1158
|
+
const state = existing
|
|
1159
|
+
? {
|
|
1160
|
+
...existing,
|
|
1161
|
+
cycle: kind === "establish_evals"
|
|
1162
|
+
? {
|
|
1163
|
+
...existing.cycle,
|
|
1164
|
+
id: `cycle-${crypto.randomUUID()}`,
|
|
1165
|
+
targetId: undefined,
|
|
1166
|
+
taskProfileId: undefined,
|
|
1167
|
+
activeCandidateId: undefined,
|
|
1168
|
+
previousCandidateId: undefined,
|
|
1169
|
+
proposedCandidateId: undefined,
|
|
1170
|
+
stage: startedStage(kind),
|
|
1171
|
+
verifiedSwapCount: 0,
|
|
1172
|
+
baselineEvidenceSufficient: false,
|
|
1173
|
+
observabilityReady: false,
|
|
1174
|
+
policy: undefined,
|
|
1175
|
+
}
|
|
1176
|
+
: { ...existing.cycle, stage: startedStage(kind) },
|
|
1177
|
+
activeIntent: intent,
|
|
1178
|
+
...(kind === "establish_evals" ? {
|
|
1179
|
+
candidates: undefined,
|
|
1180
|
+
assessments: undefined,
|
|
1181
|
+
experimentManifest: undefined,
|
|
1182
|
+
observability: undefined,
|
|
1183
|
+
lastResult: undefined,
|
|
1184
|
+
} : {}),
|
|
1185
|
+
automation: {
|
|
1186
|
+
...existing.automation,
|
|
1187
|
+
...(existing.automation.pendingIntentKind === kind
|
|
1188
|
+
? { pendingIntentKind: undefined, pendingIntentReason: undefined }
|
|
1189
|
+
: {}),
|
|
1190
|
+
},
|
|
1191
|
+
updatedAt: now,
|
|
1192
|
+
}
|
|
1193
|
+
: {
|
|
1194
|
+
schemaVersion: 1,
|
|
1195
|
+
cycle: {
|
|
1196
|
+
id: `cycle-${crypto.randomUUID()}`,
|
|
1197
|
+
stage: startedStage(kind),
|
|
1198
|
+
verifiedSwapCount: 0,
|
|
1199
|
+
baselineEvidenceSufficient: false,
|
|
1200
|
+
observabilityReady: false,
|
|
1201
|
+
},
|
|
1202
|
+
activeIntent: intent,
|
|
1203
|
+
automation: {
|
|
1204
|
+
setupNudgeShown: kind === "establish_evals",
|
|
1205
|
+
observedCompletedTasks: 0,
|
|
1206
|
+
lastAssessmentTaskCount: 0,
|
|
1207
|
+
processedSignalIds: [],
|
|
1208
|
+
},
|
|
1209
|
+
updatedAt: now,
|
|
1210
|
+
};
|
|
1211
|
+
if (existing)
|
|
1212
|
+
await this.store.writeState(state);
|
|
1213
|
+
else {
|
|
1214
|
+
await this.store.initialize(config, state);
|
|
1215
|
+
await this.store.appendEvent({
|
|
1216
|
+
schemaVersion: 1,
|
|
1217
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1218
|
+
type: "project_initialized",
|
|
1219
|
+
occurredAt: now,
|
|
1220
|
+
cycleId: state.cycle.id,
|
|
1221
|
+
data: { harnessId: this.adapter.descriptor.id },
|
|
1222
|
+
});
|
|
1223
|
+
}
|
|
1224
|
+
await this.store.appendEvent({
|
|
1225
|
+
schemaVersion: 1,
|
|
1226
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1227
|
+
type: "intent_requested",
|
|
1228
|
+
occurredAt: now,
|
|
1229
|
+
cycleId: state.cycle.id,
|
|
1230
|
+
intentId: intent.id,
|
|
1231
|
+
data: { kind, harnessId: this.adapter.descriptor.id },
|
|
1232
|
+
});
|
|
1233
|
+
const receipt = await dispatchToHarness(this.adapter, intent);
|
|
1234
|
+
if (!receipt.accepted) {
|
|
1235
|
+
await this.store.writeState({
|
|
1236
|
+
...state,
|
|
1237
|
+
cycle: { ...state.cycle, stage: existing?.cycle.stage ?? "unconfigured" },
|
|
1238
|
+
activeIntent: undefined,
|
|
1239
|
+
automation: {
|
|
1240
|
+
...state.automation,
|
|
1241
|
+
pendingIntentKind: kind,
|
|
1242
|
+
pendingIntentReason: "Harness dispatch failed; the due intent remains pending.",
|
|
1243
|
+
},
|
|
1244
|
+
updatedAt: new Date().toISOString(),
|
|
1245
|
+
});
|
|
1246
|
+
throw new Error(receipt.reason ?? "Harness rejected the OpenMerit intent.");
|
|
1247
|
+
}
|
|
1248
|
+
return { intent, state, receipt };
|
|
1249
|
+
}
|
|
1250
|
+
async reconcileAutomation(now = new Date().toISOString()) {
|
|
1251
|
+
const state = await this.store.readState();
|
|
1252
|
+
if (!state)
|
|
1253
|
+
throw new Error("OpenMerit project is not initialized.");
|
|
1254
|
+
const plan = buildAutomationPlan(state, this.adapter.descriptor);
|
|
1255
|
+
if (plan.scheduling !== "native" || !plan.nextWakeupAt || Date.parse(plan.nextWakeupAt) <= Date.parse(now)) {
|
|
1256
|
+
return { plan };
|
|
1257
|
+
}
|
|
1258
|
+
if (!this.adapter.scheduleWakeup) {
|
|
1259
|
+
return {
|
|
1260
|
+
plan: {
|
|
1261
|
+
...plan,
|
|
1262
|
+
scheduling: "external_scheduler_required",
|
|
1263
|
+
gaps: [...plan.gaps, "Harness declared native scheduling but did not implement scheduleWakeup."],
|
|
1264
|
+
},
|
|
1265
|
+
};
|
|
1266
|
+
}
|
|
1267
|
+
const receipt = await this.adapter.scheduleWakeup({
|
|
1268
|
+
at: plan.nextWakeupAt,
|
|
1269
|
+
signal: "scheduled_tick",
|
|
1270
|
+
reason: "Wake the harness so OpenMerit can evaluate the user-confirmed check policy.",
|
|
1271
|
+
});
|
|
1272
|
+
if (!receipt.accepted) {
|
|
1273
|
+
return {
|
|
1274
|
+
plan: {
|
|
1275
|
+
...plan,
|
|
1276
|
+
scheduling: "external_scheduler_required",
|
|
1277
|
+
gaps: [...plan.gaps, `Harness wakeup provisioning failed: ${receipt.reason ?? "unknown reason"}`],
|
|
1278
|
+
},
|
|
1279
|
+
receipt,
|
|
1280
|
+
};
|
|
1281
|
+
}
|
|
1282
|
+
return { plan, receipt };
|
|
1283
|
+
}
|
|
1284
|
+
async acceptResult(result) {
|
|
1285
|
+
const state = await this.store.readState();
|
|
1286
|
+
if (!state?.activeIntent)
|
|
1287
|
+
throw new Error("No durable OpenMerit intent is active.");
|
|
1288
|
+
const config = await this.store.readConfig() ?? { schemaVersion: 1, taskProfiles: [] };
|
|
1289
|
+
const interpretation = interpretIntentResult(state.activeIntent, result, state.cycle, config);
|
|
1290
|
+
if (!interpretation.valid || !interpretation.nextStage) {
|
|
1291
|
+
throw new Error(`OpenMerit rejected the harness result: ${interpretation.errors.join("; ")}`);
|
|
1292
|
+
}
|
|
1293
|
+
if (interpretation.config)
|
|
1294
|
+
await this.store.writeConfig(interpretation.config);
|
|
1295
|
+
const now = new Date().toISOString();
|
|
1296
|
+
const evaluationSpend = result.status === "succeeded" && state.activeIntent.kind === "run_challenger_trials"
|
|
1297
|
+
? result.outputs.assessments.reduce((sum, assessment) => sum + assessment.metrics
|
|
1298
|
+
.filter((metric) => metric.metricId === "total_task_cost" && metric.state === "measured")
|
|
1299
|
+
.reduce((metricSum, metric) => metricSum + (metric.value ?? 0), 0), state.automation.evaluationSpend ?? 0)
|
|
1300
|
+
: state.automation.evaluationSpend;
|
|
1301
|
+
if (config.policy && evaluationSpend !== undefined &&
|
|
1302
|
+
evaluationSpend > config.policy.evaluationBudget.maximumSpend) {
|
|
1303
|
+
throw new Error(`OpenMerit rejected challenger results: evaluation spend ${evaluationSpend.toFixed(6)} ` +
|
|
1304
|
+
`${config.policy.evaluationBudget.currency} exceeds the approved limit ` +
|
|
1305
|
+
`${config.policy.evaluationBudget.maximumSpend.toFixed(6)}.`);
|
|
1306
|
+
}
|
|
1307
|
+
let observability = state.observability;
|
|
1308
|
+
if (result.status === "succeeded" && state.activeIntent.kind === "establish_evals") {
|
|
1309
|
+
observability = result.outputs.observability;
|
|
1310
|
+
}
|
|
1311
|
+
else if (result.status === "succeeded" && state.activeIntent.kind === "instrument_observability") {
|
|
1312
|
+
const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
1313
|
+
if (profile) {
|
|
1314
|
+
const output = result.outputs;
|
|
1315
|
+
const requiredMetricIds = profile.objectives.filter((item) => item.required).map((item) => item.metricId);
|
|
1316
|
+
observability = {
|
|
1317
|
+
targetId: profile.applicationTarget.id,
|
|
1318
|
+
status: "ready",
|
|
1319
|
+
requiredMetricIds,
|
|
1320
|
+
coveredMetricIds: output.coveredMetricIds,
|
|
1321
|
+
missingMetricIds: output.missingMetricIds,
|
|
1322
|
+
checkedAt: result.completedAt,
|
|
1323
|
+
};
|
|
1324
|
+
}
|
|
1325
|
+
}
|
|
1326
|
+
const nextState = {
|
|
1327
|
+
...state,
|
|
1328
|
+
cycle: {
|
|
1329
|
+
...state.cycle,
|
|
1330
|
+
stage: interpretation.nextStage,
|
|
1331
|
+
...interpretation.cyclePatch,
|
|
1332
|
+
},
|
|
1333
|
+
activeIntent: undefined,
|
|
1334
|
+
lastResult: result,
|
|
1335
|
+
...(result.status === "succeeded" && state.activeIntent.kind === "establish_evals"
|
|
1336
|
+
? { candidates: [result.outputs.incumbent] }
|
|
1337
|
+
: {}),
|
|
1338
|
+
...(result.status === "succeeded" && state.activeIntent.kind === "discover_candidates"
|
|
1339
|
+
? { candidates: result.outputs.candidates }
|
|
1340
|
+
: {}),
|
|
1341
|
+
...(result.status === "succeeded" && state.activeIntent.kind === "run_challenger_trials"
|
|
1342
|
+
? {
|
|
1343
|
+
assessments: result.outputs.assessments,
|
|
1344
|
+
experimentManifest: result.outputs.experimentManifest,
|
|
1345
|
+
}
|
|
1346
|
+
: {}),
|
|
1347
|
+
...(observability ? { observability } : {}),
|
|
1348
|
+
automation: {
|
|
1349
|
+
...state.automation,
|
|
1350
|
+
...(result.status === "succeeded" && ["establish_evals", "instrument_observability"].includes(state.activeIntent.kind) &&
|
|
1351
|
+
interpretation.nextStage === "collecting_baseline"
|
|
1352
|
+
? {
|
|
1353
|
+
baselineStartedAt: now,
|
|
1354
|
+
lastAssessmentAt: undefined,
|
|
1355
|
+
lastAssessmentTaskCount: state.automation.observedCompletedTasks,
|
|
1356
|
+
baselineMetricWindows: [],
|
|
1357
|
+
}
|
|
1358
|
+
: {}),
|
|
1359
|
+
...(evaluationSpend !== undefined ? { evaluationSpend } : {}),
|
|
1360
|
+
...(state.activeIntent.kind === "run_assessment"
|
|
1361
|
+
? { lastAssessmentTaskCount: state.automation.observedCompletedTasks, lastAssessmentAt: now }
|
|
1362
|
+
: {}),
|
|
1363
|
+
...(state.activeIntent.kind === "calculate_frontier"
|
|
1364
|
+
? { lastReassessmentTaskCount: state.automation.observedCompletedTasks, lastReassessmentAt: now }
|
|
1365
|
+
: {}),
|
|
1366
|
+
...(state.activeIntent.kind === "investigate_regression"
|
|
1367
|
+
? { lastRegressionCheckTaskCount: state.automation.observedCompletedTasks, lastRegressionCheckAt: now }
|
|
1368
|
+
: {}),
|
|
1369
|
+
...(state.activeIntent.kind === "apply_model_swap"
|
|
1370
|
+
? { swapAppliedTaskCount: state.automation.observedCompletedTasks, swapAppliedAt: now }
|
|
1371
|
+
: {}),
|
|
1372
|
+
...(state.activeIntent.kind === "verify_model_swap" ? { lastVerificationAt: now } : {}),
|
|
1373
|
+
},
|
|
1374
|
+
updatedAt: now,
|
|
1375
|
+
};
|
|
1376
|
+
await this.store.writeState(nextState);
|
|
1377
|
+
if (result.evidence.length && nextState.cycle.taskProfileId) {
|
|
1378
|
+
await this.store.writeEvidenceManifest({
|
|
1379
|
+
schemaVersion: 1,
|
|
1380
|
+
id: `manifest-${result.intentId}`,
|
|
1381
|
+
createdAt: now,
|
|
1382
|
+
taskProfileId: nextState.cycle.taskProfileId,
|
|
1383
|
+
references: result.evidence,
|
|
1384
|
+
});
|
|
1385
|
+
}
|
|
1386
|
+
await this.store.appendEvent({
|
|
1387
|
+
schemaVersion: 1,
|
|
1388
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1389
|
+
type: "intent_completed",
|
|
1390
|
+
occurredAt: now,
|
|
1391
|
+
cycleId: nextState.cycle.id,
|
|
1392
|
+
intentId: result.intentId,
|
|
1393
|
+
data: {
|
|
1394
|
+
kind: state.activeIntent.kind,
|
|
1395
|
+
status: result.status,
|
|
1396
|
+
nextStage: interpretation.nextStage,
|
|
1397
|
+
harnessId: this.adapter.descriptor.id,
|
|
1398
|
+
summary: result.summary,
|
|
1399
|
+
evidence: result.evidence.map((item) => ({ id: item.id, source: item.source })),
|
|
1400
|
+
},
|
|
1401
|
+
});
|
|
1402
|
+
if (observability && (state.activeIntent.kind === "establish_evals" || state.activeIntent.kind === "instrument_observability")) {
|
|
1403
|
+
await this.store.appendEvent({
|
|
1404
|
+
schemaVersion: 1,
|
|
1405
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1406
|
+
type: "observability_readiness_checked",
|
|
1407
|
+
occurredAt: now,
|
|
1408
|
+
cycleId: nextState.cycle.id,
|
|
1409
|
+
intentId: result.intentId,
|
|
1410
|
+
data: {
|
|
1411
|
+
targetId: observability.targetId,
|
|
1412
|
+
status: observability.status,
|
|
1413
|
+
requiredMetricIds: observability.requiredMetricIds,
|
|
1414
|
+
coveredMetricIds: observability.coveredMetricIds,
|
|
1415
|
+
missingMetricIds: observability.missingMetricIds,
|
|
1416
|
+
},
|
|
1417
|
+
});
|
|
1418
|
+
}
|
|
1419
|
+
const automation = await this.reconcileAutomation(now);
|
|
1420
|
+
return { intentKind: state.activeIntent.kind, state: nextState, interpretation, automation };
|
|
1421
|
+
}
|
|
1422
|
+
async recordLifecycle(update) {
|
|
1423
|
+
const state = await this.store.readState();
|
|
1424
|
+
if (!state?.activeIntent || state.activeIntent.id !== update.intentId) {
|
|
1425
|
+
throw new Error(`Lifecycle update ${update.intentId} does not match the active intent.`);
|
|
1426
|
+
}
|
|
1427
|
+
await this.store.appendEvent({
|
|
1428
|
+
schemaVersion: 1,
|
|
1429
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1430
|
+
type: "intent_progressed",
|
|
1431
|
+
occurredAt: update.occurredAt,
|
|
1432
|
+
cycleId: state.cycle.id,
|
|
1433
|
+
intentId: update.intentId,
|
|
1434
|
+
data: {
|
|
1435
|
+
status: update.status,
|
|
1436
|
+
...(update.message ? { message: update.message } : {}),
|
|
1437
|
+
...(update.progress ? { progress: update.progress } : {}),
|
|
1438
|
+
...(update.details ? { details: update.details } : {}),
|
|
1439
|
+
},
|
|
1440
|
+
});
|
|
1441
|
+
}
|
|
1442
|
+
async recordTaskObservation(details = {}) {
|
|
1443
|
+
const now = new Date().toISOString();
|
|
1444
|
+
const state = await this.store.readState();
|
|
1445
|
+
if (!state?.cycle.targetId)
|
|
1446
|
+
throw new Error("A confirmed application LLM target is required.");
|
|
1447
|
+
const result = await this.recordAutomationSignal({
|
|
1448
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
1449
|
+
id: `legacy-task-${crypto.randomUUID()}`,
|
|
1450
|
+
type: "product_task_completed",
|
|
1451
|
+
occurredAt: now,
|
|
1452
|
+
targetId: state.cycle.targetId,
|
|
1453
|
+
modelId: typeof details.modelId === "string" ? details.modelId : "unknown",
|
|
1454
|
+
});
|
|
1455
|
+
return {
|
|
1456
|
+
state: result.state,
|
|
1457
|
+
assessmentDue: result.state.automation.lastAssessmentAt === now,
|
|
1458
|
+
};
|
|
1459
|
+
}
|
|
1460
|
+
async assessBaseline(now = new Date().toISOString()) {
|
|
1461
|
+
const state = await this.store.readState();
|
|
1462
|
+
if (!state || state.cycle.stage !== "collecting_baseline" || state.activeIntent) {
|
|
1463
|
+
throw new Error("Baseline assessment requires an idle collecting-baseline cycle.");
|
|
1464
|
+
}
|
|
1465
|
+
const config = await this.store.readConfig();
|
|
1466
|
+
const profile = config?.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
1467
|
+
const errors = baselineAssessmentErrors(state, profile, state.automation.baselineMetricWindows ?? []);
|
|
1468
|
+
const sufficient = errors.length === 0;
|
|
1469
|
+
const decision = sufficient
|
|
1470
|
+
? { action: "issue_intent", intentKind: "discover_candidates", reason: "Required baseline metric windows meet the confirmed sample thresholds; discover candidates." }
|
|
1471
|
+
: { action: "observe", reason: `Baseline evidence remains insufficient: ${errors.join("; ")}.` };
|
|
1472
|
+
const nextState = {
|
|
1473
|
+
...state,
|
|
1474
|
+
cycle: { ...state.cycle, baselineEvidenceSufficient: sufficient, stage: sufficient ? "baseline_ready" : "collecting_baseline" },
|
|
1475
|
+
automation: {
|
|
1476
|
+
...state.automation,
|
|
1477
|
+
lastAssessmentAt: now,
|
|
1478
|
+
lastAssessmentTaskCount: state.automation.observedCompletedTasks,
|
|
1479
|
+
pendingIntentKind: decision.action === "issue_intent" ? decision.intentKind : undefined,
|
|
1480
|
+
pendingIntentReason: decision.action === "issue_intent" ? decision.reason : undefined,
|
|
1481
|
+
},
|
|
1482
|
+
updatedAt: now,
|
|
1483
|
+
};
|
|
1484
|
+
await this.store.writeState(nextState);
|
|
1485
|
+
await this.store.appendEvent({
|
|
1486
|
+
schemaVersion: 1, id: `event-${crypto.randomUUID()}`, type: "baseline_sufficiency_checked",
|
|
1487
|
+
occurredAt: now, cycleId: state.cycle.id,
|
|
1488
|
+
data: { sufficient, errors, metricIds: (state.automation.baselineMetricWindows ?? []).map((metric) => metric.metricId), source: "manual" },
|
|
1489
|
+
});
|
|
1490
|
+
return { state: nextState, errors, decision };
|
|
1491
|
+
}
|
|
1492
|
+
async recordAutomationSignal(signal) {
|
|
1493
|
+
const state = await this.store.readState();
|
|
1494
|
+
if (!state)
|
|
1495
|
+
throw new Error("OpenMerit project is not initialized.");
|
|
1496
|
+
if (signal.protocolVersion !== PROTOCOL_VERSION)
|
|
1497
|
+
throw new Error("Automation signal protocol version does not match.");
|
|
1498
|
+
if (!signal.id.trim())
|
|
1499
|
+
throw new Error("Automation signal ID is required.");
|
|
1500
|
+
if (!state.cycle.targetId || signal.targetId !== state.cycle.targetId) {
|
|
1501
|
+
throw new Error("Automation signal target does not match the confirmed application LLM target.");
|
|
1502
|
+
}
|
|
1503
|
+
if (signal.type === "metric_window_available" || signal.type === "verification_window_completed") {
|
|
1504
|
+
const mismatched = (signal.metrics ?? []).find((metric) => metric.targetId !== state.cycle.targetId);
|
|
1505
|
+
if (mismatched)
|
|
1506
|
+
throw new Error(`Metric ${mismatched.metricId} target does not match the confirmed application LLM target.`);
|
|
1507
|
+
}
|
|
1508
|
+
const processed = state.automation.processedSignalIds ?? [];
|
|
1509
|
+
if (processed.includes(signal.id)) {
|
|
1510
|
+
await this.store.appendEvent({
|
|
1511
|
+
schemaVersion: 1,
|
|
1512
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1513
|
+
type: "automation_signal_duplicate",
|
|
1514
|
+
occurredAt: signal.occurredAt,
|
|
1515
|
+
cycleId: state.cycle.id,
|
|
1516
|
+
data: { signalId: signal.id, signalType: signal.type },
|
|
1517
|
+
});
|
|
1518
|
+
return { state, duplicate: true, decision: dueDecision(state, signal, signal.occurredAt) };
|
|
1519
|
+
}
|
|
1520
|
+
const completedTasks = state.automation.observedCompletedTasks +
|
|
1521
|
+
(signal.type === "product_task_completed" ? 1 : 0);
|
|
1522
|
+
const baselineMetricWindows = state.cycle.stage === "collecting_baseline"
|
|
1523
|
+
? updatedBaselineWindows(state.automation.baselineMetricWindows, signal)
|
|
1524
|
+
: state.automation.baselineMetricWindows ?? [];
|
|
1525
|
+
const decisionState = {
|
|
1526
|
+
...state,
|
|
1527
|
+
automation: { ...state.automation, observedCompletedTasks: completedTasks, baselineMetricWindows },
|
|
1528
|
+
};
|
|
1529
|
+
const due = dueDecision(decisionState, signal, signal.occurredAt);
|
|
1530
|
+
const baselineCheckDue = due.action === "issue_intent" && due.intentKind === "run_assessment";
|
|
1531
|
+
const config = baselineCheckDue ? await this.store.readConfig() : undefined;
|
|
1532
|
+
const profile = config?.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
1533
|
+
const baselineErrors = baselineCheckDue ? baselineAssessmentErrors(state, profile, baselineMetricWindows) : [];
|
|
1534
|
+
const baselineReady = baselineCheckDue && baselineErrors.length === 0;
|
|
1535
|
+
const decision = baselineCheckDue
|
|
1536
|
+
? baselineReady
|
|
1537
|
+
? { action: "issue_intent", intentKind: "discover_candidates", reason: "Required baseline metric windows meet the confirmed sample thresholds; discover candidates." }
|
|
1538
|
+
: { action: "observe", reason: `Baseline evidence remains insufficient: ${baselineErrors.join("; ")}.` }
|
|
1539
|
+
: due;
|
|
1540
|
+
const latestMetricValues = { ...(state.automation.latestMetricValues ?? {}) };
|
|
1541
|
+
if (signal.type === "product_task_completed" && signal.telemetry) {
|
|
1542
|
+
latestMetricValues.input_token_consumption = signal.telemetry.inputTokens;
|
|
1543
|
+
latestMetricValues.output_token_consumption = signal.telemetry.outputTokens;
|
|
1544
|
+
latestMetricValues.total_task_cost = signal.telemetry.cost;
|
|
1545
|
+
latestMetricValues.end_to_end_task_latency = signal.telemetry.durationMs;
|
|
1546
|
+
}
|
|
1547
|
+
if (signal.type === "metric_window_available" || signal.type === "verification_window_completed") {
|
|
1548
|
+
for (const metric of signal.metrics ?? []) {
|
|
1549
|
+
if (metric.state === "measured" && metric.value !== undefined)
|
|
1550
|
+
latestMetricValues[metric.metricId] = metric.value;
|
|
1551
|
+
}
|
|
1552
|
+
}
|
|
1553
|
+
const nextState = {
|
|
1554
|
+
...decisionState,
|
|
1555
|
+
cycle: baselineCheckDue
|
|
1556
|
+
? { ...decisionState.cycle, baselineEvidenceSufficient: baselineReady, stage: baselineReady ? "baseline_ready" : "collecting_baseline" }
|
|
1557
|
+
: decisionState.cycle,
|
|
1558
|
+
automation: {
|
|
1559
|
+
...decisionState.automation,
|
|
1560
|
+
processedSignalIds: [...processed, signal.id].slice(-MAX_PROCESSED_SIGNAL_IDS),
|
|
1561
|
+
latestMetricValues,
|
|
1562
|
+
...(baselineCheckDue ? {
|
|
1563
|
+
lastAssessmentAt: signal.occurredAt,
|
|
1564
|
+
lastAssessmentTaskCount: completedTasks,
|
|
1565
|
+
pendingIntentKind: decision.action === "issue_intent" ? decision.intentKind : undefined,
|
|
1566
|
+
pendingIntentReason: decision.action === "issue_intent" ? decision.reason : undefined,
|
|
1567
|
+
} : {}),
|
|
1568
|
+
...(signal.type === "model_catalog_changed" ? {
|
|
1569
|
+
modelCatalogFingerprint: signal.fingerprint,
|
|
1570
|
+
...(state.automation.modelCatalogFingerprint && state.automation.modelCatalogFingerprint !== signal.fingerprint
|
|
1571
|
+
? { modelCatalogChangedAt: signal.occurredAt }
|
|
1572
|
+
: {}),
|
|
1573
|
+
} : {}),
|
|
1574
|
+
...(decision.action === "issue_intent" ? {
|
|
1575
|
+
pendingIntentKind: decision.intentKind,
|
|
1576
|
+
pendingIntentReason: decision.reason,
|
|
1577
|
+
} : {}),
|
|
1578
|
+
},
|
|
1579
|
+
updatedAt: signal.occurredAt,
|
|
1580
|
+
};
|
|
1581
|
+
await this.store.writeState(nextState);
|
|
1582
|
+
await this.store.appendEvent({
|
|
1583
|
+
schemaVersion: 1,
|
|
1584
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1585
|
+
type: "automation_signal_recorded",
|
|
1586
|
+
occurredAt: signal.occurredAt,
|
|
1587
|
+
cycleId: state.cycle.id,
|
|
1588
|
+
data: {
|
|
1589
|
+
signalId: signal.id,
|
|
1590
|
+
signalType: signal.type,
|
|
1591
|
+
decision: decision.action,
|
|
1592
|
+
decisionReason: decision.reason,
|
|
1593
|
+
...auditSignalData(signal),
|
|
1594
|
+
},
|
|
1595
|
+
});
|
|
1596
|
+
if (baselineCheckDue) {
|
|
1597
|
+
await this.store.appendEvent({
|
|
1598
|
+
schemaVersion: 1,
|
|
1599
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1600
|
+
type: "baseline_sufficiency_checked",
|
|
1601
|
+
occurredAt: signal.occurredAt,
|
|
1602
|
+
cycleId: state.cycle.id,
|
|
1603
|
+
data: {
|
|
1604
|
+
sufficient: baselineReady,
|
|
1605
|
+
errors: baselineErrors,
|
|
1606
|
+
metricIds: baselineMetricWindows.map((metric) => metric.metricId),
|
|
1607
|
+
sourceSignalId: signal.id,
|
|
1608
|
+
},
|
|
1609
|
+
});
|
|
1610
|
+
}
|
|
1611
|
+
if (decision.action === "issue_intent") {
|
|
1612
|
+
await this.store.appendEvent({
|
|
1613
|
+
schemaVersion: 1,
|
|
1614
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1615
|
+
type: "automation_check_due",
|
|
1616
|
+
occurredAt: signal.occurredAt,
|
|
1617
|
+
cycleId: state.cycle.id,
|
|
1618
|
+
data: { signalId: signal.id, intentKind: decision.intentKind, reason: decision.reason },
|
|
1619
|
+
});
|
|
1620
|
+
}
|
|
1621
|
+
return { state: nextState, duplicate: false, decision };
|
|
1622
|
+
}
|
|
1623
|
+
async recordModelCatalogFingerprint(fingerprint) {
|
|
1624
|
+
const state = await this.store.readState();
|
|
1625
|
+
if (!state)
|
|
1626
|
+
throw new Error("OpenMerit project is not initialized.");
|
|
1627
|
+
if (!state.cycle.targetId)
|
|
1628
|
+
throw new Error("A confirmed application LLM target is required.");
|
|
1629
|
+
const previousFingerprint = state.automation.modelCatalogFingerprint;
|
|
1630
|
+
const changed = previousFingerprint !== undefined && previousFingerprint !== fingerprint;
|
|
1631
|
+
if (previousFingerprint === fingerprint) {
|
|
1632
|
+
return { state, changed: false, reassessmentDue: false };
|
|
1633
|
+
}
|
|
1634
|
+
const now = new Date().toISOString();
|
|
1635
|
+
const result = await this.recordAutomationSignal({
|
|
1636
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
1637
|
+
id: `legacy-catalog-${crypto.randomUUID()}`,
|
|
1638
|
+
type: "model_catalog_changed",
|
|
1639
|
+
occurredAt: now,
|
|
1640
|
+
targetId: state.cycle.targetId,
|
|
1641
|
+
fingerprint,
|
|
1642
|
+
});
|
|
1643
|
+
return {
|
|
1644
|
+
state: result.state,
|
|
1645
|
+
changed,
|
|
1646
|
+
reassessmentDue: result.decision.action === "issue_intent" && result.decision.intentKind === "discover_candidates",
|
|
1647
|
+
};
|
|
1648
|
+
}
|
|
1649
|
+
}
|