openmerit 0.1.4 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/README.md +83 -372
- package/dist/core/src/index.d.ts +90 -0
- package/dist/core/src/index.js +1137 -0
- package/dist/core/src/store.d.ts +35 -0
- package/dist/core/src/store.js +102 -0
- package/dist/pi/src/index.d.ts +15 -0
- package/dist/pi/src/index.js +423 -0
- package/dist/protocol/src/index.d.ts +402 -0
- package/dist/protocol/src/index.js +47 -0
- package/dist/protocol/src/schemas.d.ts +450 -0
- package/dist/protocol/src/schemas.js +224 -0
- package/docs/adapter-guide.md +172 -0
- package/docs/architecture.md +55 -0
- package/docs/automation.md +66 -0
- package/docs/getting-started.md +55 -0
- package/docs/lifecycle.md +30 -0
- package/docs/metrics-and-evidence.md +40 -0
- package/docs/operations.md +31 -0
- package/docs/pareto-spec.md +76 -0
- package/docs/pi-extension.md +44 -0
- package/docs/roadmap.md +26 -0
- package/docs/security.md +23 -0
- package/docs/testing.md +36 -0
- package/docs/ux-reference.md +32 -0
- package/package.json +45 -54
- package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
- package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
- package/benchmark/invoice_ocr/data/manifest.json +0 -97
- package/dist/benchmarks.js +0 -98
- package/dist/catalog.js +0 -61
- package/dist/cli.js +0 -188
- package/dist/daemon.js +0 -407
- package/dist/diagnostics.js +0 -227
- package/dist/frontier.js +0 -56
- package/dist/harness.js +0 -1
- package/dist/integrations.js +0 -19
- package/dist/invoice-eval.js +0 -33
- package/dist/invoice-score.js +0 -124
- package/dist/judge.js +0 -43
- package/dist/llm.js +0 -207
- package/dist/pi-config.js +0 -46
- package/dist/pi-trials.js +0 -373
- package/dist/policy.js +0 -185
- package/dist/providers.js +0 -1
- package/dist/recommend.js +0 -76
- package/dist/routes.js +0 -74
- package/dist/standalone.js +0 -224
- package/dist/store.js +0 -89
- package/dist/strategist.js +0 -68
- package/dist/task-input.js +0 -54
- package/dist/traces.js +0 -127
- package/dist/trials.js +0 -140
- package/dist/types.js +0 -2
- package/examples/invoice-prompt.txt +0 -19
- package/examples/task.example.json +0 -7
- package/extension/openmerit.ts +0 -947
- package/instructions/OPENMERIT.md +0 -63
- package/instructions/openmerit.policy.json +0 -37
- package/rules.md +0 -43
|
@@ -0,0 +1,1137 @@
|
|
|
1
|
+
import { METRIC_CATALOG, OPENMERIT_SCHEMA_DIALECT, PROTOCOL_VERSION, REQUIRED_CAPABILITIES, validateIntentOutputSchema, } from "../../protocol/src/index.js";
|
|
2
|
+
export { OPENMERIT_DIRECTORY, ProjectStore, } from "./store.js";
|
|
3
|
+
const outcomes = {
|
|
4
|
+
establish_evals: "Infer the current task profile and metric objectives, ask the user to confirm requirements, evaluation budget, check cadence, thresholds, scheduling mode, and automation permissions, establish applicable evals and observability, verify them, and return the confirmed taskProfile and policy. Do not change model configuration.",
|
|
5
|
+
instrument_observability: "Establish and verify the missing observability required by the confirmed task profile.",
|
|
6
|
+
run_assessment: "Assess whether current baseline evidence satisfies every required metric and sample threshold. Return measured, missing, and insufficient-evidence metrics explicitly.",
|
|
7
|
+
discover_candidates: "Discover credible model candidates for the confirmed task profile using current capability, reputation, price, availability, and benchmark signals without exceeding the approved evaluation budget.",
|
|
8
|
+
run_challenger_trials: "Run comparable controlled challenger trials under the approved budget and return candidate assessments with durable evidence.",
|
|
9
|
+
calculate_frontier: "Calculate the Pareto frontier using the OpenMerit conformance definition, explain every candidate classification, and return a FrontierSnapshot with calculation evidence.",
|
|
10
|
+
investigate_regression: "Investigate the observed regression and return its likely cause, affected metrics, and supporting evidence.",
|
|
11
|
+
apply_model_swap: "Apply the approved model change in the harness or product target and return proof of the resulting configuration. Do not broaden the authorized change.",
|
|
12
|
+
verify_model_swap: "Verify the applied model against the approved post-swap evaluation window and report any material regression.",
|
|
13
|
+
rollback_model_swap: "Restore the previously verified model configuration and validate the rollback.",
|
|
14
|
+
};
|
|
15
|
+
export function createOpenMeritIntent(kind, options = {}) {
|
|
16
|
+
return {
|
|
17
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
18
|
+
id: options.id ?? `intent-${crypto.randomUUID()}`,
|
|
19
|
+
kind,
|
|
20
|
+
taskProfileId: options.taskProfileId ?? "current-project",
|
|
21
|
+
authorizationPolicyId: options.authorizationPolicyId ?? "observe-and-evaluate-only",
|
|
22
|
+
requestedOutcome: outcomes[kind],
|
|
23
|
+
constraints: {
|
|
24
|
+
modelMutationAllowed: kind === "apply_model_swap" || kind === "rollback_model_swap",
|
|
25
|
+
requireVerifiedArtifacts: true,
|
|
26
|
+
distinguishSingleRunFromMultiRunClaims: true,
|
|
27
|
+
...(options.policy ? { evaluationBudget: options.policy.evaluationBudget } : {}),
|
|
28
|
+
...(options.policy ? { checkPolicy: options.policy.checkPolicy } : {}),
|
|
29
|
+
},
|
|
30
|
+
requiredEvidence: kind === "establish_evals"
|
|
31
|
+
? [
|
|
32
|
+
{ id: "task-profile", description: "The inferred task profile and the user constraints it represents." },
|
|
33
|
+
{ id: "evaluation-artifacts", description: "Runnable evaluation artifacts and verification results." },
|
|
34
|
+
{ id: "observability-coverage", description: "The observable metrics and explicit coverage gaps." },
|
|
35
|
+
]
|
|
36
|
+
: kind === "calculate_frontier"
|
|
37
|
+
? [
|
|
38
|
+
{ id: "frontier-snapshot", description: "A complete harness-calculated FrontierSnapshot." },
|
|
39
|
+
{ id: "frontier-calculation", description: "Durable evidence of the frontier calculation." },
|
|
40
|
+
]
|
|
41
|
+
: [
|
|
42
|
+
{ id: "assessment-results", description: "Assessment results tied to the active task profile." },
|
|
43
|
+
{ id: "metric-coverage", description: "Observed, derived, and insufficient-evidence metric states." },
|
|
44
|
+
],
|
|
45
|
+
requestedAt: options.requestedAt ?? new Date().toISOString(),
|
|
46
|
+
};
|
|
47
|
+
}
|
|
48
|
+
export function missingHarnessCapabilities(descriptor, kind) {
|
|
49
|
+
if (!descriptor.protocolVersions.includes(PROTOCOL_VERSION))
|
|
50
|
+
return REQUIRED_CAPABILITIES[kind];
|
|
51
|
+
const available = new Set(descriptor.capabilities);
|
|
52
|
+
return REQUIRED_CAPABILITIES[kind].filter((capability) => !available.has(capability));
|
|
53
|
+
}
|
|
54
|
+
export function validateHarnessDescriptor(descriptor) {
|
|
55
|
+
const errors = [];
|
|
56
|
+
if (!descriptor.id.trim())
|
|
57
|
+
errors.push("harness ID is required");
|
|
58
|
+
if (!descriptor.name.trim())
|
|
59
|
+
errors.push("harness name is required");
|
|
60
|
+
if (!descriptor.version.trim())
|
|
61
|
+
errors.push("harness version is required");
|
|
62
|
+
if (!descriptor.protocolVersions.includes(PROTOCOL_VERSION)) {
|
|
63
|
+
errors.push(`harness does not support OpenMerit protocol ${PROTOCOL_VERSION}`);
|
|
64
|
+
}
|
|
65
|
+
if (new Set(descriptor.capabilities).size !== descriptor.capabilities.length) {
|
|
66
|
+
errors.push("harness capabilities contain duplicates");
|
|
67
|
+
}
|
|
68
|
+
if (descriptor.executionModes.length === 0)
|
|
69
|
+
errors.push("at least one execution mode is required");
|
|
70
|
+
if (!descriptor.structuredOutput) {
|
|
71
|
+
errors.push("harness must declare structured-output behavior");
|
|
72
|
+
}
|
|
73
|
+
else if (descriptor.structuredOutput.schemaDialect !== OPENMERIT_SCHEMA_DIALECT) {
|
|
74
|
+
errors.push(`harness does not support OpenMerit schema dialect ${OPENMERIT_SCHEMA_DIALECT}`);
|
|
75
|
+
}
|
|
76
|
+
if (!descriptor.automation || descriptor.automation.supportedSignals.length === 0) {
|
|
77
|
+
errors.push("harness must declare at least one supported automation signal");
|
|
78
|
+
}
|
|
79
|
+
else if (descriptor.automation.persistentScheduling && descriptor.automation.wakeupProvisioning === "none") {
|
|
80
|
+
errors.push("persistent scheduling requires a wakeup provisioning mechanism");
|
|
81
|
+
}
|
|
82
|
+
return { valid: errors.length === 0, errors };
|
|
83
|
+
}
|
|
84
|
+
export async function dispatchToHarness(adapter, intent) {
|
|
85
|
+
const descriptorValidation = validateHarnessDescriptor(adapter.descriptor);
|
|
86
|
+
if (!descriptorValidation.valid) {
|
|
87
|
+
return { accepted: false, reason: descriptorValidation.errors.join("; ") };
|
|
88
|
+
}
|
|
89
|
+
const missing = missingHarnessCapabilities(adapter.descriptor, intent.kind);
|
|
90
|
+
if (missing.length) {
|
|
91
|
+
return { accepted: false, reason: `Harness lacks required capabilities: ${missing.join(", ")}` };
|
|
92
|
+
}
|
|
93
|
+
return adapter.dispatch(intent);
|
|
94
|
+
}
|
|
95
|
+
export function startedStage(kind) {
|
|
96
|
+
switch (kind) {
|
|
97
|
+
case "establish_evals":
|
|
98
|
+
case "instrument_observability": return "establishing_evidence";
|
|
99
|
+
case "run_assessment": return "collecting_baseline";
|
|
100
|
+
case "discover_candidates": return "discovering_candidates";
|
|
101
|
+
case "run_challenger_trials": return "running_challenger_trials";
|
|
102
|
+
case "calculate_frontier": return "calculating_frontier";
|
|
103
|
+
case "apply_model_swap": return "applying_swap";
|
|
104
|
+
case "verify_model_swap": return "verifying_swap";
|
|
105
|
+
case "rollback_model_swap": return "rolling_back";
|
|
106
|
+
case "investigate_regression": return "monitoring";
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
export function completedStage(kind, succeeded) {
|
|
110
|
+
if (!succeeded)
|
|
111
|
+
return kind === "verify_model_swap" ? "verification_failed" : "monitoring";
|
|
112
|
+
switch (kind) {
|
|
113
|
+
case "establish_evals":
|
|
114
|
+
case "instrument_observability": return "collecting_baseline";
|
|
115
|
+
case "run_assessment": return "baseline_ready";
|
|
116
|
+
case "discover_candidates": return "candidates_ready";
|
|
117
|
+
case "run_challenger_trials": return "challengers_ready";
|
|
118
|
+
case "calculate_frontier": return "frontier_ready";
|
|
119
|
+
case "apply_model_swap": return "swap_applied";
|
|
120
|
+
case "verify_model_swap": return "swap_verified";
|
|
121
|
+
case "rollback_model_swap":
|
|
122
|
+
case "investigate_regression": return "monitoring";
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
const knownMetricIds = new Set(METRIC_CATALOG.map((metric) => metric.id));
|
|
126
|
+
function isRecord(value) {
|
|
127
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
128
|
+
}
|
|
129
|
+
function validateSetupOutput(output) {
|
|
130
|
+
if (!isRecord(output) || !isRecord(output.taskProfile) || !isRecord(output.policy)) {
|
|
131
|
+
return ["setup output requires taskProfile and policy"];
|
|
132
|
+
}
|
|
133
|
+
const profile = output.taskProfile;
|
|
134
|
+
const policy = output.policy;
|
|
135
|
+
const errors = [];
|
|
136
|
+
if (typeof profile.id !== "string" || typeof profile.name !== "string" ||
|
|
137
|
+
typeof profile.goal !== "string" || !Number.isInteger(profile.revision) ||
|
|
138
|
+
typeof profile.confirmedAt !== "string" || !Array.isArray(profile.objectives) ||
|
|
139
|
+
profile.objectives.length === 0) {
|
|
140
|
+
errors.push("taskProfile must be confirmed and contain at least one objective");
|
|
141
|
+
}
|
|
142
|
+
else {
|
|
143
|
+
for (const objective of profile.objectives) {
|
|
144
|
+
if (!isRecord(objective) || typeof objective.metricId !== "string" ||
|
|
145
|
+
!knownMetricIds.has(objective.metricId) || typeof objective.required !== "boolean" ||
|
|
146
|
+
!Number.isInteger(objective.minimumSamples) || objective.minimumSamples < 1 ||
|
|
147
|
+
typeof objective.tolerance !== "number") {
|
|
148
|
+
errors.push("taskProfile contains an invalid metric objective");
|
|
149
|
+
break;
|
|
150
|
+
}
|
|
151
|
+
}
|
|
152
|
+
}
|
|
153
|
+
const budget = policy.evaluationBudget;
|
|
154
|
+
if (policy.mode !== "supervised_graduation" || !isRecord(budget) ||
|
|
155
|
+
typeof budget.currency !== "string" || typeof budget.maximumSpend !== "number" ||
|
|
156
|
+
budget.maximumSpend < 0 || typeof policy.automaticSwapsEnabled !== "boolean" ||
|
|
157
|
+
!Number.isInteger(policy.confirmationRequiredUntilVerifiedSwaps) ||
|
|
158
|
+
policy.requirePostSwapVerification !== true || typeof policy.rollbackOnRegression !== "boolean") {
|
|
159
|
+
errors.push("policy must be a valid supervised-graduation policy with an evaluation budget");
|
|
160
|
+
}
|
|
161
|
+
if (!isRecord(output.observability)) {
|
|
162
|
+
errors.push("setup output requires an observability coverage report");
|
|
163
|
+
}
|
|
164
|
+
else if (Array.isArray(profile.objectives)) {
|
|
165
|
+
errors.push(...validateObservabilityCoverage(profile, output.observability, false));
|
|
166
|
+
}
|
|
167
|
+
return errors;
|
|
168
|
+
}
|
|
169
|
+
export function validateObservabilityCoverage(profile, coverage, requireReady) {
|
|
170
|
+
const required = new Set(profile.objectives.filter((item) => item.required).map((item) => item.metricId));
|
|
171
|
+
const declaredRequired = new Set(coverage.requiredMetricIds);
|
|
172
|
+
const covered = new Set(coverage.coveredMetricIds);
|
|
173
|
+
const missing = new Set(coverage.missingMetricIds);
|
|
174
|
+
const errors = [];
|
|
175
|
+
for (const metricId of required) {
|
|
176
|
+
if (!declaredRequired.has(metricId))
|
|
177
|
+
errors.push(`observability omits required metric ${metricId}`);
|
|
178
|
+
if (!covered.has(metricId) && !missing.has(metricId))
|
|
179
|
+
errors.push(`observability does not classify required metric ${metricId}`);
|
|
180
|
+
}
|
|
181
|
+
for (const metricId of covered) {
|
|
182
|
+
if (missing.has(metricId))
|
|
183
|
+
errors.push(`observability marks ${metricId} as both covered and missing`);
|
|
184
|
+
}
|
|
185
|
+
const requiredMissing = [...required].filter((metricId) => !covered.has(metricId));
|
|
186
|
+
const expectedStatus = requiredMissing.length === 0 ? "ready" : "incomplete";
|
|
187
|
+
if (coverage.status !== expectedStatus)
|
|
188
|
+
errors.push(`observability status must be ${expectedStatus}`);
|
|
189
|
+
if (requireReady && requiredMissing.length) {
|
|
190
|
+
errors.push(`required observability remains missing: ${requiredMissing.join(", ")}`);
|
|
191
|
+
}
|
|
192
|
+
return errors;
|
|
193
|
+
}
|
|
194
|
+
export function interpretIntentResult(intent, result, cycle, config) {
|
|
195
|
+
const errors = [];
|
|
196
|
+
if (result.protocolVersion !== PROTOCOL_VERSION)
|
|
197
|
+
errors.push("result protocol version does not match");
|
|
198
|
+
if (result.intentId !== intent.id)
|
|
199
|
+
errors.push("result intent ID does not match");
|
|
200
|
+
if (result.status === "succeeded" && result.evidence.length === 0)
|
|
201
|
+
errors.push("successful result requires evidence");
|
|
202
|
+
if (result.status !== "succeeded") {
|
|
203
|
+
return {
|
|
204
|
+
valid: errors.length === 0,
|
|
205
|
+
errors,
|
|
206
|
+
nextStage: completedStage(intent.kind, false),
|
|
207
|
+
};
|
|
208
|
+
}
|
|
209
|
+
const output = result.outputs;
|
|
210
|
+
errors.push(...validateIntentOutputSchema(intent.kind, output).map((error) => `result outputs ${error}`));
|
|
211
|
+
if (errors.length) {
|
|
212
|
+
return { valid: false, errors, nextStage: completedStage(intent.kind, false) };
|
|
213
|
+
}
|
|
214
|
+
let nextStage = completedStage(intent.kind, true);
|
|
215
|
+
let nextConfig;
|
|
216
|
+
let cyclePatch;
|
|
217
|
+
switch (intent.kind) {
|
|
218
|
+
case "establish_evals": {
|
|
219
|
+
errors.push(...validateSetupOutput(output));
|
|
220
|
+
if (!errors.length && isRecord(output)) {
|
|
221
|
+
const taskProfile = output.taskProfile;
|
|
222
|
+
const policy = output.policy;
|
|
223
|
+
const observability = output.observability;
|
|
224
|
+
nextConfig = {
|
|
225
|
+
schemaVersion: 1,
|
|
226
|
+
activeTaskProfileId: taskProfile.id,
|
|
227
|
+
taskProfiles: [...config.taskProfiles.filter((item) => item.id !== taskProfile.id), taskProfile],
|
|
228
|
+
policy,
|
|
229
|
+
};
|
|
230
|
+
cyclePatch = {
|
|
231
|
+
taskProfileId: taskProfile.id,
|
|
232
|
+
policy,
|
|
233
|
+
observabilityReady: observability.status === "ready",
|
|
234
|
+
};
|
|
235
|
+
if (observability.status !== "ready")
|
|
236
|
+
nextStage = "establishing_evidence";
|
|
237
|
+
}
|
|
238
|
+
break;
|
|
239
|
+
}
|
|
240
|
+
case "instrument_observability":
|
|
241
|
+
if (!isRecord(output) || !Array.isArray(output.coveredMetricIds) || !Array.isArray(output.missingMetricIds)) {
|
|
242
|
+
errors.push("observability result requires coveredMetricIds and missingMetricIds");
|
|
243
|
+
}
|
|
244
|
+
else {
|
|
245
|
+
const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
246
|
+
if (!profile)
|
|
247
|
+
errors.push("no confirmed task profile exists for observability verification");
|
|
248
|
+
else {
|
|
249
|
+
const coverage = {
|
|
250
|
+
status: output.missingMetricIds.length ? "incomplete" : "ready",
|
|
251
|
+
requiredMetricIds: profile.objectives.filter((item) => item.required).map((item) => item.metricId),
|
|
252
|
+
coveredMetricIds: output.coveredMetricIds,
|
|
253
|
+
missingMetricIds: output.missingMetricIds,
|
|
254
|
+
checkedAt: result.completedAt,
|
|
255
|
+
};
|
|
256
|
+
errors.push(...validateObservabilityCoverage(profile, coverage, true));
|
|
257
|
+
cyclePatch = { observabilityReady: errors.length === 0 };
|
|
258
|
+
}
|
|
259
|
+
}
|
|
260
|
+
break;
|
|
261
|
+
case "run_assessment":
|
|
262
|
+
if (!isRecord(output) || typeof output.baselineEvidenceSufficient !== "boolean") {
|
|
263
|
+
errors.push("assessment result requires baselineEvidenceSufficient");
|
|
264
|
+
}
|
|
265
|
+
else {
|
|
266
|
+
cyclePatch = { baselineEvidenceSufficient: output.baselineEvidenceSufficient };
|
|
267
|
+
if (!output.baselineEvidenceSufficient)
|
|
268
|
+
nextStage = "collecting_baseline";
|
|
269
|
+
}
|
|
270
|
+
break;
|
|
271
|
+
case "discover_candidates":
|
|
272
|
+
if (!isRecord(output) || !Array.isArray(output.candidates))
|
|
273
|
+
errors.push("candidate discovery requires candidates");
|
|
274
|
+
break;
|
|
275
|
+
case "run_challenger_trials":
|
|
276
|
+
if (!isRecord(output) || !Array.isArray(output.assessments))
|
|
277
|
+
errors.push("challenger trials require assessments");
|
|
278
|
+
break;
|
|
279
|
+
case "calculate_frontier": {
|
|
280
|
+
if (!isRecord(output) || !Array.isArray(output.assessments) || !isRecord(output.frontierSnapshot)) {
|
|
281
|
+
errors.push("frontier result requires assessments and frontierSnapshot");
|
|
282
|
+
break;
|
|
283
|
+
}
|
|
284
|
+
const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
285
|
+
if (!profile) {
|
|
286
|
+
errors.push("no confirmed task profile exists for frontier verification");
|
|
287
|
+
break;
|
|
288
|
+
}
|
|
289
|
+
const verification = verifyFrontierSnapshot(profile, output.assessments, output.frontierSnapshot);
|
|
290
|
+
errors.push(...verification.errors);
|
|
291
|
+
if (output.recommendation !== undefined) {
|
|
292
|
+
if (!isRecord(output.recommendation) || typeof output.recommendation.selectedCandidateId !== "string") {
|
|
293
|
+
errors.push("frontier recommendation is invalid");
|
|
294
|
+
}
|
|
295
|
+
else if (!output.frontierSnapshot.frontierCandidateIds.includes(output.recommendation.selectedCandidateId)) {
|
|
296
|
+
errors.push("recommended candidate is not on the verified frontier");
|
|
297
|
+
}
|
|
298
|
+
else {
|
|
299
|
+
cyclePatch = { proposedCandidateId: output.recommendation.selectedCandidateId };
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
break;
|
|
303
|
+
}
|
|
304
|
+
case "investigate_regression":
|
|
305
|
+
if (!isRecord(output) || typeof output.regressionDetected !== "boolean" || !Array.isArray(output.affectedMetricIds)) {
|
|
306
|
+
errors.push("regression result requires regressionDetected and affectedMetricIds");
|
|
307
|
+
}
|
|
308
|
+
break;
|
|
309
|
+
case "apply_model_swap":
|
|
310
|
+
if (!isRecord(output) || typeof output.appliedCandidateId !== "string") {
|
|
311
|
+
errors.push("swap result requires appliedCandidateId");
|
|
312
|
+
}
|
|
313
|
+
else if (cycle.proposedCandidateId && output.appliedCandidateId !== cycle.proposedCandidateId) {
|
|
314
|
+
errors.push("applied candidate does not match the authorized proposal");
|
|
315
|
+
}
|
|
316
|
+
else {
|
|
317
|
+
cyclePatch = {
|
|
318
|
+
previousCandidateId: cycle.activeCandidateId,
|
|
319
|
+
activeCandidateId: output.appliedCandidateId,
|
|
320
|
+
};
|
|
321
|
+
}
|
|
322
|
+
break;
|
|
323
|
+
case "verify_model_swap":
|
|
324
|
+
if (!isRecord(output) || typeof output.verified !== "boolean" || typeof output.regressionDetected !== "boolean") {
|
|
325
|
+
errors.push("swap verification requires verified and regressionDetected");
|
|
326
|
+
}
|
|
327
|
+
else if (!output.verified || output.regressionDetected) {
|
|
328
|
+
nextStage = "verification_failed";
|
|
329
|
+
}
|
|
330
|
+
else {
|
|
331
|
+
cyclePatch = { verifiedSwapCount: cycle.verifiedSwapCount + 1 };
|
|
332
|
+
}
|
|
333
|
+
break;
|
|
334
|
+
case "rollback_model_swap":
|
|
335
|
+
if (!isRecord(output) || typeof output.restoredCandidateId !== "string") {
|
|
336
|
+
errors.push("rollback result requires restoredCandidateId");
|
|
337
|
+
}
|
|
338
|
+
else if (cycle.previousCandidateId && output.restoredCandidateId !== cycle.previousCandidateId) {
|
|
339
|
+
errors.push("rollback did not restore the previous verified candidate");
|
|
340
|
+
}
|
|
341
|
+
else {
|
|
342
|
+
cyclePatch = { activeCandidateId: output.restoredCandidateId };
|
|
343
|
+
}
|
|
344
|
+
break;
|
|
345
|
+
}
|
|
346
|
+
return {
|
|
347
|
+
valid: errors.length === 0,
|
|
348
|
+
errors,
|
|
349
|
+
...(errors.length ? {} : { nextStage, config: nextConfig, cyclePatch }),
|
|
350
|
+
};
|
|
351
|
+
}
|
|
352
|
+
export function nextImprovementAction(state) {
|
|
353
|
+
switch (state.stage) {
|
|
354
|
+
case "unconfigured":
|
|
355
|
+
return { action: "issue_intent", intentKind: "establish_evals", reason: "Task profile, evaluations, and observability are not configured." };
|
|
356
|
+
case "establishing_evidence":
|
|
357
|
+
return state.observabilityReady
|
|
358
|
+
? { action: "observe", reason: "The harness is establishing evaluations and observability." }
|
|
359
|
+
: { action: "issue_intent", intentKind: "instrument_observability", reason: "Required metric instrumentation is incomplete." };
|
|
360
|
+
case "collecting_baseline":
|
|
361
|
+
return state.baselineEvidenceSufficient
|
|
362
|
+
? { action: "issue_intent", intentKind: "run_assessment", reason: "Baseline evidence has reached the approved sufficiency threshold." }
|
|
363
|
+
: { action: "observe", reason: "Continue collecting baseline production evidence." };
|
|
364
|
+
case "baseline_ready":
|
|
365
|
+
return { action: "issue_intent", intentKind: "discover_candidates", reason: "The baseline is ready for budgeted candidate discovery." };
|
|
366
|
+
case "discovering_candidates":
|
|
367
|
+
case "running_challenger_trials":
|
|
368
|
+
case "calculating_frontier":
|
|
369
|
+
case "applying_swap":
|
|
370
|
+
case "verifying_swap":
|
|
371
|
+
case "rolling_back":
|
|
372
|
+
return { action: "observe", reason: "Harness work is in progress." };
|
|
373
|
+
case "candidates_ready":
|
|
374
|
+
return { action: "issue_intent", intentKind: "run_challenger_trials", reason: "Candidate models are ready for controlled comparison." };
|
|
375
|
+
case "challengers_ready":
|
|
376
|
+
return { action: "issue_intent", intentKind: "calculate_frontier", reason: "Comparable challenger evidence is ready." };
|
|
377
|
+
case "frontier_ready": {
|
|
378
|
+
if (!state.proposedCandidateId || state.proposedCandidateId === state.activeCandidateId) {
|
|
379
|
+
return { action: "observe", reason: "The harness did not identify a materially better frontier candidate." };
|
|
380
|
+
}
|
|
381
|
+
if (!state.policy)
|
|
382
|
+
return { action: "await_user", reason: "A swap policy must be approved before changing the model." };
|
|
383
|
+
const graduated = state.policy.automaticSwapsEnabled &&
|
|
384
|
+
state.verifiedSwapCount >= state.policy.confirmationRequiredUntilVerifiedSwaps;
|
|
385
|
+
return graduated
|
|
386
|
+
? { action: "issue_intent", intentKind: "apply_model_swap", reason: "Bounded automatic swapping is enabled and the graduation threshold is satisfied." }
|
|
387
|
+
: { action: "await_user", reason: "User confirmation is required for this model swap." };
|
|
388
|
+
}
|
|
389
|
+
case "awaiting_swap_approval":
|
|
390
|
+
return { action: "await_user", reason: "Waiting for the user to approve or reject the proposed model swap." };
|
|
391
|
+
case "swap_approved":
|
|
392
|
+
return { action: "issue_intent", intentKind: "apply_model_swap", reason: "The user approved the proposed model swap." };
|
|
393
|
+
case "swap_applied":
|
|
394
|
+
return { action: "observe", reason: "Collect the user-confirmed post-swap verification window before verification." };
|
|
395
|
+
case "swap_verified":
|
|
396
|
+
case "monitoring":
|
|
397
|
+
return { action: "observe", reason: "Monitor production evidence until reassessment is due." };
|
|
398
|
+
case "verification_failed":
|
|
399
|
+
return state.policy?.rollbackOnRegression
|
|
400
|
+
? { action: "issue_intent", intentKind: "rollback_model_swap", reason: "Post-swap verification detected a regression and rollback is required." }
|
|
401
|
+
: { action: "await_user", reason: "Post-swap verification failed and automatic rollback is not authorized." };
|
|
402
|
+
}
|
|
403
|
+
}
|
|
404
|
+
function metricFor(assessment, objective) {
|
|
405
|
+
return assessment.metrics.find((metric) => metric.metricId === objective.metricId);
|
|
406
|
+
}
|
|
407
|
+
function bounds(metric) {
|
|
408
|
+
if (metric.state !== "measured" || metric.value === undefined)
|
|
409
|
+
return undefined;
|
|
410
|
+
return metric.interval ?? { lower: metric.value, upper: metric.value };
|
|
411
|
+
}
|
|
412
|
+
function readinessErrors(profile, assessment) {
|
|
413
|
+
const errors = [];
|
|
414
|
+
for (const objective of profile.objectives) {
|
|
415
|
+
if (!objective.required)
|
|
416
|
+
continue;
|
|
417
|
+
const metric = metricFor(assessment, objective);
|
|
418
|
+
if (!metric)
|
|
419
|
+
errors.push(`${objective.metricId}: missing`);
|
|
420
|
+
else if (metric.state !== "measured")
|
|
421
|
+
errors.push(`${objective.metricId}: ${metric.state}`);
|
|
422
|
+
else if (metric.value === undefined)
|
|
423
|
+
errors.push(`${objective.metricId}: no numeric value`);
|
|
424
|
+
else if (metric.sampleCount < objective.minimumSamples) {
|
|
425
|
+
errors.push(`${objective.metricId}: ${metric.sampleCount}/${objective.minimumSamples} samples`);
|
|
426
|
+
}
|
|
427
|
+
}
|
|
428
|
+
return errors;
|
|
429
|
+
}
|
|
430
|
+
function feasibility(profile, assessment) {
|
|
431
|
+
if (readinessErrors(profile, assessment).length)
|
|
432
|
+
return "unresolved";
|
|
433
|
+
let unresolved = false;
|
|
434
|
+
for (const objective of profile.objectives) {
|
|
435
|
+
if (!objective.constraint)
|
|
436
|
+
continue;
|
|
437
|
+
const range = bounds(metricFor(assessment, objective));
|
|
438
|
+
if (objective.constraint.operator === "at_least") {
|
|
439
|
+
if (range.upper < objective.constraint.value)
|
|
440
|
+
return "ineligible";
|
|
441
|
+
if (range.lower < objective.constraint.value)
|
|
442
|
+
unresolved = true;
|
|
443
|
+
}
|
|
444
|
+
else {
|
|
445
|
+
if (range.lower > objective.constraint.value)
|
|
446
|
+
return "ineligible";
|
|
447
|
+
if (range.upper > objective.constraint.value)
|
|
448
|
+
unresolved = true;
|
|
449
|
+
}
|
|
450
|
+
}
|
|
451
|
+
return unresolved ? "unresolved" : "eligible";
|
|
452
|
+
}
|
|
453
|
+
function expectedComparison(profile, candidate, other) {
|
|
454
|
+
const reasons = [];
|
|
455
|
+
let strictlyBetter = false;
|
|
456
|
+
let uncertain = false;
|
|
457
|
+
for (const objective of profile.objectives) {
|
|
458
|
+
const aMetric = metricFor(candidate, objective);
|
|
459
|
+
const bMetric = metricFor(other, objective);
|
|
460
|
+
const a = aMetric && bounds(aMetric);
|
|
461
|
+
const b = bMetric && bounds(bMetric);
|
|
462
|
+
if (!aMetric || !bMetric || !a || !b || aMetric.unit !== bMetric.unit || aMetric.direction !== bMetric.direction) {
|
|
463
|
+
return {
|
|
464
|
+
candidateId: candidate.candidateId,
|
|
465
|
+
otherCandidateId: other.candidateId,
|
|
466
|
+
state: "unresolved",
|
|
467
|
+
reasons: [`${objective.metricId}: incompatible or insufficient evidence`],
|
|
468
|
+
};
|
|
469
|
+
}
|
|
470
|
+
const tolerance = objective.tolerance;
|
|
471
|
+
if (aMetric.direction === "maximize") {
|
|
472
|
+
if (a.lower + tolerance >= b.upper) {
|
|
473
|
+
if (a.lower > b.upper + tolerance)
|
|
474
|
+
strictlyBetter = true;
|
|
475
|
+
}
|
|
476
|
+
else if (a.upper + tolerance < b.lower) {
|
|
477
|
+
reasons.push(`${objective.metricId}: proven worse`);
|
|
478
|
+
return { candidateId: candidate.candidateId, otherCandidateId: other.candidateId, state: "does_not_dominate", reasons };
|
|
479
|
+
}
|
|
480
|
+
else
|
|
481
|
+
uncertain = true;
|
|
482
|
+
}
|
|
483
|
+
else {
|
|
484
|
+
if (a.upper <= b.lower + tolerance) {
|
|
485
|
+
if (a.upper + tolerance < b.lower)
|
|
486
|
+
strictlyBetter = true;
|
|
487
|
+
}
|
|
488
|
+
else if (a.lower > b.upper + tolerance) {
|
|
489
|
+
reasons.push(`${objective.metricId}: proven worse`);
|
|
490
|
+
return { candidateId: candidate.candidateId, otherCandidateId: other.candidateId, state: "does_not_dominate", reasons };
|
|
491
|
+
}
|
|
492
|
+
else
|
|
493
|
+
uncertain = true;
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
if (uncertain) {
|
|
497
|
+
return {
|
|
498
|
+
candidateId: candidate.candidateId,
|
|
499
|
+
otherCandidateId: other.candidateId,
|
|
500
|
+
state: "unresolved",
|
|
501
|
+
reasons: ["one or more objective intervals do not establish ordering"],
|
|
502
|
+
};
|
|
503
|
+
}
|
|
504
|
+
if (!strictlyBetter) {
|
|
505
|
+
return {
|
|
506
|
+
candidateId: candidate.candidateId,
|
|
507
|
+
otherCandidateId: other.candidateId,
|
|
508
|
+
state: "does_not_dominate",
|
|
509
|
+
reasons: ["no objective is materially better"],
|
|
510
|
+
};
|
|
511
|
+
}
|
|
512
|
+
return {
|
|
513
|
+
candidateId: candidate.candidateId,
|
|
514
|
+
otherCandidateId: other.candidateId,
|
|
515
|
+
state: "dominates",
|
|
516
|
+
reasons: ["proven no worse on every objective and materially better on at least one"],
|
|
517
|
+
};
|
|
518
|
+
}
|
|
519
|
+
function sameMembers(actual, expected) {
|
|
520
|
+
return actual.length === expected.length && [...actual].sort().every((item, index) => item === [...expected].sort()[index]);
|
|
521
|
+
}
|
|
522
|
+
export function verifyFrontierSnapshot(profile, assessments, snapshot) {
|
|
523
|
+
const errors = [];
|
|
524
|
+
const byCandidate = new Map(assessments.map((assessment) => [assessment.candidateId, assessment]));
|
|
525
|
+
if (snapshot.taskProfileId !== profile.id || snapshot.taskProfileRevision !== profile.revision) {
|
|
526
|
+
errors.push("snapshot task profile does not match");
|
|
527
|
+
}
|
|
528
|
+
if (snapshot.calculationEvidence.length === 0)
|
|
529
|
+
errors.push("calculation evidence is required");
|
|
530
|
+
if (!sameMembers(snapshot.assessmentIds, assessments.map((assessment) => assessment.id))) {
|
|
531
|
+
errors.push("snapshot assessment IDs do not match the supplied assessments");
|
|
532
|
+
}
|
|
533
|
+
const eligible = [];
|
|
534
|
+
const expectedIneligible = [];
|
|
535
|
+
const expectedUnresolved = new Set();
|
|
536
|
+
for (const assessment of assessments) {
|
|
537
|
+
if (assessment.taskProfileId !== profile.id || assessment.taskProfileRevision !== profile.revision) {
|
|
538
|
+
errors.push(`${assessment.candidateId}: task profile revision mismatch`);
|
|
539
|
+
continue;
|
|
540
|
+
}
|
|
541
|
+
if (assessment.productRevision !== snapshot.productRevision) {
|
|
542
|
+
errors.push(`${assessment.candidateId}: product revision mismatch`);
|
|
543
|
+
continue;
|
|
544
|
+
}
|
|
545
|
+
const state = feasibility(profile, assessment);
|
|
546
|
+
if (state === "eligible")
|
|
547
|
+
eligible.push(assessment);
|
|
548
|
+
else if (state === "ineligible")
|
|
549
|
+
expectedIneligible.push(assessment.candidateId);
|
|
550
|
+
else
|
|
551
|
+
expectedUnresolved.add(assessment.candidateId);
|
|
552
|
+
}
|
|
553
|
+
const expectedDominated = new Set();
|
|
554
|
+
const expectedComparisons = [];
|
|
555
|
+
for (const candidate of eligible) {
|
|
556
|
+
for (const other of eligible) {
|
|
557
|
+
if (candidate.candidateId === other.candidateId)
|
|
558
|
+
continue;
|
|
559
|
+
const comparison = expectedComparison(profile, candidate, other);
|
|
560
|
+
expectedComparisons.push(comparison);
|
|
561
|
+
if (comparison.state === "dominates")
|
|
562
|
+
expectedDominated.add(other.candidateId);
|
|
563
|
+
if (comparison.state === "unresolved") {
|
|
564
|
+
expectedUnresolved.add(candidate.candidateId);
|
|
565
|
+
expectedUnresolved.add(other.candidateId);
|
|
566
|
+
}
|
|
567
|
+
}
|
|
568
|
+
}
|
|
569
|
+
for (const expected of expectedComparisons) {
|
|
570
|
+
const actual = snapshot.comparisons.find((comparison) => comparison.candidateId === expected.candidateId && comparison.otherCandidateId === expected.otherCandidateId);
|
|
571
|
+
if (!actual)
|
|
572
|
+
errors.push(`missing comparison ${expected.candidateId} -> ${expected.otherCandidateId}`);
|
|
573
|
+
else if (actual.state !== expected.state) {
|
|
574
|
+
errors.push(`comparison ${expected.candidateId} -> ${expected.otherCandidateId} should be ${expected.state}, got ${actual.state}`);
|
|
575
|
+
}
|
|
576
|
+
else if (actual.reasons.length === 0) {
|
|
577
|
+
errors.push(`comparison ${expected.candidateId} -> ${expected.otherCandidateId} requires reasons`);
|
|
578
|
+
}
|
|
579
|
+
}
|
|
580
|
+
const expectedFrontier = eligible
|
|
581
|
+
.map((assessment) => assessment.candidateId)
|
|
582
|
+
.filter((candidateId) => !expectedDominated.has(candidateId) && !expectedUnresolved.has(candidateId));
|
|
583
|
+
if (!sameMembers(snapshot.frontierCandidateIds, expectedFrontier))
|
|
584
|
+
errors.push("frontier candidate set is incorrect");
|
|
585
|
+
if (!sameMembers(snapshot.dominatedCandidateIds, [...expectedDominated]))
|
|
586
|
+
errors.push("dominated candidate set is incorrect");
|
|
587
|
+
if (!sameMembers(snapshot.ineligibleCandidateIds, expectedIneligible))
|
|
588
|
+
errors.push("ineligible candidate set is incorrect");
|
|
589
|
+
if (!sameMembers(snapshot.unresolvedCandidateIds, [...expectedUnresolved]))
|
|
590
|
+
errors.push("unresolved candidate set is incorrect");
|
|
591
|
+
const allKnown = new Set(byCandidate.keys());
|
|
592
|
+
for (const id of [
|
|
593
|
+
...snapshot.frontierCandidateIds,
|
|
594
|
+
...snapshot.dominatedCandidateIds,
|
|
595
|
+
...snapshot.ineligibleCandidateIds,
|
|
596
|
+
...snapshot.unresolvedCandidateIds,
|
|
597
|
+
]) {
|
|
598
|
+
if (!allKnown.has(id))
|
|
599
|
+
errors.push(`unknown candidate ${id}`);
|
|
600
|
+
}
|
|
601
|
+
return { valid: errors.length === 0, errors };
|
|
602
|
+
}
|
|
603
|
+
const MAX_PROCESSED_SIGNAL_IDS = 256;
|
|
604
|
+
function nextAt(anchor, seconds) {
|
|
605
|
+
if (seconds === undefined)
|
|
606
|
+
return undefined;
|
|
607
|
+
return new Date(Date.parse(anchor) + seconds * 1000).toISOString();
|
|
608
|
+
}
|
|
609
|
+
function earliest(values) {
|
|
610
|
+
return values.filter((value) => value !== undefined)
|
|
611
|
+
.sort((left, right) => Date.parse(left) - Date.parse(right))[0];
|
|
612
|
+
}
|
|
613
|
+
export function buildAutomationPlan(state, descriptor) {
|
|
614
|
+
const policy = state.cycle.policy?.checkPolicy;
|
|
615
|
+
if (!policy) {
|
|
616
|
+
return {
|
|
617
|
+
mode: "active_session_only",
|
|
618
|
+
requiredSignals: [],
|
|
619
|
+
scheduling: "active_session_fallback",
|
|
620
|
+
gaps: ["No user-confirmed automation policy exists."],
|
|
621
|
+
};
|
|
622
|
+
}
|
|
623
|
+
const required = new Set();
|
|
624
|
+
const cadences = [
|
|
625
|
+
policy.baselineAssessment,
|
|
626
|
+
policy.frontierReassessment,
|
|
627
|
+
policy.regressionCheck,
|
|
628
|
+
policy.postSwapVerification,
|
|
629
|
+
];
|
|
630
|
+
if (cadences.some((cadence) => cadence.afterCompletedTasks !== undefined))
|
|
631
|
+
required.add("product_task_completed");
|
|
632
|
+
if (cadences.some((cadence) => cadence.afterElapsedSeconds !== undefined))
|
|
633
|
+
required.add("scheduled_tick");
|
|
634
|
+
if (policy.frontierReassessment.onModelCatalogChange)
|
|
635
|
+
required.add("model_catalog_changed");
|
|
636
|
+
if (policy.regressionCheck.thresholds.length)
|
|
637
|
+
required.add("metric_window_available");
|
|
638
|
+
if (policy.postSwapVerification.afterCompletedTasks === undefined &&
|
|
639
|
+
policy.postSwapVerification.afterElapsedSeconds === undefined) {
|
|
640
|
+
required.add("verification_window_completed");
|
|
641
|
+
}
|
|
642
|
+
const nextWakeupAt = state.cycle.stage === "collecting_baseline"
|
|
643
|
+
? nextAt(state.automation.lastAssessmentAt ?? state.updatedAt, policy.baselineAssessment.afterElapsedSeconds)
|
|
644
|
+
: ["monitoring", "swap_verified"].includes(state.cycle.stage)
|
|
645
|
+
? earliest([
|
|
646
|
+
nextAt(state.automation.lastReassessmentAt ?? state.updatedAt, policy.frontierReassessment.afterElapsedSeconds),
|
|
647
|
+
nextAt(state.automation.lastRegressionCheckAt ?? state.updatedAt, policy.regressionCheck.afterElapsedSeconds),
|
|
648
|
+
])
|
|
649
|
+
: state.cycle.stage === "swap_applied"
|
|
650
|
+
? nextAt(state.automation.swapAppliedAt ?? state.updatedAt, policy.postSwapVerification.afterElapsedSeconds)
|
|
651
|
+
: undefined;
|
|
652
|
+
const supported = new Set(descriptor.automation.supportedSignals);
|
|
653
|
+
const gaps = [...required]
|
|
654
|
+
.filter((signal) => !supported.has(signal))
|
|
655
|
+
.map((signal) => `Harness does not report ${signal}.`);
|
|
656
|
+
let scheduling = "active_session_fallback";
|
|
657
|
+
if (policy.execution.mode === "persistent") {
|
|
658
|
+
scheduling = descriptor.automation.persistentScheduling && descriptor.automation.backgroundExecution
|
|
659
|
+
? "native"
|
|
660
|
+
: "external_scheduler_required";
|
|
661
|
+
if (scheduling === "external_scheduler_required") {
|
|
662
|
+
gaps.push("Persistent scheduling or background execution is unavailable; configure an external scheduler.");
|
|
663
|
+
}
|
|
664
|
+
if (descriptor.automation.wakeupProvisioning === "infrastructure_change" &&
|
|
665
|
+
!policy.execution.infrastructureChangesAllowed) {
|
|
666
|
+
scheduling = "external_scheduler_required";
|
|
667
|
+
gaps.push("Wakeup provisioning requires an infrastructure change that the user has not authorized.");
|
|
668
|
+
}
|
|
669
|
+
}
|
|
670
|
+
return {
|
|
671
|
+
mode: policy.execution.mode,
|
|
672
|
+
requiredSignals: [...required],
|
|
673
|
+
...(nextWakeupAt ? { nextWakeupAt } : {}),
|
|
674
|
+
scheduling,
|
|
675
|
+
gaps,
|
|
676
|
+
};
|
|
677
|
+
}
|
|
678
|
+
function cadenceDue(cadence, completedTasks, lastTaskCount, lastAt, now) {
|
|
679
|
+
const taskDue = cadence.afterCompletedTasks !== undefined &&
|
|
680
|
+
completedTasks - lastTaskCount >= cadence.afterCompletedTasks;
|
|
681
|
+
const elapsedDue = cadence.afterElapsedSeconds !== undefined &&
|
|
682
|
+
Date.parse(now) - Date.parse(lastAt) >= cadence.afterElapsedSeconds * 1000;
|
|
683
|
+
return taskDue || elapsedDue;
|
|
684
|
+
}
|
|
685
|
+
function regressionMetric(signal, previous, policy) {
|
|
686
|
+
if (signal.type !== "metric_window_available")
|
|
687
|
+
return undefined;
|
|
688
|
+
const directions = new Map(METRIC_CATALOG.map((metric) => [metric.id, metric.direction]));
|
|
689
|
+
for (const threshold of policy.checkPolicy.regressionCheck.thresholds) {
|
|
690
|
+
const current = signal.metrics.find((metric) => metric.metricId === threshold.metricId && metric.state === "measured" && metric.value !== undefined);
|
|
691
|
+
const prior = previous[threshold.metricId];
|
|
692
|
+
if (!current || current.value === undefined || prior === undefined)
|
|
693
|
+
continue;
|
|
694
|
+
const denominator = Math.max(Math.abs(prior), Number.EPSILON);
|
|
695
|
+
const deterioration = directions.get(threshold.metricId) === "minimize"
|
|
696
|
+
? (current.value - prior) / denominator
|
|
697
|
+
: (prior - current.value) / denominator;
|
|
698
|
+
if (deterioration >= threshold.relativeChangeAtLeast)
|
|
699
|
+
return threshold.metricId;
|
|
700
|
+
}
|
|
701
|
+
return undefined;
|
|
702
|
+
}
|
|
703
|
+
function dueDecision(state, signal, now) {
|
|
704
|
+
if (state.activeIntent)
|
|
705
|
+
return { action: "observe", reason: "Harness work is already active." };
|
|
706
|
+
if (state.automation.pendingIntentKind) {
|
|
707
|
+
return {
|
|
708
|
+
action: "issue_intent",
|
|
709
|
+
intentKind: state.automation.pendingIntentKind,
|
|
710
|
+
reason: state.automation.pendingIntentReason ?? "A previously detected automation check remains due.",
|
|
711
|
+
};
|
|
712
|
+
}
|
|
713
|
+
const policy = state.cycle.policy;
|
|
714
|
+
if (!policy)
|
|
715
|
+
return { action: "observe", reason: "No user-confirmed automation policy exists." };
|
|
716
|
+
const completed = state.automation.observedCompletedTasks;
|
|
717
|
+
const cooldownAnchor = [state.automation.lastReassessmentAt, state.automation.lastRegressionCheckAt]
|
|
718
|
+
.filter((value) => value !== undefined)
|
|
719
|
+
.sort((left, right) => Date.parse(right) - Date.parse(left))[0];
|
|
720
|
+
const cooldownSatisfied = !cooldownAnchor ||
|
|
721
|
+
Date.parse(now) - Date.parse(cooldownAnchor) >= policy.checkPolicy.cooldownSeconds * 1000;
|
|
722
|
+
if (state.cycle.stage === "collecting_baseline" && cadenceDue(policy.checkPolicy.baselineAssessment, completed, state.automation.lastAssessmentTaskCount, state.automation.lastAssessmentAt ?? state.updatedAt, now)) {
|
|
723
|
+
return { action: "issue_intent", intentKind: "run_assessment", reason: "The user-confirmed baseline assessment cadence is due." };
|
|
724
|
+
}
|
|
725
|
+
if (["monitoring", "swap_verified"].includes(state.cycle.stage) && cooldownSatisfied) {
|
|
726
|
+
const regressedMetric = regressionMetric(signal, state.automation.latestMetricValues ?? {}, policy);
|
|
727
|
+
if (regressedMetric) {
|
|
728
|
+
return { action: "issue_intent", intentKind: "investigate_regression", reason: `${regressedMetric} crossed its user-confirmed regression threshold.` };
|
|
729
|
+
}
|
|
730
|
+
if (cadenceDue(policy.checkPolicy.regressionCheck, completed, state.automation.lastRegressionCheckTaskCount ?? 0, state.automation.lastRegressionCheckAt ?? state.updatedAt, now)) {
|
|
731
|
+
return { action: "issue_intent", intentKind: "investigate_regression", reason: "The user-confirmed regression check cadence is due." };
|
|
732
|
+
}
|
|
733
|
+
if (signal.type === "model_catalog_changed" && policy.checkPolicy.frontierReassessment.onModelCatalogChange &&
|
|
734
|
+
state.automation.modelCatalogFingerprint !== undefined &&
|
|
735
|
+
state.automation.modelCatalogFingerprint !== signal.fingerprint) {
|
|
736
|
+
return { action: "issue_intent", intentKind: "discover_candidates", reason: "The model catalogue changed and the user enabled catalogue-triggered reassessment." };
|
|
737
|
+
}
|
|
738
|
+
if (cadenceDue(policy.checkPolicy.frontierReassessment, completed, state.automation.lastReassessmentTaskCount ?? 0, state.automation.lastReassessmentAt ?? state.updatedAt, now)) {
|
|
739
|
+
return { action: "issue_intent", intentKind: "discover_candidates", reason: "The user-confirmed frontier reassessment cadence is due." };
|
|
740
|
+
}
|
|
741
|
+
}
|
|
742
|
+
if (state.cycle.stage === "swap_applied" &&
|
|
743
|
+
(signal.type === "verification_window_completed" || cadenceDue(policy.checkPolicy.postSwapVerification, completed, state.automation.swapAppliedTaskCount ?? completed, state.automation.swapAppliedAt ?? state.updatedAt, now))) {
|
|
744
|
+
return { action: "issue_intent", intentKind: "verify_model_swap", reason: "The user-confirmed post-swap verification window is complete." };
|
|
745
|
+
}
|
|
746
|
+
return { action: "observe", reason: "No user-confirmed automation check is due." };
|
|
747
|
+
}
|
|
748
|
+
function auditSignalData(signal) {
|
|
749
|
+
switch (signal.type) {
|
|
750
|
+
case "product_task_completed":
|
|
751
|
+
return {
|
|
752
|
+
modelId: signal.modelId,
|
|
753
|
+
contextTokens: signal.contextTokens ?? null,
|
|
754
|
+
evidenceIds: (signal.evidence ?? []).map((item) => item.id),
|
|
755
|
+
};
|
|
756
|
+
case "metric_window_available":
|
|
757
|
+
case "verification_window_completed":
|
|
758
|
+
return {
|
|
759
|
+
metrics: (signal.metrics ?? []).map((metric) => ({
|
|
760
|
+
metricId: metric.metricId,
|
|
761
|
+
state: metric.state,
|
|
762
|
+
scope: metric.scope,
|
|
763
|
+
direction: metric.direction,
|
|
764
|
+
method: metric.method,
|
|
765
|
+
value: metric.value ?? null,
|
|
766
|
+
unit: metric.unit,
|
|
767
|
+
sampleCount: metric.sampleCount,
|
|
768
|
+
...(metric.interval ? { interval: metric.interval } : {}),
|
|
769
|
+
...(metric.window ? { window: metric.window } : {}),
|
|
770
|
+
evidenceIds: metric.evidence.map((item) => item.id),
|
|
771
|
+
})),
|
|
772
|
+
};
|
|
773
|
+
case "model_catalog_changed":
|
|
774
|
+
return { fingerprint: signal.fingerprint };
|
|
775
|
+
case "scheduled_tick":
|
|
776
|
+
return {};
|
|
777
|
+
}
|
|
778
|
+
}
|
|
779
|
+
/** Harness-neutral durable coordinator. Adapters transport jobs; this class owns lifecycle state. */
|
|
780
|
+
export class OpenMeritCoordinator {
|
|
781
|
+
store;
|
|
782
|
+
adapter;
|
|
783
|
+
constructor(store, adapter) {
|
|
784
|
+
this.store = store;
|
|
785
|
+
this.adapter = adapter;
|
|
786
|
+
}
|
|
787
|
+
async issue(kind) {
|
|
788
|
+
const missing = missingHarnessCapabilities(this.adapter.descriptor, kind);
|
|
789
|
+
if (missing.length)
|
|
790
|
+
throw new Error(`Harness lacks required capabilities: ${missing.join(", ")}`);
|
|
791
|
+
const config = await this.store.readConfig() ?? { schemaVersion: 1, taskProfiles: [] };
|
|
792
|
+
const existing = await this.store.readState();
|
|
793
|
+
if (existing?.activeIntent)
|
|
794
|
+
throw new Error(`OpenMerit intent ${existing.activeIntent.id} is already active.`);
|
|
795
|
+
const intent = createOpenMeritIntent(kind, {
|
|
796
|
+
taskProfileId: config.activeTaskProfileId,
|
|
797
|
+
policy: config.policy,
|
|
798
|
+
});
|
|
799
|
+
const now = new Date().toISOString();
|
|
800
|
+
const state = existing
|
|
801
|
+
? {
|
|
802
|
+
...existing,
|
|
803
|
+
cycle: { ...existing.cycle, stage: startedStage(kind) },
|
|
804
|
+
activeIntent: intent,
|
|
805
|
+
automation: {
|
|
806
|
+
...existing.automation,
|
|
807
|
+
...(existing.automation.pendingIntentKind === kind
|
|
808
|
+
? { pendingIntentKind: undefined, pendingIntentReason: undefined }
|
|
809
|
+
: {}),
|
|
810
|
+
},
|
|
811
|
+
updatedAt: now,
|
|
812
|
+
}
|
|
813
|
+
: {
|
|
814
|
+
schemaVersion: 1,
|
|
815
|
+
cycle: {
|
|
816
|
+
id: `cycle-${crypto.randomUUID()}`,
|
|
817
|
+
stage: startedStage(kind),
|
|
818
|
+
verifiedSwapCount: 0,
|
|
819
|
+
baselineEvidenceSufficient: false,
|
|
820
|
+
observabilityReady: false,
|
|
821
|
+
},
|
|
822
|
+
activeIntent: intent,
|
|
823
|
+
automation: {
|
|
824
|
+
setupNudgeShown: kind === "establish_evals",
|
|
825
|
+
observedCompletedTasks: 0,
|
|
826
|
+
lastAssessmentTaskCount: 0,
|
|
827
|
+
processedSignalIds: [],
|
|
828
|
+
},
|
|
829
|
+
updatedAt: now,
|
|
830
|
+
};
|
|
831
|
+
if (existing)
|
|
832
|
+
await this.store.writeState(state);
|
|
833
|
+
else {
|
|
834
|
+
await this.store.initialize(config, state);
|
|
835
|
+
await this.store.appendEvent({
|
|
836
|
+
schemaVersion: 1,
|
|
837
|
+
id: `event-${crypto.randomUUID()}`,
|
|
838
|
+
type: "project_initialized",
|
|
839
|
+
occurredAt: now,
|
|
840
|
+
cycleId: state.cycle.id,
|
|
841
|
+
data: { harnessId: this.adapter.descriptor.id },
|
|
842
|
+
});
|
|
843
|
+
}
|
|
844
|
+
await this.store.appendEvent({
|
|
845
|
+
schemaVersion: 1,
|
|
846
|
+
id: `event-${crypto.randomUUID()}`,
|
|
847
|
+
type: "intent_requested",
|
|
848
|
+
occurredAt: now,
|
|
849
|
+
cycleId: state.cycle.id,
|
|
850
|
+
intentId: intent.id,
|
|
851
|
+
data: { kind, harnessId: this.adapter.descriptor.id },
|
|
852
|
+
});
|
|
853
|
+
const receipt = await dispatchToHarness(this.adapter, intent);
|
|
854
|
+
if (!receipt.accepted) {
|
|
855
|
+
await this.store.writeState({
|
|
856
|
+
...state,
|
|
857
|
+
cycle: { ...state.cycle, stage: existing?.cycle.stage ?? "unconfigured" },
|
|
858
|
+
activeIntent: undefined,
|
|
859
|
+
automation: {
|
|
860
|
+
...state.automation,
|
|
861
|
+
pendingIntentKind: kind,
|
|
862
|
+
pendingIntentReason: "Harness dispatch failed; the due intent remains pending.",
|
|
863
|
+
},
|
|
864
|
+
updatedAt: new Date().toISOString(),
|
|
865
|
+
});
|
|
866
|
+
throw new Error(receipt.reason ?? "Harness rejected the OpenMerit intent.");
|
|
867
|
+
}
|
|
868
|
+
return { intent, state, receipt };
|
|
869
|
+
}
|
|
870
|
+
async reconcileAutomation(now = new Date().toISOString()) {
|
|
871
|
+
const state = await this.store.readState();
|
|
872
|
+
if (!state)
|
|
873
|
+
throw new Error("OpenMerit project is not initialized.");
|
|
874
|
+
const plan = buildAutomationPlan(state, this.adapter.descriptor);
|
|
875
|
+
if (plan.scheduling !== "native" || !plan.nextWakeupAt || Date.parse(plan.nextWakeupAt) <= Date.parse(now)) {
|
|
876
|
+
return { plan };
|
|
877
|
+
}
|
|
878
|
+
if (!this.adapter.scheduleWakeup) {
|
|
879
|
+
return {
|
|
880
|
+
plan: {
|
|
881
|
+
...plan,
|
|
882
|
+
scheduling: "external_scheduler_required",
|
|
883
|
+
gaps: [...plan.gaps, "Harness declared native scheduling but did not implement scheduleWakeup."],
|
|
884
|
+
},
|
|
885
|
+
};
|
|
886
|
+
}
|
|
887
|
+
const receipt = await this.adapter.scheduleWakeup({
|
|
888
|
+
at: plan.nextWakeupAt,
|
|
889
|
+
signal: "scheduled_tick",
|
|
890
|
+
reason: "Wake the harness so OpenMerit can evaluate the user-confirmed check policy.",
|
|
891
|
+
});
|
|
892
|
+
return { plan, receipt };
|
|
893
|
+
}
|
|
894
|
+
async acceptResult(result) {
|
|
895
|
+
const state = await this.store.readState();
|
|
896
|
+
if (!state?.activeIntent)
|
|
897
|
+
throw new Error("No durable OpenMerit intent is active.");
|
|
898
|
+
const config = await this.store.readConfig() ?? { schemaVersion: 1, taskProfiles: [] };
|
|
899
|
+
const interpretation = interpretIntentResult(state.activeIntent, result, state.cycle, config);
|
|
900
|
+
if (!interpretation.valid || !interpretation.nextStage) {
|
|
901
|
+
throw new Error(`OpenMerit rejected the harness result: ${interpretation.errors.join("; ")}`);
|
|
902
|
+
}
|
|
903
|
+
if (interpretation.config)
|
|
904
|
+
await this.store.writeConfig(interpretation.config);
|
|
905
|
+
const now = new Date().toISOString();
|
|
906
|
+
let observability = state.observability;
|
|
907
|
+
if (result.status === "succeeded" && state.activeIntent.kind === "establish_evals") {
|
|
908
|
+
observability = result.outputs.observability;
|
|
909
|
+
}
|
|
910
|
+
else if (result.status === "succeeded" && state.activeIntent.kind === "instrument_observability") {
|
|
911
|
+
const profile = config.taskProfiles.find((item) => item.id === config.activeTaskProfileId);
|
|
912
|
+
if (profile) {
|
|
913
|
+
const output = result.outputs;
|
|
914
|
+
const requiredMetricIds = profile.objectives.filter((item) => item.required).map((item) => item.metricId);
|
|
915
|
+
observability = {
|
|
916
|
+
status: "ready",
|
|
917
|
+
requiredMetricIds,
|
|
918
|
+
coveredMetricIds: output.coveredMetricIds,
|
|
919
|
+
missingMetricIds: output.missingMetricIds,
|
|
920
|
+
checkedAt: result.completedAt,
|
|
921
|
+
};
|
|
922
|
+
}
|
|
923
|
+
}
|
|
924
|
+
const nextState = {
|
|
925
|
+
...state,
|
|
926
|
+
cycle: {
|
|
927
|
+
...state.cycle,
|
|
928
|
+
stage: interpretation.nextStage,
|
|
929
|
+
...interpretation.cyclePatch,
|
|
930
|
+
},
|
|
931
|
+
activeIntent: undefined,
|
|
932
|
+
lastResult: result,
|
|
933
|
+
...(observability ? { observability } : {}),
|
|
934
|
+
automation: {
|
|
935
|
+
...state.automation,
|
|
936
|
+
...(state.activeIntent.kind === "run_assessment"
|
|
937
|
+
? { lastAssessmentTaskCount: state.automation.observedCompletedTasks, lastAssessmentAt: now }
|
|
938
|
+
: {}),
|
|
939
|
+
...(state.activeIntent.kind === "calculate_frontier"
|
|
940
|
+
? { lastReassessmentTaskCount: state.automation.observedCompletedTasks, lastReassessmentAt: now }
|
|
941
|
+
: {}),
|
|
942
|
+
...(state.activeIntent.kind === "investigate_regression"
|
|
943
|
+
? { lastRegressionCheckTaskCount: state.automation.observedCompletedTasks, lastRegressionCheckAt: now }
|
|
944
|
+
: {}),
|
|
945
|
+
...(state.activeIntent.kind === "apply_model_swap"
|
|
946
|
+
? { swapAppliedTaskCount: state.automation.observedCompletedTasks, swapAppliedAt: now }
|
|
947
|
+
: {}),
|
|
948
|
+
...(state.activeIntent.kind === "verify_model_swap" ? { lastVerificationAt: now } : {}),
|
|
949
|
+
},
|
|
950
|
+
updatedAt: now,
|
|
951
|
+
};
|
|
952
|
+
await this.store.writeState(nextState);
|
|
953
|
+
if (result.evidence.length && nextState.cycle.taskProfileId) {
|
|
954
|
+
await this.store.writeEvidenceManifest({
|
|
955
|
+
schemaVersion: 1,
|
|
956
|
+
id: `manifest-${result.intentId}`,
|
|
957
|
+
createdAt: now,
|
|
958
|
+
taskProfileId: nextState.cycle.taskProfileId,
|
|
959
|
+
references: result.evidence,
|
|
960
|
+
});
|
|
961
|
+
}
|
|
962
|
+
await this.store.appendEvent({
|
|
963
|
+
schemaVersion: 1,
|
|
964
|
+
id: `event-${crypto.randomUUID()}`,
|
|
965
|
+
type: "intent_completed",
|
|
966
|
+
occurredAt: now,
|
|
967
|
+
cycleId: nextState.cycle.id,
|
|
968
|
+
intentId: result.intentId,
|
|
969
|
+
data: {
|
|
970
|
+
kind: state.activeIntent.kind,
|
|
971
|
+
status: result.status,
|
|
972
|
+
nextStage: interpretation.nextStage,
|
|
973
|
+
harnessId: this.adapter.descriptor.id,
|
|
974
|
+
summary: result.summary,
|
|
975
|
+
evidence: result.evidence.map((item) => ({ id: item.id, source: item.source })),
|
|
976
|
+
},
|
|
977
|
+
});
|
|
978
|
+
if (observability && (state.activeIntent.kind === "establish_evals" || state.activeIntent.kind === "instrument_observability")) {
|
|
979
|
+
await this.store.appendEvent({
|
|
980
|
+
schemaVersion: 1,
|
|
981
|
+
id: `event-${crypto.randomUUID()}`,
|
|
982
|
+
type: "observability_readiness_checked",
|
|
983
|
+
occurredAt: now,
|
|
984
|
+
cycleId: nextState.cycle.id,
|
|
985
|
+
intentId: result.intentId,
|
|
986
|
+
data: {
|
|
987
|
+
status: observability.status,
|
|
988
|
+
requiredMetricIds: observability.requiredMetricIds,
|
|
989
|
+
coveredMetricIds: observability.coveredMetricIds,
|
|
990
|
+
missingMetricIds: observability.missingMetricIds,
|
|
991
|
+
},
|
|
992
|
+
});
|
|
993
|
+
}
|
|
994
|
+
const automation = await this.reconcileAutomation(now);
|
|
995
|
+
return { intentKind: state.activeIntent.kind, state: nextState, interpretation, automation };
|
|
996
|
+
}
|
|
997
|
+
async recordLifecycle(update) {
|
|
998
|
+
const state = await this.store.readState();
|
|
999
|
+
if (!state?.activeIntent || state.activeIntent.id !== update.intentId) {
|
|
1000
|
+
throw new Error(`Lifecycle update ${update.intentId} does not match the active intent.`);
|
|
1001
|
+
}
|
|
1002
|
+
await this.store.appendEvent({
|
|
1003
|
+
schemaVersion: 1,
|
|
1004
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1005
|
+
type: "intent_progressed",
|
|
1006
|
+
occurredAt: update.occurredAt,
|
|
1007
|
+
cycleId: state.cycle.id,
|
|
1008
|
+
intentId: update.intentId,
|
|
1009
|
+
data: {
|
|
1010
|
+
status: update.status,
|
|
1011
|
+
...(update.message ? { message: update.message } : {}),
|
|
1012
|
+
...(update.progress ? { progress: update.progress } : {}),
|
|
1013
|
+
...(update.details ? { details: update.details } : {}),
|
|
1014
|
+
},
|
|
1015
|
+
});
|
|
1016
|
+
}
|
|
1017
|
+
async recordTaskObservation(details = {}) {
|
|
1018
|
+
const now = new Date().toISOString();
|
|
1019
|
+
const result = await this.recordAutomationSignal({
|
|
1020
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
1021
|
+
id: `legacy-task-${crypto.randomUUID()}`,
|
|
1022
|
+
type: "product_task_completed",
|
|
1023
|
+
occurredAt: now,
|
|
1024
|
+
modelId: typeof details.modelId === "string" ? details.modelId : "unknown",
|
|
1025
|
+
contextTokens: typeof details.contextTokens === "number" || details.contextTokens === null
|
|
1026
|
+
? details.contextTokens
|
|
1027
|
+
: undefined,
|
|
1028
|
+
});
|
|
1029
|
+
return {
|
|
1030
|
+
state: result.state,
|
|
1031
|
+
assessmentDue: result.decision.action === "issue_intent" && result.decision.intentKind === "run_assessment",
|
|
1032
|
+
};
|
|
1033
|
+
}
|
|
1034
|
+
async recordAutomationSignal(signal) {
|
|
1035
|
+
const state = await this.store.readState();
|
|
1036
|
+
if (!state)
|
|
1037
|
+
throw new Error("OpenMerit project is not initialized.");
|
|
1038
|
+
if (signal.protocolVersion !== PROTOCOL_VERSION)
|
|
1039
|
+
throw new Error("Automation signal protocol version does not match.");
|
|
1040
|
+
if (!signal.id.trim())
|
|
1041
|
+
throw new Error("Automation signal ID is required.");
|
|
1042
|
+
const processed = state.automation.processedSignalIds ?? [];
|
|
1043
|
+
if (processed.includes(signal.id)) {
|
|
1044
|
+
await this.store.appendEvent({
|
|
1045
|
+
schemaVersion: 1,
|
|
1046
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1047
|
+
type: "automation_signal_duplicate",
|
|
1048
|
+
occurredAt: signal.occurredAt,
|
|
1049
|
+
cycleId: state.cycle.id,
|
|
1050
|
+
data: { signalId: signal.id, signalType: signal.type },
|
|
1051
|
+
});
|
|
1052
|
+
return { state, duplicate: true, decision: dueDecision(state, signal, signal.occurredAt) };
|
|
1053
|
+
}
|
|
1054
|
+
const completedTasks = state.automation.observedCompletedTasks +
|
|
1055
|
+
(signal.type === "product_task_completed" ? 1 : 0);
|
|
1056
|
+
const decisionState = {
|
|
1057
|
+
...state,
|
|
1058
|
+
automation: { ...state.automation, observedCompletedTasks: completedTasks },
|
|
1059
|
+
};
|
|
1060
|
+
const decision = dueDecision(decisionState, signal, signal.occurredAt);
|
|
1061
|
+
const latestMetricValues = { ...(state.automation.latestMetricValues ?? {}) };
|
|
1062
|
+
if (signal.type === "metric_window_available" || signal.type === "verification_window_completed") {
|
|
1063
|
+
for (const metric of signal.metrics ?? []) {
|
|
1064
|
+
if (metric.state === "measured" && metric.value !== undefined)
|
|
1065
|
+
latestMetricValues[metric.metricId] = metric.value;
|
|
1066
|
+
}
|
|
1067
|
+
}
|
|
1068
|
+
const nextState = {
|
|
1069
|
+
...decisionState,
|
|
1070
|
+
automation: {
|
|
1071
|
+
...decisionState.automation,
|
|
1072
|
+
processedSignalIds: [...processed, signal.id].slice(-MAX_PROCESSED_SIGNAL_IDS),
|
|
1073
|
+
latestMetricValues,
|
|
1074
|
+
...(signal.type === "model_catalog_changed" ? {
|
|
1075
|
+
modelCatalogFingerprint: signal.fingerprint,
|
|
1076
|
+
...(state.automation.modelCatalogFingerprint && state.automation.modelCatalogFingerprint !== signal.fingerprint
|
|
1077
|
+
? { modelCatalogChangedAt: signal.occurredAt }
|
|
1078
|
+
: {}),
|
|
1079
|
+
} : {}),
|
|
1080
|
+
...(decision.action === "issue_intent" ? {
|
|
1081
|
+
pendingIntentKind: decision.intentKind,
|
|
1082
|
+
pendingIntentReason: decision.reason,
|
|
1083
|
+
} : {}),
|
|
1084
|
+
},
|
|
1085
|
+
updatedAt: signal.occurredAt,
|
|
1086
|
+
};
|
|
1087
|
+
await this.store.writeState(nextState);
|
|
1088
|
+
await this.store.appendEvent({
|
|
1089
|
+
schemaVersion: 1,
|
|
1090
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1091
|
+
type: "automation_signal_recorded",
|
|
1092
|
+
occurredAt: signal.occurredAt,
|
|
1093
|
+
cycleId: state.cycle.id,
|
|
1094
|
+
data: {
|
|
1095
|
+
signalId: signal.id,
|
|
1096
|
+
signalType: signal.type,
|
|
1097
|
+
decision: decision.action,
|
|
1098
|
+
decisionReason: decision.reason,
|
|
1099
|
+
...auditSignalData(signal),
|
|
1100
|
+
},
|
|
1101
|
+
});
|
|
1102
|
+
if (decision.action === "issue_intent") {
|
|
1103
|
+
await this.store.appendEvent({
|
|
1104
|
+
schemaVersion: 1,
|
|
1105
|
+
id: `event-${crypto.randomUUID()}`,
|
|
1106
|
+
type: "automation_check_due",
|
|
1107
|
+
occurredAt: signal.occurredAt,
|
|
1108
|
+
cycleId: state.cycle.id,
|
|
1109
|
+
data: { signalId: signal.id, intentKind: decision.intentKind, reason: decision.reason },
|
|
1110
|
+
});
|
|
1111
|
+
}
|
|
1112
|
+
return { state: nextState, duplicate: false, decision };
|
|
1113
|
+
}
|
|
1114
|
+
async recordModelCatalogFingerprint(fingerprint) {
|
|
1115
|
+
const state = await this.store.readState();
|
|
1116
|
+
if (!state)
|
|
1117
|
+
throw new Error("OpenMerit project is not initialized.");
|
|
1118
|
+
const previousFingerprint = state.automation.modelCatalogFingerprint;
|
|
1119
|
+
const changed = previousFingerprint !== undefined && previousFingerprint !== fingerprint;
|
|
1120
|
+
if (previousFingerprint === fingerprint) {
|
|
1121
|
+
return { state, changed: false, reassessmentDue: false };
|
|
1122
|
+
}
|
|
1123
|
+
const now = new Date().toISOString();
|
|
1124
|
+
const result = await this.recordAutomationSignal({
|
|
1125
|
+
protocolVersion: PROTOCOL_VERSION,
|
|
1126
|
+
id: `legacy-catalog-${crypto.randomUUID()}`,
|
|
1127
|
+
type: "model_catalog_changed",
|
|
1128
|
+
occurredAt: now,
|
|
1129
|
+
fingerprint,
|
|
1130
|
+
});
|
|
1131
|
+
return {
|
|
1132
|
+
state: result.state,
|
|
1133
|
+
changed,
|
|
1134
|
+
reassessmentDue: result.decision.action === "issue_intent" && result.decision.intentKind === "discover_candidates",
|
|
1135
|
+
};
|
|
1136
|
+
}
|
|
1137
|
+
}
|