openmerit 0.1.4 → 0.1.5
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +26 -0
- package/README.md +83 -372
- package/dist/core/src/index.d.ts +90 -0
- package/dist/core/src/index.js +1137 -0
- package/dist/core/src/store.d.ts +35 -0
- package/dist/core/src/store.js +102 -0
- package/dist/pi/src/index.d.ts +15 -0
- package/dist/pi/src/index.js +423 -0
- package/dist/protocol/src/index.d.ts +402 -0
- package/dist/protocol/src/index.js +47 -0
- package/dist/protocol/src/schemas.d.ts +450 -0
- package/dist/protocol/src/schemas.js +224 -0
- package/docs/adapter-guide.md +172 -0
- package/docs/architecture.md +55 -0
- package/docs/automation.md +66 -0
- package/docs/getting-started.md +55 -0
- package/docs/lifecycle.md +30 -0
- package/docs/metrics-and-evidence.md +40 -0
- package/docs/operations.md +31 -0
- package/docs/pareto-spec.md +76 -0
- package/docs/pi-extension.md +44 -0
- package/docs/roadmap.md +26 -0
- package/docs/security.md +23 -0
- package/docs/testing.md +36 -0
- package/docs/ux-reference.md +32 -0
- package/package.json +45 -54
- package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
- package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
- package/benchmark/invoice_ocr/data/manifest.json +0 -97
- package/dist/benchmarks.js +0 -98
- package/dist/catalog.js +0 -61
- package/dist/cli.js +0 -188
- package/dist/daemon.js +0 -407
- package/dist/diagnostics.js +0 -227
- package/dist/frontier.js +0 -56
- package/dist/harness.js +0 -1
- package/dist/integrations.js +0 -19
- package/dist/invoice-eval.js +0 -33
- package/dist/invoice-score.js +0 -124
- package/dist/judge.js +0 -43
- package/dist/llm.js +0 -207
- package/dist/pi-config.js +0 -46
- package/dist/pi-trials.js +0 -373
- package/dist/policy.js +0 -185
- package/dist/providers.js +0 -1
- package/dist/recommend.js +0 -76
- package/dist/routes.js +0 -74
- package/dist/standalone.js +0 -224
- package/dist/store.js +0 -89
- package/dist/strategist.js +0 -68
- package/dist/task-input.js +0 -54
- package/dist/traces.js +0 -127
- package/dist/trials.js +0 -140
- package/dist/types.js +0 -2
- package/examples/invoice-prompt.txt +0 -19
- package/examples/task.example.json +0 -7
- package/extension/openmerit.ts +0 -947
- package/instructions/OPENMERIT.md +0 -63
- package/instructions/openmerit.policy.json +0 -37
- package/rules.md +0 -43
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
export declare const PROTOCOL_VERSION: "0.1";
|
|
2
|
+
export { CandidateAssessmentSchema, CandidateDescriptorSchema, completionToolSchemaFor, EvidenceReferenceSchema, FrontierRecommendationSchema, FrontierSnapshotSchema, INTENT_OUTPUT_SCHEMAS, intentOutputSchemaFor, MetricObjectiveSchema, MetricRecordSchema, ObservabilityCoverageSchema, OPENMERIT_RESULT_SCHEMA_VERSION, OPENMERIT_SCHEMA_DIALECT, SupervisedGraduationPolicySchema, TaskProfileSchema, validateIntentOutputSchema, } from "./schemas.ts";
|
|
3
|
+
export type JsonPrimitive = string | number | boolean | null;
|
|
4
|
+
export type JsonValue = JsonPrimitive | {
|
|
5
|
+
readonly [key: string]: JsonValue;
|
|
6
|
+
} | readonly JsonValue[];
|
|
7
|
+
export type IntentKind = "establish_evals" | "instrument_observability" | "run_assessment" | "discover_candidates" | "run_challenger_trials" | "calculate_frontier" | "investigate_regression" | "apply_model_swap" | "verify_model_swap" | "rollback_model_swap";
|
|
8
|
+
export type HarnessCapability = "user_confirmation" | "evaluation_authoring" | "observability_instrumentation" | "production_observation" | "candidate_discovery" | "challenger_execution" | "frontier_calculation" | "model_mutation" | "post_swap_verification" | "rollback";
|
|
9
|
+
export interface HarnessDescriptor {
|
|
10
|
+
readonly id: string;
|
|
11
|
+
readonly name: string;
|
|
12
|
+
readonly version: string;
|
|
13
|
+
readonly protocolVersions: readonly string[];
|
|
14
|
+
readonly capabilities: readonly HarnessCapability[];
|
|
15
|
+
readonly executionModes: readonly ("interactive" | "non_interactive" | "remote")[];
|
|
16
|
+
readonly structuredOutput: {
|
|
17
|
+
readonly schemaDialect: typeof import("./schemas.ts").OPENMERIT_SCHEMA_DIALECT;
|
|
18
|
+
readonly presentation: "dynamic_tool" | "static_tool_union" | "adapter_validation";
|
|
19
|
+
readonly enforcement: "provider_and_harness" | "harness" | "adapter";
|
|
20
|
+
};
|
|
21
|
+
readonly automation: {
|
|
22
|
+
readonly supportedSignals: readonly AutomationSignalType[];
|
|
23
|
+
readonly persistentScheduling: boolean;
|
|
24
|
+
readonly backgroundExecution: boolean;
|
|
25
|
+
readonly wakeupProvisioning: "none" | "native" | "infrastructure_change";
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
export interface HarnessDispatchReceipt {
|
|
29
|
+
readonly accepted: boolean;
|
|
30
|
+
readonly harnessJobId?: string;
|
|
31
|
+
readonly reason?: string;
|
|
32
|
+
}
|
|
33
|
+
export interface EvaluationBudget {
|
|
34
|
+
readonly currency: string;
|
|
35
|
+
readonly maximumSpend: number;
|
|
36
|
+
readonly maximumCandidateCount?: number;
|
|
37
|
+
readonly maximumRunsPerCandidate?: number;
|
|
38
|
+
readonly expiresAt?: string;
|
|
39
|
+
}
|
|
40
|
+
export interface SupervisedGraduationPolicy {
|
|
41
|
+
readonly mode: "supervised_graduation";
|
|
42
|
+
readonly evaluationBudget: EvaluationBudget;
|
|
43
|
+
readonly automaticSwapsEnabled: boolean;
|
|
44
|
+
readonly confirmationRequiredUntilVerifiedSwaps: number;
|
|
45
|
+
readonly requirePostSwapVerification: true;
|
|
46
|
+
readonly rollbackOnRegression: boolean;
|
|
47
|
+
readonly checkPolicy: AutomationCheckPolicy;
|
|
48
|
+
}
|
|
49
|
+
export interface CheckCadence {
|
|
50
|
+
readonly afterCompletedTasks?: number;
|
|
51
|
+
readonly afterElapsedSeconds?: number;
|
|
52
|
+
}
|
|
53
|
+
export interface MetricRegressionThreshold {
|
|
54
|
+
readonly metricId: MetricId;
|
|
55
|
+
/** Fractional deterioration from the last measured value; 0.1 means 10%. */
|
|
56
|
+
readonly relativeChangeAtLeast: number;
|
|
57
|
+
}
|
|
58
|
+
export interface AutomationCheckPolicy {
|
|
59
|
+
readonly baselineAssessment: CheckCadence;
|
|
60
|
+
readonly frontierReassessment: CheckCadence & {
|
|
61
|
+
readonly onModelCatalogChange: boolean;
|
|
62
|
+
};
|
|
63
|
+
readonly regressionCheck: CheckCadence & {
|
|
64
|
+
readonly thresholds: readonly MetricRegressionThreshold[];
|
|
65
|
+
};
|
|
66
|
+
readonly postSwapVerification: CheckCadence;
|
|
67
|
+
readonly cooldownSeconds: number;
|
|
68
|
+
readonly execution: {
|
|
69
|
+
readonly mode: "active_session_only" | "persistent";
|
|
70
|
+
readonly infrastructureChangesAllowed: boolean;
|
|
71
|
+
};
|
|
72
|
+
}
|
|
73
|
+
export type IntentStatus = "requested" | "accepted" | "running" | "waiting_for_user" | "succeeded" | "failed" | "cancelled";
|
|
74
|
+
export type TerminalIntentStatus = Extract<IntentStatus, "succeeded" | "failed" | "cancelled">;
|
|
75
|
+
export interface EvidenceRequirement {
|
|
76
|
+
readonly id: string;
|
|
77
|
+
readonly description: string;
|
|
78
|
+
readonly metricProfileId?: string;
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* A durable, harness-neutral request for intelligent work.
|
|
82
|
+
*
|
|
83
|
+
* `requestedOutcome` says what must be true when the work is complete.
|
|
84
|
+
* `constraints` carries JSON-safe task constraints without prescribing how a
|
|
85
|
+
* harness should satisfy them.
|
|
86
|
+
*/
|
|
87
|
+
export interface OpenMeritIntent {
|
|
88
|
+
readonly protocolVersion: typeof PROTOCOL_VERSION;
|
|
89
|
+
readonly id: string;
|
|
90
|
+
readonly kind: IntentKind;
|
|
91
|
+
readonly taskProfileId: string;
|
|
92
|
+
readonly authorizationPolicyId: string;
|
|
93
|
+
readonly requestedOutcome: string;
|
|
94
|
+
readonly constraints: Readonly<Record<string, JsonValue>>;
|
|
95
|
+
readonly requiredEvidence: readonly EvidenceRequirement[];
|
|
96
|
+
readonly requestedAt: string;
|
|
97
|
+
}
|
|
98
|
+
export interface IntentLifecycleUpdate {
|
|
99
|
+
readonly protocolVersion: typeof PROTOCOL_VERSION;
|
|
100
|
+
readonly intentId: string;
|
|
101
|
+
readonly status: IntentStatus;
|
|
102
|
+
readonly occurredAt: string;
|
|
103
|
+
readonly message?: string;
|
|
104
|
+
readonly progress?: {
|
|
105
|
+
readonly completed: number;
|
|
106
|
+
readonly total?: number;
|
|
107
|
+
readonly unit?: string;
|
|
108
|
+
};
|
|
109
|
+
readonly details?: Readonly<Record<string, JsonValue>>;
|
|
110
|
+
}
|
|
111
|
+
export type EvidenceSource = "harness_trace" | "evaluation_artifact" | "observability_record" | "external_source" | "user_feedback";
|
|
112
|
+
/** A stable pointer to evidence stored outside the protocol message. */
|
|
113
|
+
export interface EvidenceReference {
|
|
114
|
+
readonly id: string;
|
|
115
|
+
readonly source: EvidenceSource;
|
|
116
|
+
readonly uri: string;
|
|
117
|
+
readonly mediaType?: string;
|
|
118
|
+
readonly digest?: string;
|
|
119
|
+
}
|
|
120
|
+
interface IntentResultBase {
|
|
121
|
+
readonly protocolVersion: typeof PROTOCOL_VERSION;
|
|
122
|
+
readonly intentId: string;
|
|
123
|
+
readonly completedAt: string;
|
|
124
|
+
readonly summary: string;
|
|
125
|
+
readonly evidence: readonly EvidenceReference[];
|
|
126
|
+
}
|
|
127
|
+
export type IntentResult<K extends IntentKind = IntentKind> = IntentResultBase & ({
|
|
128
|
+
readonly status: "succeeded";
|
|
129
|
+
readonly outputs: IntentOutputMap[K];
|
|
130
|
+
readonly error?: never;
|
|
131
|
+
} | {
|
|
132
|
+
readonly status: "failed";
|
|
133
|
+
readonly outputs?: Readonly<Record<string, JsonValue>>;
|
|
134
|
+
readonly error: {
|
|
135
|
+
readonly code: string;
|
|
136
|
+
readonly message: string;
|
|
137
|
+
readonly retryable: boolean;
|
|
138
|
+
readonly details?: Readonly<Record<string, JsonValue>>;
|
|
139
|
+
};
|
|
140
|
+
} | {
|
|
141
|
+
readonly status: "cancelled";
|
|
142
|
+
readonly outputs?: Readonly<Record<string, JsonValue>>;
|
|
143
|
+
readonly error?: {
|
|
144
|
+
readonly code: string;
|
|
145
|
+
readonly message: string;
|
|
146
|
+
readonly retryable: boolean;
|
|
147
|
+
readonly details?: Readonly<Record<string, JsonValue>>;
|
|
148
|
+
};
|
|
149
|
+
});
|
|
150
|
+
export type MetricId = "task_quality" | "task_success" | "consistency" | "instruction_following" | "tool_use_performance" | "structured_output_reliability" | "hallucination_rate" | "reasoning_efficiency" | "input_token_consumption" | "output_token_consumption" | "total_task_cost" | "cost_per_successful_task" | "time_to_first_useful_output" | "end_to_end_task_latency" | "throughput" | "retry_rate" | "agent_steps" | "recovery_ability" | "context_handling" | "retrieval_use_quality" | "long_horizon_performance" | "human_intervention_rate" | "preference_fit";
|
|
151
|
+
export type MetricDirection = "maximize" | "minimize";
|
|
152
|
+
export type MetricScope = "single_run" | "sample_window";
|
|
153
|
+
export type MetricState = "measured" | "not_applicable" | "not_configured" | "insufficient_evidence";
|
|
154
|
+
export type MeasurementMethod = "observed" | "derived" | "evaluator" | "human" | "external";
|
|
155
|
+
export interface MetricDefinition {
|
|
156
|
+
readonly id: MetricId;
|
|
157
|
+
readonly label: string;
|
|
158
|
+
readonly direction: MetricDirection;
|
|
159
|
+
readonly defaultUnit: string;
|
|
160
|
+
readonly requiresRepeatedRuns: boolean;
|
|
161
|
+
}
|
|
162
|
+
export declare const METRIC_CATALOG: readonly MetricDefinition[];
|
|
163
|
+
export interface MetricRecord {
|
|
164
|
+
readonly metricId: MetricId;
|
|
165
|
+
readonly state: MetricState;
|
|
166
|
+
readonly scope: MetricScope;
|
|
167
|
+
readonly direction: MetricDirection;
|
|
168
|
+
readonly method: MeasurementMethod;
|
|
169
|
+
readonly value?: number;
|
|
170
|
+
readonly unit: string;
|
|
171
|
+
readonly sampleCount: number;
|
|
172
|
+
readonly interval?: {
|
|
173
|
+
readonly lower: number;
|
|
174
|
+
readonly upper: number;
|
|
175
|
+
readonly confidenceLevel?: number;
|
|
176
|
+
};
|
|
177
|
+
readonly window?: {
|
|
178
|
+
readonly startedAt: string;
|
|
179
|
+
readonly endedAt: string;
|
|
180
|
+
};
|
|
181
|
+
readonly evidence: readonly EvidenceReference[];
|
|
182
|
+
readonly note?: string;
|
|
183
|
+
}
|
|
184
|
+
export interface MetricObjective {
|
|
185
|
+
readonly metricId: MetricId;
|
|
186
|
+
readonly required: boolean;
|
|
187
|
+
readonly minimumSamples: number;
|
|
188
|
+
readonly tolerance: number;
|
|
189
|
+
readonly constraint?: {
|
|
190
|
+
readonly operator: "at_least" | "at_most";
|
|
191
|
+
readonly value: number;
|
|
192
|
+
};
|
|
193
|
+
}
|
|
194
|
+
export interface TaskProfile {
|
|
195
|
+
readonly id: string;
|
|
196
|
+
readonly revision: number;
|
|
197
|
+
readonly name: string;
|
|
198
|
+
readonly goal: string;
|
|
199
|
+
readonly objectives: readonly MetricObjective[];
|
|
200
|
+
readonly createdAt: string;
|
|
201
|
+
readonly confirmedAt?: string;
|
|
202
|
+
}
|
|
203
|
+
export type ObservabilityReadinessStatus = "ready" | "incomplete";
|
|
204
|
+
export interface ObservabilityCoverage {
|
|
205
|
+
readonly status: ObservabilityReadinessStatus;
|
|
206
|
+
readonly requiredMetricIds: readonly MetricId[];
|
|
207
|
+
readonly coveredMetricIds: readonly MetricId[];
|
|
208
|
+
readonly missingMetricIds: readonly MetricId[];
|
|
209
|
+
readonly checkedAt: string;
|
|
210
|
+
}
|
|
211
|
+
export interface CandidateAssessment {
|
|
212
|
+
readonly id: string;
|
|
213
|
+
readonly taskProfileId: string;
|
|
214
|
+
readonly taskProfileRevision: number;
|
|
215
|
+
readonly candidateId: string;
|
|
216
|
+
readonly modelId: string;
|
|
217
|
+
readonly modelVersion?: string;
|
|
218
|
+
readonly productRevision: string;
|
|
219
|
+
readonly harnessId: string;
|
|
220
|
+
readonly startedAt: string;
|
|
221
|
+
readonly completedAt: string;
|
|
222
|
+
readonly metrics: readonly MetricRecord[];
|
|
223
|
+
}
|
|
224
|
+
export type ComparisonState = "dominates" | "does_not_dominate" | "unresolved";
|
|
225
|
+
export interface DominanceComparison {
|
|
226
|
+
readonly candidateId: string;
|
|
227
|
+
readonly otherCandidateId: string;
|
|
228
|
+
readonly state: ComparisonState;
|
|
229
|
+
readonly reasons: readonly string[];
|
|
230
|
+
}
|
|
231
|
+
export interface FrontierSnapshot {
|
|
232
|
+
readonly id: string;
|
|
233
|
+
readonly taskProfileId: string;
|
|
234
|
+
readonly taskProfileRevision: number;
|
|
235
|
+
readonly productRevision: string;
|
|
236
|
+
readonly createdAt: string;
|
|
237
|
+
readonly frontierCandidateIds: readonly string[];
|
|
238
|
+
readonly dominatedCandidateIds: readonly string[];
|
|
239
|
+
readonly ineligibleCandidateIds: readonly string[];
|
|
240
|
+
readonly unresolvedCandidateIds: readonly string[];
|
|
241
|
+
readonly comparisons: readonly DominanceComparison[];
|
|
242
|
+
readonly assessmentIds: readonly string[];
|
|
243
|
+
readonly calculationEvidence: readonly EvidenceReference[];
|
|
244
|
+
}
|
|
245
|
+
export interface FrontierRecommendation {
|
|
246
|
+
readonly frontierSnapshotId: string;
|
|
247
|
+
readonly selectedCandidateId: string;
|
|
248
|
+
readonly currentCandidateId: string;
|
|
249
|
+
readonly reasons: readonly string[];
|
|
250
|
+
readonly preferenceEvidence: readonly EvidenceReference[];
|
|
251
|
+
}
|
|
252
|
+
export interface CandidateDescriptor {
|
|
253
|
+
readonly candidateId: string;
|
|
254
|
+
readonly modelId: string;
|
|
255
|
+
readonly modelVersion?: string;
|
|
256
|
+
readonly reasons: readonly string[];
|
|
257
|
+
}
|
|
258
|
+
export interface IntentOutputMap {
|
|
259
|
+
readonly establish_evals: {
|
|
260
|
+
readonly taskProfile: TaskProfile;
|
|
261
|
+
readonly policy: SupervisedGraduationPolicy;
|
|
262
|
+
readonly observability: ObservabilityCoverage;
|
|
263
|
+
};
|
|
264
|
+
readonly instrument_observability: {
|
|
265
|
+
readonly coveredMetricIds: readonly MetricId[];
|
|
266
|
+
readonly missingMetricIds: readonly MetricId[];
|
|
267
|
+
};
|
|
268
|
+
readonly run_assessment: {
|
|
269
|
+
readonly baselineEvidenceSufficient: boolean;
|
|
270
|
+
readonly metrics?: readonly MetricRecord[];
|
|
271
|
+
};
|
|
272
|
+
readonly discover_candidates: {
|
|
273
|
+
readonly candidates: readonly CandidateDescriptor[];
|
|
274
|
+
};
|
|
275
|
+
readonly run_challenger_trials: {
|
|
276
|
+
readonly assessments: readonly CandidateAssessment[];
|
|
277
|
+
};
|
|
278
|
+
readonly calculate_frontier: {
|
|
279
|
+
readonly assessments: readonly CandidateAssessment[];
|
|
280
|
+
readonly frontierSnapshot: FrontierSnapshot;
|
|
281
|
+
readonly recommendation?: FrontierRecommendation;
|
|
282
|
+
};
|
|
283
|
+
readonly investigate_regression: {
|
|
284
|
+
readonly regressionDetected: boolean;
|
|
285
|
+
readonly affectedMetricIds: readonly MetricId[];
|
|
286
|
+
readonly likelyCause?: string;
|
|
287
|
+
};
|
|
288
|
+
readonly apply_model_swap: {
|
|
289
|
+
readonly appliedCandidateId: string;
|
|
290
|
+
readonly previousCandidateId?: string;
|
|
291
|
+
};
|
|
292
|
+
readonly verify_model_swap: {
|
|
293
|
+
readonly verified: boolean;
|
|
294
|
+
readonly regressionDetected: boolean;
|
|
295
|
+
};
|
|
296
|
+
readonly rollback_model_swap: {
|
|
297
|
+
readonly restoredCandidateId: string;
|
|
298
|
+
};
|
|
299
|
+
}
|
|
300
|
+
export type AnyIntentResult = {
|
|
301
|
+
readonly [K in IntentKind]: IntentResult<K>;
|
|
302
|
+
}[IntentKind];
|
|
303
|
+
export declare const REQUIRED_CAPABILITIES: Readonly<Record<IntentKind, readonly HarnessCapability[]>>;
|
|
304
|
+
export type ImprovementStage = "unconfigured" | "establishing_evidence" | "collecting_baseline" | "baseline_ready" | "discovering_candidates" | "candidates_ready" | "running_challenger_trials" | "challengers_ready" | "calculating_frontier" | "frontier_ready" | "awaiting_swap_approval" | "swap_approved" | "applying_swap" | "swap_applied" | "verifying_swap" | "swap_verified" | "verification_failed" | "rolling_back" | "monitoring";
|
|
305
|
+
export interface ImprovementCycleState {
|
|
306
|
+
readonly id: string;
|
|
307
|
+
readonly taskProfileId?: string;
|
|
308
|
+
readonly stage: ImprovementStage;
|
|
309
|
+
readonly activeCandidateId?: string;
|
|
310
|
+
readonly previousCandidateId?: string;
|
|
311
|
+
readonly proposedCandidateId?: string;
|
|
312
|
+
readonly verifiedSwapCount: number;
|
|
313
|
+
readonly baselineEvidenceSufficient: boolean;
|
|
314
|
+
readonly observabilityReady?: boolean;
|
|
315
|
+
readonly policy?: SupervisedGraduationPolicy;
|
|
316
|
+
}
|
|
317
|
+
export interface OpenMeritProjectConfig {
|
|
318
|
+
readonly schemaVersion: 1;
|
|
319
|
+
readonly activeTaskProfileId?: string;
|
|
320
|
+
readonly taskProfiles: readonly TaskProfile[];
|
|
321
|
+
readonly policy?: SupervisedGraduationPolicy;
|
|
322
|
+
}
|
|
323
|
+
export interface OpenMeritProjectState {
|
|
324
|
+
readonly schemaVersion: 1;
|
|
325
|
+
readonly cycle: ImprovementCycleState;
|
|
326
|
+
readonly activeIntent?: OpenMeritIntent;
|
|
327
|
+
readonly lastResult?: IntentResult;
|
|
328
|
+
readonly observability?: ObservabilityCoverage;
|
|
329
|
+
readonly automation: {
|
|
330
|
+
readonly setupNudgeShown: boolean;
|
|
331
|
+
readonly observedCompletedTasks: number;
|
|
332
|
+
readonly lastAssessmentTaskCount: number;
|
|
333
|
+
readonly modelCatalogFingerprint?: string;
|
|
334
|
+
readonly modelCatalogChangedAt?: string;
|
|
335
|
+
readonly processedSignalIds?: readonly string[];
|
|
336
|
+
readonly pendingIntentKind?: IntentKind;
|
|
337
|
+
readonly pendingIntentReason?: string;
|
|
338
|
+
readonly lastAssessmentAt?: string;
|
|
339
|
+
readonly lastReassessmentTaskCount?: number;
|
|
340
|
+
readonly lastReassessmentAt?: string;
|
|
341
|
+
readonly lastRegressionCheckTaskCount?: number;
|
|
342
|
+
readonly lastRegressionCheckAt?: string;
|
|
343
|
+
readonly swapAppliedTaskCount?: number;
|
|
344
|
+
readonly swapAppliedAt?: string;
|
|
345
|
+
readonly lastVerificationAt?: string;
|
|
346
|
+
readonly latestMetricValues?: Readonly<Partial<Record<MetricId, number>>>;
|
|
347
|
+
};
|
|
348
|
+
readonly updatedAt: string;
|
|
349
|
+
}
|
|
350
|
+
export type AuditEventType = "project_initialized" | "intent_requested" | "intent_progressed" | "intent_completed" | "requirements_confirmation_required" | "swap_confirmation_required" | "frontier_verified" | "frontier_rejected" | "cycle_advanced" | "setup_nudged" | "task_observed" | "model_catalog_changed" | "automation_signal_recorded" | "automation_signal_duplicate" | "automation_check_due" | "observability_readiness_checked";
|
|
351
|
+
interface AutomationSignalBase {
|
|
352
|
+
readonly protocolVersion: typeof PROTOCOL_VERSION;
|
|
353
|
+
readonly id: string;
|
|
354
|
+
readonly occurredAt: string;
|
|
355
|
+
readonly evidence?: readonly EvidenceReference[];
|
|
356
|
+
}
|
|
357
|
+
export type AutomationSignal = (AutomationSignalBase & {
|
|
358
|
+
readonly type: "product_task_completed";
|
|
359
|
+
readonly modelId: string;
|
|
360
|
+
readonly contextTokens?: number | null;
|
|
361
|
+
}) | (AutomationSignalBase & {
|
|
362
|
+
readonly type: "metric_window_available";
|
|
363
|
+
readonly metrics: readonly MetricRecord[];
|
|
364
|
+
}) | (AutomationSignalBase & {
|
|
365
|
+
readonly type: "scheduled_tick";
|
|
366
|
+
}) | (AutomationSignalBase & {
|
|
367
|
+
readonly type: "model_catalog_changed";
|
|
368
|
+
readonly fingerprint: string;
|
|
369
|
+
}) | (AutomationSignalBase & {
|
|
370
|
+
readonly type: "verification_window_completed";
|
|
371
|
+
readonly metrics?: readonly MetricRecord[];
|
|
372
|
+
});
|
|
373
|
+
export type AutomationSignalType = AutomationSignal["type"];
|
|
374
|
+
export interface HarnessWakeupRequest {
|
|
375
|
+
readonly at: string;
|
|
376
|
+
readonly signal: "scheduled_tick";
|
|
377
|
+
readonly reason: string;
|
|
378
|
+
}
|
|
379
|
+
export interface AutomationPlan {
|
|
380
|
+
readonly mode: AutomationCheckPolicy["execution"]["mode"];
|
|
381
|
+
readonly requiredSignals: readonly AutomationSignalType[];
|
|
382
|
+
readonly nextWakeupAt?: string;
|
|
383
|
+
readonly scheduling: "native" | "active_session_fallback" | "external_scheduler_required";
|
|
384
|
+
readonly gaps: readonly string[];
|
|
385
|
+
}
|
|
386
|
+
export interface OpenMeritAuditEvent {
|
|
387
|
+
readonly schemaVersion: 1;
|
|
388
|
+
readonly id: string;
|
|
389
|
+
readonly type: AuditEventType;
|
|
390
|
+
readonly occurredAt: string;
|
|
391
|
+
readonly cycleId: string;
|
|
392
|
+
readonly intentId?: string;
|
|
393
|
+
readonly data: Readonly<Record<string, JsonValue>>;
|
|
394
|
+
}
|
|
395
|
+
export interface EvidenceManifest {
|
|
396
|
+
readonly schemaVersion: 1;
|
|
397
|
+
readonly id: string;
|
|
398
|
+
readonly createdAt: string;
|
|
399
|
+
readonly taskProfileId: string;
|
|
400
|
+
readonly references: readonly EvidenceReference[];
|
|
401
|
+
}
|
|
402
|
+
export declare function isTerminalIntentStatus(status: IntentStatus): status is TerminalIntentStatus;
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
export const PROTOCOL_VERSION = "0.1";
|
|
2
|
+
export { CandidateAssessmentSchema, CandidateDescriptorSchema, completionToolSchemaFor, EvidenceReferenceSchema, FrontierRecommendationSchema, FrontierSnapshotSchema, INTENT_OUTPUT_SCHEMAS, intentOutputSchemaFor, MetricObjectiveSchema, MetricRecordSchema, ObservabilityCoverageSchema, OPENMERIT_RESULT_SCHEMA_VERSION, OPENMERIT_SCHEMA_DIALECT, SupervisedGraduationPolicySchema, TaskProfileSchema, validateIntentOutputSchema, } from "./schemas.js";
|
|
3
|
+
export const METRIC_CATALOG = [
|
|
4
|
+
{ id: "task_quality", label: "Task quality", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
5
|
+
{ id: "task_success", label: "Task success rate", direction: "maximize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
6
|
+
{ id: "consistency", label: "Consistency", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: true },
|
|
7
|
+
{ id: "instruction_following", label: "Instruction following", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
8
|
+
{ id: "tool_use_performance", label: "Tool-use performance", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
9
|
+
{ id: "structured_output_reliability", label: "Structured-output reliability", direction: "maximize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
10
|
+
{ id: "hallucination_rate", label: "Hallucination rate", direction: "minimize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
11
|
+
{ id: "reasoning_efficiency", label: "Reasoning efficiency", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
12
|
+
{ id: "input_token_consumption", label: "Input-token consumption", direction: "minimize", defaultUnit: "tokens", requiresRepeatedRuns: false },
|
|
13
|
+
{ id: "output_token_consumption", label: "Output-token consumption", direction: "minimize", defaultUnit: "tokens", requiresRepeatedRuns: false },
|
|
14
|
+
{ id: "total_task_cost", label: "Total task cost", direction: "minimize", defaultUnit: "usd", requiresRepeatedRuns: false },
|
|
15
|
+
{ id: "cost_per_successful_task", label: "Cost per successful task", direction: "minimize", defaultUnit: "usd", requiresRepeatedRuns: true },
|
|
16
|
+
{ id: "time_to_first_useful_output", label: "Time to first useful output", direction: "minimize", defaultUnit: "milliseconds", requiresRepeatedRuns: false },
|
|
17
|
+
{ id: "end_to_end_task_latency", label: "End-to-end task latency", direction: "minimize", defaultUnit: "milliseconds", requiresRepeatedRuns: false },
|
|
18
|
+
{ id: "throughput", label: "Throughput under workload", direction: "maximize", defaultUnit: "tasks_per_hour", requiresRepeatedRuns: true },
|
|
19
|
+
{ id: "retry_rate", label: "Retry rate", direction: "minimize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
20
|
+
{ id: "agent_steps", label: "Agent steps", direction: "minimize", defaultUnit: "count", requiresRepeatedRuns: false },
|
|
21
|
+
{ id: "recovery_ability", label: "Recovery ability", direction: "maximize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
22
|
+
{ id: "context_handling", label: "Context handling", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
23
|
+
{ id: "retrieval_use_quality", label: "Retrieval-use quality", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
24
|
+
{ id: "long_horizon_performance", label: "Long-horizon performance", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: true },
|
|
25
|
+
{ id: "human_intervention_rate", label: "Human intervention rate", direction: "minimize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
26
|
+
{ id: "preference_fit", label: "Preference and style fit", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: true },
|
|
27
|
+
];
|
|
28
|
+
export const REQUIRED_CAPABILITIES = {
|
|
29
|
+
establish_evals: ["user_confirmation", "evaluation_authoring", "observability_instrumentation"],
|
|
30
|
+
instrument_observability: ["observability_instrumentation"],
|
|
31
|
+
run_assessment: ["production_observation"],
|
|
32
|
+
discover_candidates: ["candidate_discovery"],
|
|
33
|
+
run_challenger_trials: ["challenger_execution"],
|
|
34
|
+
calculate_frontier: ["frontier_calculation"],
|
|
35
|
+
investigate_regression: ["production_observation"],
|
|
36
|
+
apply_model_swap: ["model_mutation"],
|
|
37
|
+
verify_model_swap: ["post_swap_verification"],
|
|
38
|
+
rollback_model_swap: ["rollback"],
|
|
39
|
+
};
|
|
40
|
+
const terminalStatuses = new Set([
|
|
41
|
+
"succeeded",
|
|
42
|
+
"failed",
|
|
43
|
+
"cancelled",
|
|
44
|
+
]);
|
|
45
|
+
export function isTerminalIntentStatus(status) {
|
|
46
|
+
return terminalStatuses.has(status);
|
|
47
|
+
}
|