openmerit 0.1.4 → 0.1.6-preview.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +40 -0
- package/README.md +121 -386
- package/dist/core/src/index.d.ts +101 -0
- package/dist/core/src/index.js +1649 -0
- package/dist/core/src/store.d.ts +35 -0
- package/dist/core/src/store.js +102 -0
- package/dist/pi/src/index.d.ts +32 -0
- package/dist/pi/src/index.js +794 -0
- package/dist/pi/src/scheduler.d.ts +11 -0
- package/dist/pi/src/scheduler.js +137 -0
- package/dist/pi/src/wakeup.d.ts +2 -0
- package/dist/pi/src/wakeup.js +108 -0
- package/dist/protocol/src/index.d.ts +484 -0
- package/dist/protocol/src/index.js +47 -0
- package/dist/protocol/src/schemas.d.ts +576 -0
- package/dist/protocol/src/schemas.js +280 -0
- package/dist/terminal/public/app.js +297 -0
- package/dist/terminal/public/brands/anthropic.png +0 -0
- package/dist/terminal/public/brands/baai.png +0 -0
- package/dist/terminal/public/brands/baseten.png +0 -0
- package/dist/terminal/public/brands/cerebras.png +0 -0
- package/dist/terminal/public/brands/cohere.png +0 -0
- package/dist/terminal/public/brands/deepseek.ico +0 -0
- package/dist/terminal/public/brands/google.png +0 -0
- package/dist/terminal/public/brands/groq.ico +0 -0
- package/dist/terminal/public/brands/lm-studio.png +0 -0
- package/dist/terminal/public/brands/meta.ico +0 -0
- package/dist/terminal/public/brands/mistral.png +0 -0
- package/dist/terminal/public/brands/nomic.png +0 -0
- package/dist/terminal/public/brands/ollama.png +0 -0
- package/dist/terminal/public/brands/openai.png +0 -0
- package/dist/terminal/public/brands/openrouter.png +0 -0
- package/dist/terminal/public/brands/qwen.png +0 -0
- package/dist/terminal/public/brands/vllm.ico +0 -0
- package/dist/terminal/public/brands/vllm.png +0 -0
- package/dist/terminal/public/favicon.svg +1 -0
- package/dist/terminal/public/flow.css +1 -0
- package/dist/terminal/public/flow.js +770 -0
- package/dist/terminal/public/index.html +21 -0
- package/dist/terminal/public/styles.css +779 -0
- package/dist/terminal/src/activity-merge.mjs +64 -0
- package/dist/terminal/src/browser.mjs +29 -0
- package/dist/terminal/src/cli.mjs +60 -0
- package/dist/terminal/src/collect.mjs +311 -0
- package/dist/terminal/src/discovery.mjs +93 -0
- package/dist/terminal/src/hardware.mjs +57 -0
- package/dist/terminal/src/project-activity.mjs +156 -0
- package/dist/terminal/src/sample.mjs +171 -0
- package/dist/terminal/src/server.mjs +56 -0
- package/dist/terminal/src/services.mjs +62 -0
- package/dist/terminal/src/topology.mjs +30 -0
- package/docs/adapter-guide.md +189 -0
- package/docs/architecture.md +59 -0
- package/docs/automation.md +74 -0
- package/docs/budgets.md +37 -0
- package/docs/commands.md +85 -0
- package/docs/demo-backfill.md +29 -0
- package/docs/demo-fieldkit.md +47 -0
- package/docs/demo-placement.md +30 -0
- package/docs/demo-spam.md +15 -0
- package/docs/demo-support.md +42 -0
- package/docs/demo.md +57 -0
- package/docs/first-trial.md +60 -0
- package/docs/getting-started.md +65 -0
- package/docs/index.md +40 -0
- package/docs/inference-terminal.md +439 -0
- package/docs/lifecycle.md +30 -0
- package/docs/memo.md +126 -0
- package/docs/metrics-and-evidence.md +48 -0
- package/docs/operations.md +40 -0
- package/docs/pareto-spec.md +76 -0
- package/docs/pi-extension.md +54 -0
- package/docs/roadmap.md +28 -0
- package/docs/security.md +37 -0
- package/docs/site-artwork-linocut.md +23 -0
- package/docs/site-artwork-miniature-diverse.md +28 -0
- package/docs/site-artwork-miniature.md +26 -0
- package/docs/site-demo.md +177 -0
- package/docs/site-design.md +94 -0
- package/docs/site-documentation.md +83 -0
- package/docs/site-dynamic-og.md +35 -0
- package/docs/site-faq-maintenance.md +115 -0
- package/docs/site-hero-resolution.md +60 -0
- package/docs/site-illustration-sequences.md +227 -0
- package/docs/site-inference-terminal.md +203 -0
- package/docs/site-memo.md +39 -0
- package/docs/site-og-image.md +38 -0
- package/docs/site-og-workshop.md +21 -0
- package/docs/site-section-artwork.md +56 -0
- package/docs/site-skill-review.md +57 -0
- package/docs/site-terminal-preview.md +85 -0
- package/docs/testing.md +118 -0
- package/docs/troubleshooting.md +55 -0
- package/docs/ux-reference.md +32 -0
- package/package.json +74 -42
- package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
- package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
- package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
- package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
- package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
- package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
- package/benchmark/invoice_ocr/data/manifest.json +0 -97
- package/dist/benchmarks.js +0 -98
- package/dist/catalog.js +0 -61
- package/dist/cli.js +0 -188
- package/dist/daemon.js +0 -407
- package/dist/diagnostics.js +0 -227
- package/dist/frontier.js +0 -56
- package/dist/harness.js +0 -1
- package/dist/integrations.js +0 -19
- package/dist/invoice-eval.js +0 -33
- package/dist/invoice-score.js +0 -124
- package/dist/judge.js +0 -43
- package/dist/llm.js +0 -207
- package/dist/pi-config.js +0 -46
- package/dist/pi-trials.js +0 -373
- package/dist/policy.js +0 -185
- package/dist/providers.js +0 -1
- package/dist/recommend.js +0 -76
- package/dist/routes.js +0 -74
- package/dist/standalone.js +0 -224
- package/dist/store.js +0 -89
- package/dist/strategist.js +0 -68
- package/dist/task-input.js +0 -54
- package/dist/traces.js +0 -127
- package/dist/trials.js +0 -140
- package/dist/types.js +0 -2
- package/examples/invoice-prompt.txt +0 -19
- package/examples/task.example.json +0 -7
- package/extension/openmerit.ts +0 -947
- package/instructions/OPENMERIT.md +0 -63
- package/instructions/openmerit.policy.json +0 -37
- package/rules.md +0 -43
|
@@ -0,0 +1,484 @@
|
|
|
1
|
+
export declare const PROTOCOL_VERSION: "0.1";
|
|
2
|
+
export { ApplicationLlmTargetSchema, CandidateAssessmentSchema, CandidateDescriptorSchema, completionToolSchemaFor, EvidenceReferenceSchema, FrontierRecommendationSchema, FrontierSnapshotSchema, INTENT_OUTPUT_SCHEMAS, intentOutputSchemaFor, MetricObjectiveSchema, MetricRecordSchema, ObservabilityCoverageSchema, OPENMERIT_RESULT_SCHEMA_VERSION, OPENMERIT_SCHEMA_DIALECT, SupervisedGraduationPolicySchema, TaskProfileSchema, validateIntentOutputSchema, } from "./schemas.ts";
|
|
3
|
+
export type JsonPrimitive = string | number | boolean | null;
|
|
4
|
+
export type JsonValue = JsonPrimitive | {
|
|
5
|
+
readonly [key: string]: JsonValue;
|
|
6
|
+
} | readonly JsonValue[];
|
|
7
|
+
export type IntentKind = "establish_evals" | "instrument_observability" | "run_assessment" | "discover_candidates" | "run_challenger_trials" | "calculate_frontier" | "investigate_regression" | "apply_model_swap" | "verify_model_swap" | "rollback_model_swap";
|
|
8
|
+
export type HarnessCapability = "user_confirmation" | "evaluation_authoring" | "observability_instrumentation" | "production_observation" | "candidate_discovery" | "challenger_execution" | "frontier_calculation" | "model_mutation" | "post_swap_verification" | "rollback";
|
|
9
|
+
export interface HarnessDescriptor {
|
|
10
|
+
readonly id: string;
|
|
11
|
+
readonly name: string;
|
|
12
|
+
readonly version: string;
|
|
13
|
+
readonly protocolVersions: readonly string[];
|
|
14
|
+
readonly capabilities: readonly HarnessCapability[];
|
|
15
|
+
readonly executionModes: readonly ("interactive" | "non_interactive" | "remote")[];
|
|
16
|
+
readonly structuredOutput: {
|
|
17
|
+
readonly schemaDialect: typeof import("./schemas.ts").OPENMERIT_SCHEMA_DIALECT;
|
|
18
|
+
readonly presentation: "dynamic_tool" | "static_tool_union" | "adapter_validation";
|
|
19
|
+
readonly enforcement: "provider_and_harness" | "harness" | "adapter";
|
|
20
|
+
};
|
|
21
|
+
readonly automation: {
|
|
22
|
+
readonly supportedSignals: readonly AutomationSignalType[];
|
|
23
|
+
readonly persistentScheduling: boolean;
|
|
24
|
+
readonly backgroundExecution: boolean;
|
|
25
|
+
readonly wakeupProvisioning: "none" | "native" | "infrastructure_change";
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
export interface HarnessDispatchReceipt {
|
|
29
|
+
readonly accepted: boolean;
|
|
30
|
+
readonly harnessJobId?: string;
|
|
31
|
+
readonly reason?: string;
|
|
32
|
+
}
|
|
33
|
+
export interface EvaluationBudget {
|
|
34
|
+
readonly currency: string;
|
|
35
|
+
readonly maximumSpend: number;
|
|
36
|
+
readonly maximumCandidateCount?: number;
|
|
37
|
+
readonly maximumRunsPerCandidate?: number;
|
|
38
|
+
readonly expiresAt?: string;
|
|
39
|
+
}
|
|
40
|
+
export interface SupervisedGraduationPolicy {
|
|
41
|
+
readonly mode: "supervised_graduation";
|
|
42
|
+
readonly evaluationBudget: EvaluationBudget;
|
|
43
|
+
readonly automaticSwapsEnabled: boolean;
|
|
44
|
+
readonly confirmationRequiredUntilVerifiedSwaps: number;
|
|
45
|
+
readonly requirePostSwapVerification: true;
|
|
46
|
+
readonly rollbackOnRegression: boolean;
|
|
47
|
+
readonly checkPolicy: AutomationCheckPolicy;
|
|
48
|
+
}
|
|
49
|
+
export interface CheckCadence {
|
|
50
|
+
readonly afterCompletedTasks?: number;
|
|
51
|
+
readonly afterElapsedSeconds?: number;
|
|
52
|
+
}
|
|
53
|
+
export interface MetricRegressionThreshold {
|
|
54
|
+
readonly metricId: MetricId;
|
|
55
|
+
/** Fractional deterioration from the last measured value; 0.1 means 10%. */
|
|
56
|
+
readonly relativeChangeAtLeast: number;
|
|
57
|
+
}
|
|
58
|
+
export interface AutomationCheckPolicy {
|
|
59
|
+
readonly baselineAssessment: CheckCadence;
|
|
60
|
+
readonly frontierReassessment: CheckCadence & {
|
|
61
|
+
readonly onModelCatalogChange: boolean;
|
|
62
|
+
};
|
|
63
|
+
readonly regressionCheck: CheckCadence & {
|
|
64
|
+
readonly thresholds: readonly MetricRegressionThreshold[];
|
|
65
|
+
};
|
|
66
|
+
readonly postSwapVerification: CheckCadence;
|
|
67
|
+
readonly cooldownSeconds: number;
|
|
68
|
+
readonly execution: {
|
|
69
|
+
readonly mode: "active_session_only" | "persistent";
|
|
70
|
+
readonly infrastructureChangesAllowed: boolean;
|
|
71
|
+
};
|
|
72
|
+
}
|
|
73
|
+
export type IntentStatus = "requested" | "accepted" | "running" | "waiting_for_user" | "succeeded" | "failed" | "cancelled";
|
|
74
|
+
export type TerminalIntentStatus = Extract<IntentStatus, "succeeded" | "failed" | "cancelled">;
|
|
75
|
+
export interface EvidenceRequirement {
|
|
76
|
+
readonly id: string;
|
|
77
|
+
readonly description: string;
|
|
78
|
+
readonly metricProfileId?: string;
|
|
79
|
+
}
|
|
80
|
+
/**
|
|
81
|
+
* A durable, harness-neutral request for intelligent work.
|
|
82
|
+
*
|
|
83
|
+
* `requestedOutcome` says what must be true when the work is complete.
|
|
84
|
+
* `constraints` carries JSON-safe task constraints without prescribing how a
|
|
85
|
+
* harness should satisfy them.
|
|
86
|
+
*/
|
|
87
|
+
export interface OpenMeritIntent {
|
|
88
|
+
readonly protocolVersion: typeof PROTOCOL_VERSION;
|
|
89
|
+
readonly id: string;
|
|
90
|
+
readonly kind: IntentKind;
|
|
91
|
+
readonly targetId?: string;
|
|
92
|
+
readonly taskProfileId: string;
|
|
93
|
+
readonly authorizationPolicyId: string;
|
|
94
|
+
readonly requestedOutcome: string;
|
|
95
|
+
readonly constraints: Readonly<Record<string, JsonValue>>;
|
|
96
|
+
readonly requiredEvidence: readonly EvidenceRequirement[];
|
|
97
|
+
readonly requestedAt: string;
|
|
98
|
+
}
|
|
99
|
+
export interface IntentLifecycleUpdate {
|
|
100
|
+
readonly protocolVersion: typeof PROTOCOL_VERSION;
|
|
101
|
+
readonly intentId: string;
|
|
102
|
+
readonly status: IntentStatus;
|
|
103
|
+
readonly occurredAt: string;
|
|
104
|
+
readonly message?: string;
|
|
105
|
+
readonly progress?: {
|
|
106
|
+
readonly completed: number;
|
|
107
|
+
readonly total?: number;
|
|
108
|
+
readonly unit?: string;
|
|
109
|
+
};
|
|
110
|
+
readonly details?: Readonly<Record<string, JsonValue>>;
|
|
111
|
+
}
|
|
112
|
+
export type EvidenceSource = "harness_trace" | "evaluation_artifact" | "observability_record" | "external_source" | "user_feedback";
|
|
113
|
+
/** A stable pointer to evidence stored outside the protocol message. */
|
|
114
|
+
export interface EvidenceReference {
|
|
115
|
+
readonly id: string;
|
|
116
|
+
readonly source: EvidenceSource;
|
|
117
|
+
readonly uri: string;
|
|
118
|
+
readonly mediaType?: string;
|
|
119
|
+
readonly digest?: string;
|
|
120
|
+
}
|
|
121
|
+
interface IntentResultBase {
|
|
122
|
+
readonly protocolVersion: typeof PROTOCOL_VERSION;
|
|
123
|
+
readonly intentId: string;
|
|
124
|
+
readonly completedAt: string;
|
|
125
|
+
readonly summary: string;
|
|
126
|
+
readonly evidence: readonly EvidenceReference[];
|
|
127
|
+
}
|
|
128
|
+
export type IntentResult<K extends IntentKind = IntentKind> = IntentResultBase & ({
|
|
129
|
+
readonly status: "succeeded";
|
|
130
|
+
readonly outputs: IntentOutputMap[K];
|
|
131
|
+
readonly error?: never;
|
|
132
|
+
} | {
|
|
133
|
+
readonly status: "failed";
|
|
134
|
+
readonly outputs?: Readonly<Record<string, JsonValue>>;
|
|
135
|
+
readonly error: {
|
|
136
|
+
readonly code: string;
|
|
137
|
+
readonly message: string;
|
|
138
|
+
readonly retryable: boolean;
|
|
139
|
+
readonly details?: Readonly<Record<string, JsonValue>>;
|
|
140
|
+
};
|
|
141
|
+
} | {
|
|
142
|
+
readonly status: "cancelled";
|
|
143
|
+
readonly outputs?: Readonly<Record<string, JsonValue>>;
|
|
144
|
+
readonly error?: {
|
|
145
|
+
readonly code: string;
|
|
146
|
+
readonly message: string;
|
|
147
|
+
readonly retryable: boolean;
|
|
148
|
+
readonly details?: Readonly<Record<string, JsonValue>>;
|
|
149
|
+
};
|
|
150
|
+
});
|
|
151
|
+
export type MetricId = "task_quality" | "task_success" | "consistency" | "instruction_following" | "tool_use_performance" | "structured_output_reliability" | "hallucination_rate" | "reasoning_efficiency" | "input_token_consumption" | "output_token_consumption" | "total_task_cost" | "cost_per_successful_task" | "time_to_first_useful_output" | "end_to_end_task_latency" | "throughput" | "retry_rate" | "agent_steps" | "recovery_ability" | "context_handling" | "retrieval_use_quality" | "long_horizon_performance" | "human_intervention_rate" | "preference_fit";
|
|
152
|
+
export type MetricDirection = "maximize" | "minimize";
|
|
153
|
+
export type MetricScope = "single_run" | "sample_window";
|
|
154
|
+
export type MetricState = "measured" | "not_applicable" | "not_configured" | "insufficient_evidence";
|
|
155
|
+
export type MeasurementMethod = "observed" | "derived" | "evaluator" | "human" | "external";
|
|
156
|
+
export interface MetricDefinition {
|
|
157
|
+
readonly id: MetricId;
|
|
158
|
+
readonly label: string;
|
|
159
|
+
readonly direction: MetricDirection;
|
|
160
|
+
readonly defaultUnit: string;
|
|
161
|
+
readonly requiresRepeatedRuns: boolean;
|
|
162
|
+
}
|
|
163
|
+
export declare const METRIC_CATALOG: readonly MetricDefinition[];
|
|
164
|
+
export interface MetricRecord {
|
|
165
|
+
readonly targetId: string;
|
|
166
|
+
readonly metricId: MetricId;
|
|
167
|
+
readonly state: MetricState;
|
|
168
|
+
readonly scope: MetricScope;
|
|
169
|
+
readonly direction: MetricDirection;
|
|
170
|
+
readonly method: MeasurementMethod;
|
|
171
|
+
readonly aggregation: "single_value" | "sum" | "mean" | "rate" | "distribution" | "percentile";
|
|
172
|
+
readonly value?: number;
|
|
173
|
+
readonly unit: string;
|
|
174
|
+
readonly sampleCount: number;
|
|
175
|
+
readonly interval?: {
|
|
176
|
+
readonly lower: number;
|
|
177
|
+
readonly upper: number;
|
|
178
|
+
readonly confidenceLevel?: number;
|
|
179
|
+
};
|
|
180
|
+
readonly window?: {
|
|
181
|
+
readonly startedAt: string;
|
|
182
|
+
readonly endedAt: string;
|
|
183
|
+
};
|
|
184
|
+
readonly evidence: readonly EvidenceReference[];
|
|
185
|
+
readonly note?: string;
|
|
186
|
+
}
|
|
187
|
+
export interface MetricObjective {
|
|
188
|
+
readonly metricId: MetricId;
|
|
189
|
+
readonly required: boolean;
|
|
190
|
+
readonly minimumSamples: number;
|
|
191
|
+
readonly tolerance: number;
|
|
192
|
+
readonly samplingPlan: {
|
|
193
|
+
readonly strategy: "single_run" | "repeated_same_input" | "representative_task_set" | "repeated_representative_task_set" | "load_test";
|
|
194
|
+
readonly representativeCaseCount: number;
|
|
195
|
+
readonly repetitionsPerCase: number;
|
|
196
|
+
readonly aggregation: "single_value" | "mean" | "rate" | "distribution" | "percentile";
|
|
197
|
+
readonly rationale: string;
|
|
198
|
+
};
|
|
199
|
+
readonly constraint?: {
|
|
200
|
+
readonly operator: "at_least" | "at_most";
|
|
201
|
+
readonly value: number;
|
|
202
|
+
};
|
|
203
|
+
}
|
|
204
|
+
/** A user-confirmed application LLM route. The coding harness itself is never an evaluation target. */
|
|
205
|
+
export interface ApplicationLlmTarget {
|
|
206
|
+
readonly id: string;
|
|
207
|
+
readonly kind: "application_llm_call";
|
|
208
|
+
readonly name: string;
|
|
209
|
+
readonly applicationId: string;
|
|
210
|
+
readonly callSites: readonly string[];
|
|
211
|
+
readonly routeKey: string;
|
|
212
|
+
readonly confirmedAt: string;
|
|
213
|
+
}
|
|
214
|
+
export interface TaskProfile {
|
|
215
|
+
readonly id: string;
|
|
216
|
+
readonly revision: number;
|
|
217
|
+
readonly name: string;
|
|
218
|
+
readonly goal: string;
|
|
219
|
+
readonly applicationTarget: ApplicationLlmTarget;
|
|
220
|
+
readonly objectives: readonly MetricObjective[];
|
|
221
|
+
readonly createdAt: string;
|
|
222
|
+
readonly confirmedAt?: string;
|
|
223
|
+
}
|
|
224
|
+
export type ObservabilityReadinessStatus = "ready" | "incomplete";
|
|
225
|
+
export interface ObservabilityCoverage {
|
|
226
|
+
readonly targetId: string;
|
|
227
|
+
readonly status: ObservabilityReadinessStatus;
|
|
228
|
+
readonly requiredMetricIds: readonly MetricId[];
|
|
229
|
+
readonly coveredMetricIds: readonly MetricId[];
|
|
230
|
+
readonly missingMetricIds: readonly MetricId[];
|
|
231
|
+
readonly checkedAt: string;
|
|
232
|
+
}
|
|
233
|
+
export interface CandidateAssessment {
|
|
234
|
+
readonly id: string;
|
|
235
|
+
readonly targetId: string;
|
|
236
|
+
readonly taskProfileId: string;
|
|
237
|
+
readonly taskProfileRevision: number;
|
|
238
|
+
readonly candidateId: string;
|
|
239
|
+
readonly modelId: string;
|
|
240
|
+
readonly modelVersion?: string;
|
|
241
|
+
readonly productRevision: string;
|
|
242
|
+
readonly harnessId: string;
|
|
243
|
+
readonly startedAt: string;
|
|
244
|
+
readonly completedAt: string;
|
|
245
|
+
readonly metrics: readonly MetricRecord[];
|
|
246
|
+
}
|
|
247
|
+
export type ComparisonState = "dominates" | "does_not_dominate" | "unresolved";
|
|
248
|
+
export interface DominanceComparison {
|
|
249
|
+
readonly candidateId: string;
|
|
250
|
+
readonly otherCandidateId: string;
|
|
251
|
+
readonly state: ComparisonState;
|
|
252
|
+
readonly reasons: readonly string[];
|
|
253
|
+
}
|
|
254
|
+
export interface FrontierSnapshot {
|
|
255
|
+
readonly id: string;
|
|
256
|
+
readonly targetId: string;
|
|
257
|
+
readonly taskProfileId: string;
|
|
258
|
+
readonly taskProfileRevision: number;
|
|
259
|
+
readonly productRevision: string;
|
|
260
|
+
readonly createdAt: string;
|
|
261
|
+
readonly frontierCandidateIds: readonly string[];
|
|
262
|
+
readonly dominatedCandidateIds: readonly string[];
|
|
263
|
+
readonly ineligibleCandidateIds: readonly string[];
|
|
264
|
+
readonly unresolvedCandidateIds: readonly string[];
|
|
265
|
+
readonly comparisons: readonly DominanceComparison[];
|
|
266
|
+
readonly assessmentIds: readonly string[];
|
|
267
|
+
readonly calculationEvidence: readonly EvidenceReference[];
|
|
268
|
+
}
|
|
269
|
+
export interface FrontierRecommendation {
|
|
270
|
+
readonly targetId: string;
|
|
271
|
+
readonly frontierSnapshotId: string;
|
|
272
|
+
readonly selectedCandidateId: string;
|
|
273
|
+
readonly currentCandidateId: string;
|
|
274
|
+
readonly reasons: readonly string[];
|
|
275
|
+
readonly preferenceEvidence: readonly EvidenceReference[];
|
|
276
|
+
}
|
|
277
|
+
export interface CandidateDescriptor {
|
|
278
|
+
readonly targetId: string;
|
|
279
|
+
readonly candidateId: string;
|
|
280
|
+
readonly modelId: string;
|
|
281
|
+
readonly modelVersion?: string;
|
|
282
|
+
readonly reasons: readonly string[];
|
|
283
|
+
}
|
|
284
|
+
/** Research priors only. These are never MetricRecords or frontier evidence. */
|
|
285
|
+
export interface PreliminaryModelLead {
|
|
286
|
+
readonly targetId: string;
|
|
287
|
+
readonly modelId: string;
|
|
288
|
+
readonly modelVersion?: string;
|
|
289
|
+
readonly estimatedPrice: {
|
|
290
|
+
readonly currency: string;
|
|
291
|
+
readonly inputPerMillionTokens: number;
|
|
292
|
+
readonly outputPerMillionTokens: number;
|
|
293
|
+
readonly sourceUri: string;
|
|
294
|
+
readonly checkedAt: string;
|
|
295
|
+
};
|
|
296
|
+
readonly reputation: {
|
|
297
|
+
readonly summary: string;
|
|
298
|
+
readonly sourceUri: string;
|
|
299
|
+
readonly checkedAt: string;
|
|
300
|
+
};
|
|
301
|
+
}
|
|
302
|
+
export interface IntentOutputMap {
|
|
303
|
+
readonly establish_evals: {
|
|
304
|
+
readonly taskProfile: TaskProfile;
|
|
305
|
+
/** The application model configured before any challenger trial. */
|
|
306
|
+
readonly incumbent: CandidateDescriptor;
|
|
307
|
+
readonly policy: SupervisedGraduationPolicy;
|
|
308
|
+
readonly observability: ObservabilityCoverage;
|
|
309
|
+
readonly preliminaryModelLeads?: readonly PreliminaryModelLead[];
|
|
310
|
+
};
|
|
311
|
+
readonly instrument_observability: {
|
|
312
|
+
readonly targetId: string;
|
|
313
|
+
readonly coveredMetricIds: readonly MetricId[];
|
|
314
|
+
readonly missingMetricIds: readonly MetricId[];
|
|
315
|
+
};
|
|
316
|
+
readonly run_assessment: {
|
|
317
|
+
readonly targetId: string;
|
|
318
|
+
readonly baselineEvidenceSufficient: boolean;
|
|
319
|
+
readonly metrics: readonly MetricRecord[];
|
|
320
|
+
};
|
|
321
|
+
readonly discover_candidates: {
|
|
322
|
+
readonly targetId: string;
|
|
323
|
+
readonly candidates: readonly CandidateDescriptor[];
|
|
324
|
+
};
|
|
325
|
+
readonly run_challenger_trials: {
|
|
326
|
+
readonly targetId: string;
|
|
327
|
+
readonly assessments: readonly CandidateAssessment[];
|
|
328
|
+
readonly experimentManifest: EvidenceReference;
|
|
329
|
+
};
|
|
330
|
+
readonly calculate_frontier: {
|
|
331
|
+
readonly targetId: string;
|
|
332
|
+
readonly assessments: readonly CandidateAssessment[];
|
|
333
|
+
readonly frontierSnapshot: FrontierSnapshot;
|
|
334
|
+
readonly recommendation?: FrontierRecommendation;
|
|
335
|
+
};
|
|
336
|
+
readonly investigate_regression: {
|
|
337
|
+
readonly targetId: string;
|
|
338
|
+
readonly regressionDetected: boolean;
|
|
339
|
+
readonly affectedMetricIds: readonly MetricId[];
|
|
340
|
+
readonly likelyCause?: string;
|
|
341
|
+
};
|
|
342
|
+
readonly apply_model_swap: {
|
|
343
|
+
readonly targetId: string;
|
|
344
|
+
readonly appliedCandidateId: string;
|
|
345
|
+
readonly previousCandidateId?: string;
|
|
346
|
+
};
|
|
347
|
+
readonly verify_model_swap: {
|
|
348
|
+
readonly targetId: string;
|
|
349
|
+
readonly verified: boolean;
|
|
350
|
+
readonly regressionDetected: boolean;
|
|
351
|
+
};
|
|
352
|
+
readonly rollback_model_swap: {
|
|
353
|
+
readonly targetId: string;
|
|
354
|
+
readonly restoredCandidateId: string;
|
|
355
|
+
};
|
|
356
|
+
}
|
|
357
|
+
export type AnyIntentResult = {
|
|
358
|
+
readonly [K in IntentKind]: IntentResult<K>;
|
|
359
|
+
}[IntentKind];
|
|
360
|
+
export declare const REQUIRED_CAPABILITIES: Readonly<Record<IntentKind, readonly HarnessCapability[]>>;
|
|
361
|
+
export type ImprovementStage = "unconfigured" | "establishing_evidence" | "collecting_baseline" | "baseline_ready" | "discovering_candidates" | "candidates_ready" | "running_challenger_trials" | "challengers_ready" | "calculating_frontier" | "frontier_ready" | "awaiting_swap_approval" | "swap_approved" | "applying_swap" | "swap_applied" | "verifying_swap" | "swap_verified" | "verification_failed" | "rolling_back" | "monitoring";
|
|
362
|
+
export interface ImprovementCycleState {
|
|
363
|
+
readonly id: string;
|
|
364
|
+
readonly targetId?: string;
|
|
365
|
+
readonly taskProfileId?: string;
|
|
366
|
+
readonly stage: ImprovementStage;
|
|
367
|
+
readonly activeCandidateId?: string;
|
|
368
|
+
readonly previousCandidateId?: string;
|
|
369
|
+
readonly proposedCandidateId?: string;
|
|
370
|
+
readonly verifiedSwapCount: number;
|
|
371
|
+
readonly baselineEvidenceSufficient: boolean;
|
|
372
|
+
readonly observabilityReady?: boolean;
|
|
373
|
+
readonly policy?: SupervisedGraduationPolicy;
|
|
374
|
+
}
|
|
375
|
+
export interface OpenMeritProjectConfig {
|
|
376
|
+
readonly schemaVersion: 1;
|
|
377
|
+
readonly activeTaskProfileId?: string;
|
|
378
|
+
readonly taskProfiles: readonly TaskProfile[];
|
|
379
|
+
readonly policy?: SupervisedGraduationPolicy;
|
|
380
|
+
readonly preliminaryModelLeads?: readonly PreliminaryModelLead[];
|
|
381
|
+
}
|
|
382
|
+
export interface OpenMeritProjectState {
|
|
383
|
+
readonly schemaVersion: 1;
|
|
384
|
+
readonly cycle: ImprovementCycleState;
|
|
385
|
+
readonly activeIntent?: OpenMeritIntent;
|
|
386
|
+
readonly lastResult?: IntentResult;
|
|
387
|
+
/** Durable handoffs between discovery, trials, and frontier calculation. */
|
|
388
|
+
readonly candidates?: readonly CandidateDescriptor[];
|
|
389
|
+
readonly assessments?: readonly CandidateAssessment[];
|
|
390
|
+
readonly experimentManifest?: EvidenceReference;
|
|
391
|
+
readonly observability?: ObservabilityCoverage;
|
|
392
|
+
readonly automation: {
|
|
393
|
+
readonly paused?: boolean;
|
|
394
|
+
readonly setupNudgeShown: boolean;
|
|
395
|
+
readonly observedCompletedTasks: number;
|
|
396
|
+
readonly lastAssessmentTaskCount: number;
|
|
397
|
+
readonly modelCatalogFingerprint?: string;
|
|
398
|
+
readonly modelCatalogChangedAt?: string;
|
|
399
|
+
readonly processedSignalIds?: readonly string[];
|
|
400
|
+
readonly pendingIntentKind?: IntentKind;
|
|
401
|
+
readonly pendingIntentReason?: string;
|
|
402
|
+
readonly lastAssessmentAt?: string;
|
|
403
|
+
readonly baselineStartedAt?: string;
|
|
404
|
+
readonly lastReassessmentTaskCount?: number;
|
|
405
|
+
readonly lastReassessmentAt?: string;
|
|
406
|
+
readonly lastRegressionCheckTaskCount?: number;
|
|
407
|
+
readonly lastRegressionCheckAt?: string;
|
|
408
|
+
readonly swapAppliedTaskCount?: number;
|
|
409
|
+
readonly swapAppliedAt?: string;
|
|
410
|
+
readonly lastVerificationAt?: string;
|
|
411
|
+
readonly latestMetricValues?: Readonly<Partial<Record<MetricId, number>>>;
|
|
412
|
+
/** Latest reported aggregate application-run window per metric; overlapping windows are never summed. */
|
|
413
|
+
readonly baselineMetricWindows?: readonly MetricRecord[];
|
|
414
|
+
/** Spend from controlled challenger trials; ordinary product work is excluded. */
|
|
415
|
+
readonly evaluationSpend?: number;
|
|
416
|
+
};
|
|
417
|
+
readonly updatedAt: string;
|
|
418
|
+
}
|
|
419
|
+
export type AuditEventType = "project_initialized" | "intent_requested" | "intent_progressed" | "intent_completed" | "requirements_confirmation_required" | "swap_confirmation_required" | "frontier_verified" | "frontier_rejected" | "cycle_advanced" | "setup_nudged" | "task_observed" | "model_catalog_changed" | "automation_signal_recorded" | "automation_signal_duplicate" | "automation_check_due" | "baseline_sufficiency_checked" | "observability_readiness_checked";
|
|
420
|
+
interface AutomationSignalBase {
|
|
421
|
+
readonly protocolVersion: typeof PROTOCOL_VERSION;
|
|
422
|
+
readonly id: string;
|
|
423
|
+
readonly occurredAt: string;
|
|
424
|
+
readonly targetId: string;
|
|
425
|
+
readonly evidence?: readonly EvidenceReference[];
|
|
426
|
+
}
|
|
427
|
+
export type AutomationSignal = (AutomationSignalBase & {
|
|
428
|
+
readonly type: "product_task_completed";
|
|
429
|
+
/** The model used by the application target, never the coding harness model. */
|
|
430
|
+
readonly modelId: string;
|
|
431
|
+
/** Provider telemetry captured from the application LLM call. */
|
|
432
|
+
readonly telemetry?: {
|
|
433
|
+
readonly startedAt: string;
|
|
434
|
+
readonly completedAt: string;
|
|
435
|
+
readonly durationMs: number;
|
|
436
|
+
readonly inputTokens: number;
|
|
437
|
+
readonly outputTokens: number;
|
|
438
|
+
readonly cacheReadTokens: number;
|
|
439
|
+
readonly cacheWriteTokens: number;
|
|
440
|
+
readonly cost: number;
|
|
441
|
+
readonly currency: "USD";
|
|
442
|
+
};
|
|
443
|
+
}) | (AutomationSignalBase & {
|
|
444
|
+
readonly type: "metric_window_available";
|
|
445
|
+
readonly metrics: readonly MetricRecord[];
|
|
446
|
+
}) | (AutomationSignalBase & {
|
|
447
|
+
readonly type: "scheduled_tick";
|
|
448
|
+
}) | (AutomationSignalBase & {
|
|
449
|
+
readonly type: "model_catalog_changed";
|
|
450
|
+
readonly fingerprint: string;
|
|
451
|
+
}) | (AutomationSignalBase & {
|
|
452
|
+
readonly type: "verification_window_completed";
|
|
453
|
+
readonly metrics?: readonly MetricRecord[];
|
|
454
|
+
});
|
|
455
|
+
export type AutomationSignalType = AutomationSignal["type"];
|
|
456
|
+
export interface HarnessWakeupRequest {
|
|
457
|
+
readonly at: string;
|
|
458
|
+
readonly signal: "scheduled_tick";
|
|
459
|
+
readonly reason: string;
|
|
460
|
+
}
|
|
461
|
+
export interface AutomationPlan {
|
|
462
|
+
readonly mode: AutomationCheckPolicy["execution"]["mode"];
|
|
463
|
+
readonly requiredSignals: readonly AutomationSignalType[];
|
|
464
|
+
readonly nextWakeupAt?: string;
|
|
465
|
+
readonly scheduling: "native" | "active_session_fallback" | "external_scheduler_required";
|
|
466
|
+
readonly gaps: readonly string[];
|
|
467
|
+
}
|
|
468
|
+
export interface OpenMeritAuditEvent {
|
|
469
|
+
readonly schemaVersion: 1;
|
|
470
|
+
readonly id: string;
|
|
471
|
+
readonly type: AuditEventType;
|
|
472
|
+
readonly occurredAt: string;
|
|
473
|
+
readonly cycleId: string;
|
|
474
|
+
readonly intentId?: string;
|
|
475
|
+
readonly data: Readonly<Record<string, JsonValue>>;
|
|
476
|
+
}
|
|
477
|
+
export interface EvidenceManifest {
|
|
478
|
+
readonly schemaVersion: 1;
|
|
479
|
+
readonly id: string;
|
|
480
|
+
readonly createdAt: string;
|
|
481
|
+
readonly taskProfileId: string;
|
|
482
|
+
readonly references: readonly EvidenceReference[];
|
|
483
|
+
}
|
|
484
|
+
export declare function isTerminalIntentStatus(status: IntentStatus): status is TerminalIntentStatus;
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
export const PROTOCOL_VERSION = "0.1";
|
|
2
|
+
export { ApplicationLlmTargetSchema, CandidateAssessmentSchema, CandidateDescriptorSchema, completionToolSchemaFor, EvidenceReferenceSchema, FrontierRecommendationSchema, FrontierSnapshotSchema, INTENT_OUTPUT_SCHEMAS, intentOutputSchemaFor, MetricObjectiveSchema, MetricRecordSchema, ObservabilityCoverageSchema, OPENMERIT_RESULT_SCHEMA_VERSION, OPENMERIT_SCHEMA_DIALECT, SupervisedGraduationPolicySchema, TaskProfileSchema, validateIntentOutputSchema, } from "./schemas.js";
|
|
3
|
+
export const METRIC_CATALOG = [
|
|
4
|
+
{ id: "task_quality", label: "Task quality", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
5
|
+
{ id: "task_success", label: "Task success rate", direction: "maximize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
6
|
+
{ id: "consistency", label: "Consistency", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: true },
|
|
7
|
+
{ id: "instruction_following", label: "Instruction following", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
8
|
+
{ id: "tool_use_performance", label: "Tool-use performance", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
9
|
+
{ id: "structured_output_reliability", label: "Structured-output reliability", direction: "maximize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
10
|
+
{ id: "hallucination_rate", label: "Hallucination rate", direction: "minimize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
11
|
+
{ id: "reasoning_efficiency", label: "Reasoning efficiency", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
12
|
+
{ id: "input_token_consumption", label: "Input-token consumption", direction: "minimize", defaultUnit: "tokens", requiresRepeatedRuns: false },
|
|
13
|
+
{ id: "output_token_consumption", label: "Output-token consumption", direction: "minimize", defaultUnit: "tokens", requiresRepeatedRuns: false },
|
|
14
|
+
{ id: "total_task_cost", label: "Total task cost", direction: "minimize", defaultUnit: "usd", requiresRepeatedRuns: false },
|
|
15
|
+
{ id: "cost_per_successful_task", label: "Cost per successful task", direction: "minimize", defaultUnit: "usd", requiresRepeatedRuns: true },
|
|
16
|
+
{ id: "time_to_first_useful_output", label: "Time to first useful output", direction: "minimize", defaultUnit: "milliseconds", requiresRepeatedRuns: false },
|
|
17
|
+
{ id: "end_to_end_task_latency", label: "End-to-end task latency", direction: "minimize", defaultUnit: "milliseconds", requiresRepeatedRuns: false },
|
|
18
|
+
{ id: "throughput", label: "Throughput under workload", direction: "maximize", defaultUnit: "tasks_per_hour", requiresRepeatedRuns: true },
|
|
19
|
+
{ id: "retry_rate", label: "Retry rate", direction: "minimize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
20
|
+
{ id: "agent_steps", label: "Agent steps", direction: "minimize", defaultUnit: "count", requiresRepeatedRuns: false },
|
|
21
|
+
{ id: "recovery_ability", label: "Recovery ability", direction: "maximize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
22
|
+
{ id: "context_handling", label: "Context handling", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
23
|
+
{ id: "retrieval_use_quality", label: "Retrieval-use quality", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: false },
|
|
24
|
+
{ id: "long_horizon_performance", label: "Long-horizon performance", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: true },
|
|
25
|
+
{ id: "human_intervention_rate", label: "Human intervention rate", direction: "minimize", defaultUnit: "ratio", requiresRepeatedRuns: true },
|
|
26
|
+
{ id: "preference_fit", label: "Preference and style fit", direction: "maximize", defaultUnit: "score", requiresRepeatedRuns: true },
|
|
27
|
+
];
|
|
28
|
+
export const REQUIRED_CAPABILITIES = {
|
|
29
|
+
establish_evals: ["user_confirmation", "evaluation_authoring", "observability_instrumentation"],
|
|
30
|
+
instrument_observability: ["observability_instrumentation"],
|
|
31
|
+
run_assessment: ["production_observation"],
|
|
32
|
+
discover_candidates: ["candidate_discovery"],
|
|
33
|
+
run_challenger_trials: ["challenger_execution"],
|
|
34
|
+
calculate_frontier: ["frontier_calculation"],
|
|
35
|
+
investigate_regression: ["production_observation"],
|
|
36
|
+
apply_model_swap: ["model_mutation"],
|
|
37
|
+
verify_model_swap: ["post_swap_verification"],
|
|
38
|
+
rollback_model_swap: ["rollback"],
|
|
39
|
+
};
|
|
40
|
+
const terminalStatuses = new Set([
|
|
41
|
+
"succeeded",
|
|
42
|
+
"failed",
|
|
43
|
+
"cancelled",
|
|
44
|
+
]);
|
|
45
|
+
export function isTerminalIntentStatus(status) {
|
|
46
|
+
return terminalStatuses.has(status);
|
|
47
|
+
}
|