openmerit 0.1.3 → 0.1.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/CHANGELOG.md +26 -0
  2. package/README.md +83 -312
  3. package/dist/core/src/index.d.ts +90 -0
  4. package/dist/core/src/index.js +1137 -0
  5. package/dist/core/src/store.d.ts +35 -0
  6. package/dist/core/src/store.js +102 -0
  7. package/dist/pi/src/index.d.ts +15 -0
  8. package/dist/pi/src/index.js +423 -0
  9. package/dist/protocol/src/index.d.ts +402 -0
  10. package/dist/protocol/src/index.js +47 -0
  11. package/dist/protocol/src/schemas.d.ts +450 -0
  12. package/dist/protocol/src/schemas.js +224 -0
  13. package/docs/adapter-guide.md +172 -0
  14. package/docs/architecture.md +55 -0
  15. package/docs/automation.md +66 -0
  16. package/docs/getting-started.md +55 -0
  17. package/docs/lifecycle.md +30 -0
  18. package/docs/metrics-and-evidence.md +40 -0
  19. package/docs/operations.md +31 -0
  20. package/docs/pareto-spec.md +76 -0
  21. package/docs/pi-extension.md +44 -0
  22. package/docs/roadmap.md +26 -0
  23. package/docs/security.md +23 -0
  24. package/docs/testing.md +36 -0
  25. package/docs/ux-reference.md +32 -0
  26. package/package.json +45 -54
  27. package/benchmark/invoice_ocr/data/invoice_01_ground_truth.json +0 -38
  28. package/benchmark/invoice_ocr/data/invoice_01_row_2.jpg +0 -0
  29. package/benchmark/invoice_ocr/data/invoice_02_ground_truth.json +0 -32
  30. package/benchmark/invoice_ocr/data/invoice_02_row_5.jpg +0 -0
  31. package/benchmark/invoice_ocr/data/invoice_03_ground_truth.json +0 -26
  32. package/benchmark/invoice_ocr/data/invoice_03_row_6.jpg +0 -0
  33. package/benchmark/invoice_ocr/data/invoice_04_ground_truth.json +0 -26
  34. package/benchmark/invoice_ocr/data/invoice_04_row_7.jpg +0 -0
  35. package/benchmark/invoice_ocr/data/invoice_05_ground_truth.json +0 -38
  36. package/benchmark/invoice_ocr/data/invoice_05_row_947.jpg +0 -0
  37. package/benchmark/invoice_ocr/data/invoice_06_ground_truth.json +0 -38
  38. package/benchmark/invoice_ocr/data/invoice_06_row_948.jpg +0 -0
  39. package/benchmark/invoice_ocr/data/invoice_07_ground_truth.json +0 -20
  40. package/benchmark/invoice_ocr/data/invoice_07_row_949.jpg +0 -0
  41. package/benchmark/invoice_ocr/data/invoice_08_ground_truth.json +0 -38
  42. package/benchmark/invoice_ocr/data/invoice_08_row_1888.jpg +0 -0
  43. package/benchmark/invoice_ocr/data/invoice_09_ground_truth.json +0 -26
  44. package/benchmark/invoice_ocr/data/invoice_09_row_1890.jpg +0 -0
  45. package/benchmark/invoice_ocr/data/invoice_10_ground_truth.json +0 -20
  46. package/benchmark/invoice_ocr/data/invoice_10_row_1892.jpg +0 -0
  47. package/benchmark/invoice_ocr/data/manifest.json +0 -97
  48. package/dist/benchmarks.js +0 -98
  49. package/dist/catalog.js +0 -61
  50. package/dist/cli.js +0 -188
  51. package/dist/daemon.js +0 -388
  52. package/dist/diagnostics.js +0 -194
  53. package/dist/frontier.js +0 -56
  54. package/dist/harness.js +0 -1
  55. package/dist/integrations.js +0 -19
  56. package/dist/invoice-eval.js +0 -33
  57. package/dist/invoice-score.js +0 -124
  58. package/dist/judge.js +0 -43
  59. package/dist/llm.js +0 -203
  60. package/dist/pi-trials.js +0 -366
  61. package/dist/policy.js +0 -115
  62. package/dist/providers.js +0 -1
  63. package/dist/recommend.js +0 -76
  64. package/dist/routes.js +0 -59
  65. package/dist/standalone.js +0 -220
  66. package/dist/store.js +0 -89
  67. package/dist/strategist.js +0 -68
  68. package/dist/task-input.js +0 -54
  69. package/dist/traces.js +0 -127
  70. package/dist/trials.js +0 -140
  71. package/dist/types.js +0 -2
  72. package/examples/invoice-prompt.txt +0 -19
  73. package/examples/task.example.json +0 -7
  74. package/extension/openmerit.ts +0 -820
  75. package/instructions/OPENMERIT.md +0 -54
  76. package/instructions/openmerit.policy.json +0 -33
  77. package/rules.md +0 -39
@@ -0,0 +1,450 @@
1
+ import { Type, type TSchema } from "typebox";
2
+ import type { IntentKind } from "./index.ts";
3
+ export declare const OPENMERIT_SCHEMA_DIALECT: "https://json-schema.org/draft/2020-12/schema";
4
+ export declare const OPENMERIT_RESULT_SCHEMA_VERSION: "0.2";
5
+ export declare const EvidenceReferenceSchema: Type.TObject<{
6
+ id: Type.TString;
7
+ source: Type.TUnion<Type.TLiteral<string>[]>;
8
+ uri: Type.TString;
9
+ mediaType: Type.TOptional<Type.TString>;
10
+ digest: Type.TOptional<Type.TString>;
11
+ }>;
12
+ export declare const MetricObjectiveSchema: Type.TObject<{
13
+ metricId: Type.TUnion<Type.TLiteral<string>[]>;
14
+ required: Type.TBoolean;
15
+ minimumSamples: Type.TInteger;
16
+ tolerance: Type.TNumber;
17
+ constraint: Type.TOptional<Type.TObject<{
18
+ operator: Type.TUnion<Type.TLiteral<string>[]>;
19
+ value: Type.TNumber;
20
+ }>>;
21
+ }>;
22
+ export declare const TaskProfileSchema: Type.TObject<{
23
+ id: Type.TString;
24
+ revision: Type.TInteger;
25
+ name: Type.TString;
26
+ goal: Type.TString;
27
+ objectives: Type.TArray<Type.TObject<{
28
+ metricId: Type.TUnion<Type.TLiteral<string>[]>;
29
+ required: Type.TBoolean;
30
+ minimumSamples: Type.TInteger;
31
+ tolerance: Type.TNumber;
32
+ constraint: Type.TOptional<Type.TObject<{
33
+ operator: Type.TUnion<Type.TLiteral<string>[]>;
34
+ value: Type.TNumber;
35
+ }>>;
36
+ }>>;
37
+ createdAt: Type.TString;
38
+ confirmedAt: Type.TOptional<Type.TString>;
39
+ }>;
40
+ export declare const ObservabilityCoverageSchema: Type.TObject<{
41
+ status: Type.TUnion<Type.TLiteral<string>[]>;
42
+ requiredMetricIds: Type.TArray<Type.TUnion<Type.TLiteral<string>[]>>;
43
+ coveredMetricIds: Type.TArray<Type.TUnion<Type.TLiteral<string>[]>>;
44
+ missingMetricIds: Type.TArray<Type.TUnion<Type.TLiteral<string>[]>>;
45
+ checkedAt: Type.TString;
46
+ }>;
47
+ export declare const SupervisedGraduationPolicySchema: Type.TObject<{
48
+ mode: Type.TLiteral<"supervised_graduation">;
49
+ evaluationBudget: Type.TObject<{
50
+ currency: Type.TString;
51
+ maximumSpend: Type.TNumber;
52
+ maximumCandidateCount: Type.TOptional<Type.TInteger>;
53
+ maximumRunsPerCandidate: Type.TOptional<Type.TInteger>;
54
+ expiresAt: Type.TOptional<Type.TString>;
55
+ }>;
56
+ automaticSwapsEnabled: Type.TBoolean;
57
+ confirmationRequiredUntilVerifiedSwaps: Type.TInteger;
58
+ requirePostSwapVerification: Type.TLiteral<true>;
59
+ rollbackOnRegression: Type.TBoolean;
60
+ checkPolicy: Type.TObject<{
61
+ baselineAssessment: Type.TObject<{
62
+ afterCompletedTasks: Type.TOptional<Type.TInteger>;
63
+ afterElapsedSeconds: Type.TOptional<Type.TInteger>;
64
+ }>;
65
+ frontierReassessment: Type.TObject<{
66
+ afterCompletedTasks: Type.TOptional<Type.TInteger>;
67
+ afterElapsedSeconds: Type.TOptional<Type.TInteger>;
68
+ onModelCatalogChange: Type.TBoolean;
69
+ }>;
70
+ regressionCheck: Type.TObject<{
71
+ afterCompletedTasks: Type.TOptional<Type.TInteger>;
72
+ afterElapsedSeconds: Type.TOptional<Type.TInteger>;
73
+ thresholds: Type.TArray<Type.TObject<{
74
+ metricId: Type.TUnion<Type.TLiteral<string>[]>;
75
+ relativeChangeAtLeast: Type.TNumber;
76
+ }>>;
77
+ }>;
78
+ postSwapVerification: Type.TObject<{
79
+ afterCompletedTasks: Type.TOptional<Type.TInteger>;
80
+ afterElapsedSeconds: Type.TOptional<Type.TInteger>;
81
+ }>;
82
+ cooldownSeconds: Type.TInteger;
83
+ execution: Type.TObject<{
84
+ mode: Type.TUnion<Type.TLiteral<string>[]>;
85
+ infrastructureChangesAllowed: Type.TBoolean;
86
+ }>;
87
+ }>;
88
+ }>;
89
+ export declare const MetricRecordSchema: Type.TObject<{
90
+ metricId: Type.TUnion<Type.TLiteral<string>[]>;
91
+ state: Type.TUnion<Type.TLiteral<string>[]>;
92
+ scope: Type.TUnion<Type.TLiteral<string>[]>;
93
+ direction: Type.TUnion<Type.TLiteral<string>[]>;
94
+ method: Type.TUnion<Type.TLiteral<string>[]>;
95
+ value: Type.TOptional<Type.TNumber>;
96
+ unit: Type.TString;
97
+ sampleCount: Type.TInteger;
98
+ interval: Type.TOptional<Type.TObject<{
99
+ lower: Type.TNumber;
100
+ upper: Type.TNumber;
101
+ confidenceLevel: Type.TOptional<Type.TNumber>;
102
+ }>>;
103
+ window: Type.TOptional<Type.TObject<{
104
+ startedAt: Type.TString;
105
+ endedAt: Type.TString;
106
+ }>>;
107
+ evidence: Type.TArray<Type.TObject<{
108
+ id: Type.TString;
109
+ source: Type.TUnion<Type.TLiteral<string>[]>;
110
+ uri: Type.TString;
111
+ mediaType: Type.TOptional<Type.TString>;
112
+ digest: Type.TOptional<Type.TString>;
113
+ }>>;
114
+ note: Type.TOptional<Type.TString>;
115
+ }>;
116
+ export declare const CandidateDescriptorSchema: Type.TObject<{
117
+ candidateId: Type.TString;
118
+ modelId: Type.TString;
119
+ modelVersion: Type.TOptional<Type.TString>;
120
+ reasons: Type.TArray<Type.TString>;
121
+ }>;
122
+ export declare const CandidateAssessmentSchema: Type.TObject<{
123
+ id: Type.TString;
124
+ taskProfileId: Type.TString;
125
+ taskProfileRevision: Type.TInteger;
126
+ candidateId: Type.TString;
127
+ modelId: Type.TString;
128
+ modelVersion: Type.TOptional<Type.TString>;
129
+ productRevision: Type.TString;
130
+ harnessId: Type.TString;
131
+ startedAt: Type.TString;
132
+ completedAt: Type.TString;
133
+ metrics: Type.TArray<Type.TObject<{
134
+ metricId: Type.TUnion<Type.TLiteral<string>[]>;
135
+ state: Type.TUnion<Type.TLiteral<string>[]>;
136
+ scope: Type.TUnion<Type.TLiteral<string>[]>;
137
+ direction: Type.TUnion<Type.TLiteral<string>[]>;
138
+ method: Type.TUnion<Type.TLiteral<string>[]>;
139
+ value: Type.TOptional<Type.TNumber>;
140
+ unit: Type.TString;
141
+ sampleCount: Type.TInteger;
142
+ interval: Type.TOptional<Type.TObject<{
143
+ lower: Type.TNumber;
144
+ upper: Type.TNumber;
145
+ confidenceLevel: Type.TOptional<Type.TNumber>;
146
+ }>>;
147
+ window: Type.TOptional<Type.TObject<{
148
+ startedAt: Type.TString;
149
+ endedAt: Type.TString;
150
+ }>>;
151
+ evidence: Type.TArray<Type.TObject<{
152
+ id: Type.TString;
153
+ source: Type.TUnion<Type.TLiteral<string>[]>;
154
+ uri: Type.TString;
155
+ mediaType: Type.TOptional<Type.TString>;
156
+ digest: Type.TOptional<Type.TString>;
157
+ }>>;
158
+ note: Type.TOptional<Type.TString>;
159
+ }>>;
160
+ }>;
161
+ export declare const FrontierSnapshotSchema: Type.TObject<{
162
+ id: Type.TString;
163
+ taskProfileId: Type.TString;
164
+ taskProfileRevision: Type.TInteger;
165
+ productRevision: Type.TString;
166
+ createdAt: Type.TString;
167
+ frontierCandidateIds: Type.TArray<Type.TString>;
168
+ dominatedCandidateIds: Type.TArray<Type.TString>;
169
+ ineligibleCandidateIds: Type.TArray<Type.TString>;
170
+ unresolvedCandidateIds: Type.TArray<Type.TString>;
171
+ comparisons: Type.TArray<Type.TObject<{
172
+ candidateId: Type.TString;
173
+ otherCandidateId: Type.TString;
174
+ state: Type.TUnion<Type.TLiteral<string>[]>;
175
+ reasons: Type.TArray<Type.TString>;
176
+ }>>;
177
+ assessmentIds: Type.TArray<Type.TString>;
178
+ calculationEvidence: Type.TArray<Type.TObject<{
179
+ id: Type.TString;
180
+ source: Type.TUnion<Type.TLiteral<string>[]>;
181
+ uri: Type.TString;
182
+ mediaType: Type.TOptional<Type.TString>;
183
+ digest: Type.TOptional<Type.TString>;
184
+ }>>;
185
+ }>;
186
+ export declare const FrontierRecommendationSchema: Type.TObject<{
187
+ frontierSnapshotId: Type.TString;
188
+ selectedCandidateId: Type.TString;
189
+ currentCandidateId: Type.TString;
190
+ reasons: Type.TArray<Type.TString>;
191
+ preferenceEvidence: Type.TArray<Type.TObject<{
192
+ id: Type.TString;
193
+ source: Type.TUnion<Type.TLiteral<string>[]>;
194
+ uri: Type.TString;
195
+ mediaType: Type.TOptional<Type.TString>;
196
+ digest: Type.TOptional<Type.TString>;
197
+ }>>;
198
+ }>;
199
+ export declare const INTENT_OUTPUT_SCHEMAS: {
200
+ readonly establish_evals: Type.TObject<{
201
+ taskProfile: Type.TObject<{
202
+ confirmedAt: Type.TString;
203
+ id: Type.TString;
204
+ revision: Type.TInteger;
205
+ name: Type.TString;
206
+ goal: Type.TString;
207
+ objectives: Type.TArray<Type.TObject<{
208
+ metricId: Type.TUnion<Type.TLiteral<string>[]>;
209
+ required: Type.TBoolean;
210
+ minimumSamples: Type.TInteger;
211
+ tolerance: Type.TNumber;
212
+ constraint: Type.TOptional<Type.TObject<{
213
+ operator: Type.TUnion<Type.TLiteral<string>[]>;
214
+ value: Type.TNumber;
215
+ }>>;
216
+ }>>;
217
+ createdAt: Type.TString;
218
+ }>;
219
+ policy: Type.TObject<{
220
+ mode: Type.TLiteral<"supervised_graduation">;
221
+ evaluationBudget: Type.TObject<{
222
+ currency: Type.TString;
223
+ maximumSpend: Type.TNumber;
224
+ maximumCandidateCount: Type.TOptional<Type.TInteger>;
225
+ maximumRunsPerCandidate: Type.TOptional<Type.TInteger>;
226
+ expiresAt: Type.TOptional<Type.TString>;
227
+ }>;
228
+ automaticSwapsEnabled: Type.TBoolean;
229
+ confirmationRequiredUntilVerifiedSwaps: Type.TInteger;
230
+ requirePostSwapVerification: Type.TLiteral<true>;
231
+ rollbackOnRegression: Type.TBoolean;
232
+ checkPolicy: Type.TObject<{
233
+ baselineAssessment: Type.TObject<{
234
+ afterCompletedTasks: Type.TOptional<Type.TInteger>;
235
+ afterElapsedSeconds: Type.TOptional<Type.TInteger>;
236
+ }>;
237
+ frontierReassessment: Type.TObject<{
238
+ afterCompletedTasks: Type.TOptional<Type.TInteger>;
239
+ afterElapsedSeconds: Type.TOptional<Type.TInteger>;
240
+ onModelCatalogChange: Type.TBoolean;
241
+ }>;
242
+ regressionCheck: Type.TObject<{
243
+ afterCompletedTasks: Type.TOptional<Type.TInteger>;
244
+ afterElapsedSeconds: Type.TOptional<Type.TInteger>;
245
+ thresholds: Type.TArray<Type.TObject<{
246
+ metricId: Type.TUnion<Type.TLiteral<string>[]>;
247
+ relativeChangeAtLeast: Type.TNumber;
248
+ }>>;
249
+ }>;
250
+ postSwapVerification: Type.TObject<{
251
+ afterCompletedTasks: Type.TOptional<Type.TInteger>;
252
+ afterElapsedSeconds: Type.TOptional<Type.TInteger>;
253
+ }>;
254
+ cooldownSeconds: Type.TInteger;
255
+ execution: Type.TObject<{
256
+ mode: Type.TUnion<Type.TLiteral<string>[]>;
257
+ infrastructureChangesAllowed: Type.TBoolean;
258
+ }>;
259
+ }>;
260
+ }>;
261
+ observability: Type.TObject<{
262
+ status: Type.TUnion<Type.TLiteral<string>[]>;
263
+ requiredMetricIds: Type.TArray<Type.TUnion<Type.TLiteral<string>[]>>;
264
+ coveredMetricIds: Type.TArray<Type.TUnion<Type.TLiteral<string>[]>>;
265
+ missingMetricIds: Type.TArray<Type.TUnion<Type.TLiteral<string>[]>>;
266
+ checkedAt: Type.TString;
267
+ }>;
268
+ }>;
269
+ readonly instrument_observability: Type.TObject<{
270
+ coveredMetricIds: Type.TArray<Type.TUnion<Type.TLiteral<string>[]>>;
271
+ missingMetricIds: Type.TArray<Type.TUnion<Type.TLiteral<string>[]>>;
272
+ }>;
273
+ readonly run_assessment: Type.TObject<{
274
+ baselineEvidenceSufficient: Type.TBoolean;
275
+ metrics: Type.TOptional<Type.TArray<Type.TObject<{
276
+ metricId: Type.TUnion<Type.TLiteral<string>[]>;
277
+ state: Type.TUnion<Type.TLiteral<string>[]>;
278
+ scope: Type.TUnion<Type.TLiteral<string>[]>;
279
+ direction: Type.TUnion<Type.TLiteral<string>[]>;
280
+ method: Type.TUnion<Type.TLiteral<string>[]>;
281
+ value: Type.TOptional<Type.TNumber>;
282
+ unit: Type.TString;
283
+ sampleCount: Type.TInteger;
284
+ interval: Type.TOptional<Type.TObject<{
285
+ lower: Type.TNumber;
286
+ upper: Type.TNumber;
287
+ confidenceLevel: Type.TOptional<Type.TNumber>;
288
+ }>>;
289
+ window: Type.TOptional<Type.TObject<{
290
+ startedAt: Type.TString;
291
+ endedAt: Type.TString;
292
+ }>>;
293
+ evidence: Type.TArray<Type.TObject<{
294
+ id: Type.TString;
295
+ source: Type.TUnion<Type.TLiteral<string>[]>;
296
+ uri: Type.TString;
297
+ mediaType: Type.TOptional<Type.TString>;
298
+ digest: Type.TOptional<Type.TString>;
299
+ }>>;
300
+ note: Type.TOptional<Type.TString>;
301
+ }>>>;
302
+ }>;
303
+ readonly discover_candidates: Type.TObject<{
304
+ candidates: Type.TArray<Type.TObject<{
305
+ candidateId: Type.TString;
306
+ modelId: Type.TString;
307
+ modelVersion: Type.TOptional<Type.TString>;
308
+ reasons: Type.TArray<Type.TString>;
309
+ }>>;
310
+ }>;
311
+ readonly run_challenger_trials: Type.TObject<{
312
+ assessments: Type.TArray<Type.TObject<{
313
+ id: Type.TString;
314
+ taskProfileId: Type.TString;
315
+ taskProfileRevision: Type.TInteger;
316
+ candidateId: Type.TString;
317
+ modelId: Type.TString;
318
+ modelVersion: Type.TOptional<Type.TString>;
319
+ productRevision: Type.TString;
320
+ harnessId: Type.TString;
321
+ startedAt: Type.TString;
322
+ completedAt: Type.TString;
323
+ metrics: Type.TArray<Type.TObject<{
324
+ metricId: Type.TUnion<Type.TLiteral<string>[]>;
325
+ state: Type.TUnion<Type.TLiteral<string>[]>;
326
+ scope: Type.TUnion<Type.TLiteral<string>[]>;
327
+ direction: Type.TUnion<Type.TLiteral<string>[]>;
328
+ method: Type.TUnion<Type.TLiteral<string>[]>;
329
+ value: Type.TOptional<Type.TNumber>;
330
+ unit: Type.TString;
331
+ sampleCount: Type.TInteger;
332
+ interval: Type.TOptional<Type.TObject<{
333
+ lower: Type.TNumber;
334
+ upper: Type.TNumber;
335
+ confidenceLevel: Type.TOptional<Type.TNumber>;
336
+ }>>;
337
+ window: Type.TOptional<Type.TObject<{
338
+ startedAt: Type.TString;
339
+ endedAt: Type.TString;
340
+ }>>;
341
+ evidence: Type.TArray<Type.TObject<{
342
+ id: Type.TString;
343
+ source: Type.TUnion<Type.TLiteral<string>[]>;
344
+ uri: Type.TString;
345
+ mediaType: Type.TOptional<Type.TString>;
346
+ digest: Type.TOptional<Type.TString>;
347
+ }>>;
348
+ note: Type.TOptional<Type.TString>;
349
+ }>>;
350
+ }>>;
351
+ }>;
352
+ readonly calculate_frontier: Type.TObject<{
353
+ assessments: Type.TArray<Type.TObject<{
354
+ id: Type.TString;
355
+ taskProfileId: Type.TString;
356
+ taskProfileRevision: Type.TInteger;
357
+ candidateId: Type.TString;
358
+ modelId: Type.TString;
359
+ modelVersion: Type.TOptional<Type.TString>;
360
+ productRevision: Type.TString;
361
+ harnessId: Type.TString;
362
+ startedAt: Type.TString;
363
+ completedAt: Type.TString;
364
+ metrics: Type.TArray<Type.TObject<{
365
+ metricId: Type.TUnion<Type.TLiteral<string>[]>;
366
+ state: Type.TUnion<Type.TLiteral<string>[]>;
367
+ scope: Type.TUnion<Type.TLiteral<string>[]>;
368
+ direction: Type.TUnion<Type.TLiteral<string>[]>;
369
+ method: Type.TUnion<Type.TLiteral<string>[]>;
370
+ value: Type.TOptional<Type.TNumber>;
371
+ unit: Type.TString;
372
+ sampleCount: Type.TInteger;
373
+ interval: Type.TOptional<Type.TObject<{
374
+ lower: Type.TNumber;
375
+ upper: Type.TNumber;
376
+ confidenceLevel: Type.TOptional<Type.TNumber>;
377
+ }>>;
378
+ window: Type.TOptional<Type.TObject<{
379
+ startedAt: Type.TString;
380
+ endedAt: Type.TString;
381
+ }>>;
382
+ evidence: Type.TArray<Type.TObject<{
383
+ id: Type.TString;
384
+ source: Type.TUnion<Type.TLiteral<string>[]>;
385
+ uri: Type.TString;
386
+ mediaType: Type.TOptional<Type.TString>;
387
+ digest: Type.TOptional<Type.TString>;
388
+ }>>;
389
+ note: Type.TOptional<Type.TString>;
390
+ }>>;
391
+ }>>;
392
+ frontierSnapshot: Type.TObject<{
393
+ id: Type.TString;
394
+ taskProfileId: Type.TString;
395
+ taskProfileRevision: Type.TInteger;
396
+ productRevision: Type.TString;
397
+ createdAt: Type.TString;
398
+ frontierCandidateIds: Type.TArray<Type.TString>;
399
+ dominatedCandidateIds: Type.TArray<Type.TString>;
400
+ ineligibleCandidateIds: Type.TArray<Type.TString>;
401
+ unresolvedCandidateIds: Type.TArray<Type.TString>;
402
+ comparisons: Type.TArray<Type.TObject<{
403
+ candidateId: Type.TString;
404
+ otherCandidateId: Type.TString;
405
+ state: Type.TUnion<Type.TLiteral<string>[]>;
406
+ reasons: Type.TArray<Type.TString>;
407
+ }>>;
408
+ assessmentIds: Type.TArray<Type.TString>;
409
+ calculationEvidence: Type.TArray<Type.TObject<{
410
+ id: Type.TString;
411
+ source: Type.TUnion<Type.TLiteral<string>[]>;
412
+ uri: Type.TString;
413
+ mediaType: Type.TOptional<Type.TString>;
414
+ digest: Type.TOptional<Type.TString>;
415
+ }>>;
416
+ }>;
417
+ recommendation: Type.TOptional<Type.TObject<{
418
+ frontierSnapshotId: Type.TString;
419
+ selectedCandidateId: Type.TString;
420
+ currentCandidateId: Type.TString;
421
+ reasons: Type.TArray<Type.TString>;
422
+ preferenceEvidence: Type.TArray<Type.TObject<{
423
+ id: Type.TString;
424
+ source: Type.TUnion<Type.TLiteral<string>[]>;
425
+ uri: Type.TString;
426
+ mediaType: Type.TOptional<Type.TString>;
427
+ digest: Type.TOptional<Type.TString>;
428
+ }>>;
429
+ }>>;
430
+ }>;
431
+ readonly investigate_regression: Type.TObject<{
432
+ regressionDetected: Type.TBoolean;
433
+ affectedMetricIds: Type.TArray<Type.TUnion<Type.TLiteral<string>[]>>;
434
+ likelyCause: Type.TOptional<Type.TString>;
435
+ }>;
436
+ readonly apply_model_swap: Type.TObject<{
437
+ appliedCandidateId: Type.TString;
438
+ previousCandidateId: Type.TOptional<Type.TString>;
439
+ }>;
440
+ readonly verify_model_swap: Type.TObject<{
441
+ verified: Type.TBoolean;
442
+ regressionDetected: Type.TBoolean;
443
+ }>;
444
+ readonly rollback_model_swap: Type.TObject<{
445
+ restoredCandidateId: Type.TString;
446
+ }>;
447
+ };
448
+ export declare function intentOutputSchemaFor(kind: IntentKind): TSchema;
449
+ export declare function validateIntentOutputSchema(kind: IntentKind, output: unknown): readonly string[];
450
+ export declare function completionToolSchemaFor(kind: IntentKind): TSchema;
@@ -0,0 +1,224 @@
1
+ import { Type } from "typebox";
2
+ import { Check, Errors } from "typebox/value";
3
+ export const OPENMERIT_SCHEMA_DIALECT = "https://json-schema.org/draft/2020-12/schema";
4
+ export const OPENMERIT_RESULT_SCHEMA_VERSION = "0.2";
5
+ const strictObject = (properties, options = {}) => Type.Object(properties, { additionalProperties: false, ...options });
6
+ const metricIds = [
7
+ "task_quality", "task_success", "consistency", "instruction_following",
8
+ "tool_use_performance", "structured_output_reliability", "hallucination_rate",
9
+ "reasoning_efficiency", "input_token_consumption", "output_token_consumption",
10
+ "total_task_cost", "cost_per_successful_task", "time_to_first_useful_output",
11
+ "end_to_end_task_latency", "throughput", "retry_rate", "agent_steps",
12
+ "recovery_ability", "context_handling", "retrieval_use_quality",
13
+ "long_horizon_performance", "human_intervention_rate", "preference_fit",
14
+ ];
15
+ const evidenceSources = [
16
+ "harness_trace", "evaluation_artifact", "observability_record", "external_source", "user_feedback",
17
+ ];
18
+ const stringEnum = (values) => Type.Union(values.map((value) => Type.Literal(value)));
19
+ const cadenceSchema = () => strictObject({
20
+ afterCompletedTasks: Type.Optional(Type.Integer({ minimum: 1 })),
21
+ afterElapsedSeconds: Type.Optional(Type.Integer({ minimum: 1 })),
22
+ });
23
+ export const EvidenceReferenceSchema = strictObject({
24
+ id: Type.String({ minLength: 1 }),
25
+ source: stringEnum(evidenceSources),
26
+ uri: Type.String({ minLength: 1 }),
27
+ mediaType: Type.Optional(Type.String({ minLength: 1 })),
28
+ digest: Type.Optional(Type.String({ minLength: 1 })),
29
+ });
30
+ export const MetricObjectiveSchema = strictObject({
31
+ metricId: stringEnum(metricIds),
32
+ required: Type.Boolean(),
33
+ minimumSamples: Type.Integer({ minimum: 1 }),
34
+ tolerance: Type.Number({ minimum: 0 }),
35
+ constraint: Type.Optional(strictObject({
36
+ operator: stringEnum(["at_least", "at_most"]),
37
+ value: Type.Number(),
38
+ })),
39
+ });
40
+ export const TaskProfileSchema = strictObject({
41
+ id: Type.String({ minLength: 1 }),
42
+ revision: Type.Integer({ minimum: 1 }),
43
+ name: Type.String({ minLength: 1 }),
44
+ goal: Type.String({ minLength: 1 }),
45
+ objectives: Type.Array(MetricObjectiveSchema, { minItems: 1 }),
46
+ createdAt: Type.String({ minLength: 1 }),
47
+ confirmedAt: Type.Optional(Type.String({ minLength: 1 })),
48
+ });
49
+ export const ObservabilityCoverageSchema = strictObject({
50
+ status: stringEnum(["ready", "incomplete"]),
51
+ requiredMetricIds: Type.Array(stringEnum(metricIds)),
52
+ coveredMetricIds: Type.Array(stringEnum(metricIds)),
53
+ missingMetricIds: Type.Array(stringEnum(metricIds)),
54
+ checkedAt: Type.String({ minLength: 1 }),
55
+ });
56
+ const ConfirmedTaskProfileSchema = strictObject({
57
+ ...TaskProfileSchema.properties,
58
+ confirmedAt: Type.String({ minLength: 1 }),
59
+ });
60
+ export const SupervisedGraduationPolicySchema = strictObject({
61
+ mode: Type.Literal("supervised_graduation"),
62
+ evaluationBudget: strictObject({
63
+ currency: Type.String({ minLength: 1 }),
64
+ maximumSpend: Type.Number({ minimum: 0 }),
65
+ maximumCandidateCount: Type.Optional(Type.Integer({ minimum: 1 })),
66
+ maximumRunsPerCandidate: Type.Optional(Type.Integer({ minimum: 1 })),
67
+ expiresAt: Type.Optional(Type.String({ minLength: 1 })),
68
+ }),
69
+ automaticSwapsEnabled: Type.Boolean(),
70
+ confirmationRequiredUntilVerifiedSwaps: Type.Integer({ minimum: 0 }),
71
+ requirePostSwapVerification: Type.Literal(true),
72
+ rollbackOnRegression: Type.Boolean(),
73
+ checkPolicy: strictObject({
74
+ baselineAssessment: cadenceSchema(),
75
+ frontierReassessment: strictObject({
76
+ afterCompletedTasks: Type.Optional(Type.Integer({ minimum: 1 })),
77
+ afterElapsedSeconds: Type.Optional(Type.Integer({ minimum: 1 })),
78
+ onModelCatalogChange: Type.Boolean(),
79
+ }),
80
+ regressionCheck: strictObject({
81
+ afterCompletedTasks: Type.Optional(Type.Integer({ minimum: 1 })),
82
+ afterElapsedSeconds: Type.Optional(Type.Integer({ minimum: 1 })),
83
+ thresholds: Type.Array(strictObject({
84
+ metricId: stringEnum(metricIds),
85
+ relativeChangeAtLeast: Type.Number({ exclusiveMinimum: 0 }),
86
+ })),
87
+ }),
88
+ postSwapVerification: cadenceSchema(),
89
+ cooldownSeconds: Type.Integer({ minimum: 0 }),
90
+ execution: strictObject({
91
+ mode: stringEnum(["active_session_only", "persistent"]),
92
+ infrastructureChangesAllowed: Type.Boolean(),
93
+ }),
94
+ }),
95
+ });
96
+ export const MetricRecordSchema = strictObject({
97
+ metricId: stringEnum(metricIds),
98
+ state: stringEnum(["measured", "not_applicable", "not_configured", "insufficient_evidence"]),
99
+ scope: stringEnum(["single_run", "sample_window"]),
100
+ direction: stringEnum(["maximize", "minimize"]),
101
+ method: stringEnum(["observed", "derived", "evaluator", "human", "external"]),
102
+ value: Type.Optional(Type.Number()),
103
+ unit: Type.String({ minLength: 1 }),
104
+ sampleCount: Type.Integer({ minimum: 0 }),
105
+ interval: Type.Optional(strictObject({
106
+ lower: Type.Number(),
107
+ upper: Type.Number(),
108
+ confidenceLevel: Type.Optional(Type.Number({ minimum: 0, maximum: 1 })),
109
+ })),
110
+ window: Type.Optional(strictObject({
111
+ startedAt: Type.String({ minLength: 1 }),
112
+ endedAt: Type.String({ minLength: 1 }),
113
+ })),
114
+ evidence: Type.Array(EvidenceReferenceSchema),
115
+ note: Type.Optional(Type.String()),
116
+ });
117
+ export const CandidateDescriptorSchema = strictObject({
118
+ candidateId: Type.String({ minLength: 1 }),
119
+ modelId: Type.String({ minLength: 1 }),
120
+ modelVersion: Type.Optional(Type.String({ minLength: 1 })),
121
+ reasons: Type.Array(Type.String()),
122
+ });
123
+ export const CandidateAssessmentSchema = strictObject({
124
+ id: Type.String({ minLength: 1 }),
125
+ taskProfileId: Type.String({ minLength: 1 }),
126
+ taskProfileRevision: Type.Integer({ minimum: 1 }),
127
+ candidateId: Type.String({ minLength: 1 }),
128
+ modelId: Type.String({ minLength: 1 }),
129
+ modelVersion: Type.Optional(Type.String({ minLength: 1 })),
130
+ productRevision: Type.String({ minLength: 1 }),
131
+ harnessId: Type.String({ minLength: 1 }),
132
+ startedAt: Type.String({ minLength: 1 }),
133
+ completedAt: Type.String({ minLength: 1 }),
134
+ metrics: Type.Array(MetricRecordSchema),
135
+ });
136
+ export const FrontierSnapshotSchema = strictObject({
137
+ id: Type.String({ minLength: 1 }),
138
+ taskProfileId: Type.String({ minLength: 1 }),
139
+ taskProfileRevision: Type.Integer({ minimum: 1 }),
140
+ productRevision: Type.String({ minLength: 1 }),
141
+ createdAt: Type.String({ minLength: 1 }),
142
+ frontierCandidateIds: Type.Array(Type.String()),
143
+ dominatedCandidateIds: Type.Array(Type.String()),
144
+ ineligibleCandidateIds: Type.Array(Type.String()),
145
+ unresolvedCandidateIds: Type.Array(Type.String()),
146
+ comparisons: Type.Array(strictObject({
147
+ candidateId: Type.String({ minLength: 1 }),
148
+ otherCandidateId: Type.String({ minLength: 1 }),
149
+ state: stringEnum(["dominates", "does_not_dominate", "unresolved"]),
150
+ reasons: Type.Array(Type.String()),
151
+ })),
152
+ assessmentIds: Type.Array(Type.String()),
153
+ calculationEvidence: Type.Array(EvidenceReferenceSchema, { minItems: 1 }),
154
+ });
155
+ export const FrontierRecommendationSchema = strictObject({
156
+ frontierSnapshotId: Type.String({ minLength: 1 }),
157
+ selectedCandidateId: Type.String({ minLength: 1 }),
158
+ currentCandidateId: Type.String({ minLength: 1 }),
159
+ reasons: Type.Array(Type.String()),
160
+ preferenceEvidence: Type.Array(EvidenceReferenceSchema),
161
+ });
162
+ export const INTENT_OUTPUT_SCHEMAS = {
163
+ establish_evals: strictObject({
164
+ taskProfile: ConfirmedTaskProfileSchema,
165
+ policy: SupervisedGraduationPolicySchema,
166
+ observability: ObservabilityCoverageSchema,
167
+ }),
168
+ instrument_observability: strictObject({
169
+ coveredMetricIds: Type.Array(stringEnum(metricIds)),
170
+ missingMetricIds: Type.Array(stringEnum(metricIds)),
171
+ }),
172
+ run_assessment: strictObject({
173
+ baselineEvidenceSufficient: Type.Boolean(),
174
+ metrics: Type.Optional(Type.Array(MetricRecordSchema)),
175
+ }),
176
+ discover_candidates: strictObject({
177
+ candidates: Type.Array(CandidateDescriptorSchema),
178
+ }),
179
+ run_challenger_trials: strictObject({
180
+ assessments: Type.Array(CandidateAssessmentSchema),
181
+ }),
182
+ calculate_frontier: strictObject({
183
+ assessments: Type.Array(CandidateAssessmentSchema),
184
+ frontierSnapshot: FrontierSnapshotSchema,
185
+ recommendation: Type.Optional(FrontierRecommendationSchema),
186
+ }),
187
+ investigate_regression: strictObject({
188
+ regressionDetected: Type.Boolean(),
189
+ affectedMetricIds: Type.Array(stringEnum(metricIds)),
190
+ likelyCause: Type.Optional(Type.String()),
191
+ }),
192
+ apply_model_swap: strictObject({
193
+ appliedCandidateId: Type.String({ minLength: 1 }),
194
+ previousCandidateId: Type.Optional(Type.String({ minLength: 1 })),
195
+ }),
196
+ verify_model_swap: strictObject({
197
+ verified: Type.Boolean(),
198
+ regressionDetected: Type.Boolean(),
199
+ }),
200
+ rollback_model_swap: strictObject({
201
+ restoredCandidateId: Type.String({ minLength: 1 }),
202
+ }),
203
+ };
204
+ export function intentOutputSchemaFor(kind) {
205
+ return INTENT_OUTPUT_SCHEMAS[kind];
206
+ }
207
+ export function validateIntentOutputSchema(kind, output) {
208
+ const schema = intentOutputSchemaFor(kind);
209
+ if (Check(schema, output))
210
+ return [];
211
+ return [...Errors(schema, output)].map((error) => `${error.instancePath || "/"} ${error.message}`);
212
+ }
213
+ export function completionToolSchemaFor(kind) {
214
+ return strictObject({
215
+ intentId: Type.String({ minLength: 1 }),
216
+ status: stringEnum(["succeeded", "failed", "cancelled"]),
217
+ summary: Type.String({ minLength: 1 }),
218
+ evidence: Type.Array(EvidenceReferenceSchema),
219
+ outputs: Type.Optional(intentOutputSchemaFor(kind)),
220
+ errorCode: Type.Optional(Type.String({ minLength: 1 })),
221
+ errorMessage: Type.Optional(Type.String({ minLength: 1 })),
222
+ retryable: Type.Optional(Type.Boolean()),
223
+ });
224
+ }