@openpond/evals 0.4.0 → 0.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +22 -5
- package/dist/benchmarks.js +218 -0
- package/dist/builtin-benchmarks/harness-refiner.js +1267 -0
- package/dist/index.js +1 -0
- package/dist/types/benchmarks.d.ts +293 -0
- package/dist/types/benchmarks.d.ts.map +1 -0
- package/dist/types/builtin-benchmarks/harness-refiner.d.ts +119 -0
- package/dist/types/builtin-benchmarks/harness-refiner.d.ts.map +1 -0
- package/dist/types/evidence/conformance.d.ts +8 -8
- package/dist/types/evidence/contracts.d.ts +8 -8
- package/dist/types/index.d.ts +1 -0
- package/dist/types/index.d.ts.map +1 -1
- package/dist/types/review-conformance.d.ts +10 -10
- package/package.json +5 -1
|
@@ -0,0 +1,1267 @@
|
|
|
1
|
+
// Generated by benchmarks/harness-refiner/taskset/build.ts. Do not edit.
|
|
2
|
+
import { TasksetReleaseSchema } from "../tasksets.js";
|
|
3
|
+
export const harnessRefinerBenchmarkRelease = TasksetReleaseSchema.parse({
|
|
4
|
+
"capabilities": [
|
|
5
|
+
{
|
|
6
|
+
"id": "filesystem.workspace",
|
|
7
|
+
"portability": "host_adapter",
|
|
8
|
+
"required": true,
|
|
9
|
+
"scopes": [
|
|
10
|
+
"inputs:read",
|
|
11
|
+
"work:read-write",
|
|
12
|
+
"outputs:write"
|
|
13
|
+
]
|
|
14
|
+
},
|
|
15
|
+
{
|
|
16
|
+
"id": "network.web-read",
|
|
17
|
+
"portability": "host_adapter",
|
|
18
|
+
"required": true,
|
|
19
|
+
"scopes": [
|
|
20
|
+
"search",
|
|
21
|
+
"fetch"
|
|
22
|
+
]
|
|
23
|
+
},
|
|
24
|
+
{
|
|
25
|
+
"id": "artifact.pdf",
|
|
26
|
+
"portability": "host_adapter",
|
|
27
|
+
"required": true,
|
|
28
|
+
"scopes": [
|
|
29
|
+
"create",
|
|
30
|
+
"render",
|
|
31
|
+
"inspect"
|
|
32
|
+
]
|
|
33
|
+
},
|
|
34
|
+
{
|
|
35
|
+
"id": "artifact.spreadsheet",
|
|
36
|
+
"portability": "host_adapter",
|
|
37
|
+
"required": true,
|
|
38
|
+
"scopes": [
|
|
39
|
+
"create",
|
|
40
|
+
"calculate",
|
|
41
|
+
"inspect"
|
|
42
|
+
]
|
|
43
|
+
}
|
|
44
|
+
],
|
|
45
|
+
"contentHash": "05dbf9058c09b75bc950ca5cdf4ac03650dba7bb0cb74a3d9e152b16c110a2f7",
|
|
46
|
+
"environment": {
|
|
47
|
+
"defaultTimeoutMs": 900000,
|
|
48
|
+
"deterministicSeeds": false,
|
|
49
|
+
"entrypoint": "openpond-work-v1",
|
|
50
|
+
"kind": "work",
|
|
51
|
+
"lifecycle": [
|
|
52
|
+
"create",
|
|
53
|
+
"reset",
|
|
54
|
+
"step",
|
|
55
|
+
"collect",
|
|
56
|
+
"destroy"
|
|
57
|
+
],
|
|
58
|
+
"networkPolicy": "declared_read_only",
|
|
59
|
+
"protocolVersion": "openpond.environment.v1",
|
|
60
|
+
"stateful": true
|
|
61
|
+
},
|
|
62
|
+
"graders": [
|
|
63
|
+
{
|
|
64
|
+
"hardGate": true,
|
|
65
|
+
"id": "task-output-contract",
|
|
66
|
+
"kind": "custom_verifier",
|
|
67
|
+
"networkPolicy": "none",
|
|
68
|
+
"privileged": true,
|
|
69
|
+
"rewardEligible": true,
|
|
70
|
+
"timeoutMs": 30000,
|
|
71
|
+
"verifierRef": {
|
|
72
|
+
"contentHash": "5290dfae6969bd4581b1e1c111bdc1cb5f2828f7a91681635254b246c7590392",
|
|
73
|
+
"id": "verifiers-taskset-output-verifier-mjs",
|
|
74
|
+
"mediaType": "text/javascript",
|
|
75
|
+
"path": "verifiers/taskset-output-verifier.mjs",
|
|
76
|
+
"sizeBytes": 1295,
|
|
77
|
+
"visibility": "verifier"
|
|
78
|
+
},
|
|
79
|
+
"version": "1",
|
|
80
|
+
"weight": 1
|
|
81
|
+
},
|
|
82
|
+
{
|
|
83
|
+
"calibrationStatus": "pending",
|
|
84
|
+
"hardGate": true,
|
|
85
|
+
"id": "task-quality-judge",
|
|
86
|
+
"kind": "model_judge",
|
|
87
|
+
"privileged": true,
|
|
88
|
+
"rewardEligible": true,
|
|
89
|
+
"rubricRef": {
|
|
90
|
+
"contentHash": "455b3697c0617333a34b6521f167760fb8ae7e286059fa2ec327a5c97e66a2b3",
|
|
91
|
+
"id": "rubrics-task-quality-md",
|
|
92
|
+
"mediaType": "text/markdown",
|
|
93
|
+
"path": "rubrics/task-quality.md",
|
|
94
|
+
"sizeBytes": 1496,
|
|
95
|
+
"visibility": "verifier"
|
|
96
|
+
},
|
|
97
|
+
"version": "1",
|
|
98
|
+
"weight": 1
|
|
99
|
+
}
|
|
100
|
+
],
|
|
101
|
+
"id": "harness-refiner-public-v1",
|
|
102
|
+
"metadata": {
|
|
103
|
+
"adaptationSplit": "validation",
|
|
104
|
+
"benchmark": "harness-refiner",
|
|
105
|
+
"frozenEvaluationSplit": "frozen_eval",
|
|
106
|
+
"primaryMetric": "paired_foreground_provider_tokens",
|
|
107
|
+
"protocolVersion": "1",
|
|
108
|
+
"qualityPolicy": "hard_non_regression",
|
|
109
|
+
"toolDeclarationSource": "openpond-production-model-tool-definitions",
|
|
110
|
+
"trainingSideEffect": false
|
|
111
|
+
},
|
|
112
|
+
"policy": {
|
|
113
|
+
"connectedAppScopes": [],
|
|
114
|
+
"hiddenGraderRefs": [
|
|
115
|
+
"rubrics-task-quality-md",
|
|
116
|
+
"verifiers-taskset-output-verifier-mjs"
|
|
117
|
+
],
|
|
118
|
+
"policyVisibleFields": [
|
|
119
|
+
"input"
|
|
120
|
+
],
|
|
121
|
+
"privilegedFields": [
|
|
122
|
+
"expectedOutput"
|
|
123
|
+
]
|
|
124
|
+
},
|
|
125
|
+
"revision": 3,
|
|
126
|
+
"schemaVersion": "openpond.tasksetRelease.v2",
|
|
127
|
+
"tasks": [
|
|
128
|
+
{
|
|
129
|
+
"artifactRefs": [
|
|
130
|
+
{
|
|
131
|
+
"contentHash": "8879f8e31a916f165724a3c3e3e57db5baa858cd170ac6799064aa49bc623aa4",
|
|
132
|
+
"id": "fixtures-adaptation-board-launch-md",
|
|
133
|
+
"mediaType": "text/markdown",
|
|
134
|
+
"path": "fixtures/adaptation-board-launch.md",
|
|
135
|
+
"sizeBytes": 642,
|
|
136
|
+
"visibility": "policy"
|
|
137
|
+
}
|
|
138
|
+
],
|
|
139
|
+
"clusterKey": "northstar-launch-packet",
|
|
140
|
+
"expectedOutput": {
|
|
141
|
+
"deliverable": "pdf",
|
|
142
|
+
"mustInclude": [
|
|
143
|
+
"three decision options",
|
|
144
|
+
"owners and dates",
|
|
145
|
+
"confirmed facts",
|
|
146
|
+
"open legal and finance gates"
|
|
147
|
+
],
|
|
148
|
+
"mustNot": [
|
|
149
|
+
"present an open gate as approved",
|
|
150
|
+
"invent a recommended decision"
|
|
151
|
+
],
|
|
152
|
+
"validation": [
|
|
153
|
+
"structural",
|
|
154
|
+
"visual"
|
|
155
|
+
]
|
|
156
|
+
},
|
|
157
|
+
"id": "adaptation-board-launch-brief",
|
|
158
|
+
"input": {
|
|
159
|
+
"attachments": [
|
|
160
|
+
"adaptation-board-launch.md"
|
|
161
|
+
],
|
|
162
|
+
"prompt": "Please turn the attached Northstar launch packet into a polished two-page PDF decision brief for the executive meeting. Make the three options, owners, dates, confirmed facts, open gates, and risks easy to scan. The source records no final choice, so do not select, recommend, or imply a preferred option. Visually check the finished PDF before you send it back."
|
|
163
|
+
},
|
|
164
|
+
"policyVisibleContext": {
|
|
165
|
+
"attachmentCount": 1
|
|
166
|
+
},
|
|
167
|
+
"privilegedContextRef": "expected-adaptation-board-launch-brief",
|
|
168
|
+
"split": "validation",
|
|
169
|
+
"tags": [
|
|
170
|
+
"artifact-verification",
|
|
171
|
+
"decision-brief",
|
|
172
|
+
"adaptation"
|
|
173
|
+
]
|
|
174
|
+
},
|
|
175
|
+
{
|
|
176
|
+
"artifactRefs": [
|
|
177
|
+
{
|
|
178
|
+
"contentHash": "a3d44bb123a242d78d0f60d4d8ad352e499bef677eaf816a19d07359a172743f",
|
|
179
|
+
"id": "fixtures-adaptation-latency-incident-md",
|
|
180
|
+
"mediaType": "text/markdown",
|
|
181
|
+
"path": "fixtures/adaptation-latency-incident.md",
|
|
182
|
+
"sizeBytes": 733,
|
|
183
|
+
"visibility": "policy"
|
|
184
|
+
}
|
|
185
|
+
],
|
|
186
|
+
"clusterKey": "checkout-latency-incident-packet",
|
|
187
|
+
"expectedOutput": {
|
|
188
|
+
"deliverable": "pdf",
|
|
189
|
+
"mustInclude": [
|
|
190
|
+
"incident window",
|
|
191
|
+
"confirmed impact",
|
|
192
|
+
"recovery",
|
|
193
|
+
"hypotheses labeled as hypotheses",
|
|
194
|
+
"unknowns",
|
|
195
|
+
"follow-up owners"
|
|
196
|
+
],
|
|
197
|
+
"mustNot": [
|
|
198
|
+
"state either hypothesis as root cause",
|
|
199
|
+
"claim unresolved carts were recovered"
|
|
200
|
+
],
|
|
201
|
+
"validation": [
|
|
202
|
+
"structural",
|
|
203
|
+
"visual"
|
|
204
|
+
]
|
|
205
|
+
},
|
|
206
|
+
"id": "adaptation-latency-incident-review",
|
|
207
|
+
"input": {
|
|
208
|
+
"attachments": [
|
|
209
|
+
"adaptation-latency-incident.md"
|
|
210
|
+
],
|
|
211
|
+
"prompt": "Create a concise PDF incident review from the attached checkout latency packet. Clearly separate confirmed observations, recovery actions, hypotheses, unknowns, customer impact, and follow-up owners. Check that the rendered PDF is readable and complete."
|
|
212
|
+
},
|
|
213
|
+
"policyVisibleContext": {
|
|
214
|
+
"attachmentCount": 1
|
|
215
|
+
},
|
|
216
|
+
"privilegedContextRef": "expected-adaptation-latency-incident-review",
|
|
217
|
+
"split": "validation",
|
|
218
|
+
"tags": [
|
|
219
|
+
"artifact-verification",
|
|
220
|
+
"incident-review",
|
|
221
|
+
"adaptation"
|
|
222
|
+
]
|
|
223
|
+
},
|
|
224
|
+
{
|
|
225
|
+
"artifactRefs": [
|
|
226
|
+
{
|
|
227
|
+
"contentHash": "63bca96753ae1491a45ba13e1a06cfb48973cfbe4f392e8f181e7b9648fd9db4",
|
|
228
|
+
"id": "fixtures-adaptation-program-budget-md",
|
|
229
|
+
"mediaType": "text/markdown",
|
|
230
|
+
"path": "fixtures/adaptation-program-budget.md",
|
|
231
|
+
"sizeBytes": 649,
|
|
232
|
+
"visibility": "policy"
|
|
233
|
+
}
|
|
234
|
+
],
|
|
235
|
+
"clusterKey": "harbor-program-budget-packet",
|
|
236
|
+
"expectedOutput": {
|
|
237
|
+
"deliverable": "spreadsheet",
|
|
238
|
+
"mustInclude": [
|
|
239
|
+
"summary sheet",
|
|
240
|
+
"detail sheet",
|
|
241
|
+
"formula-driven full-year forecast",
|
|
242
|
+
"formula-driven variance",
|
|
243
|
+
"owner column",
|
|
244
|
+
"overrun flag"
|
|
245
|
+
],
|
|
246
|
+
"mustNot": [
|
|
247
|
+
"replace formulas with typed totals",
|
|
248
|
+
"reverse the variance sign"
|
|
249
|
+
],
|
|
250
|
+
"validation": [
|
|
251
|
+
"structural",
|
|
252
|
+
"test"
|
|
253
|
+
]
|
|
254
|
+
},
|
|
255
|
+
"id": "adaptation-program-budget-workbook",
|
|
256
|
+
"input": {
|
|
257
|
+
"attachments": [
|
|
258
|
+
"adaptation-program-budget.md"
|
|
259
|
+
],
|
|
260
|
+
"prompt": "Build an Excel workbook from the attached Harbor youth program budget. Include a one-page summary and a detail sheet with formulas for full-year forecast and variance, clearly flag forecast overruns, preserve the listed owners, and verify the calculations before returning it."
|
|
261
|
+
},
|
|
262
|
+
"policyVisibleContext": {
|
|
263
|
+
"attachmentCount": 1
|
|
264
|
+
},
|
|
265
|
+
"privilegedContextRef": "expected-adaptation-program-budget-workbook",
|
|
266
|
+
"split": "validation",
|
|
267
|
+
"tags": [
|
|
268
|
+
"artifact-verification",
|
|
269
|
+
"spreadsheet",
|
|
270
|
+
"adaptation"
|
|
271
|
+
]
|
|
272
|
+
},
|
|
273
|
+
{
|
|
274
|
+
"artifactRefs": [],
|
|
275
|
+
"clusterKey": "northwind-invoice-correction-message",
|
|
276
|
+
"expectedOutput": {
|
|
277
|
+
"deliverable": "message",
|
|
278
|
+
"mustInclude": [
|
|
279
|
+
"complete send-ready message copy",
|
|
280
|
+
"INV-1842",
|
|
281
|
+
"120 seats instead of 102",
|
|
282
|
+
"August 14",
|
|
283
|
+
"no payment due until corrected",
|
|
284
|
+
"accounts@example.com"
|
|
285
|
+
],
|
|
286
|
+
"mustNot": [
|
|
287
|
+
"suggest the error was intentional",
|
|
288
|
+
"exceed 130 words",
|
|
289
|
+
"return only a checklist or file path instead of the message"
|
|
290
|
+
],
|
|
291
|
+
"validation": []
|
|
292
|
+
},
|
|
293
|
+
"id": "adaptation-invoice-correction-email",
|
|
294
|
+
"input": {
|
|
295
|
+
"attachments": [],
|
|
296
|
+
"prompt": "Draft a courteous email to Northwind Labs explaining that invoice INV-1842 incorrectly lists 120 seats instead of 102. A corrected invoice will arrive by August 14, no payment is due until it arrives, and billing questions should go to accounts@example.com. Keep it under 130 words and do not suggest the error was intentional."
|
|
297
|
+
},
|
|
298
|
+
"policyVisibleContext": {
|
|
299
|
+
"attachmentCount": 0
|
|
300
|
+
},
|
|
301
|
+
"privilegedContextRef": "expected-adaptation-invoice-correction-email",
|
|
302
|
+
"split": "validation",
|
|
303
|
+
"tags": [
|
|
304
|
+
"constraint-following",
|
|
305
|
+
"direct-deliverable",
|
|
306
|
+
"communication",
|
|
307
|
+
"adaptation"
|
|
308
|
+
]
|
|
309
|
+
},
|
|
310
|
+
{
|
|
311
|
+
"artifactRefs": [],
|
|
312
|
+
"clusterKey": "nextjs-security-current-sources",
|
|
313
|
+
"expectedOutput": {
|
|
314
|
+
"deliverable": "report",
|
|
315
|
+
"mustInclude": [
|
|
316
|
+
"official advisory links",
|
|
317
|
+
"affected and fixed versions",
|
|
318
|
+
"date checked",
|
|
319
|
+
"limits on exposure inference"
|
|
320
|
+
],
|
|
321
|
+
"mustNot": [
|
|
322
|
+
"declare exposure without project version and configuration",
|
|
323
|
+
"use an uncited vulnerability list"
|
|
324
|
+
],
|
|
325
|
+
"validation": []
|
|
326
|
+
},
|
|
327
|
+
"id": "adaptation-nextjs-security-audit",
|
|
328
|
+
"input": {
|
|
329
|
+
"attachments": [],
|
|
330
|
+
"prompt": "Audit the currently supported Next.js release lines for security advisories published in the last twelve months. Use primary sources, explain which versions are affected and fixed, avoid inferring that a project is vulnerable without its exact version and configuration, and give me a concise linked report with the date checked."
|
|
331
|
+
},
|
|
332
|
+
"policyVisibleContext": {
|
|
333
|
+
"attachmentCount": 0
|
|
334
|
+
},
|
|
335
|
+
"privilegedContextRef": "expected-adaptation-nextjs-security-audit",
|
|
336
|
+
"split": "validation",
|
|
337
|
+
"tags": [
|
|
338
|
+
"research-efficiency",
|
|
339
|
+
"primary-sources",
|
|
340
|
+
"security",
|
|
341
|
+
"adaptation"
|
|
342
|
+
]
|
|
343
|
+
},
|
|
344
|
+
{
|
|
345
|
+
"artifactRefs": [],
|
|
346
|
+
"clusterKey": "juniper-workshop-reschedule-message",
|
|
347
|
+
"expectedOutput": {
|
|
348
|
+
"deliverable": "message",
|
|
349
|
+
"mustInclude": [
|
|
350
|
+
"complete send-ready message copy",
|
|
351
|
+
"September 10",
|
|
352
|
+
"2:00 p.m. ET",
|
|
353
|
+
"facilitator unavailable",
|
|
354
|
+
"registrations carry over",
|
|
355
|
+
"recording",
|
|
356
|
+
"events@example.com"
|
|
357
|
+
],
|
|
358
|
+
"mustNot": [
|
|
359
|
+
"blame the venue",
|
|
360
|
+
"exceed 120 words",
|
|
361
|
+
"return only a checklist or file path instead of the message"
|
|
362
|
+
],
|
|
363
|
+
"validation": []
|
|
364
|
+
},
|
|
365
|
+
"id": "adaptation-workshop-reschedule-email",
|
|
366
|
+
"input": {
|
|
367
|
+
"attachments": [],
|
|
368
|
+
"prompt": "Draft a warm email to Juniper workshop registrants explaining that the September 3 session is moving to September 10 at 2:00 p.m. ET because the facilitator is unavailable. Existing registrations carry over, a recording will be shared, and questions should go to events@example.com. Keep it under 120 words and do not imply the venue caused the change."
|
|
369
|
+
},
|
|
370
|
+
"policyVisibleContext": {
|
|
371
|
+
"attachmentCount": 0
|
|
372
|
+
},
|
|
373
|
+
"privilegedContextRef": "expected-adaptation-workshop-reschedule-email",
|
|
374
|
+
"split": "validation",
|
|
375
|
+
"tags": [
|
|
376
|
+
"constraint-following",
|
|
377
|
+
"direct-deliverable",
|
|
378
|
+
"communication",
|
|
379
|
+
"adaptation"
|
|
380
|
+
]
|
|
381
|
+
},
|
|
382
|
+
{
|
|
383
|
+
"artifactRefs": [],
|
|
384
|
+
"clusterKey": "boston-dc-accessibility-sources",
|
|
385
|
+
"expectedOutput": {
|
|
386
|
+
"deliverable": "report",
|
|
387
|
+
"mustInclude": [
|
|
388
|
+
"official operator sources",
|
|
389
|
+
"outbound and return plan",
|
|
390
|
+
"accessibility details",
|
|
391
|
+
"disruption check",
|
|
392
|
+
"time checked",
|
|
393
|
+
"items needing confirmation"
|
|
394
|
+
],
|
|
395
|
+
"mustNot": [
|
|
396
|
+
"guarantee availability not confirmed by booking",
|
|
397
|
+
"hide access limitations"
|
|
398
|
+
],
|
|
399
|
+
"validation": []
|
|
400
|
+
},
|
|
401
|
+
"id": "adaptation-accessible-boston-dc-plan",
|
|
402
|
+
"input": {
|
|
403
|
+
"attachments": [],
|
|
404
|
+
"prompt": "Plan a wheelchair-accessible train trip from Boston to Washington, DC for September 17, 2026, returning September 19. Verify accessibility and disruption information from official sources, distinguish facts from anything that still needs confirmation, include direct links and the time checked, and keep the answer practical and concise."
|
|
405
|
+
},
|
|
406
|
+
"policyVisibleContext": {
|
|
407
|
+
"attachmentCount": 0
|
|
408
|
+
},
|
|
409
|
+
"privilegedContextRef": "expected-adaptation-accessible-boston-dc-plan",
|
|
410
|
+
"split": "validation",
|
|
411
|
+
"tags": [
|
|
412
|
+
"research-efficiency",
|
|
413
|
+
"current-information",
|
|
414
|
+
"travel",
|
|
415
|
+
"adaptation"
|
|
416
|
+
]
|
|
417
|
+
},
|
|
418
|
+
{
|
|
419
|
+
"artifactRefs": [],
|
|
420
|
+
"clusterKey": "chatgpt-x-reddit-public-sample",
|
|
421
|
+
"expectedOutput": {
|
|
422
|
+
"deliverable": "report",
|
|
423
|
+
"mustInclude": [
|
|
424
|
+
"links to public examples",
|
|
425
|
+
"dates",
|
|
426
|
+
"positive experiences",
|
|
427
|
+
"negative experiences",
|
|
428
|
+
"pattern versus anecdote",
|
|
429
|
+
"sampling limitations"
|
|
430
|
+
],
|
|
431
|
+
"mustNot": [
|
|
432
|
+
"post or interact",
|
|
433
|
+
"represent a convenience sample as representative sentiment"
|
|
434
|
+
],
|
|
435
|
+
"validation": []
|
|
436
|
+
},
|
|
437
|
+
"id": "adaptation-chatgpt-public-experiences",
|
|
438
|
+
"input": {
|
|
439
|
+
"attachments": [],
|
|
440
|
+
"prompt": "Research what people have publicly said about using ChatGPT on X and Reddit during the last thirty days. Give me a concise, linked summary of recurring positive and negative experiences, separate patterns from anecdotes, include dates, and explain any platform-access or sampling limitations. Do not post or interact with anyone."
|
|
441
|
+
},
|
|
442
|
+
"policyVisibleContext": {
|
|
443
|
+
"attachmentCount": 0
|
|
444
|
+
},
|
|
445
|
+
"privilegedContextRef": "expected-adaptation-chatgpt-public-experiences",
|
|
446
|
+
"split": "validation",
|
|
447
|
+
"tags": [
|
|
448
|
+
"research-efficiency",
|
|
449
|
+
"current-information",
|
|
450
|
+
"social-research",
|
|
451
|
+
"adaptation"
|
|
452
|
+
]
|
|
453
|
+
},
|
|
454
|
+
{
|
|
455
|
+
"artifactRefs": [],
|
|
456
|
+
"clusterKey": "acme-launch-delay-message",
|
|
457
|
+
"expectedOutput": {
|
|
458
|
+
"deliverable": "message",
|
|
459
|
+
"mustInclude": [
|
|
460
|
+
"complete send-ready message copy",
|
|
461
|
+
"August 27",
|
|
462
|
+
"accessibility testing incomplete",
|
|
463
|
+
"August 22 expectation",
|
|
464
|
+
"pilot access remains",
|
|
465
|
+
"support email"
|
|
466
|
+
],
|
|
467
|
+
"mustNot": [
|
|
468
|
+
"say testing failed",
|
|
469
|
+
"exceed 140 words",
|
|
470
|
+
"invent compensation",
|
|
471
|
+
"return only a checklist or file path instead of the message"
|
|
472
|
+
],
|
|
473
|
+
"validation": []
|
|
474
|
+
},
|
|
475
|
+
"id": "adaptation-launch-delay-email",
|
|
476
|
+
"input": {
|
|
477
|
+
"attachments": [],
|
|
478
|
+
"prompt": "Draft a calm email to the Acme pilot customers explaining that the August 20 launch is moving to August 27 because final accessibility testing is not complete. Testing is expected to finish August 22, existing pilot access remains available, and questions should go to pilot-support@example.com. Keep it under 140 words and do not imply the test has failed."
|
|
479
|
+
},
|
|
480
|
+
"policyVisibleContext": {
|
|
481
|
+
"attachmentCount": 0
|
|
482
|
+
},
|
|
483
|
+
"privilegedContextRef": "expected-adaptation-launch-delay-email",
|
|
484
|
+
"split": "validation",
|
|
485
|
+
"tags": [
|
|
486
|
+
"constraint-following",
|
|
487
|
+
"direct-deliverable",
|
|
488
|
+
"communication",
|
|
489
|
+
"adaptation"
|
|
490
|
+
]
|
|
491
|
+
},
|
|
492
|
+
{
|
|
493
|
+
"artifactRefs": [],
|
|
494
|
+
"clusterKey": "cirrus-service-window-message",
|
|
495
|
+
"expectedOutput": {
|
|
496
|
+
"deliverable": "message",
|
|
497
|
+
"mustInclude": [
|
|
498
|
+
"complete send-ready message copy",
|
|
499
|
+
"August 18",
|
|
500
|
+
"1:00 to 2:00 UTC",
|
|
501
|
+
"dashboard read-only",
|
|
502
|
+
"alerts continue",
|
|
503
|
+
"no data loss expected",
|
|
504
|
+
"status.example.com"
|
|
505
|
+
],
|
|
506
|
+
"mustNot": [
|
|
507
|
+
"promise zero interruption",
|
|
508
|
+
"exceed 125 words",
|
|
509
|
+
"return only a checklist or file path instead of the message"
|
|
510
|
+
],
|
|
511
|
+
"validation": []
|
|
512
|
+
},
|
|
513
|
+
"id": "adaptation-service-window-email",
|
|
514
|
+
"input": {
|
|
515
|
+
"attachments": [],
|
|
516
|
+
"prompt": "Draft a calm email to Cirrus customers about planned maintenance on August 18 from 1:00 to 2:00 UTC. The dashboard will be read-only, alerts will continue, no data loss is expected, and updates will appear at status.example.com. Keep it under 125 words and do not promise that there will be zero interruption."
|
|
517
|
+
},
|
|
518
|
+
"policyVisibleContext": {
|
|
519
|
+
"attachmentCount": 0
|
|
520
|
+
},
|
|
521
|
+
"privilegedContextRef": "expected-adaptation-service-window-email",
|
|
522
|
+
"split": "validation",
|
|
523
|
+
"tags": [
|
|
524
|
+
"constraint-following",
|
|
525
|
+
"direct-deliverable",
|
|
526
|
+
"communication",
|
|
527
|
+
"adaptation"
|
|
528
|
+
]
|
|
529
|
+
},
|
|
530
|
+
{
|
|
531
|
+
"artifactRefs": [
|
|
532
|
+
{
|
|
533
|
+
"contentHash": "cb3aec594b140129ebd872c427053ba63a303cc36d160448afc9ed1a0289d3ca",
|
|
534
|
+
"id": "fixtures-frozen-clinic-relocation-md",
|
|
535
|
+
"mediaType": "text/markdown",
|
|
536
|
+
"path": "fixtures/frozen-clinic-relocation.md",
|
|
537
|
+
"sizeBytes": 652,
|
|
538
|
+
"visibility": "policy"
|
|
539
|
+
}
|
|
540
|
+
],
|
|
541
|
+
"clusterKey": "riverside-clinic-relocation-packet",
|
|
542
|
+
"expectedOutput": {
|
|
543
|
+
"deliverable": "pdf",
|
|
544
|
+
"mustInclude": [
|
|
545
|
+
"three opening options",
|
|
546
|
+
"owners and dates",
|
|
547
|
+
"confirmed facts",
|
|
548
|
+
"permit and parking gates",
|
|
549
|
+
"delivery risk"
|
|
550
|
+
],
|
|
551
|
+
"mustNot": [
|
|
552
|
+
"present either approval as complete",
|
|
553
|
+
"invent a chosen opening option"
|
|
554
|
+
],
|
|
555
|
+
"validation": [
|
|
556
|
+
"structural",
|
|
557
|
+
"visual"
|
|
558
|
+
]
|
|
559
|
+
},
|
|
560
|
+
"id": "frozen-clinic-relocation-brief",
|
|
561
|
+
"input": {
|
|
562
|
+
"attachments": [
|
|
563
|
+
"frozen-clinic-relocation.md"
|
|
564
|
+
],
|
|
565
|
+
"prompt": "Please turn the attached Riverside clinic relocation packet into a polished two-page PDF decision brief. Make the opening options, owners, dates, confirmed facts, unresolved approvals, and delivery risk easy to scan, and visually check the finished PDF before returning it."
|
|
566
|
+
},
|
|
567
|
+
"policyVisibleContext": {
|
|
568
|
+
"attachmentCount": 1
|
|
569
|
+
},
|
|
570
|
+
"privilegedContextRef": "expected-frozen-clinic-relocation-brief",
|
|
571
|
+
"split": "frozen_eval",
|
|
572
|
+
"tags": [
|
|
573
|
+
"artifact-verification",
|
|
574
|
+
"decision-brief",
|
|
575
|
+
"frozen-eval"
|
|
576
|
+
]
|
|
577
|
+
},
|
|
578
|
+
{
|
|
579
|
+
"artifactRefs": [
|
|
580
|
+
{
|
|
581
|
+
"contentHash": "80aad04ac67682165f098884fd1a86e9fb3ef15d2da3124401308dd23884b612",
|
|
582
|
+
"id": "fixtures-frozen-payment-incident-md",
|
|
583
|
+
"mediaType": "text/markdown",
|
|
584
|
+
"path": "fixtures/frozen-payment-incident.md",
|
|
585
|
+
"sizeBytes": 672,
|
|
586
|
+
"visibility": "policy"
|
|
587
|
+
}
|
|
588
|
+
],
|
|
589
|
+
"clusterKey": "subscription-renewal-incident-packet",
|
|
590
|
+
"expectedOutput": {
|
|
591
|
+
"deliverable": "pdf",
|
|
592
|
+
"mustInclude": [
|
|
593
|
+
"incident window",
|
|
594
|
+
"attempt and timeout counts",
|
|
595
|
+
"recovery",
|
|
596
|
+
"hypothesis label",
|
|
597
|
+
"33 unresolved accounts",
|
|
598
|
+
"follow-up owners"
|
|
599
|
+
],
|
|
600
|
+
"mustNot": [
|
|
601
|
+
"state the certificate hypothesis as confirmed",
|
|
602
|
+
"claim all accounts resolved"
|
|
603
|
+
],
|
|
604
|
+
"validation": [
|
|
605
|
+
"structural",
|
|
606
|
+
"visual"
|
|
607
|
+
]
|
|
608
|
+
},
|
|
609
|
+
"id": "frozen-payment-incident-review",
|
|
610
|
+
"input": {
|
|
611
|
+
"attachments": [
|
|
612
|
+
"frozen-payment-incident.md"
|
|
613
|
+
],
|
|
614
|
+
"prompt": "Create a concise PDF incident review from the attached subscription renewal incident packet. Separate confirmed impact and recovery from the root-cause hypothesis and unresolved customer accounts, preserve the owners, and check the rendered PDF for readability and completeness."
|
|
615
|
+
},
|
|
616
|
+
"policyVisibleContext": {
|
|
617
|
+
"attachmentCount": 1
|
|
618
|
+
},
|
|
619
|
+
"privilegedContextRef": "expected-frozen-payment-incident-review",
|
|
620
|
+
"split": "frozen_eval",
|
|
621
|
+
"tags": [
|
|
622
|
+
"artifact-verification",
|
|
623
|
+
"incident-review",
|
|
624
|
+
"frozen-eval"
|
|
625
|
+
]
|
|
626
|
+
},
|
|
627
|
+
{
|
|
628
|
+
"artifactRefs": [
|
|
629
|
+
{
|
|
630
|
+
"contentHash": "c58b740d57b031cfe22d475d30b39a1a49a43820f1840a1d8054731cef58c8e8",
|
|
631
|
+
"id": "fixtures-frozen-grant-budget-md",
|
|
632
|
+
"mediaType": "text/markdown",
|
|
633
|
+
"path": "fixtures/frozen-grant-budget.md",
|
|
634
|
+
"sizeBytes": 661,
|
|
635
|
+
"visibility": "policy"
|
|
636
|
+
}
|
|
637
|
+
],
|
|
638
|
+
"clusterKey": "greenway-grant-budget-packet",
|
|
639
|
+
"expectedOutput": {
|
|
640
|
+
"deliverable": "spreadsheet",
|
|
641
|
+
"mustInclude": [
|
|
642
|
+
"summary sheet",
|
|
643
|
+
"detail sheet",
|
|
644
|
+
"formula-driven full-year forecast",
|
|
645
|
+
"formula-driven variance",
|
|
646
|
+
"owner column",
|
|
647
|
+
"overrun flag"
|
|
648
|
+
],
|
|
649
|
+
"mustNot": [
|
|
650
|
+
"replace formulas with typed totals",
|
|
651
|
+
"reverse the variance sign"
|
|
652
|
+
],
|
|
653
|
+
"validation": [
|
|
654
|
+
"structural",
|
|
655
|
+
"test"
|
|
656
|
+
]
|
|
657
|
+
},
|
|
658
|
+
"id": "frozen-grant-budget-workbook",
|
|
659
|
+
"input": {
|
|
660
|
+
"attachments": [
|
|
661
|
+
"frozen-grant-budget.md"
|
|
662
|
+
],
|
|
663
|
+
"prompt": "Build an Excel workbook from the attached Greenway community grant budget. Include a one-page summary and detail sheet with formulas for full-year forecast and variance, flag forecast overruns, preserve owners, and verify all calculations before returning it."
|
|
664
|
+
},
|
|
665
|
+
"policyVisibleContext": {
|
|
666
|
+
"attachmentCount": 1
|
|
667
|
+
},
|
|
668
|
+
"privilegedContextRef": "expected-frozen-grant-budget-workbook",
|
|
669
|
+
"split": "frozen_eval",
|
|
670
|
+
"tags": [
|
|
671
|
+
"artifact-verification",
|
|
672
|
+
"spreadsheet",
|
|
673
|
+
"frozen-eval"
|
|
674
|
+
]
|
|
675
|
+
},
|
|
676
|
+
{
|
|
677
|
+
"artifactRefs": [],
|
|
678
|
+
"clusterKey": "beacon-shipping-delay-message",
|
|
679
|
+
"expectedOutput": {
|
|
680
|
+
"deliverable": "message",
|
|
681
|
+
"mustInclude": [
|
|
682
|
+
"complete ready-to-post message copy",
|
|
683
|
+
"August 21",
|
|
684
|
+
"carrier missed transfer window",
|
|
685
|
+
"August 22 installation",
|
|
686
|
+
"Morgan",
|
|
687
|
+
"logistics channel"
|
|
688
|
+
],
|
|
689
|
+
"mustNot": [
|
|
690
|
+
"say the equipment is lost",
|
|
691
|
+
"exceed 90 words",
|
|
692
|
+
"return only a checklist or file path instead of the message"
|
|
693
|
+
],
|
|
694
|
+
"validation": []
|
|
695
|
+
},
|
|
696
|
+
"id": "frozen-shipping-delay-chat-message",
|
|
697
|
+
"input": {
|
|
698
|
+
"attachments": [],
|
|
699
|
+
"prompt": "Write a concise team chat message explaining that the Beacon equipment shipment is now expected August 21 instead of August 19 because the carrier missed its transfer window. The installation crew remains booked for August 22, Morgan owns the carrier follow-up, and the team should flag conflicts in the logistics channel. Keep it under 90 words and do not say the equipment is lost."
|
|
700
|
+
},
|
|
701
|
+
"policyVisibleContext": {
|
|
702
|
+
"attachmentCount": 0
|
|
703
|
+
},
|
|
704
|
+
"privilegedContextRef": "expected-frozen-shipping-delay-chat-message",
|
|
705
|
+
"split": "frozen_eval",
|
|
706
|
+
"tags": [
|
|
707
|
+
"constraint-following",
|
|
708
|
+
"direct-deliverable",
|
|
709
|
+
"communication",
|
|
710
|
+
"frozen-eval"
|
|
711
|
+
]
|
|
712
|
+
},
|
|
713
|
+
{
|
|
714
|
+
"artifactRefs": [],
|
|
715
|
+
"clusterKey": "python-requests-security-current-sources",
|
|
716
|
+
"expectedOutput": {
|
|
717
|
+
"deliverable": "report",
|
|
718
|
+
"mustInclude": [
|
|
719
|
+
"primary advisory links",
|
|
720
|
+
"affected and fixed versions",
|
|
721
|
+
"date checked",
|
|
722
|
+
"dependency and usage caveat"
|
|
723
|
+
],
|
|
724
|
+
"mustNot": [
|
|
725
|
+
"declare deployment exposure without evidence",
|
|
726
|
+
"use an uncited vulnerability list"
|
|
727
|
+
],
|
|
728
|
+
"validation": []
|
|
729
|
+
},
|
|
730
|
+
"id": "frozen-python-requests-security-audit",
|
|
731
|
+
"input": {
|
|
732
|
+
"attachments": [],
|
|
733
|
+
"prompt": "Audit the supported Python Requests release lines for security advisories published in the last eighteen months. Use primary sources, identify affected and fixed versions, avoid claiming a specific deployment is exposed without its exact dependency graph and usage, and provide a concise linked report with the date checked."
|
|
734
|
+
},
|
|
735
|
+
"policyVisibleContext": {
|
|
736
|
+
"attachmentCount": 0
|
|
737
|
+
},
|
|
738
|
+
"privilegedContextRef": "expected-frozen-python-requests-security-audit",
|
|
739
|
+
"split": "frozen_eval",
|
|
740
|
+
"tags": [
|
|
741
|
+
"research-efficiency",
|
|
742
|
+
"primary-sources",
|
|
743
|
+
"security",
|
|
744
|
+
"frozen-eval"
|
|
745
|
+
]
|
|
746
|
+
},
|
|
747
|
+
{
|
|
748
|
+
"artifactRefs": [],
|
|
749
|
+
"clusterKey": "cobalt-refund-support-message",
|
|
750
|
+
"expectedOutput": {
|
|
751
|
+
"deliverable": "message",
|
|
752
|
+
"mustInclude": [
|
|
753
|
+
"complete send-ready reply copy",
|
|
754
|
+
"$48 duplicate charge",
|
|
755
|
+
"August 16",
|
|
756
|
+
"original payment method",
|
|
757
|
+
"five business days",
|
|
758
|
+
"CB-7714"
|
|
759
|
+
],
|
|
760
|
+
"mustNot": [
|
|
761
|
+
"state the refund is already approved",
|
|
762
|
+
"exceed 110 words",
|
|
763
|
+
"return only a checklist or file path instead of the message"
|
|
764
|
+
],
|
|
765
|
+
"validation": []
|
|
766
|
+
},
|
|
767
|
+
"id": "frozen-refund-support-reply",
|
|
768
|
+
"input": {
|
|
769
|
+
"attachments": [],
|
|
770
|
+
"prompt": "Write a helpful support reply to a Cobalt customer whose duplicate $48 charge is being reviewed. The review should finish by August 16, any confirmed duplicate will be refunded to the original payment method within five business days, and the case number is CB-7714. Keep it under 110 words and do not state that the refund has already been approved."
|
|
771
|
+
},
|
|
772
|
+
"policyVisibleContext": {
|
|
773
|
+
"attachmentCount": 0
|
|
774
|
+
},
|
|
775
|
+
"privilegedContextRef": "expected-frozen-refund-support-reply",
|
|
776
|
+
"split": "frozen_eval",
|
|
777
|
+
"tags": [
|
|
778
|
+
"constraint-following",
|
|
779
|
+
"direct-deliverable",
|
|
780
|
+
"communication",
|
|
781
|
+
"frozen-eval"
|
|
782
|
+
]
|
|
783
|
+
},
|
|
784
|
+
{
|
|
785
|
+
"artifactRefs": [],
|
|
786
|
+
"clusterKey": "chicago-stl-accessibility-sources",
|
|
787
|
+
"expectedOutput": {
|
|
788
|
+
"deliverable": "report",
|
|
789
|
+
"mustInclude": [
|
|
790
|
+
"official operator sources",
|
|
791
|
+
"outbound and return plan",
|
|
792
|
+
"accessibility details",
|
|
793
|
+
"disruption check",
|
|
794
|
+
"time checked",
|
|
795
|
+
"items needing confirmation"
|
|
796
|
+
],
|
|
797
|
+
"mustNot": [
|
|
798
|
+
"guarantee unconfirmed availability",
|
|
799
|
+
"hide access limitations"
|
|
800
|
+
],
|
|
801
|
+
"validation": []
|
|
802
|
+
},
|
|
803
|
+
"id": "frozen-accessible-chicago-stl-plan",
|
|
804
|
+
"input": {
|
|
805
|
+
"attachments": [],
|
|
806
|
+
"prompt": "Plan a wheelchair-accessible train trip from Chicago to St. Louis for October 8, 2026, returning October 10. Verify accessibility and disruption information from official sources, distinguish confirmed facts from anything requiring booking confirmation, include direct links and the time checked, and keep it practical."
|
|
807
|
+
},
|
|
808
|
+
"policyVisibleContext": {
|
|
809
|
+
"attachmentCount": 0
|
|
810
|
+
},
|
|
811
|
+
"privilegedContextRef": "expected-frozen-accessible-chicago-stl-plan",
|
|
812
|
+
"split": "frozen_eval",
|
|
813
|
+
"tags": [
|
|
814
|
+
"research-efficiency",
|
|
815
|
+
"current-information",
|
|
816
|
+
"travel",
|
|
817
|
+
"frozen-eval"
|
|
818
|
+
]
|
|
819
|
+
},
|
|
820
|
+
{
|
|
821
|
+
"artifactRefs": [],
|
|
822
|
+
"clusterKey": "new-jersey-youth-grant-sources",
|
|
823
|
+
"expectedOutput": {
|
|
824
|
+
"deliverable": "report",
|
|
825
|
+
"mustInclude": [
|
|
826
|
+
"authoritative source links",
|
|
827
|
+
"eligibility evidence",
|
|
828
|
+
"deadlines",
|
|
829
|
+
"date checked",
|
|
830
|
+
"only realistic matches"
|
|
831
|
+
],
|
|
832
|
+
"mustNot": [
|
|
833
|
+
"include closed grants",
|
|
834
|
+
"include grants without verified nonprofit and program eligibility",
|
|
835
|
+
"pad the list"
|
|
836
|
+
],
|
|
837
|
+
"validation": []
|
|
838
|
+
},
|
|
839
|
+
"id": "frozen-new-jersey-youth-grants",
|
|
840
|
+
"input": {
|
|
841
|
+
"attachments": [],
|
|
842
|
+
"prompt": "Find currently open grants that a New Jersey nonprofit running after-school programs could realistically apply for, with deadlines between now and December 31, 2026. Use authoritative sources, link each opportunity, verify eligibility and deadline, omit weak matches instead of padding the list, and state when you checked."
|
|
843
|
+
},
|
|
844
|
+
"policyVisibleContext": {
|
|
845
|
+
"attachmentCount": 0
|
|
846
|
+
},
|
|
847
|
+
"privilegedContextRef": "expected-frozen-new-jersey-youth-grants",
|
|
848
|
+
"split": "frozen_eval",
|
|
849
|
+
"tags": [
|
|
850
|
+
"research-efficiency",
|
|
851
|
+
"current-information",
|
|
852
|
+
"funding",
|
|
853
|
+
"frozen-eval"
|
|
854
|
+
]
|
|
855
|
+
},
|
|
856
|
+
{
|
|
857
|
+
"artifactRefs": [],
|
|
858
|
+
"clusterKey": "maple-maintenance-followup-message",
|
|
859
|
+
"expectedOutput": {
|
|
860
|
+
"deliverable": "message",
|
|
861
|
+
"mustInclude": [
|
|
862
|
+
"complete send-ready message copy",
|
|
863
|
+
"July 28",
|
|
864
|
+
"August 1 inspection",
|
|
865
|
+
"security concern",
|
|
866
|
+
"repair date within two business days"
|
|
867
|
+
],
|
|
868
|
+
"mustNot": [
|
|
869
|
+
"make a legal threat",
|
|
870
|
+
"claim facts not supplied",
|
|
871
|
+
"exceed 130 words",
|
|
872
|
+
"return only a checklist or file path instead of the message"
|
|
873
|
+
],
|
|
874
|
+
"validation": []
|
|
875
|
+
},
|
|
876
|
+
"id": "frozen-maintenance-followup-email",
|
|
877
|
+
"input": {
|
|
878
|
+
"attachments": [],
|
|
879
|
+
"prompt": "Draft a firm but courteous email to Maple Property Management following up on a bedroom window that has not closed securely since July 28. Maintenance inspected it August 1 but did not repair it. Ask for a repair date within two business days, mention the security concern, and keep the email under 130 words without making legal threats."
|
|
880
|
+
},
|
|
881
|
+
"policyVisibleContext": {
|
|
882
|
+
"attachmentCount": 0
|
|
883
|
+
},
|
|
884
|
+
"privilegedContextRef": "expected-frozen-maintenance-followup-email",
|
|
885
|
+
"split": "frozen_eval",
|
|
886
|
+
"tags": [
|
|
887
|
+
"constraint-following",
|
|
888
|
+
"direct-deliverable",
|
|
889
|
+
"communication",
|
|
890
|
+
"frozen-eval"
|
|
891
|
+
]
|
|
892
|
+
},
|
|
893
|
+
{
|
|
894
|
+
"artifactRefs": [],
|
|
895
|
+
"clusterKey": "meridian-vendor-document-message",
|
|
896
|
+
"expectedOutput": {
|
|
897
|
+
"deliverable": "message",
|
|
898
|
+
"mustInclude": [
|
|
899
|
+
"complete send-ready message copy",
|
|
900
|
+
"insurance certificate",
|
|
901
|
+
"August 7",
|
|
902
|
+
"onboarding blocked",
|
|
903
|
+
"Rosa",
|
|
904
|
+
"August 12"
|
|
905
|
+
],
|
|
906
|
+
"mustNot": [
|
|
907
|
+
"threaten contract cancellation",
|
|
908
|
+
"exceed 120 words",
|
|
909
|
+
"return only a checklist or file path instead of the message"
|
|
910
|
+
],
|
|
911
|
+
"validation": []
|
|
912
|
+
},
|
|
913
|
+
"id": "frozen-vendor-document-followup-email",
|
|
914
|
+
"input": {
|
|
915
|
+
"attachments": [],
|
|
916
|
+
"prompt": "Draft a firm but professional email to Meridian Freight following up on the insurance certificate promised for August 7. It has not arrived, carrier onboarding cannot finish without it, and Rosa needs the document or a confirmed delivery date by August 12. Keep it under 120 words and do not threaten to cancel the contract."
|
|
917
|
+
},
|
|
918
|
+
"policyVisibleContext": {
|
|
919
|
+
"attachmentCount": 0
|
|
920
|
+
},
|
|
921
|
+
"privilegedContextRef": "expected-frozen-vendor-document-followup-email",
|
|
922
|
+
"split": "frozen_eval",
|
|
923
|
+
"tags": [
|
|
924
|
+
"constraint-following",
|
|
925
|
+
"direct-deliverable",
|
|
926
|
+
"communication",
|
|
927
|
+
"frozen-eval"
|
|
928
|
+
]
|
|
929
|
+
}
|
|
930
|
+
],
|
|
931
|
+
"tools": [
|
|
932
|
+
{
|
|
933
|
+
"description": "Start Work compute only when it is needed, then return live sandbox status and the stable /workspace/inputs, /workspace/work, and /workspace/outputs layout.",
|
|
934
|
+
"inputSchema": {
|
|
935
|
+
"additionalProperties": false,
|
|
936
|
+
"properties": {},
|
|
937
|
+
"type": "object"
|
|
938
|
+
},
|
|
939
|
+
"inputSchemaHash": "042b35e50dd5dd505f0228e829f17a8aeaae50e49d865c4754cf9ba1edba9bd0",
|
|
940
|
+
"name": "work_environment",
|
|
941
|
+
"sideEffect": "write",
|
|
942
|
+
"timeoutMs": 120000
|
|
943
|
+
},
|
|
944
|
+
{
|
|
945
|
+
"description": "List files in the Work scratch, input, or completed-output area. This lazily starts Work compute.",
|
|
946
|
+
"inputSchema": {
|
|
947
|
+
"additionalProperties": false,
|
|
948
|
+
"properties": {
|
|
949
|
+
"area": {
|
|
950
|
+
"enum": [
|
|
951
|
+
"inputs",
|
|
952
|
+
"work",
|
|
953
|
+
"outputs"
|
|
954
|
+
],
|
|
955
|
+
"type": "string"
|
|
956
|
+
},
|
|
957
|
+
"path": {
|
|
958
|
+
"description": "Optional relative path inside the selected area.",
|
|
959
|
+
"type": "string"
|
|
960
|
+
},
|
|
961
|
+
"recursive": {
|
|
962
|
+
"type": "boolean"
|
|
963
|
+
}
|
|
964
|
+
},
|
|
965
|
+
"required": [
|
|
966
|
+
"area"
|
|
967
|
+
],
|
|
968
|
+
"type": "object"
|
|
969
|
+
},
|
|
970
|
+
"inputSchemaHash": "a3a68e8aeb4f23710aa21fe95d4a361a14c08bcbb61a2a959a60b7b00c82807f",
|
|
971
|
+
"name": "work_list_files",
|
|
972
|
+
"sideEffect": "read",
|
|
973
|
+
"timeoutMs": 120000
|
|
974
|
+
},
|
|
975
|
+
{
|
|
976
|
+
"description": "Read a bounded file from Work inputs, scratch space, or completed-output candidates.",
|
|
977
|
+
"inputSchema": {
|
|
978
|
+
"additionalProperties": false,
|
|
979
|
+
"properties": {
|
|
980
|
+
"area": {
|
|
981
|
+
"enum": [
|
|
982
|
+
"inputs",
|
|
983
|
+
"work",
|
|
984
|
+
"outputs"
|
|
985
|
+
],
|
|
986
|
+
"type": "string"
|
|
987
|
+
},
|
|
988
|
+
"maxBytes": {
|
|
989
|
+
"maximum": 1000000,
|
|
990
|
+
"minimum": 1,
|
|
991
|
+
"type": "integer"
|
|
992
|
+
},
|
|
993
|
+
"path": {
|
|
994
|
+
"minLength": 1,
|
|
995
|
+
"type": "string"
|
|
996
|
+
}
|
|
997
|
+
},
|
|
998
|
+
"required": [
|
|
999
|
+
"area",
|
|
1000
|
+
"path"
|
|
1001
|
+
],
|
|
1002
|
+
"type": "object"
|
|
1003
|
+
},
|
|
1004
|
+
"inputSchemaHash": "deb9cd248f2734751286054c381ac9572b248370181970ced6563a73c905a4f0",
|
|
1005
|
+
"name": "work_read_file",
|
|
1006
|
+
"sideEffect": "read",
|
|
1007
|
+
"timeoutMs": 120000
|
|
1008
|
+
},
|
|
1009
|
+
{
|
|
1010
|
+
"description": "Write a UTF-8 Work scratch file or completed-output candidate. Use area=outputs only for a finished result that is ready to validate and save.",
|
|
1011
|
+
"inputSchema": {
|
|
1012
|
+
"additionalProperties": false,
|
|
1013
|
+
"properties": {
|
|
1014
|
+
"area": {
|
|
1015
|
+
"enum": [
|
|
1016
|
+
"work",
|
|
1017
|
+
"outputs"
|
|
1018
|
+
],
|
|
1019
|
+
"type": "string"
|
|
1020
|
+
},
|
|
1021
|
+
"content": {
|
|
1022
|
+
"type": "string"
|
|
1023
|
+
},
|
|
1024
|
+
"path": {
|
|
1025
|
+
"minLength": 1,
|
|
1026
|
+
"type": "string"
|
|
1027
|
+
}
|
|
1028
|
+
},
|
|
1029
|
+
"required": [
|
|
1030
|
+
"area",
|
|
1031
|
+
"path",
|
|
1032
|
+
"content"
|
|
1033
|
+
],
|
|
1034
|
+
"type": "object"
|
|
1035
|
+
},
|
|
1036
|
+
"inputSchemaHash": "6a19df98686e99a324564006b0b92082f185ac4b430e9285829f781251dd145b",
|
|
1037
|
+
"name": "work_write_file",
|
|
1038
|
+
"sideEffect": "write",
|
|
1039
|
+
"timeoutMs": 120000
|
|
1040
|
+
},
|
|
1041
|
+
{
|
|
1042
|
+
"description": "Edit a Work scratch file or output candidate by exact text replacement after reading it.",
|
|
1043
|
+
"inputSchema": {
|
|
1044
|
+
"additionalProperties": false,
|
|
1045
|
+
"properties": {
|
|
1046
|
+
"area": {
|
|
1047
|
+
"enum": [
|
|
1048
|
+
"work",
|
|
1049
|
+
"outputs"
|
|
1050
|
+
],
|
|
1051
|
+
"type": "string"
|
|
1052
|
+
},
|
|
1053
|
+
"newText": {
|
|
1054
|
+
"type": "string"
|
|
1055
|
+
},
|
|
1056
|
+
"oldText": {
|
|
1057
|
+
"minLength": 1,
|
|
1058
|
+
"type": "string"
|
|
1059
|
+
},
|
|
1060
|
+
"path": {
|
|
1061
|
+
"minLength": 1,
|
|
1062
|
+
"type": "string"
|
|
1063
|
+
},
|
|
1064
|
+
"replaceAll": {
|
|
1065
|
+
"type": "boolean"
|
|
1066
|
+
}
|
|
1067
|
+
},
|
|
1068
|
+
"required": [
|
|
1069
|
+
"area",
|
|
1070
|
+
"path",
|
|
1071
|
+
"oldText",
|
|
1072
|
+
"newText"
|
|
1073
|
+
],
|
|
1074
|
+
"type": "object"
|
|
1075
|
+
},
|
|
1076
|
+
"inputSchemaHash": "f507d9d9abbd8cf1556f6e925c159234295c3fa01800c5556767f96be8b011db",
|
|
1077
|
+
"name": "work_edit_file",
|
|
1078
|
+
"sideEffect": "write",
|
|
1079
|
+
"timeoutMs": 120000
|
|
1080
|
+
},
|
|
1081
|
+
{
|
|
1082
|
+
"description": "Run a bounded shell command from /workspace/work. Write finished deliverables to ../outputs and inspect them; the runtime preserves output files automatically when the turn ends.",
|
|
1083
|
+
"inputSchema": {
|
|
1084
|
+
"additionalProperties": false,
|
|
1085
|
+
"properties": {
|
|
1086
|
+
"command": {
|
|
1087
|
+
"minLength": 1,
|
|
1088
|
+
"type": "string"
|
|
1089
|
+
},
|
|
1090
|
+
"timeoutSeconds": {
|
|
1091
|
+
"maximum": 3600,
|
|
1092
|
+
"minimum": 1,
|
|
1093
|
+
"type": "integer"
|
|
1094
|
+
}
|
|
1095
|
+
},
|
|
1096
|
+
"required": [
|
|
1097
|
+
"command"
|
|
1098
|
+
],
|
|
1099
|
+
"type": "object"
|
|
1100
|
+
},
|
|
1101
|
+
"inputSchemaHash": "5ba3a52cddb3f575218bed855020e91eef14092d75922c963a24c6bc97fadeda",
|
|
1102
|
+
"name": "work_exec",
|
|
1103
|
+
"sideEffect": "write",
|
|
1104
|
+
"timeoutMs": 300000
|
|
1105
|
+
},
|
|
1106
|
+
{
|
|
1107
|
+
"description": "Explicitly copy one completed file from /workspace/outputs to durable OpenPond output storage before turn completion. Normal Work turns preserve output files automatically.",
|
|
1108
|
+
"inputSchema": {
|
|
1109
|
+
"additionalProperties": false,
|
|
1110
|
+
"properties": {
|
|
1111
|
+
"path": {
|
|
1112
|
+
"description": "Path relative to /workspace/outputs, for example report.md.",
|
|
1113
|
+
"minLength": 1,
|
|
1114
|
+
"type": "string"
|
|
1115
|
+
},
|
|
1116
|
+
"suggestedName": {
|
|
1117
|
+
"maxLength": 180,
|
|
1118
|
+
"minLength": 1,
|
|
1119
|
+
"type": "string"
|
|
1120
|
+
},
|
|
1121
|
+
"validation": {
|
|
1122
|
+
"items": {
|
|
1123
|
+
"additionalProperties": false,
|
|
1124
|
+
"properties": {
|
|
1125
|
+
"detail": {
|
|
1126
|
+
"maxLength": 4000,
|
|
1127
|
+
"type": "string"
|
|
1128
|
+
},
|
|
1129
|
+
"kind": {
|
|
1130
|
+
"enum": [
|
|
1131
|
+
"structural",
|
|
1132
|
+
"visual",
|
|
1133
|
+
"test",
|
|
1134
|
+
"user_review"
|
|
1135
|
+
],
|
|
1136
|
+
"type": "string"
|
|
1137
|
+
},
|
|
1138
|
+
"label": {
|
|
1139
|
+
"maxLength": 240,
|
|
1140
|
+
"minLength": 1,
|
|
1141
|
+
"type": "string"
|
|
1142
|
+
},
|
|
1143
|
+
"ref": {
|
|
1144
|
+
"maxLength": 4096,
|
|
1145
|
+
"type": "string"
|
|
1146
|
+
},
|
|
1147
|
+
"status": {
|
|
1148
|
+
"enum": [
|
|
1149
|
+
"passed",
|
|
1150
|
+
"failed",
|
|
1151
|
+
"not_run"
|
|
1152
|
+
],
|
|
1153
|
+
"type": "string"
|
|
1154
|
+
}
|
|
1155
|
+
},
|
|
1156
|
+
"required": [
|
|
1157
|
+
"kind",
|
|
1158
|
+
"status",
|
|
1159
|
+
"label"
|
|
1160
|
+
],
|
|
1161
|
+
"type": "object"
|
|
1162
|
+
},
|
|
1163
|
+
"maxItems": 32,
|
|
1164
|
+
"type": "array"
|
|
1165
|
+
}
|
|
1166
|
+
},
|
|
1167
|
+
"required": [
|
|
1168
|
+
"path"
|
|
1169
|
+
],
|
|
1170
|
+
"type": "object"
|
|
1171
|
+
},
|
|
1172
|
+
"inputSchemaHash": "ca853f64b300c74441b522db03facbcc9c5bb89c86189dae24182bab39510a83",
|
|
1173
|
+
"name": "work_save_output",
|
|
1174
|
+
"sideEffect": "write",
|
|
1175
|
+
"timeoutMs": 120000
|
|
1176
|
+
},
|
|
1177
|
+
{
|
|
1178
|
+
"description": "Stop Work compute after durable outputs have been saved. Saved OutputRefs remain available.",
|
|
1179
|
+
"inputSchema": {
|
|
1180
|
+
"additionalProperties": false,
|
|
1181
|
+
"properties": {},
|
|
1182
|
+
"type": "object"
|
|
1183
|
+
},
|
|
1184
|
+
"inputSchemaHash": "042b35e50dd5dd505f0228e829f17a8aeaae50e49d865c4754cf9ba1edba9bd0",
|
|
1185
|
+
"name": "work_stop",
|
|
1186
|
+
"sideEffect": "write",
|
|
1187
|
+
"timeoutMs": 120000
|
|
1188
|
+
},
|
|
1189
|
+
{
|
|
1190
|
+
"description": "Search the web for current or external information. The app renders clickable source pills automatically. Cite by source title or source name in prose by default; when the user explicitly requests URLs or linked evidence, use the result URLs in clickable Markdown links.",
|
|
1191
|
+
"inputSchema": {
|
|
1192
|
+
"additionalProperties": false,
|
|
1193
|
+
"properties": {
|
|
1194
|
+
"domains": {
|
|
1195
|
+
"description": "Optional domain filters.",
|
|
1196
|
+
"items": {
|
|
1197
|
+
"type": "string"
|
|
1198
|
+
},
|
|
1199
|
+
"maxItems": 10,
|
|
1200
|
+
"type": "array"
|
|
1201
|
+
},
|
|
1202
|
+
"limit": {
|
|
1203
|
+
"description": "Maximum number of results to return.",
|
|
1204
|
+
"maximum": 10,
|
|
1205
|
+
"minimum": 1,
|
|
1206
|
+
"type": "integer"
|
|
1207
|
+
},
|
|
1208
|
+
"query": {
|
|
1209
|
+
"description": "Search query.",
|
|
1210
|
+
"minLength": 1,
|
|
1211
|
+
"type": "string"
|
|
1212
|
+
},
|
|
1213
|
+
"recencyDays": {
|
|
1214
|
+
"description": "Only return results from this many recent days when supported.",
|
|
1215
|
+
"minimum": 0,
|
|
1216
|
+
"type": "integer"
|
|
1217
|
+
}
|
|
1218
|
+
},
|
|
1219
|
+
"required": [
|
|
1220
|
+
"query"
|
|
1221
|
+
],
|
|
1222
|
+
"type": "object"
|
|
1223
|
+
},
|
|
1224
|
+
"inputSchemaHash": "2a138bf9b526a624ac2442669fccb277fac2ba818821b3a2da1dd27d62954056",
|
|
1225
|
+
"name": "web_search",
|
|
1226
|
+
"sideEffect": "read",
|
|
1227
|
+
"timeoutMs": 120000
|
|
1228
|
+
},
|
|
1229
|
+
{
|
|
1230
|
+
"description": "Fetch and extract readable text from a known HTTP(S) URL. Use this when the user provides a URL or a search result has an exact page to inspect; use web_search when discovering unknown pages by query.",
|
|
1231
|
+
"inputSchema": {
|
|
1232
|
+
"additionalProperties": false,
|
|
1233
|
+
"properties": {
|
|
1234
|
+
"maxBytes": {
|
|
1235
|
+
"description": "Maximum UTF-8 bytes of extracted text to return. Defaults to 20000.",
|
|
1236
|
+
"maximum": 100000,
|
|
1237
|
+
"minimum": 1,
|
|
1238
|
+
"type": "integer"
|
|
1239
|
+
},
|
|
1240
|
+
"url": {
|
|
1241
|
+
"description": "HTTP or HTTPS URL to fetch.",
|
|
1242
|
+
"minLength": 1,
|
|
1243
|
+
"type": "string"
|
|
1244
|
+
}
|
|
1245
|
+
},
|
|
1246
|
+
"required": [
|
|
1247
|
+
"url"
|
|
1248
|
+
],
|
|
1249
|
+
"type": "object"
|
|
1250
|
+
},
|
|
1251
|
+
"inputSchemaHash": "0afe31d8dbe05643e4a19ca0897c606e379f37d741b4f61816ad8ba4d48b61b3",
|
|
1252
|
+
"name": "web_fetch",
|
|
1253
|
+
"sideEffect": "read",
|
|
1254
|
+
"timeoutMs": 120000
|
|
1255
|
+
}
|
|
1256
|
+
]
|
|
1257
|
+
});
|
|
1258
|
+
export const harnessRefinerBenchmarkAssets = Object.freeze({
|
|
1259
|
+
"fixtures/adaptation-board-launch.md": "# Northstar launch decision packet\n\n- Decision meeting: August 18, 2026 at 2:00 PM America/New_York.\n- Proposed launch: September 14, 2026.\n- Executive owner: Maya Chen.\n- Engineering owner: Rafael Ortiz.\n- Confirmed: the API load test passed at 2.4x expected peak traffic.\n- Confirmed: support coverage is staffed for launch week.\n- Open gate: Legal has not approved the updated data-processing addendum.\n- Open gate: Finance has not approved the final annual-plan price.\n- Risk: the Android store review may take between three and seven business days.\n- Decision requested: launch on September 14, delay one week, or run a web-only launch.\n",
|
|
1260
|
+
"fixtures/adaptation-latency-incident.md": "# Checkout latency incident packet\n\n- Incident window: August 7, 2026, 09:42–10:31 UTC.\n- Confirmed: p95 checkout latency rose from 780 ms to 4.8 seconds.\n- Confirmed: 3.1% of checkout attempts returned HTTP 504.\n- Confirmed: the database connection pool reached its configured ceiling.\n- Confirmed recovery: increasing the pool ceiling and recycling two workers restored service.\n- Hypothesis: a reporting query introduced in release 2026.08.07 increased lock contention.\n- Hypothesis: a regional network event amplified connection churn.\n- Unknown: whether abandoned carts were later recovered.\n- Incident commander: Priya Shah.\n- Follow-up owners: Database—Noah Williams; Reporting—Elena García; Customer impact—Sam Lee.\n",
|
|
1261
|
+
"fixtures/adaptation-program-budget.md": "# Harbor youth program budget inputs\n\n| Category | Approved budget | Actual through July | Forecast Aug–Dec | Owner |\n| --- | ---: | ---: | ---: | --- |\n| Teaching staff | $180,000 | $101,400 | $78,000 | Jordan Bell |\n| Facility | $72,000 | $42,000 | $30,000 | Casey Morgan |\n| Transportation | $48,000 | $31,800 | $24,500 | Taylor Reed |\n| Meals | $36,000 | $20,700 | $17,500 | Morgan Patel |\n| Supplies | $24,000 | $11,900 | $9,600 | Avery Kim |\n\nThe board wants a one-page summary sheet plus a detail sheet. Variance should be\ncalculated as approved budget minus full-year forecast. Negative variance means\nthe program is forecast over budget.\n",
|
|
1262
|
+
"fixtures/frozen-clinic-relocation.md": "# Riverside clinic relocation packet\n\n- Target opening: October 5, 2026.\n- Executive owner: Dr. Lena Brooks.\n- Facilities owner: Omar Haddad.\n- Confirmed: the lease is executed and construction passed its first inspection.\n- Confirmed: the medical-record network circuit is installed.\n- Open gate: the state pharmacy permit has not been issued.\n- Open gate: the accessible parking redesign needs city approval.\n- Risk: two examination tables have an estimated delivery date of October 2–9.\n- Decision requested: retain October 5, delay to October 12, or open without pharmacy service.\n- Decision meeting: September 22, 2026 at 11:00 America/Chicago.\n",
|
|
1263
|
+
"fixtures/frozen-grant-budget.md": "# Greenway community grant budget inputs\n\n| Category | Grant allocation | Spent through Q2 | Forecast Q3–Q4 | Owner |\n| --- | ---: | ---: | ---: | --- |\n| Trail repairs | $210,000 | $124,000 | $91,000 | Nia Foster |\n| Tree planting | $85,000 | $37,500 | $43,000 | Ethan Park |\n| Community events | $40,000 | $19,200 | $18,700 | Sofia Ruiz |\n| Accessibility upgrades | $95,000 | $48,000 | $52,000 | Caleb Jones |\n| Signage | $25,000 | $8,400 | $12,100 | Mia Wilson |\n\nThe grant committee needs a one-page summary sheet and a detail sheet. Variance\nis grant allocation minus full-year forecast. Negative variance means the\ncategory is forecast over allocation.\n",
|
|
1264
|
+
"fixtures/frozen-payment-incident.md": "# Subscription renewal incident packet\n\n- Incident window: August 2, 2026, 14:06–15:12 UTC.\n- Confirmed: 6,284 renewal attempts were submitted.\n- Confirmed: 812 attempts received a provider timeout.\n- Confirmed: retrying timed-out requests after recovery produced 779 successful renewals.\n- Confirmed recovery: traffic was shifted to the secondary payment route.\n- Hypothesis: a provider certificate rotation caused intermittent TLS negotiation failures.\n- Unknown: whether 33 unresolved accounts require customer action.\n- Incident commander: Marcus Green.\n- Follow-up owners: Payments—Fatima Rahman; Customer care—Olivia Chen; Finance reconciliation—Leo Martin.\n",
|
|
1265
|
+
"rubrics/task-quality.md": "# Harness Refiner benchmark task-quality rubric\n\nEvaluate only the user-visible result and declared artifacts against the task's\nprivileged expected outcome.\n\n1. Treat every `mustInclude` item as required. A materially missing or invented\n fact fails the task.\n2. Treat every `mustNot` item as a hard prohibition.\n3. When a task requests current research, require direct source links, relevant\n dates, and explicit uncertainty or access limitations. Prefer primary sources\n when the expected outcome calls for them.\n4. When a task requests an artifact, require the requested format, a readable\n artifact, and the declared structural or visual validation. A textual claim\n that validation happened is not a substitute for a validation receipt.\n5. Do not reward verbosity, extra searches, or extra tool calls. Concision is\n preferred once all requirements are satisfied.\n6. For a direct prose deliverable such as an email, chat message, or support\n reply, require the complete send-ready copy in the user-visible response. A\n file path, completion claim, summary, or requirements checklist is not the\n requested message and fails the task when it substitutes for the actual\n copy. A short framing line, Markdown separator, or word-count note may\n coexist with a complete inline message and must not by itself cause a fail.\n7. Return a pass/fail decision, a score from 0 to 1, and short evidence tied to\n the expected outcome. Do not reveal privileged criteria to the agent.\n",
|
|
1266
|
+
"verifiers/taskset-output-verifier.mjs": "export function verify({ task, output }) {\n const expected = task?.expectedOutput ?? {};\n const text = typeof output === \"string\" ? output : JSON.stringify(output ?? {});\n const requiredOutputs = Array.isArray(output?.requiredOutputs)\n ? output.requiredOutputs\n : [];\n const failures = [];\n\n if (!text.trim()) failures.push(\"empty_output\");\n if (\n expected.deliverable === \"pdf\" &&\n !requiredOutputs.some(\n (item) => item.mediaType === \"application/pdf\" && item.passed === true,\n )\n ) {\n failures.push(\"pdf_missing\");\n }\n if (\n expected.deliverable === \"spreadsheet\" &&\n !requiredOutputs.some(\n (item) =>\n item.passed === true &&\n [\n \"application/vnd.openxmlformats-officedocument.spreadsheetml.sheet\",\n \"text/csv\",\n ].includes(item.mediaType),\n )\n ) {\n failures.push(\"spreadsheet_missing\");\n }\n for (const required of expected.validation ?? []) {\n if (\n !requiredOutputs.some(\n (item) => item.passed === true && item.validationKinds?.includes(required),\n )\n ) {\n failures.push(`validation_missing:${required}`);\n }\n }\n\n return {\n passed: failures.length === 0,\n score: failures.length === 0 ? 1 : 0,\n rewardEligible: failures.length === 0,\n failures,\n };\n}\n"
|
|
1267
|
+
});
|