@danceiny/gotry 0.0.1-rc.16 → 0.0.1-rc.17

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (229) hide show
  1. package/README.md +47 -13
  2. package/README.zh-CN.md +20 -10
  3. package/bin/gotry-booking-copilot.js +53 -0
  4. package/bin/gotry-bootstrap.js +111 -51
  5. package/bin/gotry-inner.js +442 -60
  6. package/bin/gotry-runtime-resolution.d.ts +27 -0
  7. package/bin/gotry-runtime-resolution.js +50 -0
  8. package/bin/gotry.js +1 -1
  9. package/cordis.gotry-patch.yml +5 -1
  10. package/dist/capabilities/agent-reach-deep.js +1 -1
  11. package/dist/capabilities/agent-reach.js +1 -1
  12. package/dist/capabilities/anything.js +1 -1
  13. package/dist/capabilities/artifacts.js +1 -1
  14. package/dist/capabilities/effect.js +1 -1
  15. package/dist/capabilities/fact-log.js +1 -1
  16. package/dist/capabilities/flyai.js +20 -6
  17. package/dist/capabilities/hbcli.js +1 -1
  18. package/dist/capabilities/incident-log.js +1 -1
  19. package/dist/capabilities/model-override.js +18 -0
  20. package/dist/capabilities/opensky.js +1 -1
  21. package/dist/capabilities/resilience.js +1 -1
  22. package/dist/capabilities/session/action-cache.js +1 -1
  23. package/dist/capabilities/session/adapters/ctrip-flight.js +1 -1
  24. package/dist/capabilities/session/adapters/meituan-local.js +1 -1
  25. package/dist/capabilities/session/benchmark.js +1 -1
  26. package/dist/capabilities/session/extension-bridge.js +19 -6
  27. package/dist/capabilities/session/extension-channel.js +3 -3
  28. package/dist/capabilities/session/extension-distribution.js +234 -0
  29. package/dist/capabilities/session/extract.js +1 -1
  30. package/dist/capabilities/session/golden-score.js +92 -0
  31. package/dist/capabilities/session/health-watch.js +1 -1
  32. package/dist/capabilities/session/read-guard.js +1 -1
  33. package/dist/capabilities/session/static-flight-golden.js +137 -0
  34. package/dist/capabilities/session/transport.js +1 -1
  35. package/dist/capabilities/session/wizard.js +21 -251
  36. package/dist/capabilities/session-consent.js +1 -1
  37. package/dist/capabilities/session-login.js +11 -5
  38. package/dist/capabilities/session-search.js +46 -4
  39. package/dist/capabilities/weather.js +168 -46
  40. package/dist/data/session-golden-20.json +25 -0
  41. package/dist/data/sf-golden-manifest.json +102 -0
  42. package/dist/data/sf-static-routes.json +91 -0
  43. package/dist/scripts/action-cache-tests.js +1 -1
  44. package/dist/scripts/agent-planning-budget-e2e.js +227 -0
  45. package/dist/scripts/agent-planning-budget-tests.js +173 -0
  46. package/dist/scripts/agent-reach-deep-tests.js +1 -1
  47. package/dist/scripts/agent-reach-tests.js +1 -1
  48. package/dist/scripts/agent-reach-wrapper-tests.js +1 -1
  49. package/dist/scripts/anything-tests.js +1 -1
  50. package/dist/scripts/async-collect.js +1 -1
  51. package/dist/scripts/benchmark-environment-bridge-e2e.js +981 -0
  52. package/dist/scripts/benchmark-environment-bridge-tests.js +2544 -0
  53. package/dist/scripts/booking-copilot-availability-ledger-binding-tests.js +335 -0
  54. package/dist/scripts/booking-copilot-availability-policy-v2-tests.js +1355 -0
  55. package/dist/scripts/booking-copilot-bin-proof-tests.js +89 -0
  56. package/dist/scripts/booking-copilot-crossrepo-fixture-server.js +113 -0
  57. package/dist/scripts/booking-copilot-dsh-core-proof-tests.js +356 -0
  58. package/dist/scripts/booking-copilot-dsh-planner-proof-tests.js +557 -0
  59. package/dist/scripts/booking-copilot-dsh-plugin-proof-tests.js +140 -0
  60. package/dist/scripts/booking-copilot-event-sequence-concurrency-proof-tests.js +202 -0
  61. package/dist/scripts/booking-copilot-gap-code-contract-proof-tests.js +102 -0
  62. package/dist/scripts/booking-copilot-operation-ledger-concurrency-proof-tests.js +374 -0
  63. package/dist/scripts/booking-copilot-receipt-ledger-concurrency-proof-tests.js +447 -0
  64. package/dist/scripts/booking-copilot-runtime-proof-tests.js +205 -0
  65. package/dist/scripts/booking-copilot-server-proof-tests.js +317 -0
  66. package/dist/scripts/booking-copilot-startup-proof-tests.js +283 -0
  67. package/dist/scripts/booking-copilot-v2-runtime-proof-tests.js +3935 -0
  68. package/dist/scripts/booking-saga-tests.js +1 -1
  69. package/dist/scripts/booking-surface-contract-proof-tests.js +393 -0
  70. package/dist/scripts/booking-surface-v2-contract-proof-tests.js +1174 -0
  71. package/dist/scripts/bootstrap-tests.js +30 -10
  72. package/dist/scripts/build-changelog.js +1 -1
  73. package/dist/scripts/changelog-tests.js +1 -1
  74. package/dist/scripts/companion-tests.js +1 -1
  75. package/dist/scripts/diff-test.js +1 -1
  76. package/dist/scripts/dsh-runtime-closure-tests.js +225 -0
  77. package/dist/scripts/dsh-runtime-closure.js +150 -0
  78. package/dist/scripts/effect-tests.js +1 -1
  79. package/dist/scripts/engine-run.js +1 -1
  80. package/dist/scripts/engine-tests.js +1 -1
  81. package/dist/scripts/evaluation-cadence-tests.js +347 -0
  82. package/dist/scripts/evaluation-contract-tests.js +574 -0
  83. package/dist/scripts/extension-distribution-cli.js +35 -0
  84. package/dist/scripts/extension-distribution-tests.js +341 -0
  85. package/dist/scripts/extension-tests.js +128 -23
  86. package/dist/scripts/fact-gate-tests.js +1 -1
  87. package/dist/scripts/flyai-tests.js +1 -1
  88. package/dist/scripts/hbcli-e2e-tests.js +1 -1
  89. package/dist/scripts/hbcli-tests.js +1 -1
  90. package/dist/scripts/health-watch-cli.js +1 -1
  91. package/dist/scripts/i18n-tests.js +1 -1
  92. package/dist/scripts/incident-tests.js +1 -1
  93. package/dist/scripts/journey-tests.js +1 -1
  94. package/dist/scripts/ledger-tests.js +1 -1
  95. package/dist/scripts/ledger-workflow-crash.js +1 -1
  96. package/dist/scripts/memory-capture-tests.js +1 -1
  97. package/dist/scripts/memory-decay-tests.js +1 -1
  98. package/dist/scripts/memory-metrics.js +1 -1
  99. package/dist/scripts/memory-value-report.js +1 -1
  100. package/dist/scripts/model-override-e2e.js +176 -0
  101. package/dist/scripts/nightly-evidence-tests.js +1 -1
  102. package/dist/scripts/nightly-evidence.js +1 -1
  103. package/dist/scripts/nudge-digest.js +1 -1
  104. package/dist/scripts/onboarding-tests.js +21 -53
  105. package/dist/scripts/opensky-check.js +1 -1
  106. package/dist/scripts/opensky-tests.js +1 -1
  107. package/dist/scripts/pnpm-dsh-closure-proof.js +20 -0
  108. package/dist/scripts/price-drift-tests.js +1 -1
  109. package/dist/scripts/price-drift-watch.js +1 -1
  110. package/dist/scripts/probe-poi-tests.js +1 -1
  111. package/dist/scripts/product-metrics.js +1 -1
  112. package/dist/scripts/publish-preverify.js +40 -4
  113. package/dist/scripts/realtime-pricing-tests.js +1 -1
  114. package/dist/scripts/replay-async.js +1 -1
  115. package/dist/scripts/replay-real.js +1 -1
  116. package/dist/scripts/replay.js +1 -1
  117. package/dist/scripts/session-attach-diagnose.js +1 -1
  118. package/dist/scripts/session-attach-poc.js +1 -1
  119. package/dist/scripts/session-benchmark.js +1 -1
  120. package/dist/scripts/session-extract-tests.js +1 -1
  121. package/dist/scripts/session-login.js +1 -1
  122. package/dist/scripts/session-tests.js +70 -20
  123. package/dist/scripts/sf-live-benchmark.js +338 -0
  124. package/dist/scripts/sf-live-cli-tests.js +21 -0
  125. package/dist/scripts/sf-soft-score-tests.js +108 -0
  126. package/dist/scripts/sf-summary.js +93 -0
  127. package/dist/scripts/skeleton-check.js +1 -1
  128. package/dist/scripts/skeleton-integration-test.js +1 -1
  129. package/dist/scripts/skills-contract-tests.js +1 -1
  130. package/dist/scripts/smoke-session-gate-tests.js +29 -0
  131. package/dist/scripts/smoke.js +79 -33
  132. package/dist/scripts/state-cli-tests.js +1 -1
  133. package/dist/scripts/state-cli.js +1 -1
  134. package/dist/scripts/static-golden-tests.js +299 -0
  135. package/dist/scripts/time-eval-tests.js +1 -1
  136. package/dist/scripts/travel-timeline-tests.js +1 -1
  137. package/dist/scripts/unified-tests.js +1 -1
  138. package/dist/scripts/weather-tests.js +694 -44
  139. package/dist/scripts/z3-race-tests.js +1 -1
  140. package/dist/src/artifact-gate.js +1 -1
  141. package/dist/src/benchmark-agent-conformance.js +370 -0
  142. package/dist/src/benchmark-environment-bridge.js +384 -0
  143. package/dist/src/benchmark-headless-child-diagnostics.js +173 -0
  144. package/dist/src/benchmark-tool-isolation.js +124 -0
  145. package/dist/src/bookable-facts.js +1 -1
  146. package/dist/src/booking-saga.js +1 -1
  147. package/dist/src/booking-surface/availability-policy-v2.js +830 -0
  148. package/dist/src/booking-surface/canonical-schema.js +113 -0
  149. package/dist/src/booking-surface/contracts-v2.js +89 -0
  150. package/dist/src/booking-surface/contracts.js +46 -0
  151. package/dist/src/booking-surface/dsh-planner.js +453 -0
  152. package/dist/src/booking-surface/dsh-plugin.js +93 -0
  153. package/dist/src/booking-surface/error-codes.js +94 -0
  154. package/dist/src/booking-surface/index.js +15 -0
  155. package/dist/src/booking-surface/profile.js +68 -0
  156. package/dist/src/booking-surface/runtime-v2.js +1771 -0
  157. package/dist/src/booking-surface/runtime.js +351 -0
  158. package/dist/src/booking-surface/server-v2.js +334 -0
  159. package/dist/src/booking-surface/server.js +302 -0
  160. package/dist/src/booking-surface/startup.js +159 -0
  161. package/dist/src/booking-surface/validation-v2.js +319 -0
  162. package/dist/src/booking-surface/validation.js +809 -0
  163. package/dist/src/bridge.js +1 -1
  164. package/dist/src/companions.js +1 -1
  165. package/dist/src/contracts.js +1 -1
  166. package/dist/src/dsh-llm.js +1 -1
  167. package/dist/src/engine.js +1 -1
  168. package/dist/src/evaluation-cadence.js +234 -0
  169. package/dist/src/evaluation-contracts.js +906 -0
  170. package/dist/src/i18n.js +1 -1
  171. package/dist/src/index.js +50 -19
  172. package/dist/src/journey.js +1 -1
  173. package/dist/src/loop.js +1 -1
  174. package/dist/src/memory-capture.js +1 -1
  175. package/dist/src/memory-decay.js +1 -1
  176. package/dist/src/memory-utility.js +1 -1
  177. package/dist/src/mock-llm.js +1 -1
  178. package/dist/src/model.js +1 -1
  179. package/dist/src/realtime-pricing.js +1 -1
  180. package/dist/src/slot-spec.js +1 -1
  181. package/dist/src/state-ledger.js +2 -1
  182. package/dist/src/time-anchor.js +1 -1
  183. package/dist/src/tool-budget.js +136 -0
  184. package/dist/src/tool-packet.js +1 -1
  185. package/dist/src/travel-slots.js +1 -1
  186. package/dist/src/travel-timeline.js +1 -1
  187. package/dist/src/unified.js +1 -1
  188. package/dist/src/wish-pool.js +1 -1
  189. package/dist/src/z3-shared.js +1 -1
  190. package/extension/README.md +31 -7
  191. package/package.json +286 -11
  192. package/schemas/booking.surface.v1.schema.json +927 -0
  193. package/schemas/booking.surface.v2.schema.json +61 -0
  194. package/ts/capabilities/flyai.ts +16 -3
  195. package/ts/capabilities/session/extension-bridge.ts +37 -14
  196. package/ts/capabilities/session/extension-channel.ts +6 -3
  197. package/ts/capabilities/session/extension-distribution.ts +264 -0
  198. package/ts/capabilities/session/golden-score.ts +139 -0
  199. package/ts/capabilities/session/health-watch.ts +1 -1
  200. package/ts/capabilities/session/static-flight-golden.ts +209 -0
  201. package/ts/capabilities/session/wizard.ts +34 -176
  202. package/ts/capabilities/session-login.ts +15 -5
  203. package/ts/capabilities/session-search.ts +40 -3
  204. package/ts/capabilities/weather.ts +141 -52
  205. package/ts/package.json +3 -3
  206. package/ts/src/benchmark-agent-conformance.ts +448 -0
  207. package/ts/src/benchmark-environment-bridge.ts +348 -0
  208. package/ts/src/benchmark-headless-child-diagnostics.ts +184 -0
  209. package/ts/src/benchmark-tool-isolation.ts +166 -0
  210. package/ts/src/booking-surface/availability-policy-v2.ts +523 -0
  211. package/ts/src/booking-surface/canonical-schema.js +113 -0
  212. package/ts/src/booking-surface/contracts-v2.ts +118 -0
  213. package/ts/src/booking-surface/contracts.ts +380 -0
  214. package/ts/src/booking-surface/dsh-planner.ts +452 -0
  215. package/ts/src/booking-surface/dsh-plugin.js +93 -0
  216. package/ts/src/booking-surface/error-codes.ts +101 -0
  217. package/ts/src/booking-surface/index.ts +12 -0
  218. package/ts/src/booking-surface/profile.ts +42 -0
  219. package/ts/src/booking-surface/runtime-v2.ts +1466 -0
  220. package/ts/src/booking-surface/runtime.ts +483 -0
  221. package/ts/src/booking-surface/server-v2.ts +247 -0
  222. package/ts/src/booking-surface/server.ts +324 -0
  223. package/ts/src/booking-surface/startup.ts +196 -0
  224. package/ts/src/booking-surface/validation-v2.ts +205 -0
  225. package/ts/src/booking-surface/validation.ts +453 -0
  226. package/ts/src/index.ts +64 -11
  227. package/ts/src/state-ledger.ts +1 -0
  228. package/ts/src/tool-budget.ts +165 -0
  229. package/dist/scripts/wizard-bootstrap.js +0 -32
@@ -0,0 +1,906 @@
1
+ import { createHash } from 'node:crypto';
2
+ export const BENCHMARK_IDS = [
3
+ 'trek',
4
+ 'travelplanner',
5
+ 'chinatravel',
6
+ 'travelbench',
7
+ 'tau2',
8
+ 'locomo',
9
+ 'bfcl'
10
+ ];
11
+ export const FAILURE_CATEGORIES = [
12
+ 'schema_or_format',
13
+ 'grounding_or_provenance',
14
+ 'constraint_or_feasibility',
15
+ 'time_or_location_continuity',
16
+ 'cost_or_cardinality',
17
+ 'tool_selection_or_arguments',
18
+ 'policy_or_write_safety',
19
+ 'preference_elicitation',
20
+ 'memory_retrieval_or_temporal_reasoning',
21
+ 'reliability_cost_or_latency'
22
+ ];
23
+ function obj(value, label) {
24
+ if (!value || typeof value !== 'object' || Array.isArray(value)) throw new Error(`${label} must be an object`);
25
+ return value;
26
+ }
27
+ function exact(value, label, keys) {
28
+ const unknown = Object.keys(value).filter((key)=>!keys.includes(key));
29
+ const missing = keys.filter((key)=>!(key in value));
30
+ if (unknown.length) throw new Error(`${label} contains undeclared fields: ${unknown.join(',')}`);
31
+ if (missing.length) throw new Error(`${label} missing fields: ${missing.join(',')}`);
32
+ }
33
+ function text(value, label) {
34
+ if (typeof value !== 'string' || value.length === 0) throw new Error(`${label} must be non-empty text`);
35
+ return value;
36
+ }
37
+ function id(value, label) {
38
+ const out = text(value, label);
39
+ if (!/^[a-z0-9][a-z0-9:._-]+$/.test(out)) throw new Error(`${label} must be a stable identifier`);
40
+ return out;
41
+ }
42
+ function lit(value, label, values) {
43
+ const out = text(value, label);
44
+ if (!values.includes(out)) throw new Error(`${label} invalid literal`);
45
+ return out;
46
+ }
47
+ function boolean(value, label) {
48
+ if (typeof value !== 'boolean') throw new Error(`${label} must be boolean`);
49
+ return value;
50
+ }
51
+ function finite(value, label, min = 0) {
52
+ if (typeof value !== 'number' || !Number.isFinite(value) || value < min) throw new Error(`${label} must be a finite number >= ${min}`);
53
+ return value;
54
+ }
55
+ function integer(value, label, min = 0) {
56
+ const out = finite(value, label, min);
57
+ if (!Number.isInteger(out)) throw new Error(`${label} must be an integer`);
58
+ return out;
59
+ }
60
+ function strings(value, label, empty = false) {
61
+ if (!Array.isArray(value) || !empty && value.length === 0) throw new Error(`${label} must be an array`);
62
+ const out = value.map((item, index)=>text(item, `${label}[${index}]`));
63
+ if (new Set(out).size !== out.length) throw new Error(`${label} must be unique`);
64
+ return out;
65
+ }
66
+ function digest(value, label) {
67
+ const out = text(value, label);
68
+ if (!/^[0-9a-f]{64}$/.test(out)) throw new Error(`${label} must be lowercase SHA-256`);
69
+ return out;
70
+ }
71
+ function commit(value, label) {
72
+ const out = text(value, label);
73
+ if (!/^[0-9a-f]{40}$/.test(out)) throw new Error(`${label} must be a lowercase 40-character Git commit`);
74
+ return out;
75
+ }
76
+ function url(value, label) {
77
+ const out = text(value, label);
78
+ let parsed;
79
+ try {
80
+ parsed = new URL(out);
81
+ } catch {
82
+ throw new Error(`${label} must be a valid URL`);
83
+ }
84
+ ;
85
+ if (parsed.protocol !== 'https:') throw new Error(`${label} must use https`);
86
+ if (parsed.username || parsed.password) throw new Error(`${label} must not contain credentials`);
87
+ return out;
88
+ }
89
+ function instant(value, label) {
90
+ const out = text(value, label);
91
+ const date = new Date(out);
92
+ if (!Number.isFinite(date.valueOf()) || date.toISOString() !== out) throw new Error(`${label} must be canonical ISO UTC with milliseconds`);
93
+ return out;
94
+ }
95
+ function zone(value, label) {
96
+ const out = text(value, label);
97
+ if (out === 'UTC') return out;
98
+ try {
99
+ new Intl.DateTimeFormat('en', {
100
+ timeZone: out
101
+ }).format(new Date(0));
102
+ } catch {
103
+ throw new Error(`${label} must be UTC or an IANA timezone`);
104
+ }
105
+ ;
106
+ if (!out.includes('/')) throw new Error(`${label} must be UTC or an IANA timezone`);
107
+ return out;
108
+ }
109
+ function revision(value, label) {
110
+ const root = obj(value, label);
111
+ exact(root, label, [
112
+ 'kind',
113
+ 'value'
114
+ ]);
115
+ const kind = lit(root.kind, `${label}.kind`, [
116
+ 'git_commit',
117
+ 'content_sha256',
118
+ 'not_separately_declared'
119
+ ]);
120
+ if (kind === 'not_separately_declared') {
121
+ if (root.value !== null) throw new Error(`${label}.value must be null`);
122
+ return {
123
+ kind,
124
+ value: null
125
+ };
126
+ }
127
+ ;
128
+ return {
129
+ kind,
130
+ value: kind === 'git_commit' ? commit(root.value, `${label}.value`) : digest(root.value, `${label}.value`)
131
+ };
132
+ }
133
+ function pin(value, label) {
134
+ const root = obj(value, label);
135
+ exact(root, label, [
136
+ 'url',
137
+ 'revision',
138
+ 'source_scope'
139
+ ]);
140
+ return {
141
+ url: url(root.url, `${label}.url`),
142
+ revision: revision(root.revision, `${label}.revision`),
143
+ source_scope: text(root.source_scope, `${label}.source_scope`)
144
+ };
145
+ }
146
+ function licenseDetermination(value, label) {
147
+ const root = obj(value, label);
148
+ exact(root, label, [
149
+ 'value',
150
+ 'determination',
151
+ 'source_url'
152
+ ]);
153
+ const determination = lit(root.determination, `${label}.determination`, [
154
+ 'declared',
155
+ 'not_separately_declared'
156
+ ]);
157
+ const resolved = text(root.value, `${label}.value`);
158
+ if (determination === 'not_separately_declared' && resolved !== 'not_separately_declared') throw new Error(`${label}.value must be not_separately_declared`);
159
+ return {
160
+ value: resolved,
161
+ determination,
162
+ source_url: url(root.source_url, `${label}.source_url`)
163
+ };
164
+ }
165
+ export function stableEvaluationJson(value) {
166
+ const sort = (item)=>Array.isArray(item) ? item.map(sort) : item && typeof item === 'object' ? Object.fromEntries(Object.entries(item).sort(([a], [b])=>a.localeCompare(b)).map(([key, child])=>[
167
+ key,
168
+ sort(child)
169
+ ])) : item;
170
+ return `${JSON.stringify(sort(value), null, 2)}\n`;
171
+ }
172
+ export function evaluationFingerprint(value) {
173
+ return createHash('sha256').update(stableEvaluationJson(value)).digest('hex');
174
+ }
175
+ const sensitiveKeys = new Set([
176
+ 'apikey',
177
+ 'accesstoken',
178
+ 'refreshtoken',
179
+ 'authorization',
180
+ 'password',
181
+ 'clientsecret',
182
+ 'credentials',
183
+ 'cookie',
184
+ 'rawprompt',
185
+ 'goldanswer',
186
+ 'oraclepayload',
187
+ 'credential',
188
+ 'rawanswer',
189
+ 'privatepayload',
190
+ 'secretpayload',
191
+ 'trajectorypayload'
192
+ ]);
193
+ export function assertPublicArtifactSafe(value, label = 'artifact') {
194
+ const walk = (item, path)=>{
195
+ if (Array.isArray(item)) return item.forEach((child, index)=>walk(child, `${path}[${index}]`));
196
+ if (typeof item === 'string') {
197
+ const withoutHttpsUrls = item.replace(/https:\/\/[^\s]+/g, '');
198
+ const highConfidenceSecret = /Bearer\s+[A-Za-z0-9._~-]{24,}|\bsk-[A-Za-z0-9_-]{20,}|\bgh[pousr]_[A-Za-z0-9_]{20,}|\bAKIA[0-9A-Z]{16}\b/;
199
+ if (/(?:^|[^A-Za-z0-9])(?:\/|~\/)|[A-Za-z]:[\\/]|file:\/\//.test(withoutHttpsUrls) || highConfidenceSecret.test(item)) throw new Error(`${path} contains an absolute path or secret`);
200
+ return;
201
+ }
202
+ if (!item || typeof item !== 'object') return;
203
+ for (const [key, child] of Object.entries(item)){
204
+ if (sensitiveKeys.has(key.replace(/[^a-z0-9]/gi, '').toLowerCase()) && child !== '' && child !== null) throw new Error(`${path}.${key} contains credentials or raw sensitive payload`);
205
+ walk(child, `${path}.${key}`);
206
+ }
207
+ };
208
+ walk(value, label);
209
+ }
210
+ function registryEntry(value, label) {
211
+ const root = obj(value, label);
212
+ exact(root, label, [
213
+ 'schema_version',
214
+ 'benchmark_id',
215
+ 'provenance',
216
+ 'license',
217
+ 'task_scopes',
218
+ 'native_metrics',
219
+ 'source_fence',
220
+ 'countability_default'
221
+ ]);
222
+ const provenance = obj(root.provenance, `${label}.provenance`);
223
+ exact(provenance, `${label}.provenance`, [
224
+ 'official_entry',
225
+ 'data',
226
+ 'evaluator'
227
+ ]);
228
+ const license = obj(root.license, `${label}.license`);
229
+ exact(license, `${label}.license`, [
230
+ 'upstream_rights',
231
+ 'repo_storage_policy'
232
+ ]);
233
+ const rights = obj(license.upstream_rights, `${label}.license.upstream_rights`);
234
+ exact(rights, `${label}.license.upstream_rights`, [
235
+ 'code',
236
+ 'data',
237
+ 'evaluator'
238
+ ]);
239
+ const metrics = obj(root.native_metrics, `${label}.native_metrics`);
240
+ exact(metrics, `${label}.native_metrics`, [
241
+ 'status',
242
+ 'values'
243
+ ]);
244
+ if (!Array.isArray(metrics.values)) throw new Error(`${label}.native_metrics.values must be an array`);
245
+ const status = lit(metrics.status, `${label}.native_metrics.status`, [
246
+ 'declared',
247
+ 'not_separately_declared'
248
+ ]);
249
+ const values = metrics.values.map((raw, index)=>{
250
+ const metric = obj(raw, `${label}.metric[${index}]`);
251
+ exact(metric, `${label}.metric[${index}]`, [
252
+ 'receipt_key',
253
+ 'upstream_label',
254
+ 'scope',
255
+ 'source_url'
256
+ ]);
257
+ return {
258
+ receipt_key: id(metric.receipt_key, `${label}.metric.receipt_key`),
259
+ upstream_label: text(metric.upstream_label, `${label}.metric.upstream_label`),
260
+ scope: text(metric.scope, `${label}.metric.scope`),
261
+ source_url: url(metric.source_url, `${label}.metric.source_url`)
262
+ };
263
+ });
264
+ if (status === 'declared' && values.length === 0) throw new Error(`${label} declared metrics must be non-empty`);
265
+ if (status === 'not_separately_declared' && values.length !== 0) throw new Error(`${label} undeclared metrics must be empty`);
266
+ const fence = obj(root.source_fence, `${label}.source_fence`);
267
+ exact(fence, `${label}.source_fence`, [
268
+ 'solver_allowed_field_classes',
269
+ 'solver_forbidden_field_classes'
270
+ ]);
271
+ return {
272
+ schema_version: lit(root.schema_version, `${label}.schema_version`, [
273
+ 'gotry_benchmark_registry_entry_v0'
274
+ ]),
275
+ benchmark_id: lit(root.benchmark_id, `${label}.benchmark_id`, BENCHMARK_IDS),
276
+ provenance: {
277
+ official_entry: pin(provenance.official_entry, `${label}.official_entry`),
278
+ data: pin(provenance.data, `${label}.data`),
279
+ evaluator: pin(provenance.evaluator, `${label}.evaluator`)
280
+ },
281
+ license: {
282
+ upstream_rights: {
283
+ code: licenseDetermination(rights.code, `${label}.rights.code`),
284
+ data: licenseDetermination(rights.data, `${label}.rights.data`),
285
+ evaluator: licenseDetermination(rights.evaluator, `${label}.rights.evaluator`)
286
+ },
287
+ repo_storage_policy: lit(license.repo_storage_policy, `${label}.storage`, [
288
+ 'metadata_only_no_upstream_payload'
289
+ ])
290
+ },
291
+ task_scopes: strings(root.task_scopes, `${label}.task_scopes`),
292
+ native_metrics: status === 'declared' ? {
293
+ status,
294
+ values
295
+ } : {
296
+ status,
297
+ values: []
298
+ },
299
+ source_fence: {
300
+ solver_allowed_field_classes: strings(fence.solver_allowed_field_classes, `${label}.allowed`),
301
+ solver_forbidden_field_classes: strings(fence.solver_forbidden_field_classes, `${label}.forbidden`)
302
+ },
303
+ countability_default: lit(root.countability_default, `${label}.countability_default`, [
304
+ 'countable_if_qualified',
305
+ 'diagnostic_only'
306
+ ])
307
+ };
308
+ }
309
+ export function parseBenchmarkRegistry(value) {
310
+ if (!Array.isArray(value) || value.length !== 7) throw new Error('registry must contain exactly seven entries');
311
+ const out = value.map((item, index)=>registryEntry(item, `registry[${index}]`));
312
+ if (out.some((item, index)=>item.benchmark_id !== BENCHMARK_IDS[index])) throw new Error('registry must use canonical unique order');
313
+ return out;
314
+ }
315
+ export function parseEvalCase(value) {
316
+ const root = obj(value, 'case');
317
+ exact(root, 'case', [
318
+ 'schema_version',
319
+ 'case_id',
320
+ 'benchmark_id',
321
+ 'input_ref',
322
+ 'isolation',
323
+ 'clock',
324
+ 'allowed_effects',
325
+ 'forbidden_effects',
326
+ 'budget',
327
+ 'scorer_revision',
328
+ 'public_safety'
329
+ ]);
330
+ const input = obj(root.input_ref, 'case.input_ref');
331
+ exact(input, 'case.input_ref', [
332
+ 'kind',
333
+ 'revision',
334
+ 'digest_sha256'
335
+ ]);
336
+ const isolation = obj(root.isolation, 'case.isolation');
337
+ exact(isolation, 'case.isolation', [
338
+ 'state_root',
339
+ 'network',
340
+ 'writes'
341
+ ]);
342
+ const clock = obj(root.clock, 'case.clock');
343
+ exact(clock, 'case.clock', [
344
+ 'now',
345
+ 'timezone'
346
+ ]);
347
+ const budget = obj(root.budget, 'case.budget');
348
+ exact(budget, 'case.budget', [
349
+ 'max_seconds',
350
+ 'max_cost_usd',
351
+ 'max_tool_calls',
352
+ 'max_turns'
353
+ ]);
354
+ const safety = obj(root.public_safety, 'case.public_safety');
355
+ exact(safety, 'case.public_safety', [
356
+ 'contains_third_party_prompt',
357
+ 'contains_gold',
358
+ 'contains_oracle',
359
+ 'contains_private_data'
360
+ ]);
361
+ for (const [key, flag] of Object.entries(safety))if (boolean(flag, `case.public_safety.${key}`) !== false) throw new Error(`case.public_safety.${key} must be false`);
362
+ const out = {
363
+ schema_version: lit(root.schema_version, 'case.schema_version', [
364
+ 'gotry_eval_case_v0'
365
+ ]),
366
+ case_id: id(root.case_id, 'case.case_id'),
367
+ benchmark_id: lit(root.benchmark_id, 'case.benchmark_id', BENCHMARK_IDS),
368
+ input_ref: {
369
+ kind: lit(input.kind, 'case.input_ref.kind', [
370
+ 'gotry_owned_synthetic',
371
+ 'external_opaque_reference'
372
+ ]),
373
+ revision: revision(input.revision, 'case.input_ref.revision'),
374
+ digest_sha256: digest(input.digest_sha256, 'case.input_ref.digest_sha256')
375
+ },
376
+ isolation: {
377
+ state_root: lit(isolation.state_root, 'case.state_root', [
378
+ 'ephemeral'
379
+ ]),
380
+ network: lit(isolation.network, 'case.network', [
381
+ 'denied',
382
+ 'benchmark_declared'
383
+ ]),
384
+ writes: lit(isolation.writes, 'case.writes', [
385
+ 'forbidden'
386
+ ])
387
+ },
388
+ clock: {
389
+ now: instant(clock.now, 'case.clock.now'),
390
+ timezone: zone(clock.timezone, 'case.clock.timezone')
391
+ },
392
+ allowed_effects: strings(root.allowed_effects, 'case.allowed_effects', true),
393
+ forbidden_effects: strings(root.forbidden_effects, 'case.forbidden_effects'),
394
+ budget: {
395
+ max_seconds: finite(budget.max_seconds, 'case.max_seconds'),
396
+ max_cost_usd: finite(budget.max_cost_usd, 'case.max_cost_usd'),
397
+ max_tool_calls: integer(budget.max_tool_calls, 'case.max_tool_calls'),
398
+ max_turns: integer(budget.max_turns, 'case.max_turns')
399
+ },
400
+ scorer_revision: revision(root.scorer_revision, 'case.scorer_revision'),
401
+ public_safety: {
402
+ contains_third_party_prompt: false,
403
+ contains_gold: false,
404
+ contains_oracle: false,
405
+ contains_private_data: false
406
+ }
407
+ };
408
+ assertPublicArtifactSafe(out, 'case');
409
+ return out;
410
+ }
411
+ function optionalDigest(value, label) {
412
+ return value === null ? null : digest(value, label);
413
+ }
414
+ function numberMap(value, label, integers = false) {
415
+ const root = obj(value, label);
416
+ return Object.fromEntries(Object.entries(root).map(([key, item])=>[
417
+ id(key, `${label}.key`),
418
+ integers ? integer(item, `${label}.${key}`) : finite(item, `${label}.${key}`, Number.NEGATIVE_INFINITY)
419
+ ]));
420
+ }
421
+ export function parseEvalRunReceipt(value) {
422
+ const root = obj(value, 'run');
423
+ exact(root, 'run', [
424
+ 'schema_version',
425
+ 'run_id',
426
+ 'benchmark_id',
427
+ 'case_id',
428
+ 'evidence_kind',
429
+ 'pairing',
430
+ 'status',
431
+ 'started_at',
432
+ 'finished_at',
433
+ 'gotry_sha',
434
+ 'model',
435
+ 'controls',
436
+ 'qualification',
437
+ 'experiment',
438
+ 'native_metrics',
439
+ 'guardrails',
440
+ 'evidence_summary'
441
+ ]);
442
+ const model = obj(root.model, 'run.model');
443
+ exact(model, 'run.model', [
444
+ 'provider',
445
+ 'model'
446
+ ]);
447
+ const controls = obj(root.controls, 'run.controls');
448
+ const controlKeys = [
449
+ 'case_set_sha256',
450
+ 'protocol_control_sha256',
451
+ 'model_parameters_sha256',
452
+ 'scorer_sha256',
453
+ 'tool_snapshot_sha256',
454
+ 'source_fence_sha256',
455
+ 'integrity_sha256',
456
+ 'official_evaluator_sha256'
457
+ ];
458
+ exact(controls, 'run.controls', controlKeys);
459
+ const qualification = obj(root.qualification, 'run.qualification');
460
+ exact(qualification, 'run.qualification', [
461
+ 'official_result',
462
+ 'source_fence_passed',
463
+ 'integrity_passed',
464
+ 'evidence_receipts'
465
+ ]);
466
+ const receipts = obj(qualification.evidence_receipts, 'run.qualification.evidence_receipts');
467
+ exact(receipts, 'run.qualification.evidence_receipts', [
468
+ 'official_evaluator_output_sha256',
469
+ 'source_fence_audit_sha256',
470
+ 'integrity_audit_sha256'
471
+ ]);
472
+ const experiment = obj(root.experiment, 'run.experiment');
473
+ exact(experiment, 'run.experiment', [
474
+ 'changed_variables',
475
+ 'candidate_sha256'
476
+ ]);
477
+ const guard = obj(root.guardrails, 'run.guardrails');
478
+ exact(guard, 'run.guardrails', [
479
+ 'hard_violation_count',
480
+ 'forbidden_leakage_hits',
481
+ 'latency_ms',
482
+ 'cost_usd',
483
+ 'tool_calls',
484
+ 'turns'
485
+ ]);
486
+ const summary = obj(root.evidence_summary, 'run.evidence_summary');
487
+ exact(summary, 'run.evidence_summary', [
488
+ 'artifact_classification',
489
+ 'sha256',
490
+ 'counts',
491
+ 'reason_codes',
492
+ 'fixture_only',
493
+ 'statement'
494
+ ]);
495
+ let pairing = null;
496
+ if (root.pairing !== null) {
497
+ const item = obj(root.pairing, 'run.pairing');
498
+ exact(item, 'run.pairing', [
499
+ 'pair_id',
500
+ 'role',
501
+ 'counterpart_run_id'
502
+ ]);
503
+ pairing = {
504
+ pair_id: id(item.pair_id, 'run.pair_id'),
505
+ role: lit(item.role, 'run.role', [
506
+ 'baseline',
507
+ 'treatment'
508
+ ]),
509
+ counterpart_run_id: id(item.counterpart_run_id, 'run.counterpart')
510
+ };
511
+ }
512
+ const evidence_kind = lit(root.evidence_kind, 'run.evidence_kind', [
513
+ 'synthetic_fixture',
514
+ 'observed_external'
515
+ ]);
516
+ const status = lit(root.status, 'run.status', [
517
+ 'running',
518
+ 'succeeded',
519
+ 'failed',
520
+ 'blocked'
521
+ ]);
522
+ const finished_at = root.finished_at === null ? null : instant(root.finished_at, 'run.finished_at');
523
+ if (status === 'running' !== (finished_at === null)) throw new Error('run.finished_at must be null only while running');
524
+ const out = {
525
+ schema_version: lit(root.schema_version, 'run.schema_version', [
526
+ 'gotry_eval_run_receipt_v0'
527
+ ]),
528
+ run_id: id(root.run_id, 'run.run_id'),
529
+ benchmark_id: lit(root.benchmark_id, 'run.benchmark_id', BENCHMARK_IDS),
530
+ case_id: id(root.case_id, 'run.case_id'),
531
+ evidence_kind,
532
+ pairing,
533
+ status,
534
+ started_at: instant(root.started_at, 'run.started_at'),
535
+ finished_at,
536
+ gotry_sha: commit(root.gotry_sha, 'run.gotry_sha'),
537
+ model: {
538
+ provider: text(model.provider, 'run.provider'),
539
+ model: text(model.model, 'run.model')
540
+ },
541
+ controls: Object.fromEntries(controlKeys.map((key)=>[
542
+ key,
543
+ digest(controls[key], `run.controls.${key}`)
544
+ ])),
545
+ qualification: {
546
+ official_result: boolean(qualification.official_result, 'run.official_result'),
547
+ source_fence_passed: boolean(qualification.source_fence_passed, 'run.source_fence_passed'),
548
+ integrity_passed: boolean(qualification.integrity_passed, 'run.integrity_passed'),
549
+ evidence_receipts: {
550
+ official_evaluator_output_sha256: optionalDigest(receipts.official_evaluator_output_sha256, 'run.evaluator_receipt'),
551
+ source_fence_audit_sha256: optionalDigest(receipts.source_fence_audit_sha256, 'run.fence_receipt'),
552
+ integrity_audit_sha256: optionalDigest(receipts.integrity_audit_sha256, 'run.integrity_receipt')
553
+ }
554
+ },
555
+ experiment: {
556
+ changed_variables: strings(experiment.changed_variables, 'run.changed_variables', true),
557
+ candidate_sha256: digest(experiment.candidate_sha256, 'run.candidate')
558
+ },
559
+ native_metrics: numberMap(root.native_metrics, 'run.native_metrics'),
560
+ guardrails: {
561
+ hard_violation_count: integer(guard.hard_violation_count, 'run.hard'),
562
+ forbidden_leakage_hits: integer(guard.forbidden_leakage_hits, 'run.leakage'),
563
+ latency_ms: finite(guard.latency_ms, 'run.latency'),
564
+ cost_usd: finite(guard.cost_usd, 'run.cost'),
565
+ tool_calls: integer(guard.tool_calls, 'run.tools'),
566
+ turns: integer(guard.turns, 'run.turns')
567
+ },
568
+ evidence_summary: {
569
+ artifact_classification: lit(summary.artifact_classification, 'run.classification', [
570
+ 'public_safe'
571
+ ]),
572
+ sha256: strings(summary.sha256, 'run.sha256', true).map((item, index)=>digest(item, `run.sha256[${index}]`)),
573
+ counts: numberMap(summary.counts, 'run.counts', true),
574
+ reason_codes: strings(summary.reason_codes, 'run.reason_codes', true),
575
+ fixture_only: boolean(summary.fixture_only, 'run.fixture_only'),
576
+ statement: text(summary.statement, 'run.statement')
577
+ }
578
+ };
579
+ if (out.finished_at !== null && new Date(out.finished_at).valueOf() < new Date(out.started_at).valueOf()) throw new Error('run.finished_at must not precede started_at');
580
+ if (evidence_kind === 'synthetic_fixture' && (pairing !== null || out.qualification.official_result || out.qualification.source_fence_passed || out.qualification.integrity_passed || Object.values(out.qualification.evidence_receipts).some(Boolean) || !out.evidence_summary.fixture_only)) throw new Error('synthetic fixture must be unmatched, unqualified, receipt-free, and fixture-only');
581
+ if (evidence_kind === 'observed_external' && out.evidence_summary.fixture_only) throw new Error('observed_external receipt must set fixture_only=false');
582
+ assertPublicArtifactSafe(out, 'run');
583
+ return out;
584
+ }
585
+ export function parseEvalFailureCluster(value) {
586
+ const root = obj(value, 'failure');
587
+ exact(root, 'failure', [
588
+ 'schema_version',
589
+ 'cluster_id',
590
+ 'category',
591
+ 'severity',
592
+ 'benchmark_ids',
593
+ 'case_ids',
594
+ 'run_ids',
595
+ 'reproduction',
596
+ 'falsifiable_hypothesis',
597
+ 'gotry_regression_id',
598
+ 'suggested_surface',
599
+ 'state'
600
+ ]);
601
+ const reproduction = obj(root.reproduction, 'failure.reproduction');
602
+ exact(reproduction, 'failure.reproduction', [
603
+ 'condition',
604
+ 'minimum_repetitions',
605
+ 'observed_repetitions'
606
+ ]);
607
+ const out = {
608
+ schema_version: lit(root.schema_version, 'failure.schema_version', [
609
+ 'gotry_eval_failure_cluster_v0'
610
+ ]),
611
+ cluster_id: id(root.cluster_id, 'failure.cluster_id'),
612
+ category: lit(root.category, 'failure.category', FAILURE_CATEGORIES),
613
+ severity: lit(root.severity, 'failure.severity', [
614
+ 'P0',
615
+ 'P1',
616
+ 'P2',
617
+ 'P3'
618
+ ]),
619
+ benchmark_ids: strings(root.benchmark_ids, 'failure.benchmark_ids').map((item)=>lit(item, 'failure.benchmark', BENCHMARK_IDS)),
620
+ case_ids: strings(root.case_ids, 'failure.case_ids'),
621
+ run_ids: strings(root.run_ids, 'failure.run_ids'),
622
+ reproduction: {
623
+ condition: text(reproduction.condition, 'failure.condition'),
624
+ minimum_repetitions: integer(reproduction.minimum_repetitions, 'failure.minimum', 1),
625
+ observed_repetitions: integer(reproduction.observed_repetitions, 'failure.observed')
626
+ },
627
+ falsifiable_hypothesis: text(root.falsifiable_hypothesis, 'failure.hypothesis'),
628
+ gotry_regression_id: id(root.gotry_regression_id, 'failure.regression'),
629
+ suggested_surface: lit(root.suggested_surface, 'failure.surface', [
630
+ 'agent',
631
+ 'tool_policy',
632
+ 'deterministic',
633
+ 'evaluation'
634
+ ]),
635
+ state: lit(root.state, 'failure.state', [
636
+ 'observed',
637
+ 'confirmed',
638
+ 'resolved',
639
+ 'rejected'
640
+ ])
641
+ };
642
+ if (out.state === 'confirmed' && out.reproduction.observed_repetitions < out.reproduction.minimum_repetitions) throw new Error('confirmed failure lacks repetitions');
643
+ assertPublicArtifactSafe(out, 'failure');
644
+ return out;
645
+ }
646
+ export function parseEvaluationFoundation(value) {
647
+ const root = obj(value, 'foundation');
648
+ exact(root, 'foundation', [
649
+ 'registry',
650
+ 'cases',
651
+ 'run_receipts',
652
+ 'failure_clusters'
653
+ ]);
654
+ const registry = parseBenchmarkRegistry(root.registry);
655
+ if (!Array.isArray(root.cases) || !Array.isArray(root.run_receipts) || !Array.isArray(root.failure_clusters)) throw new Error('foundation collections must be arrays');
656
+ const cases = root.cases.map(parseEvalCase), runs = root.run_receipts.map(parseEvalRunReceipt), failures = root.failure_clusters.map(parseEvalFailureCluster);
657
+ const unique = (values, label)=>{
658
+ if (new Set(values).size !== values.length) throw new Error(`${label} ids must be unique`);
659
+ };
660
+ unique(cases.map((item)=>item.case_id), 'case');
661
+ unique(runs.map((item)=>item.run_id), 'run');
662
+ unique(failures.map((item)=>item.cluster_id), 'failure');
663
+ const caseById = new Map(cases.map((item)=>[
664
+ item.case_id,
665
+ item
666
+ ])), runById = new Map(runs.map((item)=>[
667
+ item.run_id,
668
+ item
669
+ ])), registryById = new Map(registry.map((item)=>[
670
+ item.benchmark_id,
671
+ item
672
+ ]));
673
+ for (const run of runs){
674
+ const linked = caseById.get(run.case_id), entry = registryById.get(run.benchmark_id);
675
+ if (!linked || linked.benchmark_id !== run.benchmark_id || !entry) throw new Error(`run ${run.run_id} must close to case and registry`);
676
+ const metricKeys = Object.keys(run.native_metrics).sort(), declared = entry.native_metrics.values.map((metric)=>metric.receipt_key).sort();
677
+ if (JSON.stringify(metricKeys) !== JSON.stringify(declared)) throw new Error(`run ${run.run_id} metric keys must equal registry receipt keys`);
678
+ }
679
+ for (const failure of failures){
680
+ const linkedCases = failure.case_ids.map((key)=>caseById.get(key)), linkedRuns = failure.run_ids.map((key)=>runById.get(key));
681
+ if (linkedCases.some((item)=>!item) || linkedRuns.some((item)=>!item)) throw new Error(`failure ${failure.cluster_id} has unclosed links`);
682
+ const actual = new Set([
683
+ ...linkedCases,
684
+ ...linkedRuns
685
+ ].map((item)=>item.benchmark_id));
686
+ if (actual.size !== failure.benchmark_ids.length || failure.benchmark_ids.some((key)=>!actual.has(key))) throw new Error(`failure ${failure.cluster_id} benchmark closure must equal linked case and run benchmarks`);
687
+ }
688
+ return {
689
+ registry,
690
+ cases,
691
+ run_receipts: runs,
692
+ failure_clusters: failures
693
+ };
694
+ }
695
+ function runBinding(run) {
696
+ const { evidence_receipts: _receipts, ...qualification } = run.qualification;
697
+ return evaluationFingerprint({
698
+ ...run,
699
+ qualification
700
+ });
701
+ }
702
+ function resolveArtifact(resolver, run, kind, evalCase) {
703
+ const expected = kind === 'official_evaluator' ? run.qualification.evidence_receipts.official_evaluator_output_sha256 : kind === 'source_fence_audit' ? run.qualification.evidence_receipts.source_fence_audit_sha256 : run.qualification.evidence_receipts.integrity_audit_sha256;
704
+ if (!expected) throw new Error(`pair ${run.run_id} lacks qualification evidence receipts`);
705
+ const value = obj(resolver.resolve(expected), `${kind} artifact`);
706
+ const common = [
707
+ 'schema_version',
708
+ 'artifact_kind',
709
+ 'run_id',
710
+ 'benchmark_id',
711
+ 'case_id',
712
+ 'run_binding_sha256'
713
+ ];
714
+ const specific = kind === 'official_evaluator' ? [
715
+ 'evaluator_sha256',
716
+ 'native_metrics_sha256',
717
+ 'native_metrics',
718
+ 'official_result'
719
+ ] : kind === 'source_fence_audit' ? [
720
+ 'source_fence_sha256',
721
+ 'input_digest_sha256',
722
+ 'source_fence_passed',
723
+ 'forbidden_field_hits'
724
+ ] : [
725
+ 'integrity_sha256',
726
+ 'candidate_sha256',
727
+ 'integrity_passed'
728
+ ];
729
+ exact(value, `${kind} artifact`, [
730
+ ...common,
731
+ ...specific
732
+ ]);
733
+ lit(value.schema_version, `${kind}.schema_version`, [
734
+ 'gotry_eval_evidence_artifact_v0'
735
+ ]);
736
+ if (value.artifact_kind !== kind) throw new Error(`${kind} artifact kind mismatch`);
737
+ if (typeof value.run_id !== 'string' || typeof value.benchmark_id !== 'string' || typeof value.case_id !== 'string' || typeof value.run_binding_sha256 !== 'string') throw new Error(`${kind} artifact schema types mismatch`);
738
+ assertPublicArtifactSafe(value, ` artifact`);
739
+ if (kind === 'official_evaluator' && (typeof value.official_result !== 'boolean' || typeof value.native_metrics_sha256 !== 'string')) throw new Error(`${kind} artifact schema types mismatch`);
740
+ if (kind === 'source_fence_audit' && (typeof value.source_fence_passed !== 'boolean' || typeof value.forbidden_field_hits !== 'number' || !Number.isInteger(value.forbidden_field_hits))) throw new Error(`${kind} artifact schema types mismatch`);
741
+ if (kind === 'integrity_audit' && typeof value.integrity_passed !== 'boolean') throw new Error(`${kind} artifact schema types mismatch`);
742
+ if (evaluationFingerprint(value) !== expected) throw new Error(`pair ${run.run_id} ${kind} artifact fingerprint mismatch`);
743
+ if (value.run_id !== run.run_id || value.benchmark_id !== run.benchmark_id || value.case_id !== run.case_id || value.run_binding_sha256 !== runBinding(run)) throw new Error(`pair ${run.run_id} ${kind} artifact binding mismatch`);
744
+ if (kind === 'official_evaluator' && (value.evaluator_sha256 !== run.controls.official_evaluator_sha256 || value.native_metrics_sha256 !== evaluationFingerprint(run.native_metrics) || value.official_result !== true || JSON.stringify(value.native_metrics) !== JSON.stringify(run.native_metrics))) throw new Error(`pair ${run.run_id} evaluator artifact mismatch`);
745
+ if (kind === 'source_fence_audit' && (value.source_fence_sha256 !== run.controls.source_fence_sha256 || value.source_fence_passed !== true || value.forbidden_field_hits !== 0 || evalCase !== undefined && value.input_digest_sha256 !== evalCase.input_ref.digest_sha256)) throw new Error(`pair ${run.run_id} source-fence artifact mismatch`);
746
+ if (kind === 'integrity_audit' && (value.integrity_sha256 !== run.controls.integrity_sha256 || value.candidate_sha256 !== run.experiment.candidate_sha256 || value.integrity_passed !== true)) throw new Error(`pair ${run.run_id} integrity artifact mismatch`);
747
+ return value;
748
+ }
749
+ export function deriveMatchedPairs(foundation, resolver) {
750
+ const groups = new Map();
751
+ for (const run of foundation.run_receipts)if (run.pairing) groups.set(run.pairing.pair_id, [
752
+ ...groups.get(run.pairing.pair_id) ?? [],
753
+ run
754
+ ]);
755
+ const registry = new Map(foundation.registry.map((item)=>[
756
+ item.benchmark_id,
757
+ item
758
+ ])), cases = new Map(foundation.cases.map((item)=>[
759
+ item.case_id,
760
+ item
761
+ ]));
762
+ return [
763
+ ...groups.entries()
764
+ ].sort(([a], [b])=>a.localeCompare(b)).map(([pairId, members])=>{
765
+ if (members.length !== 2) throw new Error(`pair ${pairId} must contain exactly two receipts`);
766
+ const baseline = members.find((item)=>item.pairing?.role === 'baseline'), treatment = members.find((item)=>item.pairing?.role === 'treatment');
767
+ if (!baseline || !treatment) throw new Error(`pair ${pairId} must contain opposite baseline and treatment roles`);
768
+ if (baseline.pairing.counterpart_run_id !== treatment.run_id || treatment.pairing.counterpart_run_id !== baseline.run_id) throw new Error(`pair ${pairId} counterpart references must be reciprocal`);
769
+ if (baseline.benchmark_id !== treatment.benchmark_id || baseline.case_id !== treatment.case_id) throw new Error(`pair ${pairId} must share benchmark and case`);
770
+ const entry = registry.get(baseline.benchmark_id), evalCase = cases.get(baseline.case_id);
771
+ if (entry.countability_default !== 'countable_if_qualified') throw new Error(`pair ${pairId} benchmark is diagnostic only`);
772
+ if (evalCase.input_ref.kind !== 'external_opaque_reference') throw new Error(`pair ${pairId} requires an external opaque case reference`);
773
+ if ([
774
+ baseline,
775
+ treatment
776
+ ].some((run)=>run.evidence_kind !== 'observed_external')) throw new Error(`pair ${pairId} requires observed_external evidence`);
777
+ if (baseline.status !== 'succeeded' || treatment.status !== 'succeeded') throw new Error(`pair ${pairId} requires succeeded terminal receipts`);
778
+ if (baseline.model.provider !== treatment.model.provider || baseline.model.model !== treatment.model.model) throw new Error(`pair ${pairId} model identity must match`);
779
+ for (const key of [
780
+ 'case_set_sha256',
781
+ 'protocol_control_sha256',
782
+ 'model_parameters_sha256',
783
+ 'scorer_sha256',
784
+ 'tool_snapshot_sha256',
785
+ 'source_fence_sha256',
786
+ 'integrity_sha256',
787
+ 'official_evaluator_sha256'
788
+ ])if (baseline.controls[key] !== treatment.controls[key]) throw new Error(`pair ${pairId} control mismatch: ${key}`);
789
+ if (baseline.controls.scorer_sha256 !== evaluationFingerprint(evalCase.scorer_revision)) throw new Error(`pair ${pairId} scorer fingerprint does not close`);
790
+ if (baseline.controls.case_set_sha256 !== evaluationFingerprint(foundation.cases.filter((item)=>item.benchmark_id === baseline.benchmark_id))) throw new Error(`pair ${pairId} case-set fingerprint does not close`);
791
+ if (baseline.controls.source_fence_sha256 !== evaluationFingerprint(entry.source_fence)) throw new Error(`pair ${pairId} source-fence fingerprint does not close`);
792
+ if (baseline.controls.official_evaluator_sha256 !== evaluationFingerprint(entry.provenance.evaluator)) throw new Error(`pair ${pairId} evaluator fingerprint does not close`);
793
+ if (baseline.experiment.changed_variables.length !== 0 || JSON.stringify(treatment.experiment.changed_variables) !== JSON.stringify([
794
+ 'gotry_sha'
795
+ ])) throw new Error(`pair ${pairId} treatment variable must be exactly gotry_sha`);
796
+ if (baseline.gotry_sha === treatment.gotry_sha) throw new Error(`pair ${pairId} baseline and treatment gotry_sha must differ`);
797
+ if (baseline.guardrails.hard_violation_count !== 0 || treatment.guardrails.hard_violation_count !== 0 || baseline.guardrails.forbidden_leakage_hits !== 0 || treatment.guardrails.forbidden_leakage_hits !== 0) throw new Error(`pair ${pairId} hard violation or leakage`);
798
+ for (const run of [
799
+ baseline,
800
+ treatment
801
+ ])if (run.experiment.candidate_sha256 !== evaluationFingerprint({
802
+ treatment_variable: 'gotry_sha',
803
+ gotry_sha: run.gotry_sha
804
+ })) throw new Error(`pair ${pairId} candidate fingerprint does not close to gotry_sha`);
805
+ for (const run of [
806
+ baseline,
807
+ treatment
808
+ ]){
809
+ const receipts = Object.values(run.qualification.evidence_receipts);
810
+ if (!run.qualification.official_result || !run.qualification.source_fence_passed || !run.qualification.integrity_passed || receipts.some((item)=>item === null)) throw new Error(`pair ${pairId} lacks qualification evidence receipts`);
811
+ const evaluator = resolveArtifact(resolver, run, 'official_evaluator'), fence = resolveArtifact(resolver, run, 'source_fence_audit', evalCase), integrity = resolveArtifact(resolver, run, 'integrity_audit');
812
+ }
813
+ return {
814
+ schema_version: 'gotry_eval_matched_pair_derived_v0',
815
+ pair_id: pairId,
816
+ benchmark_id: baseline.benchmark_id,
817
+ case_id: baseline.case_id,
818
+ baseline_run_id: baseline.run_id,
819
+ treatment_run_id: treatment.run_id,
820
+ treatment_variable: 'gotry_sha',
821
+ matched_pair_countable: true
822
+ };
823
+ });
824
+ }
825
+ export function parseMutationVectors(value) {
826
+ if (!Array.isArray(value)) throw new Error('mutation vectors must be an array');
827
+ return value.map((raw, index)=>{
828
+ const root = obj(raw, `mutation[${index}]`), kind = lit(root.value_kind, 'mutation.value_kind', [
829
+ 'literal',
830
+ 'synthetic_absolute_path',
831
+ 'nan',
832
+ 'duplicate_treatment_receipt'
833
+ ]), keys = kind === 'literal' ? [
834
+ 'id',
835
+ 'foundation_kind',
836
+ 'target',
837
+ 'operation',
838
+ 'path',
839
+ 'value_kind',
840
+ 'value',
841
+ 'expected_error'
842
+ ] : [
843
+ 'id',
844
+ 'foundation_kind',
845
+ 'target',
846
+ 'operation',
847
+ 'path',
848
+ 'value_kind',
849
+ 'expected_error'
850
+ ];
851
+ exact(root, `mutation[${index}]`, keys);
852
+ return {
853
+ id: id(root.id, 'mutation.id'),
854
+ foundation_kind: lit(root.foundation_kind, 'mutation.foundation_kind', [
855
+ 'diagnostic_repository',
856
+ 'countable_test_only'
857
+ ]),
858
+ target: lit(root.target, 'mutation.target', [
859
+ 'registry',
860
+ 'case',
861
+ 'run',
862
+ 'failure',
863
+ 'foundation'
864
+ ]),
865
+ operation: lit(root.operation, 'mutation.operation', [
866
+ 'replace',
867
+ 'remove',
868
+ 'append'
869
+ ]),
870
+ path: text(root.path, 'mutation.path'),
871
+ value_kind: kind,
872
+ ...kind === 'literal' ? {
873
+ value: root.value
874
+ } : {},
875
+ expected_error: text(root.expected_error, 'mutation.expected_error')
876
+ };
877
+ });
878
+ }
879
+ export function applyMutationVector(foundation, vector) {
880
+ const cloned = JSON.parse(JSON.stringify(foundation));
881
+ const selected = vector.target === 'foundation' ? cloned : vector.target === 'registry' ? cloned.registry : vector.target === 'case' ? cloned.cases[0] : vector.target === 'run' ? cloned.run_receipts[0] : cloned.failure_clusters[0];
882
+ let value = vector.value;
883
+ if (vector.value_kind === 'synthetic_absolute_path') value = [
884
+ '',
885
+ 'Users',
886
+ 'fixture',
887
+ 'private.json'
888
+ ].join('/');
889
+ if (vector.value_kind === 'nan') value = Number.NaN;
890
+ if (vector.value_kind === 'duplicate_treatment_receipt') {
891
+ const run = structuredClone(cloned.run_receipts[1]);
892
+ run.run_id = 'run:trek:treatment-duplicate';
893
+ value = run;
894
+ }
895
+ const parts = vector.path.split('.').filter(Boolean);
896
+ let parent = selected;
897
+ for (const part of parts.slice(0, -1))parent = parent[part];
898
+ const leaf = parts.at(-1);
899
+ if (vector.operation === 'remove') Array.isArray(parent) ? parent.splice(Number(leaf), 1) : delete parent[leaf];
900
+ else if (vector.operation === 'append') parent[leaf].push(value);
901
+ else parent[leaf] = value;
902
+ return selected;
903
+ }
904
+
905
+
906
+ //# sourceURL=ts/src/evaluation-contracts.ts