@danceiny/gotry 0.0.1-rc.16 → 0.0.1-rc.18

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (229) hide show
  1. package/README.md +161 -116
  2. package/README.zh-CN.md +173 -144
  3. package/bin/gotry-booking-copilot.js +53 -0
  4. package/bin/gotry-bootstrap.js +113 -66
  5. package/bin/gotry-inner.js +436 -70
  6. package/bin/gotry-runtime-resolution.d.ts +27 -0
  7. package/bin/gotry-runtime-resolution.js +50 -0
  8. package/bin/gotry.js +1 -1
  9. package/cordis.gotry-patch.yml +5 -1
  10. package/dist/capabilities/agent-reach-deep.js +1 -1
  11. package/dist/capabilities/agent-reach.js +1 -1
  12. package/dist/capabilities/anything.js +1 -1
  13. package/dist/capabilities/artifacts.js +1 -1
  14. package/dist/capabilities/effect.js +1 -1
  15. package/dist/capabilities/fact-log.js +1 -1
  16. package/dist/capabilities/flyai.js +20 -6
  17. package/dist/capabilities/hbcli.js +1 -1
  18. package/dist/capabilities/incident-log.js +1 -1
  19. package/dist/capabilities/model-override.js +18 -0
  20. package/dist/capabilities/opensky.js +1 -1
  21. package/dist/capabilities/resilience.js +1 -1
  22. package/dist/capabilities/session/action-cache.js +1 -1
  23. package/dist/capabilities/session/adapters/ctrip-flight.js +1 -1
  24. package/dist/capabilities/session/adapters/meituan-local.js +1 -1
  25. package/dist/capabilities/session/benchmark.js +1 -1
  26. package/dist/capabilities/session/extension-bridge.js +19 -6
  27. package/dist/capabilities/session/extension-channel.js +3 -3
  28. package/dist/capabilities/session/extension-distribution.js +234 -0
  29. package/dist/capabilities/session/extract.js +1 -1
  30. package/dist/capabilities/session/golden-score.js +92 -0
  31. package/dist/capabilities/session/health-watch.js +1 -1
  32. package/dist/capabilities/session/read-guard.js +1 -1
  33. package/dist/capabilities/session/static-flight-golden.js +137 -0
  34. package/dist/capabilities/session/transport.js +1 -1
  35. package/dist/capabilities/session/wizard.js +21 -251
  36. package/dist/capabilities/session-consent.js +1 -1
  37. package/dist/capabilities/session-login.js +11 -5
  38. package/dist/capabilities/session-search.js +46 -4
  39. package/dist/capabilities/weather.js +168 -46
  40. package/dist/data/session-golden-20.json +25 -0
  41. package/dist/data/sf-golden-manifest.json +102 -0
  42. package/dist/data/sf-static-routes.json +91 -0
  43. package/dist/scripts/action-cache-tests.js +1 -1
  44. package/dist/scripts/agent-planning-budget-e2e.js +227 -0
  45. package/dist/scripts/agent-planning-budget-tests.js +173 -0
  46. package/dist/scripts/agent-reach-deep-tests.js +1 -1
  47. package/dist/scripts/agent-reach-tests.js +1 -1
  48. package/dist/scripts/agent-reach-wrapper-tests.js +1 -1
  49. package/dist/scripts/anything-tests.js +1 -1
  50. package/dist/scripts/async-collect.js +1 -1
  51. package/dist/scripts/benchmark-environment-bridge-e2e.js +981 -0
  52. package/dist/scripts/benchmark-environment-bridge-tests.js +2544 -0
  53. package/dist/scripts/booking-copilot-availability-ledger-binding-tests.js +335 -0
  54. package/dist/scripts/booking-copilot-availability-policy-v2-tests.js +1355 -0
  55. package/dist/scripts/booking-copilot-bin-proof-tests.js +89 -0
  56. package/dist/scripts/booking-copilot-crossrepo-fixture-server.js +113 -0
  57. package/dist/scripts/booking-copilot-dsh-core-proof-tests.js +356 -0
  58. package/dist/scripts/booking-copilot-dsh-planner-proof-tests.js +557 -0
  59. package/dist/scripts/booking-copilot-dsh-plugin-proof-tests.js +140 -0
  60. package/dist/scripts/booking-copilot-event-sequence-concurrency-proof-tests.js +202 -0
  61. package/dist/scripts/booking-copilot-gap-code-contract-proof-tests.js +102 -0
  62. package/dist/scripts/booking-copilot-operation-ledger-concurrency-proof-tests.js +374 -0
  63. package/dist/scripts/booking-copilot-receipt-ledger-concurrency-proof-tests.js +447 -0
  64. package/dist/scripts/booking-copilot-runtime-proof-tests.js +205 -0
  65. package/dist/scripts/booking-copilot-server-proof-tests.js +317 -0
  66. package/dist/scripts/booking-copilot-startup-proof-tests.js +283 -0
  67. package/dist/scripts/booking-copilot-v2-runtime-proof-tests.js +3935 -0
  68. package/dist/scripts/booking-saga-tests.js +1 -1
  69. package/dist/scripts/booking-surface-contract-proof-tests.js +393 -0
  70. package/dist/scripts/booking-surface-v2-contract-proof-tests.js +1174 -0
  71. package/dist/scripts/bootstrap-tests.js +45 -39
  72. package/dist/scripts/build-changelog.js +1 -1
  73. package/dist/scripts/changelog-tests.js +1 -1
  74. package/dist/scripts/companion-tests.js +1 -1
  75. package/dist/scripts/diff-test.js +1 -1
  76. package/dist/scripts/dsh-runtime-closure-tests.js +225 -0
  77. package/dist/scripts/dsh-runtime-closure.js +150 -0
  78. package/dist/scripts/effect-tests.js +1 -1
  79. package/dist/scripts/engine-run.js +1 -1
  80. package/dist/scripts/engine-tests.js +1 -1
  81. package/dist/scripts/evaluation-cadence-tests.js +347 -0
  82. package/dist/scripts/evaluation-contract-tests.js +574 -0
  83. package/dist/scripts/extension-distribution-cli.js +35 -0
  84. package/dist/scripts/extension-distribution-tests.js +341 -0
  85. package/dist/scripts/extension-tests.js +128 -23
  86. package/dist/scripts/fact-gate-tests.js +1 -1
  87. package/dist/scripts/flyai-tests.js +1 -1
  88. package/dist/scripts/hbcli-e2e-tests.js +1 -1
  89. package/dist/scripts/hbcli-tests.js +1 -1
  90. package/dist/scripts/health-watch-cli.js +1 -1
  91. package/dist/scripts/i18n-tests.js +1 -1
  92. package/dist/scripts/incident-tests.js +1 -1
  93. package/dist/scripts/journey-tests.js +1 -1
  94. package/dist/scripts/ledger-tests.js +1 -1
  95. package/dist/scripts/ledger-workflow-crash.js +1 -1
  96. package/dist/scripts/memory-capture-tests.js +1 -1
  97. package/dist/scripts/memory-decay-tests.js +1 -1
  98. package/dist/scripts/memory-metrics.js +1 -1
  99. package/dist/scripts/memory-value-report.js +1 -1
  100. package/dist/scripts/model-override-e2e.js +176 -0
  101. package/dist/scripts/nightly-evidence-tests.js +1 -1
  102. package/dist/scripts/nightly-evidence.js +1 -1
  103. package/dist/scripts/nudge-digest.js +1 -1
  104. package/dist/scripts/onboarding-tests.js +21 -53
  105. package/dist/scripts/opensky-check.js +1 -1
  106. package/dist/scripts/opensky-tests.js +1 -1
  107. package/dist/scripts/pnpm-dsh-closure-proof.js +20 -0
  108. package/dist/scripts/price-drift-tests.js +1 -1
  109. package/dist/scripts/price-drift-watch.js +1 -1
  110. package/dist/scripts/probe-poi-tests.js +1 -1
  111. package/dist/scripts/product-metrics.js +1 -1
  112. package/dist/scripts/publish-preverify.js +40 -4
  113. package/dist/scripts/realtime-pricing-tests.js +1 -1
  114. package/dist/scripts/replay-async.js +1 -1
  115. package/dist/scripts/replay-real.js +1 -1
  116. package/dist/scripts/replay.js +1 -1
  117. package/dist/scripts/session-attach-diagnose.js +1 -1
  118. package/dist/scripts/session-attach-poc.js +1 -1
  119. package/dist/scripts/session-benchmark.js +1 -1
  120. package/dist/scripts/session-extract-tests.js +1 -1
  121. package/dist/scripts/session-login.js +1 -1
  122. package/dist/scripts/session-tests.js +70 -20
  123. package/dist/scripts/sf-live-benchmark.js +338 -0
  124. package/dist/scripts/sf-live-cli-tests.js +21 -0
  125. package/dist/scripts/sf-soft-score-tests.js +108 -0
  126. package/dist/scripts/sf-summary.js +93 -0
  127. package/dist/scripts/skeleton-check.js +1 -1
  128. package/dist/scripts/skeleton-integration-test.js +1 -1
  129. package/dist/scripts/skills-contract-tests.js +1 -1
  130. package/dist/scripts/smoke-session-gate-tests.js +29 -0
  131. package/dist/scripts/smoke.js +79 -33
  132. package/dist/scripts/state-cli-tests.js +1 -1
  133. package/dist/scripts/state-cli.js +1 -1
  134. package/dist/scripts/static-golden-tests.js +299 -0
  135. package/dist/scripts/time-eval-tests.js +1 -1
  136. package/dist/scripts/travel-timeline-tests.js +1 -1
  137. package/dist/scripts/unified-tests.js +1 -1
  138. package/dist/scripts/weather-tests.js +694 -44
  139. package/dist/scripts/z3-race-tests.js +1 -1
  140. package/dist/src/artifact-gate.js +1 -1
  141. package/dist/src/benchmark-agent-conformance.js +370 -0
  142. package/dist/src/benchmark-environment-bridge.js +384 -0
  143. package/dist/src/benchmark-headless-child-diagnostics.js +173 -0
  144. package/dist/src/benchmark-tool-isolation.js +124 -0
  145. package/dist/src/bookable-facts.js +1 -1
  146. package/dist/src/booking-saga.js +1 -1
  147. package/dist/src/booking-surface/availability-policy-v2.js +830 -0
  148. package/dist/src/booking-surface/canonical-schema.js +113 -0
  149. package/dist/src/booking-surface/contracts-v2.js +89 -0
  150. package/dist/src/booking-surface/contracts.js +46 -0
  151. package/dist/src/booking-surface/dsh-planner.js +453 -0
  152. package/dist/src/booking-surface/dsh-plugin.js +93 -0
  153. package/dist/src/booking-surface/error-codes.js +94 -0
  154. package/dist/src/booking-surface/index.js +15 -0
  155. package/dist/src/booking-surface/profile.js +68 -0
  156. package/dist/src/booking-surface/runtime-v2.js +1771 -0
  157. package/dist/src/booking-surface/runtime.js +351 -0
  158. package/dist/src/booking-surface/server-v2.js +334 -0
  159. package/dist/src/booking-surface/server.js +302 -0
  160. package/dist/src/booking-surface/startup.js +159 -0
  161. package/dist/src/booking-surface/validation-v2.js +319 -0
  162. package/dist/src/booking-surface/validation.js +809 -0
  163. package/dist/src/bridge.js +1 -1
  164. package/dist/src/companions.js +1 -1
  165. package/dist/src/contracts.js +1 -1
  166. package/dist/src/dsh-llm.js +1 -1
  167. package/dist/src/engine.js +1 -1
  168. package/dist/src/evaluation-cadence.js +234 -0
  169. package/dist/src/evaluation-contracts.js +906 -0
  170. package/dist/src/i18n.js +1 -1
  171. package/dist/src/index.js +50 -19
  172. package/dist/src/journey.js +1 -1
  173. package/dist/src/loop.js +1 -1
  174. package/dist/src/memory-capture.js +1 -1
  175. package/dist/src/memory-decay.js +1 -1
  176. package/dist/src/memory-utility.js +1 -1
  177. package/dist/src/mock-llm.js +1 -1
  178. package/dist/src/model.js +1 -1
  179. package/dist/src/realtime-pricing.js +1 -1
  180. package/dist/src/slot-spec.js +1 -1
  181. package/dist/src/state-ledger.js +2 -1
  182. package/dist/src/time-anchor.js +1 -1
  183. package/dist/src/tool-budget.js +136 -0
  184. package/dist/src/tool-packet.js +1 -1
  185. package/dist/src/travel-slots.js +1 -1
  186. package/dist/src/travel-timeline.js +1 -1
  187. package/dist/src/unified.js +1 -1
  188. package/dist/src/wish-pool.js +1 -1
  189. package/dist/src/z3-shared.js +1 -1
  190. package/extension/README.md +31 -7
  191. package/package.json +286 -11
  192. package/schemas/booking.surface.v1.schema.json +927 -0
  193. package/schemas/booking.surface.v2.schema.json +61 -0
  194. package/ts/capabilities/flyai.ts +16 -3
  195. package/ts/capabilities/session/extension-bridge.ts +37 -14
  196. package/ts/capabilities/session/extension-channel.ts +6 -3
  197. package/ts/capabilities/session/extension-distribution.ts +264 -0
  198. package/ts/capabilities/session/golden-score.ts +139 -0
  199. package/ts/capabilities/session/health-watch.ts +1 -1
  200. package/ts/capabilities/session/static-flight-golden.ts +209 -0
  201. package/ts/capabilities/session/wizard.ts +34 -176
  202. package/ts/capabilities/session-login.ts +15 -5
  203. package/ts/capabilities/session-search.ts +40 -3
  204. package/ts/capabilities/weather.ts +141 -52
  205. package/ts/package.json +3 -3
  206. package/ts/src/benchmark-agent-conformance.ts +448 -0
  207. package/ts/src/benchmark-environment-bridge.ts +348 -0
  208. package/ts/src/benchmark-headless-child-diagnostics.ts +184 -0
  209. package/ts/src/benchmark-tool-isolation.ts +166 -0
  210. package/ts/src/booking-surface/availability-policy-v2.ts +523 -0
  211. package/ts/src/booking-surface/canonical-schema.js +113 -0
  212. package/ts/src/booking-surface/contracts-v2.ts +118 -0
  213. package/ts/src/booking-surface/contracts.ts +380 -0
  214. package/ts/src/booking-surface/dsh-planner.ts +452 -0
  215. package/ts/src/booking-surface/dsh-plugin.js +93 -0
  216. package/ts/src/booking-surface/error-codes.ts +101 -0
  217. package/ts/src/booking-surface/index.ts +12 -0
  218. package/ts/src/booking-surface/profile.ts +42 -0
  219. package/ts/src/booking-surface/runtime-v2.ts +1466 -0
  220. package/ts/src/booking-surface/runtime.ts +483 -0
  221. package/ts/src/booking-surface/server-v2.ts +247 -0
  222. package/ts/src/booking-surface/server.ts +324 -0
  223. package/ts/src/booking-surface/startup.ts +196 -0
  224. package/ts/src/booking-surface/validation-v2.ts +205 -0
  225. package/ts/src/booking-surface/validation.ts +453 -0
  226. package/ts/src/index.ts +64 -11
  227. package/ts/src/state-ledger.ts +1 -0
  228. package/ts/src/tool-budget.ts +165 -0
  229. package/dist/scripts/wizard-bootstrap.js +0 -32
@@ -0,0 +1,574 @@
1
+ import assert from 'node:assert/strict';
2
+ import { createHash } from 'node:crypto';
3
+ import { readFileSync } from 'node:fs';
4
+ import { BENCHMARK_IDS, applyMutationVector, assertPublicArtifactSafe, deriveMatchedPairs, evaluationFingerprint, parseBenchmarkRegistry, parseEvalCase, parseEvalFailureCluster, parseEvalRunReceipt, parseEvaluationFoundation, parseMutationVectors, stableEvaluationJson } from '../src/evaluation-contracts.js';
5
+ const load = (file)=>JSON.parse(readFileSync(file, 'utf8'));
6
+ const registry = parseBenchmarkRegistry(load('data/evaluation/benchmark-registry.json'));
7
+ assert.deepEqual(registry.map((item)=>item.benchmark_id), [
8
+ ...BENCHMARK_IDS
9
+ ]);
10
+ assert.equal(registry.length, 7);
11
+ assert.deepEqual(Object.fromEntries(registry.map((item)=>[
12
+ item.benchmark_id,
13
+ item.native_metrics.values.map((metric)=>metric.receipt_key)
14
+ ])), {
15
+ trek: [
16
+ 'task_perfect_feasible',
17
+ 'task_perfect_infeasible',
18
+ 'cat_efficiency'
19
+ ],
20
+ travelplanner: [
21
+ 'commonsense_micro_pass_rate',
22
+ 'commonsense_macro_pass_rate',
23
+ 'hard_micro_pass_rate',
24
+ 'hard_macro_pass_rate',
25
+ 'final_pass_rate'
26
+ ],
27
+ chinatravel: [
28
+ 'epr_micro',
29
+ 'epr_macro',
30
+ 'c_lpr',
31
+ 'fpr',
32
+ 'dav',
33
+ 'att',
34
+ 'ddr',
35
+ 'overall_score'
36
+ ],
37
+ travelbench: [
38
+ 'reasoning_planning_score',
39
+ 'summarization_extraction_score',
40
+ 'presentation_score',
41
+ 'user_interaction_score',
42
+ 'average_score',
43
+ 'unsolved_accuracy'
44
+ ],
45
+ tau2: [
46
+ 'avg_reward',
47
+ 'pass_hat_1'
48
+ ],
49
+ locomo: [
50
+ 'qa_f1'
51
+ ],
52
+ bfcl: [
53
+ 'category_accuracy',
54
+ 'overall_accuracy'
55
+ ]
56
+ });
57
+ const registryFacts = Object.fromEntries(registry.map((entry)=>[
58
+ entry.benchmark_id,
59
+ {
60
+ owner: new URL(entry.provenance.official_entry.url).pathname.split('/')[1],
61
+ pins: Object.entries(entry.provenance).map(([kind, pin])=>`${kind}|${pin.url}|${pin.revision.kind}|${pin.revision.value ?? 'null'}|${pin.source_scope}`),
62
+ rights: Object.entries(entry.license.upstream_rights).map(([kind, right])=>`${kind}|${right.value}|${right.determination}|${right.source_url}`),
63
+ metrics: entry.native_metrics.values.map((metric)=>`${metric.receipt_key}|${metric.upstream_label}|${metric.scope}|${metric.source_url}`)
64
+ }
65
+ ]));
66
+ assert.deepEqual(registryFacts, {
67
+ "trek": {
68
+ "owner": "TonyQJH",
69
+ "pins": [
70
+ "official_entry|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/README.md|git_commit|6ceb4ebb2debd69c5c7c4ba34b5b17524756912b|official definition",
71
+ "data|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/tree/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/api/data/v2|git_commit|6ceb4ebb2debd69c5c7c4ba34b5b17524756912b|task data v2",
72
+ "evaluator|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/scoring.py|git_commit|6ceb4ebb2debd69c5c7c4ba34b5b17524756912b|official nine-dimension scoring implementation"
73
+ ],
74
+ "rights": [
75
+ "code|MIT|declared|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/LICENSE",
76
+ "data|CC-BY-4.0|declared|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/README.md#data-and-license",
77
+ "evaluator|MIT|declared|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/LICENSE"
78
+ ],
79
+ "metrics": [
80
+ "task_perfect_feasible|task_perfect_feasible|task-perfect feasible|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/scoring.py",
81
+ "task_perfect_infeasible|task_perfect_infeasible|task-perfect infeasible|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/scoring.py",
82
+ "cat_efficiency|cat_efficiency|category efficiency|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/scoring.py"
83
+ ]
84
+ },
85
+ "travelplanner": {
86
+ "owner": "OSU-NLP-Group",
87
+ "pins": [
88
+ "official_entry|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/README.md|git_commit|e52c87f4ac348a3410c46dc3553c519db5ec5e23|official definition",
89
+ "data|https://huggingface.co/datasets/osunlp/TravelPlanner/tree/8736504ecfc31b7f8b7e40122873c337e83fff7c|git_commit|8736504ecfc31b7f8b7e40122873c337e83fff7c|official Hugging Face dataset git revision",
90
+ "evaluator|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py|git_commit|e52c87f4ac348a3410c46dc3553c519db5ec5e23|official evaluator"
91
+ ],
92
+ "rights": [
93
+ "code|MIT|declared|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/LICENSE",
94
+ "data|CC-BY-4.0|declared|https://huggingface.co/datasets/osunlp/TravelPlanner/blob/8736504ecfc31b7f8b7e40122873c337e83fff7c/README.md",
95
+ "evaluator|MIT|declared|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/LICENSE"
96
+ ],
97
+ "metrics": [
98
+ "commonsense_micro_pass_rate|Commonsense Constraint Micro Pass Rate|commonsense micro|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py",
99
+ "commonsense_macro_pass_rate|Commonsense Constraint Macro Pass Rate|commonsense macro|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py",
100
+ "hard_micro_pass_rate|Hard Constraint Micro Pass Rate|hard-constraint micro|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py",
101
+ "hard_macro_pass_rate|Hard Constraint Macro Pass Rate|hard-constraint macro|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py",
102
+ "final_pass_rate|Final Pass Rate|complete itinerary|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py"
103
+ ]
104
+ },
105
+ "chinatravel": {
106
+ "owner": "chinatravel-competition",
107
+ "pins": [
108
+ "official_entry|https://github.com/chinatravel-competition/IJCAI2026/blob/49d02bc322dda7ffbf53dfb7c3d2ced6b4bd4e8b/index.html|git_commit|49d02bc322dda7ffbf53dfb7c3d2ced6b4bd4e8b|official competition entry",
109
+ "data|https://github.com/chinatravel-competition/IJCAI2026/blob/49d02bc322dda7ffbf53dfb7c3d2ced6b4bd4e8b/TPC_IJCAI_2026_phase2_familiar_100_data.zip|git_commit|49d02bc322dda7ffbf53dfb7c3d2ced6b4bd4e8b|public familiar-track archive",
110
+ "evaluator|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py|git_commit|b071db251905b14002ec98e8b36afca7b6d6cd04|official TPC evaluator implementation"
111
+ ],
112
+ "rights": [
113
+ "code|not_separately_declared|not_separately_declared|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/README.md",
114
+ "data|CC-BY-NC-SA-4.0|declared|https://huggingface.co/datasets/LAMDA-NeSy/ChinaTravel/blob/44d5dbf3bba26bdf9a212c3e76d3242b67f0d349/README.md",
115
+ "evaluator|not_separately_declared|not_separately_declared|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/README.md"
116
+ ],
117
+ "metrics": [
118
+ "epr_micro|EPR-micro|element pass micro|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
119
+ "epr_macro|EPR-macro|element pass macro|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
120
+ "c_lpr|C-LPR|constraint-level pass|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
121
+ "fpr|FPR|final pass|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
122
+ "dav|DAV|Daily Average Attractions Visited|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
123
+ "att|ATT|Averaged Transportation Time|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
124
+ "ddr|DDR|Daily Dining Recommendations|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
125
+ "overall_score|Overall Score|weighted overall score|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py"
126
+ ]
127
+ },
128
+ "travelbench": {
129
+ "owner": "small-xiangcheng",
130
+ "pins": [
131
+ "official_entry|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/README.md|git_commit|445a29d9a9b6457fc95fe647c532b6b79e21c43f|official definition",
132
+ "data|https://github.com/small-xiangcheng/TravelBench/tree/445a29d9a9b6457fc95fe647c532b6b79e21c43f/datas|git_commit|445a29d9a9b6457fc95fe647c532b6b79e21c43f|official data directory",
133
+ "evaluator|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py|git_commit|445a29d9a9b6457fc95fe647c532b6b79e21c43f|official evaluator"
134
+ ],
135
+ "rights": [
136
+ "code|MIT|declared|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/LICENSE",
137
+ "data|CC-BY-NC-4.0|declared|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/datas/LICENSE",
138
+ "evaluator|MIT|declared|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/LICENSE"
139
+ ],
140
+ "metrics": [
141
+ "reasoning_planning_score|reasoning_planning_score|reasoning and planning|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py",
142
+ "summarization_extraction_score|summarization_extraction_score|summarization and extraction|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py",
143
+ "presentation_score|presentation_score|presentation|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py",
144
+ "user_interaction_score|user_interaction_score|user interaction|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py",
145
+ "average_score|average_score|average across dimensions|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py",
146
+ "unsolved_accuracy|unsolved_accuracy|unsolved cases|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate_unsolved.py"
147
+ ]
148
+ },
149
+ "tau2": {
150
+ "owner": "sierra-research",
151
+ "pins": [
152
+ "official_entry|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/README.md|git_commit|a2c024725189473d2d7cea3a5cfdbcc67478e41f|official definition",
153
+ "data|https://github.com/sierra-research/tau2-bench/tree/a2c024725189473d2d7cea3a5cfdbcc67478e41f/data/tau2|git_commit|a2c024725189473d2d7cea3a5cfdbcc67478e41f|official data",
154
+ "evaluator|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/src/tau2/metrics/agent_metrics.py|git_commit|a2c024725189473d2d7cea3a5cfdbcc67478e41f|official agent metrics implementation"
155
+ ],
156
+ "rights": [
157
+ "code|MIT|declared|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/LICENSE",
158
+ "data|not_separately_declared|not_separately_declared|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/README.md",
159
+ "evaluator|MIT|declared|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/LICENSE"
160
+ ],
161
+ "metrics": [
162
+ "avg_reward|avg_reward|average reward|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/src/tau2/metrics/agent_metrics.py",
163
+ "pass_hat_1|pass^1|mean task pass^1|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/src/tau2/metrics/agent_metrics.py"
164
+ ]
165
+ },
166
+ "locomo": {
167
+ "owner": "snap-research",
168
+ "pins": [
169
+ "official_entry|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/README.MD|git_commit|3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376|official definition",
170
+ "data|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/data/locomo10.json|git_commit|3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376|official data",
171
+ "evaluator|https://github.com/snap-research/locomo/tree/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/task_eval|git_commit|3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376|official task evaluation directory"
172
+ ],
173
+ "rights": [
174
+ "code|CC-BY-NC-4.0|declared|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/LICENSE.txt",
175
+ "data|CC-BY-NC-4.0|declared|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/LICENSE.txt",
176
+ "evaluator|CC-BY-NC-4.0|declared|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/LICENSE.txt"
177
+ ],
178
+ "metrics": [
179
+ "qa_f1|F1|question-answering token F1|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/task_eval/evaluate_qa.py"
180
+ ]
181
+ },
182
+ "bfcl": {
183
+ "owner": "ShishirPatil",
184
+ "pins": [
185
+ "official_entry|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/README.md|git_commit|6ea57973c7a6097fd7c5915698c54c17c5b1b6c8|official BFCL definition",
186
+ "data|https://github.com/ShishirPatil/gorilla/tree/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/bfcl_eval/data|git_commit|6ea57973c7a6097fd7c5915698c54c17c5b1b6c8|BFCL data",
187
+ "evaluator|https://github.com/ShishirPatil/gorilla/tree/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/bfcl_eval|git_commit|6ea57973c7a6097fd7c5915698c54c17c5b1b6c8|BFCL evaluator"
188
+ ],
189
+ "rights": [
190
+ "code|Apache-2.0|declared|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/LICENSE",
191
+ "data|Apache-2.0|declared|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/README.md",
192
+ "evaluator|Apache-2.0|declared|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/LICENSE"
193
+ ],
194
+ "metrics": [
195
+ "category_accuracy|accuracy|per test_category|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/bfcl_eval/eval_checker/eval_runner_helper.py",
196
+ "overall_accuracy|Overall Acc|overall accuracy|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/README.md"
197
+ ]
198
+ }
199
+ });
200
+ const raw = load('data/evaluation/known-good.json');
201
+ const diagnostic = parseEvaluationFoundation({
202
+ registry,
203
+ ...raw
204
+ });
205
+ assert.equal(diagnostic.cases.length, 1);
206
+ assert.equal(diagnostic.run_receipts.length, 1);
207
+ assert.equal(diagnostic.run_receipts[0].evidence_kind, 'synthetic_fixture');
208
+ assert.equal(diagnostic.run_receipts[0].pairing, null);
209
+ assert.equal(diagnostic.run_receipts[0].qualification.official_result, false);
210
+ const emptyResolver = {
211
+ resolve: ()=>undefined
212
+ };
213
+ assert.deepEqual(deriveMatchedPairs(diagnostic, emptyResolver), []);
214
+ for (const item of diagnostic.cases)parseEvalCase(item);
215
+ for (const item of diagnostic.run_receipts)parseEvalRunReceipt(item);
216
+ for (const item of diagnostic.failure_clusters)parseEvalFailureCluster(item);
217
+ for (const item of [
218
+ ...diagnostic.cases,
219
+ ...diagnostic.run_receipts,
220
+ ...diagnostic.failure_clusters
221
+ ])assertPublicArtifactSafe(item, 'repository fixture');
222
+ const trek = registry.find((item)=>item.benchmark_id === 'trek');
223
+ const evalCase = diagnostic.cases[0];
224
+ const observedCase = {
225
+ ...evalCase,
226
+ input_ref: {
227
+ ...evalCase.input_ref,
228
+ kind: 'external_opaque_reference'
229
+ }
230
+ };
231
+ const seed = diagnostic.run_receipts[0];
232
+ const controls = {
233
+ ...seed.controls,
234
+ case_set_sha256: evaluationFingerprint([
235
+ observedCase
236
+ ]),
237
+ scorer_sha256: evaluationFingerprint(observedCase.scorer_revision),
238
+ source_fence_sha256: evaluationFingerprint(trek.source_fence),
239
+ official_evaluator_sha256: evaluationFingerprint(trek.provenance.evaluator)
240
+ };
241
+ const observed = (role)=>{
242
+ const run = {
243
+ ...seed,
244
+ run_id: `run:trek:${role}-test-only`,
245
+ evidence_kind: 'observed_external',
246
+ gotry_sha: role === 'baseline' ? '1111111111111111111111111111111111111111' : '2222222222222222222222222222222222222222',
247
+ pairing: {
248
+ pair_id: 'pair:trek:test-only',
249
+ role,
250
+ counterpart_run_id: `run:trek:${role === 'baseline' ? 'treatment' : 'baseline'}-test-only`
251
+ },
252
+ model: {
253
+ provider: 'test-provider',
254
+ model: 'test-model'
255
+ },
256
+ controls,
257
+ qualification: {
258
+ official_result: true,
259
+ source_fence_passed: true,
260
+ integrity_passed: true,
261
+ evidence_receipts: {
262
+ official_evaluator_output_sha256: null,
263
+ source_fence_audit_sha256: null,
264
+ integrity_audit_sha256: null
265
+ }
266
+ },
267
+ experiment: {
268
+ changed_variables: role === 'baseline' ? [] : [
269
+ 'gotry_sha'
270
+ ],
271
+ candidate_sha256: evaluationFingerprint({
272
+ treatment_variable: 'gotry_sha',
273
+ gotry_sha: role === 'baseline' ? '1111111111111111111111111111111111111111' : '2222222222222222222222222222222222222222'
274
+ })
275
+ },
276
+ evidence_summary: {
277
+ ...seed.evidence_summary,
278
+ fixture_only: false,
279
+ statement: 'test-only observed-external aggregate admission object'
280
+ }
281
+ };
282
+ const { evidence_receipts: _receipts, ...qualification } = run.qualification;
283
+ const bound = {
284
+ ...run,
285
+ qualification
286
+ };
287
+ const base = {
288
+ schema_version: 'gotry_eval_evidence_artifact_v0',
289
+ run_id: run.run_id,
290
+ benchmark_id: run.benchmark_id,
291
+ case_id: run.case_id,
292
+ run_binding_sha256: evaluationFingerprint(bound)
293
+ };
294
+ const artifacts = {
295
+ official_evaluator: {
296
+ ...base,
297
+ artifact_kind: 'official_evaluator',
298
+ evaluator_sha256: run.controls.official_evaluator_sha256,
299
+ native_metrics_sha256: evaluationFingerprint(run.native_metrics),
300
+ native_metrics: run.native_metrics,
301
+ official_result: true
302
+ },
303
+ source_fence_audit: {
304
+ ...base,
305
+ artifact_kind: 'source_fence_audit',
306
+ source_fence_sha256: run.controls.source_fence_sha256,
307
+ input_digest_sha256: observedCase.input_ref.digest_sha256,
308
+ source_fence_passed: true,
309
+ forbidden_field_hits: 0
310
+ },
311
+ integrity_audit: {
312
+ ...base,
313
+ artifact_kind: 'integrity_audit',
314
+ integrity_sha256: run.controls.integrity_sha256,
315
+ candidate_sha256: run.experiment.candidate_sha256,
316
+ integrity_passed: true
317
+ }
318
+ };
319
+ run.qualification.evidence_receipts = {
320
+ official_evaluator_output_sha256: evaluationFingerprint(artifacts.official_evaluator),
321
+ source_fence_audit_sha256: evaluationFingerprint(artifacts.source_fence_audit),
322
+ integrity_audit_sha256: evaluationFingerprint(artifacts.integrity_audit)
323
+ };
324
+ return run;
325
+ };
326
+ const countable = parseEvaluationFoundation({
327
+ registry,
328
+ cases: [
329
+ observedCase
330
+ ],
331
+ run_receipts: [
332
+ observed('baseline'),
333
+ observed('treatment')
334
+ ],
335
+ failure_clusters: [
336
+ {
337
+ ...diagnostic.failure_clusters[0],
338
+ run_ids: [
339
+ 'run:trek:baseline-test-only',
340
+ 'run:trek:treatment-test-only'
341
+ ]
342
+ }
343
+ ]
344
+ });
345
+ const artifactResolver = {
346
+ resolve (sha256) {
347
+ for (const run of countable.run_receipts){
348
+ const { evidence_receipts: _receipts, ...qualification } = run.qualification;
349
+ const bound = {
350
+ ...run,
351
+ qualification
352
+ };
353
+ const base = {
354
+ schema_version: 'gotry_eval_evidence_artifact_v0',
355
+ run_id: run.run_id,
356
+ benchmark_id: run.benchmark_id,
357
+ case_id: run.case_id,
358
+ run_binding_sha256: evaluationFingerprint(bound)
359
+ };
360
+ const artifacts = [
361
+ {
362
+ ...base,
363
+ artifact_kind: 'official_evaluator',
364
+ evaluator_sha256: run.controls.official_evaluator_sha256,
365
+ native_metrics_sha256: evaluationFingerprint(run.native_metrics),
366
+ native_metrics: run.native_metrics,
367
+ official_result: true
368
+ },
369
+ {
370
+ ...base,
371
+ artifact_kind: 'source_fence_audit',
372
+ source_fence_sha256: run.controls.source_fence_sha256,
373
+ input_digest_sha256: observedCase.input_ref.digest_sha256,
374
+ source_fence_passed: true,
375
+ forbidden_field_hits: 0
376
+ },
377
+ {
378
+ ...base,
379
+ artifact_kind: 'integrity_audit',
380
+ integrity_sha256: run.controls.integrity_sha256,
381
+ candidate_sha256: run.experiment.candidate_sha256,
382
+ integrity_passed: true
383
+ }
384
+ ];
385
+ const found = artifacts.find((item)=>evaluationFingerprint(item) === sha256);
386
+ if (found) return found;
387
+ }
388
+ return undefined;
389
+ }
390
+ };
391
+ const fingerprintMismatchResolver = {
392
+ resolve (sha256) {
393
+ const artifact = artifactResolver.resolve(sha256);
394
+ if (!artifact) return undefined;
395
+ const changed = structuredClone(artifact);
396
+ if (changed.artifact_kind === 'official_evaluator') changed.official_result = false;
397
+ return changed;
398
+ }
399
+ };
400
+ assert.throws(()=>deriveMatchedPairs(countable, fingerprintMismatchResolver), /fingerprint mismatch/);
401
+ const pairs = deriveMatchedPairs(countable, artifactResolver);
402
+ assert.deepEqual(pairs, [
403
+ {
404
+ schema_version: 'gotry_eval_matched_pair_derived_v0',
405
+ pair_id: 'pair:trek:test-only',
406
+ benchmark_id: 'trek',
407
+ case_id: 'gotry:foundation:case-001',
408
+ baseline_run_id: 'run:trek:baseline-test-only',
409
+ treatment_run_id: 'run:trek:treatment-test-only',
410
+ treatment_variable: 'gotry_sha',
411
+ matched_pair_countable: true
412
+ }
413
+ ]);
414
+ const exerciseArtifact = (runIndex, kind, mutate, expected)=>{
415
+ const foundation = structuredClone(countable);
416
+ const run = foundation.run_receipts[runIndex];
417
+ const receiptKey = kind === 'official_evaluator' ? 'official_evaluator_output_sha256' : kind === 'source_fence_audit' ? 'source_fence_audit_sha256' : 'integrity_audit_sha256';
418
+ const original = artifactResolver.resolve(run.qualification.evidence_receipts[receiptKey]);
419
+ const artifact = structuredClone(original);
420
+ mutate(artifact);
421
+ const digest = evaluationFingerprint(artifact);
422
+ run.qualification.evidence_receipts[receiptKey] = digest;
423
+ const resolver = {
424
+ resolve (sha256) {
425
+ return sha256 === digest ? artifact : artifactResolver.resolve(sha256);
426
+ }
427
+ };
428
+ assert.throws(()=>deriveMatchedPairs(foundation, resolver), expected);
429
+ };
430
+ assert.throws(()=>deriveMatchedPairs(countable, {
431
+ resolve: ()=>undefined
432
+ }), /artifact/);
433
+ exerciseArtifact(0, 'official_evaluator', (a)=>{
434
+ a.evaluator_sha256 = 'f'.repeat(64);
435
+ }, /evaluator artifact mismatch/);
436
+ exerciseArtifact(0, 'official_evaluator', (a)=>{
437
+ a.native_metrics = {
438
+ altered: 1
439
+ };
440
+ a.native_metrics_sha256 = evaluationFingerprint(a.native_metrics);
441
+ }, /evaluator artifact mismatch/);
442
+ exerciseArtifact(0, 'source_fence_audit', (a)=>{
443
+ a.input_digest_sha256 = 'f'.repeat(64);
444
+ }, /source-fence artifact mismatch/);
445
+ exerciseArtifact(0, 'source_fence_audit', (a)=>{
446
+ a.source_fence_passed = false;
447
+ }, /source-fence artifact mismatch/);
448
+ exerciseArtifact(0, 'source_fence_audit', (a)=>{
449
+ a.forbidden_field_hits = 1;
450
+ }, /source-fence artifact mismatch/);
451
+ exerciseArtifact(0, 'integrity_audit', (a)=>{
452
+ a.candidate_sha256 = 'f'.repeat(64);
453
+ }, /integrity artifact mismatch/);
454
+ exerciseArtifact(0, 'integrity_audit', (a)=>{
455
+ a.integrity_passed = false;
456
+ }, /integrity artifact mismatch/);
457
+ const treatmentReuse = structuredClone(countable);
458
+ const baselineReceipt = treatmentReuse.run_receipts[0].qualification.evidence_receipts.official_evaluator_output_sha256;
459
+ treatmentReuse.run_receipts[1].qualification.evidence_receipts.official_evaluator_output_sha256 = baselineReceipt;
460
+ assert.throws(()=>deriveMatchedPairs(treatmentReuse, artifactResolver), /fingerprint mismatch|binding mismatch/);
461
+ const canonical1 = stableEvaluationJson(diagnostic);
462
+ const canonical2 = stableEvaluationJson(parseEvaluationFoundation(JSON.parse(canonical1)));
463
+ const digest1 = createHash('sha256').update(canonical1).digest('hex');
464
+ const digest2 = createHash('sha256').update(canonical2).digest('hex');
465
+ assert.equal(canonical1, canonical2);
466
+ assert.equal(digest1, digest2);
467
+ assert.match(digest1, /^[0-9a-f]{64}$/);
468
+ const vectors = parseMutationVectors(load('data/evaluation/known-bad.json'));
469
+ assert.equal(vectors.length, 39);
470
+ for (const vector of vectors){
471
+ const base = vector.foundation_kind === 'countable_test_only' ? countable : diagnostic;
472
+ const mutated = applyMutationVector(base, vector);
473
+ assert.throws(()=>vector.target === 'foundation' ? deriveMatchedPairs(parseEvaluationFoundation(mutated), vector.foundation_kind === 'countable_test_only' ? artifactResolver : emptyResolver) : vector.target === 'registry' ? parseBenchmarkRegistry(mutated) : vector.target === 'case' ? parseEvalCase(mutated) : vector.target === 'run' ? parseEvalRunReceipt(mutated) : parseEvalFailureCluster(mutated), new RegExp(vector.expected_error), vector.id);
474
+ }
475
+ const runAll = readFileSync('../scripts/run-all-tests.sh', 'utf8');
476
+ assert.match(runAll, /=== 46\. Evaluation Phase 0 foundation/);
477
+ assert.match(runAll, /npx tsx scripts\/evaluation-contract-tests\.ts/);
478
+ assert.throws(()=>deriveMatchedPairs(countable, emptyResolver), /artifact/);
479
+ for (const [key, value] of [
480
+ [
481
+ 'apikey',
482
+ 'secret'
483
+ ],
484
+ [
485
+ 'access-token',
486
+ 'secret'
487
+ ],
488
+ [
489
+ 'apiKey',
490
+ 'secret'
491
+ ],
492
+ [
493
+ 'ACCESS-TOKEN',
494
+ 'secret'
495
+ ],
496
+ [
497
+ 'client_secret',
498
+ 'secret'
499
+ ],
500
+ [
501
+ 'authorization',
502
+ 'Bearer abcdefghijklmnopqrstuvwxyz123456'
503
+ ],
504
+ [
505
+ 'value',
506
+ '/private/file'
507
+ ],
508
+ [
509
+ 'value',
510
+ '~/private/file'
511
+ ],
512
+ [
513
+ 'value',
514
+ 'C:\\private\\file'
515
+ ],
516
+ [
517
+ 'value',
518
+ 'file:///private/file'
519
+ ],
520
+ [
521
+ 'value',
522
+ 'sk-abcdefghijklmnopqrstuv'
523
+ ],
524
+ [
525
+ 'value',
526
+ 'ghp_abcdefghijklmnopqrstuvwxyz123456'
527
+ ],
528
+ [
529
+ 'value',
530
+ 'AKIA1234567890ABCDEF'
531
+ ]
532
+ ])assert.throws(()=>assertPublicArtifactSafe({
533
+ [key]: value
534
+ }, 'adversarial'), /absolute path or secret|credentials/);
535
+ assert.doesNotThrow(()=>assertPublicArtifactSafe({
536
+ url: 'https://github.com/org/repo/blob/main/README.md',
537
+ label: 'question-answering token F1'
538
+ }, 'benign'));
539
+ assert.throws(()=>assertPublicArtifactSafe({
540
+ value: 'https://example.test/?token=sk-abcdefghijklmnopqrstuv'
541
+ }, 'https-url-secret'), /absolute path or secret/);
542
+ for (const value of [
543
+ 'log=/Users/a/private.json',
544
+ 'C:\\Users\\a\\secret',
545
+ 'file:///tmp/x',
546
+ '~/x',
547
+ 'https://github.com/org/repo log=/Users/a/x',
548
+ 'Bearer abcdefghijklmnopqrstuvwxyz123456',
549
+ 'sk-abcdefghijklmnopqrstuv',
550
+ 'ghp_abcdefghijklmnopqrstuvwxyz123456',
551
+ 'AKIA1234567890ABCDEF'
552
+ ])assert.throws(()=>assertPublicArtifactSafe({
553
+ value
554
+ }, 'path-or-secret'), /absolute path or secret/);
555
+ assert.doesNotThrow(()=>assertPublicArtifactSafe({
556
+ github: 'https://github.com/org/repo',
557
+ huggingface: 'https://huggingface.co/datasets/org/name'
558
+ }, 'public urls'));
559
+ for (const key of [
560
+ 'source_fence_passed',
561
+ 'integrity_passed'
562
+ ]){
563
+ const syntheticFlags = JSON.parse(JSON.stringify(load('data/evaluation/known-good.json')));
564
+ const run = syntheticFlags.run_receipts[0];
565
+ const qualification = run.qualification;
566
+ qualification[key] = true;
567
+ assert.equal(qualification[key === 'source_fence_passed' ? 'integrity_passed' : 'source_fence_passed'], false);
568
+ assert.throws(()=>parseEvalRunReceipt(run), /synthetic fixture/);
569
+ }
570
+ console.log(`canonical sha256: ${digest1}`);
571
+ console.log(`evaluation-contract tests: ${registry.length} registry, ${pairs.length} test-only matched pair, ${vectors.length} negative vectors green`);
572
+
573
+
574
+ //# sourceURL=ts/scripts/evaluation-contract-tests.ts
@@ -0,0 +1,35 @@
1
+ import process from 'node:process';
2
+ import { homedir } from 'node:os';
3
+ import { join, dirname } from 'node:path';
4
+ import { fileURLToPath } from 'node:url';
5
+ import { DEFAULT_RELEASE_BASE, installExtensionFromGithub } from '../capabilities/session/extension-distribution.js';
6
+ function argValue(argv, name) {
7
+ const i = argv.indexOf(name);
8
+ return i >= 0 ? argv[i + 1] : undefined;
9
+ }
10
+ const argv = process.argv.slice(2);
11
+ const repoRoot = join(dirname(fileURLToPath(import.meta.url)), '..', '..');
12
+ const dest = argValue(argv, '--dest') ?? join(homedir(), '.gotry', 'extension');
13
+ const sourceDir = argValue(argv, '--source-dir') ?? join(repoRoot, 'extension');
14
+ const releaseBase = argValue(argv, '--release-base') ?? DEFAULT_RELEASE_BASE;
15
+ const checkOnly = argv.includes('--check-only');
16
+ try {
17
+ const r = await installExtensionFromGithub({
18
+ destDir: dest,
19
+ pinnedSourceDir: sourceDir,
20
+ releaseBase,
21
+ checkOnly
22
+ });
23
+ process.stdout.write(`${JSON.stringify(r)}\n`);
24
+ process.exit(r.ok ? 0 : 2);
25
+ } catch (e) {
26
+ process.stdout.write(`${JSON.stringify({
27
+ ok: false,
28
+ action: 'fallback-bundled',
29
+ error: `CLI 异常 ${e.message}`
30
+ })}\n`);
31
+ process.exit(2);
32
+ }
33
+
34
+
35
+ //# sourceURL=ts/scripts/extension-distribution-cli.ts