@danceiny/gotry 0.0.1-rc.16 → 0.0.1-rc.18
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +161 -116
- package/README.zh-CN.md +173 -144
- package/bin/gotry-booking-copilot.js +53 -0
- package/bin/gotry-bootstrap.js +113 -66
- package/bin/gotry-inner.js +436 -70
- package/bin/gotry-runtime-resolution.d.ts +27 -0
- package/bin/gotry-runtime-resolution.js +50 -0
- package/bin/gotry.js +1 -1
- package/cordis.gotry-patch.yml +5 -1
- package/dist/capabilities/agent-reach-deep.js +1 -1
- package/dist/capabilities/agent-reach.js +1 -1
- package/dist/capabilities/anything.js +1 -1
- package/dist/capabilities/artifacts.js +1 -1
- package/dist/capabilities/effect.js +1 -1
- package/dist/capabilities/fact-log.js +1 -1
- package/dist/capabilities/flyai.js +20 -6
- package/dist/capabilities/hbcli.js +1 -1
- package/dist/capabilities/incident-log.js +1 -1
- package/dist/capabilities/model-override.js +18 -0
- package/dist/capabilities/opensky.js +1 -1
- package/dist/capabilities/resilience.js +1 -1
- package/dist/capabilities/session/action-cache.js +1 -1
- package/dist/capabilities/session/adapters/ctrip-flight.js +1 -1
- package/dist/capabilities/session/adapters/meituan-local.js +1 -1
- package/dist/capabilities/session/benchmark.js +1 -1
- package/dist/capabilities/session/extension-bridge.js +19 -6
- package/dist/capabilities/session/extension-channel.js +3 -3
- package/dist/capabilities/session/extension-distribution.js +234 -0
- package/dist/capabilities/session/extract.js +1 -1
- package/dist/capabilities/session/golden-score.js +92 -0
- package/dist/capabilities/session/health-watch.js +1 -1
- package/dist/capabilities/session/read-guard.js +1 -1
- package/dist/capabilities/session/static-flight-golden.js +137 -0
- package/dist/capabilities/session/transport.js +1 -1
- package/dist/capabilities/session/wizard.js +21 -251
- package/dist/capabilities/session-consent.js +1 -1
- package/dist/capabilities/session-login.js +11 -5
- package/dist/capabilities/session-search.js +46 -4
- package/dist/capabilities/weather.js +168 -46
- package/dist/data/session-golden-20.json +25 -0
- package/dist/data/sf-golden-manifest.json +102 -0
- package/dist/data/sf-static-routes.json +91 -0
- package/dist/scripts/action-cache-tests.js +1 -1
- package/dist/scripts/agent-planning-budget-e2e.js +227 -0
- package/dist/scripts/agent-planning-budget-tests.js +173 -0
- package/dist/scripts/agent-reach-deep-tests.js +1 -1
- package/dist/scripts/agent-reach-tests.js +1 -1
- package/dist/scripts/agent-reach-wrapper-tests.js +1 -1
- package/dist/scripts/anything-tests.js +1 -1
- package/dist/scripts/async-collect.js +1 -1
- package/dist/scripts/benchmark-environment-bridge-e2e.js +981 -0
- package/dist/scripts/benchmark-environment-bridge-tests.js +2544 -0
- package/dist/scripts/booking-copilot-availability-ledger-binding-tests.js +335 -0
- package/dist/scripts/booking-copilot-availability-policy-v2-tests.js +1355 -0
- package/dist/scripts/booking-copilot-bin-proof-tests.js +89 -0
- package/dist/scripts/booking-copilot-crossrepo-fixture-server.js +113 -0
- package/dist/scripts/booking-copilot-dsh-core-proof-tests.js +356 -0
- package/dist/scripts/booking-copilot-dsh-planner-proof-tests.js +557 -0
- package/dist/scripts/booking-copilot-dsh-plugin-proof-tests.js +140 -0
- package/dist/scripts/booking-copilot-event-sequence-concurrency-proof-tests.js +202 -0
- package/dist/scripts/booking-copilot-gap-code-contract-proof-tests.js +102 -0
- package/dist/scripts/booking-copilot-operation-ledger-concurrency-proof-tests.js +374 -0
- package/dist/scripts/booking-copilot-receipt-ledger-concurrency-proof-tests.js +447 -0
- package/dist/scripts/booking-copilot-runtime-proof-tests.js +205 -0
- package/dist/scripts/booking-copilot-server-proof-tests.js +317 -0
- package/dist/scripts/booking-copilot-startup-proof-tests.js +283 -0
- package/dist/scripts/booking-copilot-v2-runtime-proof-tests.js +3935 -0
- package/dist/scripts/booking-saga-tests.js +1 -1
- package/dist/scripts/booking-surface-contract-proof-tests.js +393 -0
- package/dist/scripts/booking-surface-v2-contract-proof-tests.js +1174 -0
- package/dist/scripts/bootstrap-tests.js +45 -39
- package/dist/scripts/build-changelog.js +1 -1
- package/dist/scripts/changelog-tests.js +1 -1
- package/dist/scripts/companion-tests.js +1 -1
- package/dist/scripts/diff-test.js +1 -1
- package/dist/scripts/dsh-runtime-closure-tests.js +225 -0
- package/dist/scripts/dsh-runtime-closure.js +150 -0
- package/dist/scripts/effect-tests.js +1 -1
- package/dist/scripts/engine-run.js +1 -1
- package/dist/scripts/engine-tests.js +1 -1
- package/dist/scripts/evaluation-cadence-tests.js +347 -0
- package/dist/scripts/evaluation-contract-tests.js +574 -0
- package/dist/scripts/extension-distribution-cli.js +35 -0
- package/dist/scripts/extension-distribution-tests.js +341 -0
- package/dist/scripts/extension-tests.js +128 -23
- package/dist/scripts/fact-gate-tests.js +1 -1
- package/dist/scripts/flyai-tests.js +1 -1
- package/dist/scripts/hbcli-e2e-tests.js +1 -1
- package/dist/scripts/hbcli-tests.js +1 -1
- package/dist/scripts/health-watch-cli.js +1 -1
- package/dist/scripts/i18n-tests.js +1 -1
- package/dist/scripts/incident-tests.js +1 -1
- package/dist/scripts/journey-tests.js +1 -1
- package/dist/scripts/ledger-tests.js +1 -1
- package/dist/scripts/ledger-workflow-crash.js +1 -1
- package/dist/scripts/memory-capture-tests.js +1 -1
- package/dist/scripts/memory-decay-tests.js +1 -1
- package/dist/scripts/memory-metrics.js +1 -1
- package/dist/scripts/memory-value-report.js +1 -1
- package/dist/scripts/model-override-e2e.js +176 -0
- package/dist/scripts/nightly-evidence-tests.js +1 -1
- package/dist/scripts/nightly-evidence.js +1 -1
- package/dist/scripts/nudge-digest.js +1 -1
- package/dist/scripts/onboarding-tests.js +21 -53
- package/dist/scripts/opensky-check.js +1 -1
- package/dist/scripts/opensky-tests.js +1 -1
- package/dist/scripts/pnpm-dsh-closure-proof.js +20 -0
- package/dist/scripts/price-drift-tests.js +1 -1
- package/dist/scripts/price-drift-watch.js +1 -1
- package/dist/scripts/probe-poi-tests.js +1 -1
- package/dist/scripts/product-metrics.js +1 -1
- package/dist/scripts/publish-preverify.js +40 -4
- package/dist/scripts/realtime-pricing-tests.js +1 -1
- package/dist/scripts/replay-async.js +1 -1
- package/dist/scripts/replay-real.js +1 -1
- package/dist/scripts/replay.js +1 -1
- package/dist/scripts/session-attach-diagnose.js +1 -1
- package/dist/scripts/session-attach-poc.js +1 -1
- package/dist/scripts/session-benchmark.js +1 -1
- package/dist/scripts/session-extract-tests.js +1 -1
- package/dist/scripts/session-login.js +1 -1
- package/dist/scripts/session-tests.js +70 -20
- package/dist/scripts/sf-live-benchmark.js +338 -0
- package/dist/scripts/sf-live-cli-tests.js +21 -0
- package/dist/scripts/sf-soft-score-tests.js +108 -0
- package/dist/scripts/sf-summary.js +93 -0
- package/dist/scripts/skeleton-check.js +1 -1
- package/dist/scripts/skeleton-integration-test.js +1 -1
- package/dist/scripts/skills-contract-tests.js +1 -1
- package/dist/scripts/smoke-session-gate-tests.js +29 -0
- package/dist/scripts/smoke.js +79 -33
- package/dist/scripts/state-cli-tests.js +1 -1
- package/dist/scripts/state-cli.js +1 -1
- package/dist/scripts/static-golden-tests.js +299 -0
- package/dist/scripts/time-eval-tests.js +1 -1
- package/dist/scripts/travel-timeline-tests.js +1 -1
- package/dist/scripts/unified-tests.js +1 -1
- package/dist/scripts/weather-tests.js +694 -44
- package/dist/scripts/z3-race-tests.js +1 -1
- package/dist/src/artifact-gate.js +1 -1
- package/dist/src/benchmark-agent-conformance.js +370 -0
- package/dist/src/benchmark-environment-bridge.js +384 -0
- package/dist/src/benchmark-headless-child-diagnostics.js +173 -0
- package/dist/src/benchmark-tool-isolation.js +124 -0
- package/dist/src/bookable-facts.js +1 -1
- package/dist/src/booking-saga.js +1 -1
- package/dist/src/booking-surface/availability-policy-v2.js +830 -0
- package/dist/src/booking-surface/canonical-schema.js +113 -0
- package/dist/src/booking-surface/contracts-v2.js +89 -0
- package/dist/src/booking-surface/contracts.js +46 -0
- package/dist/src/booking-surface/dsh-planner.js +453 -0
- package/dist/src/booking-surface/dsh-plugin.js +93 -0
- package/dist/src/booking-surface/error-codes.js +94 -0
- package/dist/src/booking-surface/index.js +15 -0
- package/dist/src/booking-surface/profile.js +68 -0
- package/dist/src/booking-surface/runtime-v2.js +1771 -0
- package/dist/src/booking-surface/runtime.js +351 -0
- package/dist/src/booking-surface/server-v2.js +334 -0
- package/dist/src/booking-surface/server.js +302 -0
- package/dist/src/booking-surface/startup.js +159 -0
- package/dist/src/booking-surface/validation-v2.js +319 -0
- package/dist/src/booking-surface/validation.js +809 -0
- package/dist/src/bridge.js +1 -1
- package/dist/src/companions.js +1 -1
- package/dist/src/contracts.js +1 -1
- package/dist/src/dsh-llm.js +1 -1
- package/dist/src/engine.js +1 -1
- package/dist/src/evaluation-cadence.js +234 -0
- package/dist/src/evaluation-contracts.js +906 -0
- package/dist/src/i18n.js +1 -1
- package/dist/src/index.js +50 -19
- package/dist/src/journey.js +1 -1
- package/dist/src/loop.js +1 -1
- package/dist/src/memory-capture.js +1 -1
- package/dist/src/memory-decay.js +1 -1
- package/dist/src/memory-utility.js +1 -1
- package/dist/src/mock-llm.js +1 -1
- package/dist/src/model.js +1 -1
- package/dist/src/realtime-pricing.js +1 -1
- package/dist/src/slot-spec.js +1 -1
- package/dist/src/state-ledger.js +2 -1
- package/dist/src/time-anchor.js +1 -1
- package/dist/src/tool-budget.js +136 -0
- package/dist/src/tool-packet.js +1 -1
- package/dist/src/travel-slots.js +1 -1
- package/dist/src/travel-timeline.js +1 -1
- package/dist/src/unified.js +1 -1
- package/dist/src/wish-pool.js +1 -1
- package/dist/src/z3-shared.js +1 -1
- package/extension/README.md +31 -7
- package/package.json +286 -11
- package/schemas/booking.surface.v1.schema.json +927 -0
- package/schemas/booking.surface.v2.schema.json +61 -0
- package/ts/capabilities/flyai.ts +16 -3
- package/ts/capabilities/session/extension-bridge.ts +37 -14
- package/ts/capabilities/session/extension-channel.ts +6 -3
- package/ts/capabilities/session/extension-distribution.ts +264 -0
- package/ts/capabilities/session/golden-score.ts +139 -0
- package/ts/capabilities/session/health-watch.ts +1 -1
- package/ts/capabilities/session/static-flight-golden.ts +209 -0
- package/ts/capabilities/session/wizard.ts +34 -176
- package/ts/capabilities/session-login.ts +15 -5
- package/ts/capabilities/session-search.ts +40 -3
- package/ts/capabilities/weather.ts +141 -52
- package/ts/package.json +3 -3
- package/ts/src/benchmark-agent-conformance.ts +448 -0
- package/ts/src/benchmark-environment-bridge.ts +348 -0
- package/ts/src/benchmark-headless-child-diagnostics.ts +184 -0
- package/ts/src/benchmark-tool-isolation.ts +166 -0
- package/ts/src/booking-surface/availability-policy-v2.ts +523 -0
- package/ts/src/booking-surface/canonical-schema.js +113 -0
- package/ts/src/booking-surface/contracts-v2.ts +118 -0
- package/ts/src/booking-surface/contracts.ts +380 -0
- package/ts/src/booking-surface/dsh-planner.ts +452 -0
- package/ts/src/booking-surface/dsh-plugin.js +93 -0
- package/ts/src/booking-surface/error-codes.ts +101 -0
- package/ts/src/booking-surface/index.ts +12 -0
- package/ts/src/booking-surface/profile.ts +42 -0
- package/ts/src/booking-surface/runtime-v2.ts +1466 -0
- package/ts/src/booking-surface/runtime.ts +483 -0
- package/ts/src/booking-surface/server-v2.ts +247 -0
- package/ts/src/booking-surface/server.ts +324 -0
- package/ts/src/booking-surface/startup.ts +196 -0
- package/ts/src/booking-surface/validation-v2.ts +205 -0
- package/ts/src/booking-surface/validation.ts +453 -0
- package/ts/src/index.ts +64 -11
- package/ts/src/state-ledger.ts +1 -0
- package/ts/src/tool-budget.ts +165 -0
- package/dist/scripts/wizard-bootstrap.js +0 -32
|
@@ -0,0 +1,574 @@
|
|
|
1
|
+
import assert from 'node:assert/strict';
|
|
2
|
+
import { createHash } from 'node:crypto';
|
|
3
|
+
import { readFileSync } from 'node:fs';
|
|
4
|
+
import { BENCHMARK_IDS, applyMutationVector, assertPublicArtifactSafe, deriveMatchedPairs, evaluationFingerprint, parseBenchmarkRegistry, parseEvalCase, parseEvalFailureCluster, parseEvalRunReceipt, parseEvaluationFoundation, parseMutationVectors, stableEvaluationJson } from '../src/evaluation-contracts.js';
|
|
5
|
+
const load = (file)=>JSON.parse(readFileSync(file, 'utf8'));
|
|
6
|
+
const registry = parseBenchmarkRegistry(load('data/evaluation/benchmark-registry.json'));
|
|
7
|
+
assert.deepEqual(registry.map((item)=>item.benchmark_id), [
|
|
8
|
+
...BENCHMARK_IDS
|
|
9
|
+
]);
|
|
10
|
+
assert.equal(registry.length, 7);
|
|
11
|
+
assert.deepEqual(Object.fromEntries(registry.map((item)=>[
|
|
12
|
+
item.benchmark_id,
|
|
13
|
+
item.native_metrics.values.map((metric)=>metric.receipt_key)
|
|
14
|
+
])), {
|
|
15
|
+
trek: [
|
|
16
|
+
'task_perfect_feasible',
|
|
17
|
+
'task_perfect_infeasible',
|
|
18
|
+
'cat_efficiency'
|
|
19
|
+
],
|
|
20
|
+
travelplanner: [
|
|
21
|
+
'commonsense_micro_pass_rate',
|
|
22
|
+
'commonsense_macro_pass_rate',
|
|
23
|
+
'hard_micro_pass_rate',
|
|
24
|
+
'hard_macro_pass_rate',
|
|
25
|
+
'final_pass_rate'
|
|
26
|
+
],
|
|
27
|
+
chinatravel: [
|
|
28
|
+
'epr_micro',
|
|
29
|
+
'epr_macro',
|
|
30
|
+
'c_lpr',
|
|
31
|
+
'fpr',
|
|
32
|
+
'dav',
|
|
33
|
+
'att',
|
|
34
|
+
'ddr',
|
|
35
|
+
'overall_score'
|
|
36
|
+
],
|
|
37
|
+
travelbench: [
|
|
38
|
+
'reasoning_planning_score',
|
|
39
|
+
'summarization_extraction_score',
|
|
40
|
+
'presentation_score',
|
|
41
|
+
'user_interaction_score',
|
|
42
|
+
'average_score',
|
|
43
|
+
'unsolved_accuracy'
|
|
44
|
+
],
|
|
45
|
+
tau2: [
|
|
46
|
+
'avg_reward',
|
|
47
|
+
'pass_hat_1'
|
|
48
|
+
],
|
|
49
|
+
locomo: [
|
|
50
|
+
'qa_f1'
|
|
51
|
+
],
|
|
52
|
+
bfcl: [
|
|
53
|
+
'category_accuracy',
|
|
54
|
+
'overall_accuracy'
|
|
55
|
+
]
|
|
56
|
+
});
|
|
57
|
+
const registryFacts = Object.fromEntries(registry.map((entry)=>[
|
|
58
|
+
entry.benchmark_id,
|
|
59
|
+
{
|
|
60
|
+
owner: new URL(entry.provenance.official_entry.url).pathname.split('/')[1],
|
|
61
|
+
pins: Object.entries(entry.provenance).map(([kind, pin])=>`${kind}|${pin.url}|${pin.revision.kind}|${pin.revision.value ?? 'null'}|${pin.source_scope}`),
|
|
62
|
+
rights: Object.entries(entry.license.upstream_rights).map(([kind, right])=>`${kind}|${right.value}|${right.determination}|${right.source_url}`),
|
|
63
|
+
metrics: entry.native_metrics.values.map((metric)=>`${metric.receipt_key}|${metric.upstream_label}|${metric.scope}|${metric.source_url}`)
|
|
64
|
+
}
|
|
65
|
+
]));
|
|
66
|
+
assert.deepEqual(registryFacts, {
|
|
67
|
+
"trek": {
|
|
68
|
+
"owner": "TonyQJH",
|
|
69
|
+
"pins": [
|
|
70
|
+
"official_entry|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/README.md|git_commit|6ceb4ebb2debd69c5c7c4ba34b5b17524756912b|official definition",
|
|
71
|
+
"data|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/tree/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/api/data/v2|git_commit|6ceb4ebb2debd69c5c7c4ba34b5b17524756912b|task data v2",
|
|
72
|
+
"evaluator|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/scoring.py|git_commit|6ceb4ebb2debd69c5c7c4ba34b5b17524756912b|official nine-dimension scoring implementation"
|
|
73
|
+
],
|
|
74
|
+
"rights": [
|
|
75
|
+
"code|MIT|declared|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/LICENSE",
|
|
76
|
+
"data|CC-BY-4.0|declared|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/README.md#data-and-license",
|
|
77
|
+
"evaluator|MIT|declared|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/LICENSE"
|
|
78
|
+
],
|
|
79
|
+
"metrics": [
|
|
80
|
+
"task_perfect_feasible|task_perfect_feasible|task-perfect feasible|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/scoring.py",
|
|
81
|
+
"task_perfect_infeasible|task_perfect_infeasible|task-perfect infeasible|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/scoring.py",
|
|
82
|
+
"cat_efficiency|cat_efficiency|category efficiency|https://github.com/TonyQJH/TREK-A-Travel-Reasoning-and-Evaluation-Kit-for-LLM-Agents-in-Complex-Trip-Planning/blob/6ceb4ebb2debd69c5c7c4ba34b5b17524756912b/scoring.py"
|
|
83
|
+
]
|
|
84
|
+
},
|
|
85
|
+
"travelplanner": {
|
|
86
|
+
"owner": "OSU-NLP-Group",
|
|
87
|
+
"pins": [
|
|
88
|
+
"official_entry|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/README.md|git_commit|e52c87f4ac348a3410c46dc3553c519db5ec5e23|official definition",
|
|
89
|
+
"data|https://huggingface.co/datasets/osunlp/TravelPlanner/tree/8736504ecfc31b7f8b7e40122873c337e83fff7c|git_commit|8736504ecfc31b7f8b7e40122873c337e83fff7c|official Hugging Face dataset git revision",
|
|
90
|
+
"evaluator|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py|git_commit|e52c87f4ac348a3410c46dc3553c519db5ec5e23|official evaluator"
|
|
91
|
+
],
|
|
92
|
+
"rights": [
|
|
93
|
+
"code|MIT|declared|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/LICENSE",
|
|
94
|
+
"data|CC-BY-4.0|declared|https://huggingface.co/datasets/osunlp/TravelPlanner/blob/8736504ecfc31b7f8b7e40122873c337e83fff7c/README.md",
|
|
95
|
+
"evaluator|MIT|declared|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/LICENSE"
|
|
96
|
+
],
|
|
97
|
+
"metrics": [
|
|
98
|
+
"commonsense_micro_pass_rate|Commonsense Constraint Micro Pass Rate|commonsense micro|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py",
|
|
99
|
+
"commonsense_macro_pass_rate|Commonsense Constraint Macro Pass Rate|commonsense macro|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py",
|
|
100
|
+
"hard_micro_pass_rate|Hard Constraint Micro Pass Rate|hard-constraint micro|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py",
|
|
101
|
+
"hard_macro_pass_rate|Hard Constraint Macro Pass Rate|hard-constraint macro|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py",
|
|
102
|
+
"final_pass_rate|Final Pass Rate|complete itinerary|https://github.com/OSU-NLP-Group/TravelPlanner/blob/e52c87f4ac348a3410c46dc3553c519db5ec5e23/evaluation/eval.py"
|
|
103
|
+
]
|
|
104
|
+
},
|
|
105
|
+
"chinatravel": {
|
|
106
|
+
"owner": "chinatravel-competition",
|
|
107
|
+
"pins": [
|
|
108
|
+
"official_entry|https://github.com/chinatravel-competition/IJCAI2026/blob/49d02bc322dda7ffbf53dfb7c3d2ced6b4bd4e8b/index.html|git_commit|49d02bc322dda7ffbf53dfb7c3d2ced6b4bd4e8b|official competition entry",
|
|
109
|
+
"data|https://github.com/chinatravel-competition/IJCAI2026/blob/49d02bc322dda7ffbf53dfb7c3d2ced6b4bd4e8b/TPC_IJCAI_2026_phase2_familiar_100_data.zip|git_commit|49d02bc322dda7ffbf53dfb7c3d2ced6b4bd4e8b|public familiar-track archive",
|
|
110
|
+
"evaluator|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py|git_commit|b071db251905b14002ec98e8b36afca7b6d6cd04|official TPC evaluator implementation"
|
|
111
|
+
],
|
|
112
|
+
"rights": [
|
|
113
|
+
"code|not_separately_declared|not_separately_declared|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/README.md",
|
|
114
|
+
"data|CC-BY-NC-SA-4.0|declared|https://huggingface.co/datasets/LAMDA-NeSy/ChinaTravel/blob/44d5dbf3bba26bdf9a212c3e76d3242b67f0d349/README.md",
|
|
115
|
+
"evaluator|not_separately_declared|not_separately_declared|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/README.md"
|
|
116
|
+
],
|
|
117
|
+
"metrics": [
|
|
118
|
+
"epr_micro|EPR-micro|element pass micro|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
|
|
119
|
+
"epr_macro|EPR-macro|element pass macro|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
|
|
120
|
+
"c_lpr|C-LPR|constraint-level pass|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
|
|
121
|
+
"fpr|FPR|final pass|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
|
|
122
|
+
"dav|DAV|Daily Average Attractions Visited|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
|
|
123
|
+
"att|ATT|Averaged Transportation Time|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
|
|
124
|
+
"ddr|DDR|Daily Dining Recommendations|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py",
|
|
125
|
+
"overall_score|Overall Score|weighted overall score|https://github.com/LAMDA-NeSy/ChinaTravel/blob/b071db251905b14002ec98e8b36afca7b6d6cd04/eval_tpc.py"
|
|
126
|
+
]
|
|
127
|
+
},
|
|
128
|
+
"travelbench": {
|
|
129
|
+
"owner": "small-xiangcheng",
|
|
130
|
+
"pins": [
|
|
131
|
+
"official_entry|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/README.md|git_commit|445a29d9a9b6457fc95fe647c532b6b79e21c43f|official definition",
|
|
132
|
+
"data|https://github.com/small-xiangcheng/TravelBench/tree/445a29d9a9b6457fc95fe647c532b6b79e21c43f/datas|git_commit|445a29d9a9b6457fc95fe647c532b6b79e21c43f|official data directory",
|
|
133
|
+
"evaluator|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py|git_commit|445a29d9a9b6457fc95fe647c532b6b79e21c43f|official evaluator"
|
|
134
|
+
],
|
|
135
|
+
"rights": [
|
|
136
|
+
"code|MIT|declared|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/LICENSE",
|
|
137
|
+
"data|CC-BY-NC-4.0|declared|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/datas/LICENSE",
|
|
138
|
+
"evaluator|MIT|declared|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/LICENSE"
|
|
139
|
+
],
|
|
140
|
+
"metrics": [
|
|
141
|
+
"reasoning_planning_score|reasoning_planning_score|reasoning and planning|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py",
|
|
142
|
+
"summarization_extraction_score|summarization_extraction_score|summarization and extraction|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py",
|
|
143
|
+
"presentation_score|presentation_score|presentation|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py",
|
|
144
|
+
"user_interaction_score|user_interaction_score|user interaction|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py",
|
|
145
|
+
"average_score|average_score|average across dimensions|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate.py",
|
|
146
|
+
"unsolved_accuracy|unsolved_accuracy|unsolved cases|https://github.com/small-xiangcheng/TravelBench/blob/445a29d9a9b6457fc95fe647c532b6b79e21c43f/travelbench/evaluation/evaluate_unsolved.py"
|
|
147
|
+
]
|
|
148
|
+
},
|
|
149
|
+
"tau2": {
|
|
150
|
+
"owner": "sierra-research",
|
|
151
|
+
"pins": [
|
|
152
|
+
"official_entry|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/README.md|git_commit|a2c024725189473d2d7cea3a5cfdbcc67478e41f|official definition",
|
|
153
|
+
"data|https://github.com/sierra-research/tau2-bench/tree/a2c024725189473d2d7cea3a5cfdbcc67478e41f/data/tau2|git_commit|a2c024725189473d2d7cea3a5cfdbcc67478e41f|official data",
|
|
154
|
+
"evaluator|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/src/tau2/metrics/agent_metrics.py|git_commit|a2c024725189473d2d7cea3a5cfdbcc67478e41f|official agent metrics implementation"
|
|
155
|
+
],
|
|
156
|
+
"rights": [
|
|
157
|
+
"code|MIT|declared|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/LICENSE",
|
|
158
|
+
"data|not_separately_declared|not_separately_declared|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/README.md",
|
|
159
|
+
"evaluator|MIT|declared|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/LICENSE"
|
|
160
|
+
],
|
|
161
|
+
"metrics": [
|
|
162
|
+
"avg_reward|avg_reward|average reward|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/src/tau2/metrics/agent_metrics.py",
|
|
163
|
+
"pass_hat_1|pass^1|mean task pass^1|https://github.com/sierra-research/tau2-bench/blob/a2c024725189473d2d7cea3a5cfdbcc67478e41f/src/tau2/metrics/agent_metrics.py"
|
|
164
|
+
]
|
|
165
|
+
},
|
|
166
|
+
"locomo": {
|
|
167
|
+
"owner": "snap-research",
|
|
168
|
+
"pins": [
|
|
169
|
+
"official_entry|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/README.MD|git_commit|3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376|official definition",
|
|
170
|
+
"data|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/data/locomo10.json|git_commit|3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376|official data",
|
|
171
|
+
"evaluator|https://github.com/snap-research/locomo/tree/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/task_eval|git_commit|3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376|official task evaluation directory"
|
|
172
|
+
],
|
|
173
|
+
"rights": [
|
|
174
|
+
"code|CC-BY-NC-4.0|declared|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/LICENSE.txt",
|
|
175
|
+
"data|CC-BY-NC-4.0|declared|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/LICENSE.txt",
|
|
176
|
+
"evaluator|CC-BY-NC-4.0|declared|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/LICENSE.txt"
|
|
177
|
+
],
|
|
178
|
+
"metrics": [
|
|
179
|
+
"qa_f1|F1|question-answering token F1|https://github.com/snap-research/locomo/blob/3eb6f2c585f5e1699204e3c3bdf7adc5c28cb376/task_eval/evaluate_qa.py"
|
|
180
|
+
]
|
|
181
|
+
},
|
|
182
|
+
"bfcl": {
|
|
183
|
+
"owner": "ShishirPatil",
|
|
184
|
+
"pins": [
|
|
185
|
+
"official_entry|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/README.md|git_commit|6ea57973c7a6097fd7c5915698c54c17c5b1b6c8|official BFCL definition",
|
|
186
|
+
"data|https://github.com/ShishirPatil/gorilla/tree/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/bfcl_eval/data|git_commit|6ea57973c7a6097fd7c5915698c54c17c5b1b6c8|BFCL data",
|
|
187
|
+
"evaluator|https://github.com/ShishirPatil/gorilla/tree/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/bfcl_eval|git_commit|6ea57973c7a6097fd7c5915698c54c17c5b1b6c8|BFCL evaluator"
|
|
188
|
+
],
|
|
189
|
+
"rights": [
|
|
190
|
+
"code|Apache-2.0|declared|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/LICENSE",
|
|
191
|
+
"data|Apache-2.0|declared|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/README.md",
|
|
192
|
+
"evaluator|Apache-2.0|declared|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/LICENSE"
|
|
193
|
+
],
|
|
194
|
+
"metrics": [
|
|
195
|
+
"category_accuracy|accuracy|per test_category|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/bfcl_eval/eval_checker/eval_runner_helper.py",
|
|
196
|
+
"overall_accuracy|Overall Acc|overall accuracy|https://github.com/ShishirPatil/gorilla/blob/6ea57973c7a6097fd7c5915698c54c17c5b1b6c8/berkeley-function-call-leaderboard/README.md"
|
|
197
|
+
]
|
|
198
|
+
}
|
|
199
|
+
});
|
|
200
|
+
const raw = load('data/evaluation/known-good.json');
|
|
201
|
+
const diagnostic = parseEvaluationFoundation({
|
|
202
|
+
registry,
|
|
203
|
+
...raw
|
|
204
|
+
});
|
|
205
|
+
assert.equal(diagnostic.cases.length, 1);
|
|
206
|
+
assert.equal(diagnostic.run_receipts.length, 1);
|
|
207
|
+
assert.equal(diagnostic.run_receipts[0].evidence_kind, 'synthetic_fixture');
|
|
208
|
+
assert.equal(diagnostic.run_receipts[0].pairing, null);
|
|
209
|
+
assert.equal(diagnostic.run_receipts[0].qualification.official_result, false);
|
|
210
|
+
const emptyResolver = {
|
|
211
|
+
resolve: ()=>undefined
|
|
212
|
+
};
|
|
213
|
+
assert.deepEqual(deriveMatchedPairs(diagnostic, emptyResolver), []);
|
|
214
|
+
for (const item of diagnostic.cases)parseEvalCase(item);
|
|
215
|
+
for (const item of diagnostic.run_receipts)parseEvalRunReceipt(item);
|
|
216
|
+
for (const item of diagnostic.failure_clusters)parseEvalFailureCluster(item);
|
|
217
|
+
for (const item of [
|
|
218
|
+
...diagnostic.cases,
|
|
219
|
+
...diagnostic.run_receipts,
|
|
220
|
+
...diagnostic.failure_clusters
|
|
221
|
+
])assertPublicArtifactSafe(item, 'repository fixture');
|
|
222
|
+
const trek = registry.find((item)=>item.benchmark_id === 'trek');
|
|
223
|
+
const evalCase = diagnostic.cases[0];
|
|
224
|
+
const observedCase = {
|
|
225
|
+
...evalCase,
|
|
226
|
+
input_ref: {
|
|
227
|
+
...evalCase.input_ref,
|
|
228
|
+
kind: 'external_opaque_reference'
|
|
229
|
+
}
|
|
230
|
+
};
|
|
231
|
+
const seed = diagnostic.run_receipts[0];
|
|
232
|
+
const controls = {
|
|
233
|
+
...seed.controls,
|
|
234
|
+
case_set_sha256: evaluationFingerprint([
|
|
235
|
+
observedCase
|
|
236
|
+
]),
|
|
237
|
+
scorer_sha256: evaluationFingerprint(observedCase.scorer_revision),
|
|
238
|
+
source_fence_sha256: evaluationFingerprint(trek.source_fence),
|
|
239
|
+
official_evaluator_sha256: evaluationFingerprint(trek.provenance.evaluator)
|
|
240
|
+
};
|
|
241
|
+
const observed = (role)=>{
|
|
242
|
+
const run = {
|
|
243
|
+
...seed,
|
|
244
|
+
run_id: `run:trek:${role}-test-only`,
|
|
245
|
+
evidence_kind: 'observed_external',
|
|
246
|
+
gotry_sha: role === 'baseline' ? '1111111111111111111111111111111111111111' : '2222222222222222222222222222222222222222',
|
|
247
|
+
pairing: {
|
|
248
|
+
pair_id: 'pair:trek:test-only',
|
|
249
|
+
role,
|
|
250
|
+
counterpart_run_id: `run:trek:${role === 'baseline' ? 'treatment' : 'baseline'}-test-only`
|
|
251
|
+
},
|
|
252
|
+
model: {
|
|
253
|
+
provider: 'test-provider',
|
|
254
|
+
model: 'test-model'
|
|
255
|
+
},
|
|
256
|
+
controls,
|
|
257
|
+
qualification: {
|
|
258
|
+
official_result: true,
|
|
259
|
+
source_fence_passed: true,
|
|
260
|
+
integrity_passed: true,
|
|
261
|
+
evidence_receipts: {
|
|
262
|
+
official_evaluator_output_sha256: null,
|
|
263
|
+
source_fence_audit_sha256: null,
|
|
264
|
+
integrity_audit_sha256: null
|
|
265
|
+
}
|
|
266
|
+
},
|
|
267
|
+
experiment: {
|
|
268
|
+
changed_variables: role === 'baseline' ? [] : [
|
|
269
|
+
'gotry_sha'
|
|
270
|
+
],
|
|
271
|
+
candidate_sha256: evaluationFingerprint({
|
|
272
|
+
treatment_variable: 'gotry_sha',
|
|
273
|
+
gotry_sha: role === 'baseline' ? '1111111111111111111111111111111111111111' : '2222222222222222222222222222222222222222'
|
|
274
|
+
})
|
|
275
|
+
},
|
|
276
|
+
evidence_summary: {
|
|
277
|
+
...seed.evidence_summary,
|
|
278
|
+
fixture_only: false,
|
|
279
|
+
statement: 'test-only observed-external aggregate admission object'
|
|
280
|
+
}
|
|
281
|
+
};
|
|
282
|
+
const { evidence_receipts: _receipts, ...qualification } = run.qualification;
|
|
283
|
+
const bound = {
|
|
284
|
+
...run,
|
|
285
|
+
qualification
|
|
286
|
+
};
|
|
287
|
+
const base = {
|
|
288
|
+
schema_version: 'gotry_eval_evidence_artifact_v0',
|
|
289
|
+
run_id: run.run_id,
|
|
290
|
+
benchmark_id: run.benchmark_id,
|
|
291
|
+
case_id: run.case_id,
|
|
292
|
+
run_binding_sha256: evaluationFingerprint(bound)
|
|
293
|
+
};
|
|
294
|
+
const artifacts = {
|
|
295
|
+
official_evaluator: {
|
|
296
|
+
...base,
|
|
297
|
+
artifact_kind: 'official_evaluator',
|
|
298
|
+
evaluator_sha256: run.controls.official_evaluator_sha256,
|
|
299
|
+
native_metrics_sha256: evaluationFingerprint(run.native_metrics),
|
|
300
|
+
native_metrics: run.native_metrics,
|
|
301
|
+
official_result: true
|
|
302
|
+
},
|
|
303
|
+
source_fence_audit: {
|
|
304
|
+
...base,
|
|
305
|
+
artifact_kind: 'source_fence_audit',
|
|
306
|
+
source_fence_sha256: run.controls.source_fence_sha256,
|
|
307
|
+
input_digest_sha256: observedCase.input_ref.digest_sha256,
|
|
308
|
+
source_fence_passed: true,
|
|
309
|
+
forbidden_field_hits: 0
|
|
310
|
+
},
|
|
311
|
+
integrity_audit: {
|
|
312
|
+
...base,
|
|
313
|
+
artifact_kind: 'integrity_audit',
|
|
314
|
+
integrity_sha256: run.controls.integrity_sha256,
|
|
315
|
+
candidate_sha256: run.experiment.candidate_sha256,
|
|
316
|
+
integrity_passed: true
|
|
317
|
+
}
|
|
318
|
+
};
|
|
319
|
+
run.qualification.evidence_receipts = {
|
|
320
|
+
official_evaluator_output_sha256: evaluationFingerprint(artifacts.official_evaluator),
|
|
321
|
+
source_fence_audit_sha256: evaluationFingerprint(artifacts.source_fence_audit),
|
|
322
|
+
integrity_audit_sha256: evaluationFingerprint(artifacts.integrity_audit)
|
|
323
|
+
};
|
|
324
|
+
return run;
|
|
325
|
+
};
|
|
326
|
+
const countable = parseEvaluationFoundation({
|
|
327
|
+
registry,
|
|
328
|
+
cases: [
|
|
329
|
+
observedCase
|
|
330
|
+
],
|
|
331
|
+
run_receipts: [
|
|
332
|
+
observed('baseline'),
|
|
333
|
+
observed('treatment')
|
|
334
|
+
],
|
|
335
|
+
failure_clusters: [
|
|
336
|
+
{
|
|
337
|
+
...diagnostic.failure_clusters[0],
|
|
338
|
+
run_ids: [
|
|
339
|
+
'run:trek:baseline-test-only',
|
|
340
|
+
'run:trek:treatment-test-only'
|
|
341
|
+
]
|
|
342
|
+
}
|
|
343
|
+
]
|
|
344
|
+
});
|
|
345
|
+
const artifactResolver = {
|
|
346
|
+
resolve (sha256) {
|
|
347
|
+
for (const run of countable.run_receipts){
|
|
348
|
+
const { evidence_receipts: _receipts, ...qualification } = run.qualification;
|
|
349
|
+
const bound = {
|
|
350
|
+
...run,
|
|
351
|
+
qualification
|
|
352
|
+
};
|
|
353
|
+
const base = {
|
|
354
|
+
schema_version: 'gotry_eval_evidence_artifact_v0',
|
|
355
|
+
run_id: run.run_id,
|
|
356
|
+
benchmark_id: run.benchmark_id,
|
|
357
|
+
case_id: run.case_id,
|
|
358
|
+
run_binding_sha256: evaluationFingerprint(bound)
|
|
359
|
+
};
|
|
360
|
+
const artifacts = [
|
|
361
|
+
{
|
|
362
|
+
...base,
|
|
363
|
+
artifact_kind: 'official_evaluator',
|
|
364
|
+
evaluator_sha256: run.controls.official_evaluator_sha256,
|
|
365
|
+
native_metrics_sha256: evaluationFingerprint(run.native_metrics),
|
|
366
|
+
native_metrics: run.native_metrics,
|
|
367
|
+
official_result: true
|
|
368
|
+
},
|
|
369
|
+
{
|
|
370
|
+
...base,
|
|
371
|
+
artifact_kind: 'source_fence_audit',
|
|
372
|
+
source_fence_sha256: run.controls.source_fence_sha256,
|
|
373
|
+
input_digest_sha256: observedCase.input_ref.digest_sha256,
|
|
374
|
+
source_fence_passed: true,
|
|
375
|
+
forbidden_field_hits: 0
|
|
376
|
+
},
|
|
377
|
+
{
|
|
378
|
+
...base,
|
|
379
|
+
artifact_kind: 'integrity_audit',
|
|
380
|
+
integrity_sha256: run.controls.integrity_sha256,
|
|
381
|
+
candidate_sha256: run.experiment.candidate_sha256,
|
|
382
|
+
integrity_passed: true
|
|
383
|
+
}
|
|
384
|
+
];
|
|
385
|
+
const found = artifacts.find((item)=>evaluationFingerprint(item) === sha256);
|
|
386
|
+
if (found) return found;
|
|
387
|
+
}
|
|
388
|
+
return undefined;
|
|
389
|
+
}
|
|
390
|
+
};
|
|
391
|
+
const fingerprintMismatchResolver = {
|
|
392
|
+
resolve (sha256) {
|
|
393
|
+
const artifact = artifactResolver.resolve(sha256);
|
|
394
|
+
if (!artifact) return undefined;
|
|
395
|
+
const changed = structuredClone(artifact);
|
|
396
|
+
if (changed.artifact_kind === 'official_evaluator') changed.official_result = false;
|
|
397
|
+
return changed;
|
|
398
|
+
}
|
|
399
|
+
};
|
|
400
|
+
assert.throws(()=>deriveMatchedPairs(countable, fingerprintMismatchResolver), /fingerprint mismatch/);
|
|
401
|
+
const pairs = deriveMatchedPairs(countable, artifactResolver);
|
|
402
|
+
assert.deepEqual(pairs, [
|
|
403
|
+
{
|
|
404
|
+
schema_version: 'gotry_eval_matched_pair_derived_v0',
|
|
405
|
+
pair_id: 'pair:trek:test-only',
|
|
406
|
+
benchmark_id: 'trek',
|
|
407
|
+
case_id: 'gotry:foundation:case-001',
|
|
408
|
+
baseline_run_id: 'run:trek:baseline-test-only',
|
|
409
|
+
treatment_run_id: 'run:trek:treatment-test-only',
|
|
410
|
+
treatment_variable: 'gotry_sha',
|
|
411
|
+
matched_pair_countable: true
|
|
412
|
+
}
|
|
413
|
+
]);
|
|
414
|
+
const exerciseArtifact = (runIndex, kind, mutate, expected)=>{
|
|
415
|
+
const foundation = structuredClone(countable);
|
|
416
|
+
const run = foundation.run_receipts[runIndex];
|
|
417
|
+
const receiptKey = kind === 'official_evaluator' ? 'official_evaluator_output_sha256' : kind === 'source_fence_audit' ? 'source_fence_audit_sha256' : 'integrity_audit_sha256';
|
|
418
|
+
const original = artifactResolver.resolve(run.qualification.evidence_receipts[receiptKey]);
|
|
419
|
+
const artifact = structuredClone(original);
|
|
420
|
+
mutate(artifact);
|
|
421
|
+
const digest = evaluationFingerprint(artifact);
|
|
422
|
+
run.qualification.evidence_receipts[receiptKey] = digest;
|
|
423
|
+
const resolver = {
|
|
424
|
+
resolve (sha256) {
|
|
425
|
+
return sha256 === digest ? artifact : artifactResolver.resolve(sha256);
|
|
426
|
+
}
|
|
427
|
+
};
|
|
428
|
+
assert.throws(()=>deriveMatchedPairs(foundation, resolver), expected);
|
|
429
|
+
};
|
|
430
|
+
assert.throws(()=>deriveMatchedPairs(countable, {
|
|
431
|
+
resolve: ()=>undefined
|
|
432
|
+
}), /artifact/);
|
|
433
|
+
exerciseArtifact(0, 'official_evaluator', (a)=>{
|
|
434
|
+
a.evaluator_sha256 = 'f'.repeat(64);
|
|
435
|
+
}, /evaluator artifact mismatch/);
|
|
436
|
+
exerciseArtifact(0, 'official_evaluator', (a)=>{
|
|
437
|
+
a.native_metrics = {
|
|
438
|
+
altered: 1
|
|
439
|
+
};
|
|
440
|
+
a.native_metrics_sha256 = evaluationFingerprint(a.native_metrics);
|
|
441
|
+
}, /evaluator artifact mismatch/);
|
|
442
|
+
exerciseArtifact(0, 'source_fence_audit', (a)=>{
|
|
443
|
+
a.input_digest_sha256 = 'f'.repeat(64);
|
|
444
|
+
}, /source-fence artifact mismatch/);
|
|
445
|
+
exerciseArtifact(0, 'source_fence_audit', (a)=>{
|
|
446
|
+
a.source_fence_passed = false;
|
|
447
|
+
}, /source-fence artifact mismatch/);
|
|
448
|
+
exerciseArtifact(0, 'source_fence_audit', (a)=>{
|
|
449
|
+
a.forbidden_field_hits = 1;
|
|
450
|
+
}, /source-fence artifact mismatch/);
|
|
451
|
+
exerciseArtifact(0, 'integrity_audit', (a)=>{
|
|
452
|
+
a.candidate_sha256 = 'f'.repeat(64);
|
|
453
|
+
}, /integrity artifact mismatch/);
|
|
454
|
+
exerciseArtifact(0, 'integrity_audit', (a)=>{
|
|
455
|
+
a.integrity_passed = false;
|
|
456
|
+
}, /integrity artifact mismatch/);
|
|
457
|
+
const treatmentReuse = structuredClone(countable);
|
|
458
|
+
const baselineReceipt = treatmentReuse.run_receipts[0].qualification.evidence_receipts.official_evaluator_output_sha256;
|
|
459
|
+
treatmentReuse.run_receipts[1].qualification.evidence_receipts.official_evaluator_output_sha256 = baselineReceipt;
|
|
460
|
+
assert.throws(()=>deriveMatchedPairs(treatmentReuse, artifactResolver), /fingerprint mismatch|binding mismatch/);
|
|
461
|
+
const canonical1 = stableEvaluationJson(diagnostic);
|
|
462
|
+
const canonical2 = stableEvaluationJson(parseEvaluationFoundation(JSON.parse(canonical1)));
|
|
463
|
+
const digest1 = createHash('sha256').update(canonical1).digest('hex');
|
|
464
|
+
const digest2 = createHash('sha256').update(canonical2).digest('hex');
|
|
465
|
+
assert.equal(canonical1, canonical2);
|
|
466
|
+
assert.equal(digest1, digest2);
|
|
467
|
+
assert.match(digest1, /^[0-9a-f]{64}$/);
|
|
468
|
+
const vectors = parseMutationVectors(load('data/evaluation/known-bad.json'));
|
|
469
|
+
assert.equal(vectors.length, 39);
|
|
470
|
+
for (const vector of vectors){
|
|
471
|
+
const base = vector.foundation_kind === 'countable_test_only' ? countable : diagnostic;
|
|
472
|
+
const mutated = applyMutationVector(base, vector);
|
|
473
|
+
assert.throws(()=>vector.target === 'foundation' ? deriveMatchedPairs(parseEvaluationFoundation(mutated), vector.foundation_kind === 'countable_test_only' ? artifactResolver : emptyResolver) : vector.target === 'registry' ? parseBenchmarkRegistry(mutated) : vector.target === 'case' ? parseEvalCase(mutated) : vector.target === 'run' ? parseEvalRunReceipt(mutated) : parseEvalFailureCluster(mutated), new RegExp(vector.expected_error), vector.id);
|
|
474
|
+
}
|
|
475
|
+
const runAll = readFileSync('../scripts/run-all-tests.sh', 'utf8');
|
|
476
|
+
assert.match(runAll, /=== 46\. Evaluation Phase 0 foundation/);
|
|
477
|
+
assert.match(runAll, /npx tsx scripts\/evaluation-contract-tests\.ts/);
|
|
478
|
+
assert.throws(()=>deriveMatchedPairs(countable, emptyResolver), /artifact/);
|
|
479
|
+
for (const [key, value] of [
|
|
480
|
+
[
|
|
481
|
+
'apikey',
|
|
482
|
+
'secret'
|
|
483
|
+
],
|
|
484
|
+
[
|
|
485
|
+
'access-token',
|
|
486
|
+
'secret'
|
|
487
|
+
],
|
|
488
|
+
[
|
|
489
|
+
'apiKey',
|
|
490
|
+
'secret'
|
|
491
|
+
],
|
|
492
|
+
[
|
|
493
|
+
'ACCESS-TOKEN',
|
|
494
|
+
'secret'
|
|
495
|
+
],
|
|
496
|
+
[
|
|
497
|
+
'client_secret',
|
|
498
|
+
'secret'
|
|
499
|
+
],
|
|
500
|
+
[
|
|
501
|
+
'authorization',
|
|
502
|
+
'Bearer abcdefghijklmnopqrstuvwxyz123456'
|
|
503
|
+
],
|
|
504
|
+
[
|
|
505
|
+
'value',
|
|
506
|
+
'/private/file'
|
|
507
|
+
],
|
|
508
|
+
[
|
|
509
|
+
'value',
|
|
510
|
+
'~/private/file'
|
|
511
|
+
],
|
|
512
|
+
[
|
|
513
|
+
'value',
|
|
514
|
+
'C:\\private\\file'
|
|
515
|
+
],
|
|
516
|
+
[
|
|
517
|
+
'value',
|
|
518
|
+
'file:///private/file'
|
|
519
|
+
],
|
|
520
|
+
[
|
|
521
|
+
'value',
|
|
522
|
+
'sk-abcdefghijklmnopqrstuv'
|
|
523
|
+
],
|
|
524
|
+
[
|
|
525
|
+
'value',
|
|
526
|
+
'ghp_abcdefghijklmnopqrstuvwxyz123456'
|
|
527
|
+
],
|
|
528
|
+
[
|
|
529
|
+
'value',
|
|
530
|
+
'AKIA1234567890ABCDEF'
|
|
531
|
+
]
|
|
532
|
+
])assert.throws(()=>assertPublicArtifactSafe({
|
|
533
|
+
[key]: value
|
|
534
|
+
}, 'adversarial'), /absolute path or secret|credentials/);
|
|
535
|
+
assert.doesNotThrow(()=>assertPublicArtifactSafe({
|
|
536
|
+
url: 'https://github.com/org/repo/blob/main/README.md',
|
|
537
|
+
label: 'question-answering token F1'
|
|
538
|
+
}, 'benign'));
|
|
539
|
+
assert.throws(()=>assertPublicArtifactSafe({
|
|
540
|
+
value: 'https://example.test/?token=sk-abcdefghijklmnopqrstuv'
|
|
541
|
+
}, 'https-url-secret'), /absolute path or secret/);
|
|
542
|
+
for (const value of [
|
|
543
|
+
'log=/Users/a/private.json',
|
|
544
|
+
'C:\\Users\\a\\secret',
|
|
545
|
+
'file:///tmp/x',
|
|
546
|
+
'~/x',
|
|
547
|
+
'https://github.com/org/repo log=/Users/a/x',
|
|
548
|
+
'Bearer abcdefghijklmnopqrstuvwxyz123456',
|
|
549
|
+
'sk-abcdefghijklmnopqrstuv',
|
|
550
|
+
'ghp_abcdefghijklmnopqrstuvwxyz123456',
|
|
551
|
+
'AKIA1234567890ABCDEF'
|
|
552
|
+
])assert.throws(()=>assertPublicArtifactSafe({
|
|
553
|
+
value
|
|
554
|
+
}, 'path-or-secret'), /absolute path or secret/);
|
|
555
|
+
assert.doesNotThrow(()=>assertPublicArtifactSafe({
|
|
556
|
+
github: 'https://github.com/org/repo',
|
|
557
|
+
huggingface: 'https://huggingface.co/datasets/org/name'
|
|
558
|
+
}, 'public urls'));
|
|
559
|
+
for (const key of [
|
|
560
|
+
'source_fence_passed',
|
|
561
|
+
'integrity_passed'
|
|
562
|
+
]){
|
|
563
|
+
const syntheticFlags = JSON.parse(JSON.stringify(load('data/evaluation/known-good.json')));
|
|
564
|
+
const run = syntheticFlags.run_receipts[0];
|
|
565
|
+
const qualification = run.qualification;
|
|
566
|
+
qualification[key] = true;
|
|
567
|
+
assert.equal(qualification[key === 'source_fence_passed' ? 'integrity_passed' : 'source_fence_passed'], false);
|
|
568
|
+
assert.throws(()=>parseEvalRunReceipt(run), /synthetic fixture/);
|
|
569
|
+
}
|
|
570
|
+
console.log(`canonical sha256: ${digest1}`);
|
|
571
|
+
console.log(`evaluation-contract tests: ${registry.length} registry, ${pairs.length} test-only matched pair, ${vectors.length} negative vectors green`);
|
|
572
|
+
|
|
573
|
+
|
|
574
|
+
//# sourceURL=ts/scripts/evaluation-contract-tests.ts
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
import process from 'node:process';
|
|
2
|
+
import { homedir } from 'node:os';
|
|
3
|
+
import { join, dirname } from 'node:path';
|
|
4
|
+
import { fileURLToPath } from 'node:url';
|
|
5
|
+
import { DEFAULT_RELEASE_BASE, installExtensionFromGithub } from '../capabilities/session/extension-distribution.js';
|
|
6
|
+
function argValue(argv, name) {
|
|
7
|
+
const i = argv.indexOf(name);
|
|
8
|
+
return i >= 0 ? argv[i + 1] : undefined;
|
|
9
|
+
}
|
|
10
|
+
const argv = process.argv.slice(2);
|
|
11
|
+
const repoRoot = join(dirname(fileURLToPath(import.meta.url)), '..', '..');
|
|
12
|
+
const dest = argValue(argv, '--dest') ?? join(homedir(), '.gotry', 'extension');
|
|
13
|
+
const sourceDir = argValue(argv, '--source-dir') ?? join(repoRoot, 'extension');
|
|
14
|
+
const releaseBase = argValue(argv, '--release-base') ?? DEFAULT_RELEASE_BASE;
|
|
15
|
+
const checkOnly = argv.includes('--check-only');
|
|
16
|
+
try {
|
|
17
|
+
const r = await installExtensionFromGithub({
|
|
18
|
+
destDir: dest,
|
|
19
|
+
pinnedSourceDir: sourceDir,
|
|
20
|
+
releaseBase,
|
|
21
|
+
checkOnly
|
|
22
|
+
});
|
|
23
|
+
process.stdout.write(`${JSON.stringify(r)}\n`);
|
|
24
|
+
process.exit(r.ok ? 0 : 2);
|
|
25
|
+
} catch (e) {
|
|
26
|
+
process.stdout.write(`${JSON.stringify({
|
|
27
|
+
ok: false,
|
|
28
|
+
action: 'fallback-bundled',
|
|
29
|
+
error: `CLI 异常 ${e.message}`
|
|
30
|
+
})}\n`);
|
|
31
|
+
process.exit(2);
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
//# sourceURL=ts/scripts/extension-distribution-cli.ts
|