@evomap/evolver 1.90.0 → 2.0.0-beta.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +27 -564
- package/bin/evolver-llm-proxy.js +3 -0
- package/bin/evolver-mcp.js +2 -0
- package/bin/evolver-proxy.js +3 -0
- package/bin/evolver.js +4 -0
- package/index.js +1 -3670
- package/package.json +32 -62
- package/CONTRIBUTING.md +0 -19
- package/LICENSE +0 -641
- package/README.ja-JP.md +0 -521
- package/README.ko-KR.md +0 -520
- package/README.zh-CN.md +0 -531
- package/SKILL.md +0 -365
- package/assets/cover.png +0 -0
- package/assets/gep/genes.seed.json +0 -496
- package/conformance/savings-core/constants.json +0 -30
- package/conformance/savings-core/golden-vectors.json +0 -333
- package/scripts/a2a_export.js +0 -63
- package/scripts/a2a_ingest.js +0 -79
- package/scripts/a2a_promote.js +0 -118
- package/scripts/analyze_by_skill.js +0 -121
- package/scripts/build_binaries.js +0 -479
- package/scripts/check-changelog.js +0 -166
- package/scripts/extract_log.js +0 -85
- package/scripts/generate_history.js +0 -75
- package/scripts/gep_append_event.js +0 -96
- package/scripts/gep_personality_report.js +0 -234
- package/scripts/human_report.js +0 -147
- package/scripts/recall-verify-report.js +0 -234
- package/scripts/recover_loop.js +0 -61
- package/scripts/refresh_stars_badge.js +0 -168
- package/scripts/seed-merchants.js +0 -91
- package/scripts/skill2recipes.js +0 -118
- package/scripts/suggest_version.js +0 -89
- package/scripts/validate-modules.js +0 -38
- package/scripts/validate-suite.js +0 -78
- package/src/adapters/claudeCode.js +0 -194
- package/src/adapters/codex.js +0 -216
- package/src/adapters/cursor.js +0 -91
- package/src/adapters/hookAdapter.js +0 -469
- package/src/adapters/kiro.js +0 -195
- package/src/adapters/opencode.js +0 -326
- package/src/adapters/scripts/_lockPaths.js +0 -74
- package/src/adapters/scripts/_memoryFiltering.js +0 -35
- package/src/adapters/scripts/_runtimePaths.js +0 -440
- package/src/adapters/scripts/evolver-session-end.js +0 -321
- package/src/adapters/scripts/evolver-session-start.js +0 -587
- package/src/adapters/scripts/evolver-signal-detect.js +0 -98
- package/src/adapters/scripts/evolver-task-recall.js +0 -173
- package/src/atp/atpExecute.js +0 -283
- package/src/atp/atpTaskPickup.js +0 -233
- package/src/atp/autoBuyer.js +0 -382
- package/src/atp/autoDeliver.js +0 -215
- package/src/atp/cli.js +0 -354
- package/src/atp/cliAutobuyPrompt.js +0 -154
- package/src/atp/consumerAgent.js +0 -157
- package/src/atp/defaultHandler.js +0 -69
- package/src/atp/heartbeatSignalsHandler.js +0 -254
- package/src/atp/hubClient.js +0 -317
- package/src/atp/index.js +0 -38
- package/src/atp/merchantAgent.js +0 -118
- package/src/atp/protocol.js +0 -41
- package/src/atp/questionComposer.js +0 -133
- package/src/atp/serviceHelper.js +0 -92
- package/src/canary.js +0 -13
- package/src/config.js +0 -351
- package/src/evolve/guards.js +0 -1
- package/src/evolve/pipeline/collect.js +0 -1
- package/src/evolve/pipeline/dispatch.js +0 -1
- package/src/evolve/pipeline/enrich.js +0 -1
- package/src/evolve/pipeline/hub.js +0 -1
- package/src/evolve/pipeline/select.js +0 -1
- package/src/evolve/pipeline/signals.js +0 -1
- package/src/evolve/utils.js +0 -1
- package/src/evolve.js +0 -1
- package/src/experiment/agentRunner.js +0 -229
- package/src/experiment/cli.js +0 -159
- package/src/experiment/comparison.js +0 -233
- package/src/experiment/metrics.js +0 -75
- package/src/forceUpdate.js +0 -991
- package/src/gep/a2a.js +0 -173
- package/src/gep/a2aProtocol.js +0 -1
- package/src/gep/analyzer.js +0 -35
- package/src/gep/antiAbuseTelemetry.js +0 -1
- package/src/gep/assetCallLog.js +0 -197
- package/src/gep/assetStore.js +0 -723
- package/src/gep/assets.js +0 -36
- package/src/gep/autoDistillConv.js +0 -1
- package/src/gep/autoDistillLlm.js +0 -1
- package/src/gep/bridge.js +0 -138
- package/src/gep/candidateEval.js +0 -1
- package/src/gep/candidates.js +0 -1
- package/src/gep/claimNudge.js +0 -121
- package/src/gep/cliContracts.js +0 -1190
- package/src/gep/contentHash.js +0 -1
- package/src/gep/conversationDistiller.js +0 -1
- package/src/gep/conversationSniffer.js +0 -1
- package/src/gep/crypto.js +0 -1
- package/src/gep/curriculum.js +0 -1
- package/src/gep/deviceId.js +0 -1
- package/src/gep/directoryClient.js +0 -115
- package/src/gep/envFingerprint.js +0 -1
- package/src/gep/epigenetics.js +0 -1
- package/src/gep/execBridge.js +0 -1
- package/src/gep/executionTrace.js +0 -291
- package/src/gep/explore.js +0 -1
- package/src/gep/featureFlags.js +0 -121
- package/src/gep/gitOps.js +0 -265
- package/src/gep/hash.js +0 -1
- package/src/gep/hostErrorClassifier.js +0 -34
- package/src/gep/hubFetch.js +0 -1
- package/src/gep/hubReview.js +0 -1
- package/src/gep/hubSearch.js +0 -1
- package/src/gep/hubVerify.js +0 -1
- package/src/gep/idleScheduler.js +0 -400
- package/src/gep/issueReporter.js +0 -416
- package/src/gep/learningSignals.js +0 -1
- package/src/gep/llmReview.js +0 -92
- package/src/gep/localStateAwareness.js +0 -243
- package/src/gep/mailboxTransport.js +0 -119
- package/src/gep/memoryGraph.js +0 -1
- package/src/gep/memoryGraphAdapter.js +0 -1
- package/src/gep/mutation.js +0 -1
- package/src/gep/narrativeMemory.js +0 -1
- package/src/gep/oauthLogin.js +0 -181
- package/src/gep/openPRRegistry.js +0 -1
- package/src/gep/paths.js +0 -522
- package/src/gep/personality.js +0 -1
- package/src/gep/policyCheck.js +0 -1
- package/src/gep/portable.js +0 -103
- package/src/gep/privacyClient.js +0 -235
- package/src/gep/prompt.js +0 -1
- package/src/gep/questionGenerator.js +0 -518
- package/src/gep/recallInject.js +0 -1
- package/src/gep/recallVerifier.js +0 -1
- package/src/gep/reflection.js +0 -1
- package/src/gep/sanitize.js +0 -264
- package/src/gep/savingsCore.js +0 -1
- package/src/gep/schemas/capsule.js +0 -170
- package/src/gep/schemas/gene.js +0 -154
- package/src/gep/schemas/index.js +0 -8
- package/src/gep/schemas/protocol.js +0 -51
- package/src/gep/schemas/task.js +0 -74
- package/src/gep/selector.js +0 -1
- package/src/gep/selfPR.js +0 -469
- package/src/gep/signals.js +0 -776
- package/src/gep/skill2gep.js +0 -1056
- package/src/gep/skill2gepAudit.js +0 -303
- package/src/gep/skill2recipes.js +0 -511
- package/src/gep/skillDistiller.js +0 -1
- package/src/gep/skillPublisher.js +0 -358
- package/src/gep/solidify.js +0 -1
- package/src/gep/strategy.js +0 -1
- package/src/gep/taskReceiver.js +0 -575
- package/src/gep/tokenSavings.js +0 -1
- package/src/gep/trajectoryExport.js +0 -1
- package/src/gep/validationReport.js +0 -55
- package/src/gep/validator/index.js +0 -411
- package/src/gep/validator/reporter.js +0 -210
- package/src/gep/validator/sandboxExecutor.js +0 -480
- package/src/gep/validator/stakeBootstrap.js +0 -357
- package/src/gep/workspaceKeychain.js +0 -1
- package/src/ops/cleanup.js +0 -80
- package/src/ops/commentary.js +0 -60
- package/src/ops/health_check.js +0 -104
- package/src/ops/index.js +0 -11
- package/src/ops/innovation.js +0 -67
- package/src/ops/lifecycle.js +0 -798
- package/src/ops/self_repair.js +0 -76
- package/src/ops/skills_monitor.js +0 -147
- package/src/ops/trigger.js +0 -33
- package/src/proxy/clientSettings.js +0 -405
- package/src/proxy/envelope.js +0 -59
- package/src/proxy/extensions/dmHandler.js +0 -45
- package/src/proxy/extensions/sessionHandler.js +0 -141
- package/src/proxy/extensions/skillUpdater.js +0 -64
- package/src/proxy/extensions/traceControl.js +0 -1
- package/src/proxy/index.js +0 -1395
- package/src/proxy/inject.js +0 -1
- package/src/proxy/lifecycle/manager.js +0 -1568
- package/src/proxy/mailbox/state.js +0 -207
- package/src/proxy/mailbox/store.js +0 -602
- package/src/proxy/router/cache_passthrough.js +0 -26
- package/src/proxy/router/features.js +0 -84
- package/src/proxy/router/gemini_route.js +0 -154
- package/src/proxy/router/messages_route.js +0 -535
- package/src/proxy/router/model_router.js +0 -113
- package/src/proxy/router/models_route.js +0 -52
- package/src/proxy/router/ollama_route.js +0 -103
- package/src/proxy/router/responses_route.js +0 -170
- package/src/proxy/router/vertex_route.js +0 -110
- package/src/proxy/server/http.js +0 -363
- package/src/proxy/server/routes.js +0 -558
- package/src/proxy/server/settings.js +0 -115
- package/src/proxy/sync/engine.js +0 -179
- package/src/proxy/sync/inbound.js +0 -211
- package/src/proxy/sync/outbound.js +0 -361
- package/src/proxy/task/monitor.js +0 -131
- package/src/proxy/trace/extractor.js +0 -1
- package/src/proxy/trace/usage.js +0 -1
- package/src/solo/breaker.js +0 -25
- package/src/solo/gitGuard.js +0 -65
- package/src/webui/client/clientJs/assets.js +0 -111
- package/src/webui/client/clientJs/bootstrap.js +0 -92
- package/src/webui/client/clientJs/common.js +0 -77
- package/src/webui/client/clientJs/i18n.js +0 -366
- package/src/webui/client/clientJs/index.js +0 -35
- package/src/webui/client/clientJs/interactions.js +0 -351
- package/src/webui/client/clientJs/overview.js +0 -152
- package/src/webui/client/clientJs/personality.js +0 -285
- package/src/webui/client/clientJs/pipelines.js +0 -330
- package/src/webui/client/indexHtml.js +0 -221
- package/src/webui/client/static.js +0 -23
- package/src/webui/client/stylesCss.js +0 -639
- package/src/webui/client/vendor/README.md +0 -15
- package/src/webui/client/vendor/echarts.min.js +0 -45
- package/src/webui/index.js +0 -14
- package/src/webui/observer/assets.js +0 -146
- package/src/webui/observer/index.js +0 -37
- package/src/webui/observer/interactions.js +0 -127
- package/src/webui/observer/jsonl.js +0 -75
- package/src/webui/observer/paths.js +0 -46
- package/src/webui/observer/personality.js +0 -43
- package/src/webui/observer/pipelineEvents.js +0 -58
- package/src/webui/observer/redact.js +0 -63
- package/src/webui/observer/runs.js +0 -356
- package/src/webui/observer/safety.js +0 -57
- package/src/webui/observer/skills.js +0 -70
- package/src/webui/observer/status.js +0 -71
- package/src/webui/server/http.js +0 -138
- package/src/webui/server/routes.js +0 -41
|
@@ -1,233 +0,0 @@
|
|
|
1
|
-
// src/experiment/comparison.js
|
|
2
|
-
//
|
|
3
|
-
// Thin orchestrator for a comparative experiment: run the SAME task twice --
|
|
4
|
-
// a baseline arm (plain task) and a variant arm (task + the reused gene's
|
|
5
|
-
// strategy injected) -- through a pluggable agent runner, collect real
|
|
6
|
-
// metrics (duration / rounds / tokens / pass-rate), and emit a versioned
|
|
7
|
-
// comparison result.
|
|
8
|
-
//
|
|
9
|
-
// Design notes:
|
|
10
|
-
// - This module NEVER requires child_process. The agent runner, gene loader,
|
|
11
|
-
// and sandbox runner are all injectable, so unit tests stay deterministic
|
|
12
|
-
// (no LLM, no network, no subprocess). Production defaults are lazy-loaded.
|
|
13
|
-
// - A failed arm never fabricates a score: if either arm is !ok the winner is
|
|
14
|
-
// 'inconclusive' and improvement is null, while still recording whatever
|
|
15
|
-
// partial metrics were captured.
|
|
16
|
-
'use strict';
|
|
17
|
-
|
|
18
|
-
const { deriveMetric, scoreArm, num, round } = require('./metrics');
|
|
19
|
-
|
|
20
|
-
const SCHEMA = 'evolver.experiment.comparison.v1';
|
|
21
|
-
const RESULT_TEXT_CAP = 2000;
|
|
22
|
-
const EPS = 1e-9;
|
|
23
|
-
|
|
24
|
-
// Build the variant prompt by appending the reused gene's strategy, mirroring
|
|
25
|
-
// the numbered-list format used in src/gep/prompt.js (`${i+1}. ${s}`).
|
|
26
|
-
function buildVariantPrompt(task, gene) {
|
|
27
|
-
if (!gene || !Array.isArray(gene.strategy) || gene.strategy.length === 0) return task;
|
|
28
|
-
const steps = gene.strategy.map((s, i) => `${i + 1}. ${s}`).join('\n');
|
|
29
|
-
return (
|
|
30
|
-
task +
|
|
31
|
-
'\n\n## Reuse the following proven strategy\n' +
|
|
32
|
-
steps +
|
|
33
|
-
'\n\nApply the strategy above while completing the task.'
|
|
34
|
-
);
|
|
35
|
-
}
|
|
36
|
-
|
|
37
|
-
// Coerce whatever the agent runner returned into the canonical arm shape.
|
|
38
|
-
function normalizeArm(label, raw) {
|
|
39
|
-
raw = raw || {};
|
|
40
|
-
const tokensIn = num(raw.tokensIn);
|
|
41
|
-
const tokensOut = num(raw.tokensOut);
|
|
42
|
-
const tokensTotal = Number.isFinite(Number(raw.tokensTotal)) ? num(raw.tokensTotal) : tokensIn + tokensOut;
|
|
43
|
-
return {
|
|
44
|
-
label: String(label == null ? '' : label),
|
|
45
|
-
ok: !!raw.ok,
|
|
46
|
-
error: raw.error != null ? String(raw.error) : null,
|
|
47
|
-
durationMs: num(raw.durationMs),
|
|
48
|
-
rounds: num(raw.rounds),
|
|
49
|
-
tokensIn,
|
|
50
|
-
tokensOut,
|
|
51
|
-
tokensTotal,
|
|
52
|
-
costUsd: num(raw.costUsd),
|
|
53
|
-
passRate: Number.isFinite(Number(raw.passRate)) ? num(raw.passRate) : (raw.ok ? 1 : 0),
|
|
54
|
-
resultText: typeof raw.resultText === 'string' ? raw.resultText.slice(0, RESULT_TEXT_CAP) : '',
|
|
55
|
-
exitCode: Number.isFinite(Number(raw.exitCode)) ? num(raw.exitCode) : null,
|
|
56
|
-
timedOut: !!raw.timedOut,
|
|
57
|
-
};
|
|
58
|
-
}
|
|
59
|
-
|
|
60
|
-
// Pass-rate for ONE arm: run its `node <script>` validation commands INSIDE that
|
|
61
|
-
// arm's own workspace (where its agent just ran), so two arms whose agents
|
|
62
|
-
// produced different output get different pass-rates -- the metric is linked to
|
|
63
|
-
// the arm, not to a shared empty sandbox. Each command is a `node <script>`
|
|
64
|
-
// vetted by sandboxExecutor's allowlist (runSingleCommand rejects anything else).
|
|
65
|
-
async function passRateInDir(commands, cwd, runSingleCommand, timeoutMs, warnings) {
|
|
66
|
-
let passed = 0;
|
|
67
|
-
let total = 0;
|
|
68
|
-
for (const cmd of commands) {
|
|
69
|
-
total += 1;
|
|
70
|
-
try {
|
|
71
|
-
const r = await runSingleCommand(cmd, { cwd, timeoutMs });
|
|
72
|
-
if (r && r.ok) passed += 1;
|
|
73
|
-
} catch (e) {
|
|
74
|
-
warnings.push('passrate_command_error: ' + (e && e.message ? e.message : String(e)));
|
|
75
|
-
}
|
|
76
|
-
}
|
|
77
|
-
return total > 0 ? round(passed / total, 4) : 0;
|
|
78
|
-
}
|
|
79
|
-
|
|
80
|
-
/**
|
|
81
|
-
* Run a two-arm comparison.
|
|
82
|
-
*
|
|
83
|
-
* @param {object} params
|
|
84
|
-
* @param {string} params.task 自然语言任务(必填)
|
|
85
|
-
* @param {string} [params.baseline='baseline'] 对照臂标签
|
|
86
|
-
* @param {string} [params.variant='variant'] 实验臂标签
|
|
87
|
-
* @param {string} params.metric 评估指标(必填)
|
|
88
|
-
* @param {string} [params.geneId] 变体臂复用的基因 id
|
|
89
|
-
* @param {string[]}[params.validationCommands] 自包含 `node <script>` 校验命令
|
|
90
|
-
* @param {number} [params.timeoutMs] 单臂超时
|
|
91
|
-
* @param {function}[params.agentRunner] (prompt, opts) => Promise<AgentResult>
|
|
92
|
-
* @param {function}[params.geneLoader] () => Gene[]
|
|
93
|
-
* @param {object} [params.sandbox] { createSandboxDir, cleanupDir, runSingleCommand } (default: sandboxExecutor)
|
|
94
|
-
* @returns {Promise<object>} versioned ComparisonResult (see SCHEMA)
|
|
95
|
-
*/
|
|
96
|
-
async function runComparison(params) {
|
|
97
|
-
const p = params || {};
|
|
98
|
-
const task = String(p.task == null ? '' : p.task).trim();
|
|
99
|
-
const baseline = p.baseline ? String(p.baseline) : 'baseline';
|
|
100
|
-
const variant = p.variant ? String(p.variant) : 'variant';
|
|
101
|
-
const metric = String(p.metric == null ? '' : p.metric);
|
|
102
|
-
const geneId = p.geneId ? String(p.geneId) : null;
|
|
103
|
-
const validationCommands = Array.isArray(p.validationCommands)
|
|
104
|
-
? p.validationCommands.filter((c) => typeof c === 'string' && c.trim())
|
|
105
|
-
: null;
|
|
106
|
-
const timeoutMs = Number.isFinite(Number(p.timeoutMs)) ? Number(p.timeoutMs) : undefined;
|
|
107
|
-
|
|
108
|
-
if (!task) throw new Error('task is required');
|
|
109
|
-
if (!metric) throw new Error('metric is required');
|
|
110
|
-
|
|
111
|
-
const agentRunner = typeof p.agentRunner === 'function'
|
|
112
|
-
? p.agentRunner
|
|
113
|
-
: require('./agentRunner').runAgentTask;
|
|
114
|
-
const geneLoader = typeof p.geneLoader === 'function'
|
|
115
|
-
? p.geneLoader
|
|
116
|
-
: require('../gep/assetStore').loadGenes;
|
|
117
|
-
const sandbox = p.sandbox && typeof p.sandbox === 'object'
|
|
118
|
-
? p.sandbox
|
|
119
|
-
: require('../gep/validator/sandboxExecutor');
|
|
120
|
-
|
|
121
|
-
const startedAt = new Date().toISOString();
|
|
122
|
-
const t0 = Date.now();
|
|
123
|
-
const warnings = [];
|
|
124
|
-
|
|
125
|
-
const metricInfo = deriveMetric(metric);
|
|
126
|
-
if (!metricInfo.recognized) warnings.push('metric_unrecognized: ' + metric);
|
|
127
|
-
|
|
128
|
-
// Look up the reused gene (variant arm). Without a resolved gene the variant
|
|
129
|
-
// prompt is identical to the baseline task, so the two arms are NOT a strategy
|
|
130
|
-
// comparison -- record an explicit warning so identical arms aren't mistaken
|
|
131
|
-
// for one.
|
|
132
|
-
let gene = null;
|
|
133
|
-
if (geneId) {
|
|
134
|
-
let genes = [];
|
|
135
|
-
try {
|
|
136
|
-
genes = geneLoader() || [];
|
|
137
|
-
} catch (e) {
|
|
138
|
-
warnings.push('gene_load_error: ' + (e && e.message ? e.message : String(e)));
|
|
139
|
-
}
|
|
140
|
-
gene = genes.find((g) => g && String(g.id) === geneId) || null;
|
|
141
|
-
if (!gene) warnings.push('gene_not_found: ' + geneId + ' (variant arm equals baseline)');
|
|
142
|
-
} else {
|
|
143
|
-
warnings.push('no_gene: variant arm equals baseline (no strategy injected)');
|
|
144
|
-
}
|
|
145
|
-
|
|
146
|
-
const hasValidation = !!(validationCommands && validationCommands.length);
|
|
147
|
-
if (!hasValidation) warnings.push('passrate_degraded_no_validation');
|
|
148
|
-
|
|
149
|
-
let metaRunner = null;
|
|
150
|
-
let metaCommand = null;
|
|
151
|
-
|
|
152
|
-
const runArm = async (label, prompt) => {
|
|
153
|
-
// Each arm runs in its OWN fresh sandbox dir, so the agent works in
|
|
154
|
-
// isolation (never the evolver repo / process.cwd()) and its pass-rate
|
|
155
|
-
// validation reads that arm's own output, not a shared empty directory.
|
|
156
|
-
const workdir = sandbox.createSandboxDir();
|
|
157
|
-
let raw;
|
|
158
|
-
try {
|
|
159
|
-
raw = await agentRunner(prompt, { timeoutMs, cwd: workdir });
|
|
160
|
-
} catch (e) {
|
|
161
|
-
raw = { ok: false, error: 'agent_runner_threw: ' + (e && e.message ? e.message : String(e)) };
|
|
162
|
-
}
|
|
163
|
-
if (raw) {
|
|
164
|
-
if (metaRunner == null && raw.runnerName) metaRunner = String(raw.runnerName);
|
|
165
|
-
if (metaCommand == null && raw.agentCommand) metaCommand = String(raw.agentCommand);
|
|
166
|
-
}
|
|
167
|
-
const arm = normalizeArm(label, raw);
|
|
168
|
-
if (hasValidation) {
|
|
169
|
-
arm.passRate = await passRateInDir(validationCommands, workdir, sandbox.runSingleCommand, timeoutMs, warnings);
|
|
170
|
-
}
|
|
171
|
-
try { sandbox.cleanupDir(workdir); } catch (_) { /* best-effort cleanup */ }
|
|
172
|
-
return arm;
|
|
173
|
-
};
|
|
174
|
-
|
|
175
|
-
// Arms run sequentially: two real agent CLIs in parallel would contend for
|
|
176
|
-
// local resources / provider rate limits and muddy the duration metric.
|
|
177
|
-
const armBaseline = await runArm(baseline, task);
|
|
178
|
-
const armVariant = await runArm(variant, buildVariantPrompt(task, gene));
|
|
179
|
-
|
|
180
|
-
const baselineScore = scoreArm(armBaseline, metricInfo.metricField);
|
|
181
|
-
const variantScore = scoreArm(armVariant, metricInfo.metricField);
|
|
182
|
-
|
|
183
|
-
// Pass-rate is only a real measurement when validation commands ran. Without
|
|
184
|
-
// them it's a synthetic ok?1:0, so a pass-rate comparison would falsely tie
|
|
185
|
-
// (both arms 1.0) — report it as inconclusive instead of a fake tie.
|
|
186
|
-
const passRateNotMeasured = metricInfo.metricField === 'passRate' && !hasValidation;
|
|
187
|
-
let winner;
|
|
188
|
-
let improvement;
|
|
189
|
-
if (!armBaseline.ok || !armVariant.ok || passRateNotMeasured) {
|
|
190
|
-
winner = 'inconclusive';
|
|
191
|
-
improvement = null;
|
|
192
|
-
} else if (Math.abs(baselineScore - variantScore) <= EPS) {
|
|
193
|
-
winner = 'tie';
|
|
194
|
-
improvement = 0;
|
|
195
|
-
} else {
|
|
196
|
-
const variantBetter = metricInfo.lowerIsBetter
|
|
197
|
-
? variantScore < baselineScore
|
|
198
|
-
: variantScore > baselineScore;
|
|
199
|
-
winner = variantBetter ? 'variant' : 'baseline';
|
|
200
|
-
if (baselineScore === 0) {
|
|
201
|
-
improvement = null;
|
|
202
|
-
} else {
|
|
203
|
-
const ratio = metricInfo.lowerIsBetter
|
|
204
|
-
? (baselineScore - variantScore) / Math.abs(baselineScore)
|
|
205
|
-
: (variantScore - baselineScore) / Math.abs(baselineScore);
|
|
206
|
-
improvement = round(ratio, 4);
|
|
207
|
-
}
|
|
208
|
-
}
|
|
209
|
-
|
|
210
|
-
return {
|
|
211
|
-
schema: SCHEMA,
|
|
212
|
-
task,
|
|
213
|
-
metric,
|
|
214
|
-
metricField: metricInfo.metricField,
|
|
215
|
-
lowerIsBetter: metricInfo.lowerIsBetter,
|
|
216
|
-
scoreUnit: metricInfo.scoreUnit,
|
|
217
|
-
geneId,
|
|
218
|
-
baselineScore,
|
|
219
|
-
variantScore,
|
|
220
|
-
winner,
|
|
221
|
-
improvement,
|
|
222
|
-
arms: { baseline: armBaseline, variant: armVariant },
|
|
223
|
-
meta: {
|
|
224
|
-
runner: metaRunner || 'unknown',
|
|
225
|
-
agentCommand: metaCommand || null,
|
|
226
|
-
startedAt,
|
|
227
|
-
durationMs: Date.now() - t0,
|
|
228
|
-
warnings,
|
|
229
|
-
},
|
|
230
|
-
};
|
|
231
|
-
}
|
|
232
|
-
|
|
233
|
-
module.exports = { runComparison, buildVariantPrompt, normalizeArm, SCHEMA };
|
|
@@ -1,75 +0,0 @@
|
|
|
1
|
-
// src/experiment/metrics.js
|
|
2
|
-
//
|
|
3
|
-
// Pure, table-driven mapping from a human metric label (e.g. "完成耗时 (s)",
|
|
4
|
-
// "轮次", "token", "通过率") onto a per-arm field + comparison direction.
|
|
5
|
-
// No I/O, no side effects -- safe to unit-test in isolation.
|
|
6
|
-
'use strict';
|
|
7
|
-
|
|
8
|
-
function num(v, fallback) {
|
|
9
|
-
const n = Number(v);
|
|
10
|
-
return Number.isFinite(n) ? n : (fallback === undefined ? 0 : fallback);
|
|
11
|
-
}
|
|
12
|
-
|
|
13
|
-
function round(n, digits) {
|
|
14
|
-
const f = Math.pow(10, digits);
|
|
15
|
-
return Math.round((num(n) + Number.EPSILON) * f) / f;
|
|
16
|
-
}
|
|
17
|
-
|
|
18
|
-
// Ordered rules. The FIRST rule whose any keyword is a (case-insensitive)
|
|
19
|
-
// substring of the metric label wins. Order matters: pass-rate / rounds /
|
|
20
|
-
// tokens / cost are checked before duration so a label like "通过率" is not
|
|
21
|
-
// swallowed by a looser rule.
|
|
22
|
-
const METRIC_RULES = [
|
|
23
|
-
{ keys: ['通过率', 'pass', 'success', 'accuracy', '准确', '正确率'], field: 'passRate', lowerIsBetter: false },
|
|
24
|
-
{ keys: ['轮次', 'turn', 'round', 'step', 'iteration', '迭代'], field: 'rounds', lowerIsBetter: true },
|
|
25
|
-
{ keys: ['token', '令牌'], field: 'tokensTotal', lowerIsBetter: true },
|
|
26
|
-
{ keys: ['成本', 'cost', 'usd', '价格', '费用'], field: 'costUsd', lowerIsBetter: true },
|
|
27
|
-
{ keys: ['耗时', 'duration', 'latency', '延迟', '秒', 'second', '(s)', 'time'], field: 'durationMs', lowerIsBetter: true },
|
|
28
|
-
];
|
|
29
|
-
|
|
30
|
-
/**
|
|
31
|
-
* Resolve a metric label to the per-arm field used for scoring, the
|
|
32
|
-
* comparison direction, and the display unit.
|
|
33
|
-
*
|
|
34
|
-
* @param {string} metricStr
|
|
35
|
-
* @returns {{ metricField: string, lowerIsBetter: boolean, scoreUnit: string, recognized: boolean }}
|
|
36
|
-
*/
|
|
37
|
-
function deriveMetric(metricStr) {
|
|
38
|
-
const m = String(metricStr || '').toLowerCase();
|
|
39
|
-
for (const rule of METRIC_RULES) {
|
|
40
|
-
if (rule.keys.some((k) => m.includes(String(k).toLowerCase()))) {
|
|
41
|
-
if (rule.field === 'durationMs') {
|
|
42
|
-
// Seconds-flavoured labels ("(s)", "秒", "seconds") -> report in seconds.
|
|
43
|
-
if (/\(s\)|秒|second/.test(m)) {
|
|
44
|
-
return { metricField: 'durationSec', lowerIsBetter: true, scoreUnit: 'seconds', recognized: true };
|
|
45
|
-
}
|
|
46
|
-
return { metricField: 'durationMs', lowerIsBetter: true, scoreUnit: 'ms', recognized: true };
|
|
47
|
-
}
|
|
48
|
-
return { metricField: rule.field, lowerIsBetter: rule.lowerIsBetter, scoreUnit: 'raw', recognized: true };
|
|
49
|
-
}
|
|
50
|
-
}
|
|
51
|
-
// Unrecognized -> degrade to pass-rate (higher is better). Caller records a warning.
|
|
52
|
-
return { metricField: 'passRate', lowerIsBetter: false, scoreUnit: 'raw', recognized: false };
|
|
53
|
-
}
|
|
54
|
-
|
|
55
|
-
/**
|
|
56
|
-
* Pull the scalar score for one arm given the resolved metric field.
|
|
57
|
-
*
|
|
58
|
-
* @param {object} arm a normalized arm (see comparison.normalizeArm)
|
|
59
|
-
* @param {string} metricField
|
|
60
|
-
* @returns {number}
|
|
61
|
-
*/
|
|
62
|
-
function scoreArm(arm, metricField) {
|
|
63
|
-
if (!arm) return 0;
|
|
64
|
-
switch (metricField) {
|
|
65
|
-
case 'durationSec': return round(num(arm.durationMs) / 1000, 2);
|
|
66
|
-
case 'durationMs': return num(arm.durationMs);
|
|
67
|
-
case 'rounds': return num(arm.rounds);
|
|
68
|
-
case 'tokensTotal': return num(arm.tokensTotal);
|
|
69
|
-
case 'costUsd': return round(num(arm.costUsd), 4);
|
|
70
|
-
case 'passRate': return round(num(arm.passRate), 4);
|
|
71
|
-
default: return round(num(arm.passRate), 4);
|
|
72
|
-
}
|
|
73
|
-
}
|
|
74
|
-
|
|
75
|
-
module.exports = { deriveMetric, scoreArm, METRIC_RULES, round, num };
|