@tangle-network/agent-eval 0.123.0 → 0.123.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +9 -0
- package/README.md +151 -161
- package/dist/analyst/index.d.ts +9 -1
- package/dist/analyst/index.js +5 -5
- package/dist/authenticity/index.js +3 -2
- package/dist/authenticity/index.js.map +1 -1
- package/dist/belief-state/index.d.ts +45 -5
- package/dist/belief-state/index.js +41 -3
- package/dist/belief-state/index.js.map +1 -1
- package/dist/benchmarks/index.d.ts +2 -1
- package/dist/benchmarks/index.js +6 -6
- package/dist/campaign/index.d.ts +29 -33
- package/dist/campaign/index.js +6 -6
- package/dist/{chunk-A5S77LSE.js → chunk-4SOQ4ND2.js} +2 -2
- package/dist/{chunk-VJ7T5WIO.js → chunk-5YMKIFYP.js} +3 -3
- package/dist/{chunk-U5CHZ5M3.js → chunk-DNVPOYUS.js} +4 -4
- package/dist/{chunk-6WX7CBAR.js → chunk-E3HAD4A3.js} +19 -8
- package/dist/chunk-E3HAD4A3.js.map +1 -0
- package/dist/{chunk-LBAHQOBI.js → chunk-EBDOTTZJ.js} +37 -11
- package/dist/chunk-EBDOTTZJ.js.map +1 -0
- package/dist/{chunk-XJYR7XFV.js → chunk-GC4ATIKK.js} +1 -1
- package/dist/chunk-GC4ATIKK.js.map +1 -0
- package/dist/{chunk-HZJF4IUO.js → chunk-HQY7LBV2.js} +3 -3
- package/dist/{chunk-NJC7U437.js → chunk-J7S4YM27.js} +6 -5
- package/dist/chunk-J7S4YM27.js.map +1 -0
- package/dist/{chunk-S3UZOQ5Y.js → chunk-LOW3U7JZ.js} +2 -2
- package/dist/{chunk-OYZAPX5G.js → chunk-R226UZOI.js} +2 -2
- package/dist/{chunk-GS3FJGUF.js → chunk-RQP5UTK5.js} +120 -14
- package/dist/chunk-RQP5UTK5.js.map +1 -0
- package/dist/{chunk-DTJ6QUQB.js → chunk-VGRCHJON.js} +39 -7
- package/dist/chunk-VGRCHJON.js.map +1 -0
- package/dist/{chunk-G2GPNLSX.js → chunk-WMJR67FX.js} +3 -3
- package/dist/{chunk-FC5NDO3E.js → chunk-WXQTVEKM.js} +3 -3
- package/dist/cli.js +100 -10
- package/dist/cli.js.map +1 -1
- package/dist/contract/index.d.ts +97 -5
- package/dist/contract/index.js +9 -7
- package/dist/contract/index.js.map +1 -1
- package/dist/control.js +3 -3
- package/dist/fuzz.js +3 -2
- package/dist/fuzz.js.map +1 -1
- package/dist/hosted/index.d.ts +8 -2
- package/dist/index.d.ts +10 -2
- package/dist/index.js +13 -13
- package/dist/index.js.map +1 -1
- package/dist/openapi.json +1 -1
- package/dist/rl.d.ts +48 -12
- package/dist/rl.js +5 -5
- package/dist/storyboard/index.js +1 -1
- package/dist/storyboard/index.js.map +1 -1
- package/dist/traces.js +3 -3
- package/dist/wire/index.d.ts +61 -4
- package/dist/wire/index.js +2 -2
- package/docs/adapters-observability.md +6 -6
- package/docs/building-doctrine.md +5 -5
- package/docs/concepts.md +29 -29
- package/docs/customer-journeys.md +80 -155
- package/docs/design/loop-taxonomy.md +26 -27
- package/docs/design.md +70 -0
- package/docs/distributed-driver.md +14 -14
- package/docs/eval-surface-map.md +11 -11
- package/docs/hosted-ingest-spec.md +4 -4
- package/docs/improvement-glossary.md +38 -38
- package/docs/insight-report.md +32 -27
- package/docs/multi-shot-optimization.md +8 -8
- package/docs/research-report-methodology.md +9 -9
- package/docs/self-improvement-map.md +13 -13
- package/docs/trace-analysis.md +2 -2
- package/docs/wire-protocol.md +16 -16
- package/package.json +2 -1
- package/dist/chunk-6WX7CBAR.js.map +0 -1
- package/dist/chunk-DTJ6QUQB.js.map +0 -1
- package/dist/chunk-GS3FJGUF.js.map +0 -1
- package/dist/chunk-LBAHQOBI.js.map +0 -1
- package/dist/chunk-NJC7U437.js.map +0 -1
- package/dist/chunk-XJYR7XFV.js.map +0 -1
- package/docs/auto-research-loop-end-to-end.md +0 -186
- /package/dist/{chunk-A5S77LSE.js.map → chunk-4SOQ4ND2.js.map} +0 -0
- /package/dist/{chunk-VJ7T5WIO.js.map → chunk-5YMKIFYP.js.map} +0 -0
- /package/dist/{chunk-U5CHZ5M3.js.map → chunk-DNVPOYUS.js.map} +0 -0
- /package/dist/{chunk-HZJF4IUO.js.map → chunk-HQY7LBV2.js.map} +0 -0
- /package/dist/{chunk-S3UZOQ5Y.js.map → chunk-LOW3U7JZ.js.map} +0 -0
- /package/dist/{chunk-OYZAPX5G.js.map → chunk-R226UZOI.js.map} +0 -0
- /package/dist/{chunk-G2GPNLSX.js.map → chunk-WMJR67FX.js.map} +0 -0
- /package/dist/{chunk-FC5NDO3E.js.map → chunk-WXQTVEKM.js.map} +0 -0
|
@@ -74,8 +74,18 @@ function selfNormalizedImportanceWeighting(trajectories, opts = {}) {
|
|
|
74
74
|
function doublyRobust(trajectories, opts = {}) {
|
|
75
75
|
const cap = opts.weightCap ?? Infinity;
|
|
76
76
|
const clip = opts.rewardClip ?? { low: 0, high: 1 };
|
|
77
|
-
if (trajectories.length === 0)
|
|
77
|
+
if (trajectories.length === 0) {
|
|
78
|
+
return {
|
|
79
|
+
...zeroEstimate(),
|
|
80
|
+
contributionCounts: { dr: 0, ipsFallback: 0, legacyScalar: 0 }
|
|
81
|
+
};
|
|
82
|
+
}
|
|
78
83
|
const contributions = [];
|
|
84
|
+
const contributionCounts = {
|
|
85
|
+
dr: 0,
|
|
86
|
+
ipsFallback: 0,
|
|
87
|
+
legacyScalar: 0
|
|
88
|
+
};
|
|
79
89
|
let maxW = 0;
|
|
80
90
|
let sumW = 0;
|
|
81
91
|
let sumW2 = 0;
|
|
@@ -85,11 +95,32 @@ function doublyRobust(trajectories, opts = {}) {
|
|
|
85
95
|
}
|
|
86
96
|
const w = Math.min(cap, t.targetProb / t.behaviorProb);
|
|
87
97
|
const r = clamp(t.reward, clip.low, clip.high);
|
|
88
|
-
const
|
|
89
|
-
|
|
90
|
-
|
|
98
|
+
const rawQHatChosen = t.qHatChosen;
|
|
99
|
+
const rawVHatTarget = t.vHatTarget;
|
|
100
|
+
const hasQHatChosen = rawQHatChosen !== null && rawQHatChosen !== void 0;
|
|
101
|
+
const hasVHatTarget = rawVHatTarget !== null && rawVHatTarget !== void 0;
|
|
102
|
+
if (hasQHatChosen !== hasVHatTarget) {
|
|
103
|
+
throw new ValidationError(
|
|
104
|
+
`doublyRobust: qHatChosen and vHatTarget must be supplied together (runId=${t.runId})`
|
|
105
|
+
);
|
|
106
|
+
}
|
|
107
|
+
if (hasQHatChosen && hasVHatTarget) {
|
|
108
|
+
if (!Number.isFinite(rawQHatChosen) || !Number.isFinite(rawVHatTarget)) {
|
|
109
|
+
throw new ValidationError(
|
|
110
|
+
`doublyRobust: qHatChosen and vHatTarget must be finite (runId=${t.runId})`
|
|
111
|
+
);
|
|
112
|
+
}
|
|
113
|
+
const qHatChosen = clamp(rawQHatChosen, clip.low, clip.high);
|
|
114
|
+
const vHatTarget = clamp(rawVHatTarget, clip.low, clip.high);
|
|
115
|
+
contributions.push(vHatTarget + w * (r - qHatChosen));
|
|
116
|
+
contributionCounts.dr += 1;
|
|
117
|
+
} else if (typeof t.qHat === "number" && Number.isFinite(t.qHat)) {
|
|
118
|
+
const qHat = clamp(t.qHat, clip.low, clip.high);
|
|
119
|
+
contributions.push(qHat + w * (r - qHat));
|
|
120
|
+
contributionCounts.legacyScalar += 1;
|
|
91
121
|
} else {
|
|
92
|
-
contributions.push(
|
|
122
|
+
contributions.push(w * r);
|
|
123
|
+
contributionCounts.ipsFallback += 1;
|
|
93
124
|
}
|
|
94
125
|
if (w > maxW) maxW = w;
|
|
95
126
|
sumW += w;
|
|
@@ -104,7 +135,8 @@ function doublyRobust(trajectories, opts = {}) {
|
|
|
104
135
|
standardError: Math.sqrt(variance / n),
|
|
105
136
|
effectiveSampleSize: effN,
|
|
106
137
|
n,
|
|
107
|
-
maxImportanceWeight: maxW
|
|
138
|
+
maxImportanceWeight: maxW,
|
|
139
|
+
contributionCounts
|
|
108
140
|
};
|
|
109
141
|
}
|
|
110
142
|
function offPolicyEstimateAll(trajectories, opts = {}) {
|
|
@@ -128,4 +160,4 @@ export {
|
|
|
128
160
|
doublyRobust,
|
|
129
161
|
offPolicyEstimateAll
|
|
130
162
|
};
|
|
131
|
-
//# sourceMappingURL=chunk-
|
|
163
|
+
//# sourceMappingURL=chunk-VGRCHJON.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"sources":["../src/rl/off-policy.ts"],"sourcesContent":["/**\n * Off-policy evaluation primitives.\n *\n * Standard inverse-probability-weighted (IPS), self-normalized\n * importance-weighted (SNIPS), and doubly-robust (DR) estimators for the\n * value of a *target* policy given trajectories collected under a\n * *behavior* policy. This is the canonical RL eval task: \"we have last\n * week's runs, we changed the policy — how would the new one do without\n * re-running?\"\n *\n * The math here is textbook (Dudík, Langford, Li 2011 for DR; Swaminathan\n * & Joachims 2015 for SNIPS) but the *application* to LLM-agent\n * evaluation needs care:\n *\n * - The \"policy\" is the (prompt, tool config, model snapshot) triple.\n * Two policies have the same probability over an action *iff* their\n * LLM call would emit the same token with the same probability —\n * which is generally unknowable without the model log-probs.\n * - For LLM agents, propensity scores must be supplied by the caller\n * (logged in the trace, recovered from token log-probs, or estimated\n * via a learned propensity model). We do NOT estimate propensity here.\n * - Doubly-robust requires two outputs from a Q-function: its prediction\n * for the logged action and its expectation under the target policy.\n * Consumers compute these with a tabular estimate, regression fit, or\n * learned reward model before constructing the trajectories.\n *\n * Bias / variance tradeoffs:\n * - IPS: unbiased; high variance for small overlap, infinite variance\n * when target has support outside behavior.\n * - SNIPS: lower variance, slight bias; usually preferred in practice.\n * - DR: doubly-robust — unbiased if either propensity OR Q-function is\n * correct. Lowest practical variance when Q is decent. Use this.\n *\n * Caveat the panel will land: on the LLM-agent setting, propensity scores\n * recovered from token log-probs are noisy, the action space is enormous,\n * and overlap is often poor. These estimators are useful but not magic;\n * complement with `replayCampaign` (exact replay where the request hashes\n * match) for high-confidence answers and OPE for the gap.\n */\n\nimport { ValidationError } from '../errors'\n\nexport interface OffPolicyTrajectory {\n /** Stable id, for traceability through the dataset. */\n runId: string\n /** Reward observed under the behavior policy (the realized outcome). */\n reward: number\n /**\n * Behavior-policy probability of the action that was taken. For LLM\n * agents this is typically `exp(sum(token_log_probs))` over the chosen\n * trajectory. Must be in (0, 1].\n */\n behaviorProb: number\n /**\n * Target-policy probability of the same action. For replay-style\n * counterfactual evaluation this is what the *new* policy would have\n * assigned to the *old* trajectory. Must be in [0, 1].\n */\n targetProb: number\n /**\n * Model-based reward prediction for the action selected by the behavior\n * policy: `Q_hat(context, loggedAction)`. Supply this together with\n * `vHatTarget` for contextual-bandit doubly-robust estimation.\n */\n qHatChosen?: number | null\n /**\n * Expected model-based reward under the target policy:\n * `sum_action targetPolicy(action | context) * Q_hat(context, action)`.\n * Supply this together with `qHatChosen`. For an honest evaluation, both\n * values must come from a model cross-fitted or trained outside this row.\n */\n vHatTarget?: number | null\n /**\n * @deprecated Use `qHatChosen` and `vHatTarget` together. When the new pair\n * is absent, this scalar is used as both terms to preserve existing results.\n * When the new pair is present, this field is ignored.\n */\n qHat?: number | null\n}\n\nexport interface OffPolicyContributionCounts {\n /** Contributions using the contextual-bandit doubly-robust formula. */\n dr: number\n /** Contributions using exact IPS because no reward-model estimate was supplied. */\n ipsFallback: number\n /** Contributions using the deprecated single-scalar formula. */\n legacyScalar: number\n}\n\nexport interface OffPolicyEstimate {\n /** Estimated value of the target policy. */\n value: number\n /** Standard error of the estimate. */\n standardError: number\n /** Effective sample size (Kong 1992). Lower = more reliance on a few high-weight samples. */\n effectiveSampleSize: number\n /** Number of trajectories used. */\n n: number\n /**\n * Diagnostic: maximum importance weight observed. Large values (>>10x\n * mean) are a red flag — variance is dominated by a few outliers.\n */\n maxImportanceWeight: number\n /** Populated by `doublyRobust` to expose which formula each row used. */\n contributionCounts?: OffPolicyContributionCounts\n}\n\nexport interface OffPolicyOptions {\n /**\n * Cap importance weights at this value (Ionides 2008 truncated IS) to\n * trade unbiasedness for variance reduction. Default `Infinity` (no cap).\n * Set e.g. `10` for stable estimates when the policies are close.\n */\n weightCap?: number\n /** Reward clipping range. Default `[0, 1]`. */\n rewardClip?: { low: number; high: number }\n}\n\n/**\n * Inverse Probability Weighting (Horvitz-Thompson). Unbiased estimator\n * of E[reward under target policy]. Variance scales with the spread of\n * target/behavior ratios.\n */\nexport function inverseProbabilityWeighting(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n\n if (trajectories.length === 0) {\n return zeroEstimate()\n }\n\n const weights: number[] = []\n const weightedRewards: number[] = []\n let maxW = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(\n `inverseProbabilityWeighting: behaviorProb must be > 0 (runId=${t.runId})`,\n )\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n const r = clamp(t.reward, clip.low, clip.high)\n weights.push(w)\n weightedRewards.push(w * r)\n if (w > maxW) maxW = w\n }\n const n = weights.length\n const value = weightedRewards.reduce((s, x) => s + x, 0) / n\n const variance = weightedRewards.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1)\n const sumW = weights.reduce((s, w) => s + w, 0)\n const sumW2 = weights.reduce((s, w) => s + w * w, 0)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n\n return {\n value,\n standardError: Math.sqrt(variance / n),\n effectiveSampleSize: effN,\n n,\n maxImportanceWeight: maxW,\n }\n}\n\n/**\n * Self-Normalized Importance Sampling. Lower variance than vanilla IPS at\n * the cost of small bias (vanishing as N grows). The right default for\n * LLM-agent evaluation where overlap is often poor.\n */\nexport function selfNormalizedImportanceWeighting(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n if (trajectories.length === 0) return zeroEstimate()\n\n const weights: number[] = []\n const rewards: number[] = []\n let maxW = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(\n `selfNormalizedImportanceWeighting: behaviorProb must be > 0 (runId=${t.runId})`,\n )\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n weights.push(w)\n rewards.push(clamp(t.reward, clip.low, clip.high))\n if (w > maxW) maxW = w\n }\n const sumW = weights.reduce((s, w) => s + w, 0)\n const sumWR = weights.reduce((s, w, i) => s + w * rewards[i]!, 0)\n const value = sumW === 0 ? 0 : sumWR / sumW\n const sumW2 = weights.reduce((s, w) => s + w * w, 0)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n // Influence-function-based SE for SNIPS (Owen 2013, Ch. 9).\n const phi = weights.map((w, i) => w * (rewards[i]! - value))\n const variance = phi.reduce((s, x) => s + x * x, 0) / Math.max(1, sumW * sumW)\n return {\n value,\n standardError: Math.sqrt(variance),\n effectiveSampleSize: effN,\n n: trajectories.length,\n maxImportanceWeight: maxW,\n }\n}\n\n/**\n * Doubly-robust off-policy estimator (Dudík, Langford, Li 2011).\n *\n * V_DR = (1/N) * sum_i [ v_hat_target_i\n * + (target_prob_i / behavior_prob_i) * (r_i - q_hat_chosen_i) ]\n *\n * Unbiased if EITHER:\n * - the importance ratios are correct (IPS-style validity), OR\n * - the Q-hat function is correct (model-based validity).\n *\n * In practice both are imperfect, but the residual bias is the *product*\n * of both errors — much smaller than either alone. This is why DR is the\n * default in production OPE pipelines.\n *\n * `qHatChosen` and `vHatTarget` must be supplied together. Rows with neither\n * use the exact IPS contribution. Deprecated `qHat` rows preserve the scalar\n * formula, and a complete new pair takes precedence when both forms exist.\n * `contributionCounts` makes the mix explicit in the result.\n * Callers must cross-fit the Q-function or train it on independent rows;\n * fitting and evaluating Q on the same outcomes leaks the answer.\n */\nexport function doublyRobust(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): OffPolicyEstimate {\n const cap = opts.weightCap ?? Infinity\n const clip = opts.rewardClip ?? { low: 0, high: 1 }\n if (trajectories.length === 0) {\n return {\n ...zeroEstimate(),\n contributionCounts: { dr: 0, ipsFallback: 0, legacyScalar: 0 },\n }\n }\n\n const contributions: number[] = []\n const contributionCounts: OffPolicyContributionCounts = {\n dr: 0,\n ipsFallback: 0,\n legacyScalar: 0,\n }\n let maxW = 0\n let sumW = 0\n let sumW2 = 0\n for (const t of trajectories) {\n if (t.behaviorProb <= 0) {\n throw new ValidationError(`doublyRobust: behaviorProb must be > 0 (runId=${t.runId})`)\n }\n const w = Math.min(cap, t.targetProb / t.behaviorProb)\n const r = clamp(t.reward, clip.low, clip.high)\n const rawQHatChosen = t.qHatChosen\n const rawVHatTarget = t.vHatTarget\n const hasQHatChosen = rawQHatChosen !== null && rawQHatChosen !== undefined\n const hasVHatTarget = rawVHatTarget !== null && rawVHatTarget !== undefined\n if (hasQHatChosen !== hasVHatTarget) {\n throw new ValidationError(\n `doublyRobust: qHatChosen and vHatTarget must be supplied together (runId=${t.runId})`,\n )\n }\n\n if (hasQHatChosen && hasVHatTarget) {\n if (!Number.isFinite(rawQHatChosen) || !Number.isFinite(rawVHatTarget)) {\n throw new ValidationError(\n `doublyRobust: qHatChosen and vHatTarget must be finite (runId=${t.runId})`,\n )\n }\n const qHatChosen = clamp(rawQHatChosen, clip.low, clip.high)\n const vHatTarget = clamp(rawVHatTarget, clip.low, clip.high)\n contributions.push(vHatTarget + w * (r - qHatChosen))\n contributionCounts.dr += 1\n } else if (typeof t.qHat === 'number' && Number.isFinite(t.qHat)) {\n const qHat = clamp(t.qHat, clip.low, clip.high)\n contributions.push(qHat + w * (r - qHat))\n contributionCounts.legacyScalar += 1\n } else {\n contributions.push(w * r)\n contributionCounts.ipsFallback += 1\n }\n if (w > maxW) maxW = w\n sumW += w\n sumW2 += w * w\n }\n const n = contributions.length\n const value = contributions.reduce((s, x) => s + x, 0) / n\n const variance = contributions.reduce((s, x) => s + (x - value) ** 2, 0) / Math.max(1, n - 1)\n const effN = sumW === 0 ? 0 : (sumW * sumW) / sumW2\n return {\n value,\n standardError: Math.sqrt(variance / n),\n effectiveSampleSize: effN,\n n,\n maxImportanceWeight: maxW,\n contributionCounts,\n }\n}\n\n/**\n * Convenience: run all three estimators and return them side-by-side.\n * The recommended diagnostic — agreement across estimators is a much\n * stronger signal than any single one.\n */\nexport function offPolicyEstimateAll(\n trajectories: OffPolicyTrajectory[],\n opts: OffPolicyOptions = {},\n): { ips: OffPolicyEstimate; snips: OffPolicyEstimate; dr: OffPolicyEstimate } {\n return {\n ips: inverseProbabilityWeighting(trajectories, opts),\n snips: selfNormalizedImportanceWeighting(trajectories, opts),\n dr: doublyRobust(trajectories, opts),\n }\n}\n\n// ── Helpers ──────────────────────────────────────────────────────────────\n\nfunction zeroEstimate(): OffPolicyEstimate {\n return { value: 0, standardError: 0, effectiveSampleSize: 0, n: 0, maxImportanceWeight: 0 }\n}\n\nfunction clamp(x: number, lo: number, hi: number): number {\n if (!Number.isFinite(x)) return lo\n return Math.max(lo, Math.min(hi, x))\n}\n"],"mappings":";;;;;AA2HO,SAAS,4BACd,cACA,OAAyB,CAAC,GACP;AACnB,QAAM,MAAM,KAAK,aAAa;AAC9B,QAAM,OAAO,KAAK,cAAc,EAAE,KAAK,GAAG,MAAM,EAAE;AAElD,MAAI,aAAa,WAAW,GAAG;AAC7B,WAAO,aAAa;AAAA,EACtB;AAEA,QAAM,UAAoB,CAAC;AAC3B,QAAM,kBAA4B,CAAC;AACnC,MAAI,OAAO;AACX,aAAW,KAAK,cAAc;AAC5B,QAAI,EAAE,gBAAgB,GAAG;AACvB,YAAM,IAAI;AAAA,QACR,gEAAgE,EAAE,KAAK;AAAA,MACzE;AAAA,IACF;AACA,UAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;AACrD,UAAM,IAAI,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI;AAC7C,YAAQ,KAAK,CAAC;AACd,oBAAgB,KAAK,IAAI,CAAC;AAC1B,QAAI,IAAI,KAAM,QAAO;AAAA,EACvB;AACA,QAAM,IAAI,QAAQ;AAClB,QAAM,QAAQ,gBAAgB,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;AAC3D,QAAM,WAAW,gBAAgB,OAAO,CAAC,GAAG,MAAM,KAAK,IAAI,UAAU,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,IAAI,CAAC;AAC9F,QAAM,OAAO,QAAQ,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC;AAC9C,QAAM,QAAQ,QAAQ,OAAO,CAAC,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC;AACnD,QAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;AAE9C,SAAO;AAAA,IACL;AAAA,IACA,eAAe,KAAK,KAAK,WAAW,CAAC;AAAA,IACrC,qBAAqB;AAAA,IACrB;AAAA,IACA,qBAAqB;AAAA,EACvB;AACF;AAOO,SAAS,kCACd,cACA,OAAyB,CAAC,GACP;AACnB,QAAM,MAAM,KAAK,aAAa;AAC9B,QAAM,OAAO,KAAK,cAAc,EAAE,KAAK,GAAG,MAAM,EAAE;AAClD,MAAI,aAAa,WAAW,EAAG,QAAO,aAAa;AAEnD,QAAM,UAAoB,CAAC;AAC3B,QAAM,UAAoB,CAAC;AAC3B,MAAI,OAAO;AACX,aAAW,KAAK,cAAc;AAC5B,QAAI,EAAE,gBAAgB,GAAG;AACvB,YAAM,IAAI;AAAA,QACR,sEAAsE,EAAE,KAAK;AAAA,MAC/E;AAAA,IACF;AACA,UAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;AACrD,YAAQ,KAAK,CAAC;AACd,YAAQ,KAAK,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI,CAAC;AACjD,QAAI,IAAI,KAAM,QAAO;AAAA,EACvB;AACA,QAAM,OAAO,QAAQ,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC;AAC9C,QAAM,QAAQ,QAAQ,OAAO,CAAC,GAAG,GAAG,MAAM,IAAI,IAAI,QAAQ,CAAC,GAAI,CAAC;AAChE,QAAM,QAAQ,SAAS,IAAI,IAAI,QAAQ;AACvC,QAAM,QAAQ,QAAQ,OAAO,CAAC,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC;AACnD,QAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;AAE9C,QAAM,MAAM,QAAQ,IAAI,CAAC,GAAG,MAAM,KAAK,QAAQ,CAAC,IAAK,MAAM;AAC3D,QAAM,WAAW,IAAI,OAAO,CAAC,GAAG,MAAM,IAAI,IAAI,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,OAAO,IAAI;AAC7E,SAAO;AAAA,IACL;AAAA,IACA,eAAe,KAAK,KAAK,QAAQ;AAAA,IACjC,qBAAqB;AAAA,IACrB,GAAG,aAAa;AAAA,IAChB,qBAAqB;AAAA,EACvB;AACF;AAuBO,SAAS,aACd,cACA,OAAyB,CAAC,GACP;AACnB,QAAM,MAAM,KAAK,aAAa;AAC9B,QAAM,OAAO,KAAK,cAAc,EAAE,KAAK,GAAG,MAAM,EAAE;AAClD,MAAI,aAAa,WAAW,GAAG;AAC7B,WAAO;AAAA,MACL,GAAG,aAAa;AAAA,MAChB,oBAAoB,EAAE,IAAI,GAAG,aAAa,GAAG,cAAc,EAAE;AAAA,IAC/D;AAAA,EACF;AAEA,QAAM,gBAA0B,CAAC;AACjC,QAAM,qBAAkD;AAAA,IACtD,IAAI;AAAA,IACJ,aAAa;AAAA,IACb,cAAc;AAAA,EAChB;AACA,MAAI,OAAO;AACX,MAAI,OAAO;AACX,MAAI,QAAQ;AACZ,aAAW,KAAK,cAAc;AAC5B,QAAI,EAAE,gBAAgB,GAAG;AACvB,YAAM,IAAI,gBAAgB,iDAAiD,EAAE,KAAK,GAAG;AAAA,IACvF;AACA,UAAM,IAAI,KAAK,IAAI,KAAK,EAAE,aAAa,EAAE,YAAY;AACrD,UAAM,IAAI,MAAM,EAAE,QAAQ,KAAK,KAAK,KAAK,IAAI;AAC7C,UAAM,gBAAgB,EAAE;AACxB,UAAM,gBAAgB,EAAE;AACxB,UAAM,gBAAgB,kBAAkB,QAAQ,kBAAkB;AAClE,UAAM,gBAAgB,kBAAkB,QAAQ,kBAAkB;AAClE,QAAI,kBAAkB,eAAe;AACnC,YAAM,IAAI;AAAA,QACR,4EAA4E,EAAE,KAAK;AAAA,MACrF;AAAA,IACF;AAEA,QAAI,iBAAiB,eAAe;AAClC,UAAI,CAAC,OAAO,SAAS,aAAa,KAAK,CAAC,OAAO,SAAS,aAAa,GAAG;AACtE,cAAM,IAAI;AAAA,UACR,iEAAiE,EAAE,KAAK;AAAA,QAC1E;AAAA,MACF;AACA,YAAM,aAAa,MAAM,eAAe,KAAK,KAAK,KAAK,IAAI;AAC3D,YAAM,aAAa,MAAM,eAAe,KAAK,KAAK,KAAK,IAAI;AAC3D,oBAAc,KAAK,aAAa,KAAK,IAAI,WAAW;AACpD,yBAAmB,MAAM;AAAA,IAC3B,WAAW,OAAO,EAAE,SAAS,YAAY,OAAO,SAAS,EAAE,IAAI,GAAG;AAChE,YAAM,OAAO,MAAM,EAAE,MAAM,KAAK,KAAK,KAAK,IAAI;AAC9C,oBAAc,KAAK,OAAO,KAAK,IAAI,KAAK;AACxC,yBAAmB,gBAAgB;AAAA,IACrC,OAAO;AACL,oBAAc,KAAK,IAAI,CAAC;AACxB,yBAAmB,eAAe;AAAA,IACpC;AACA,QAAI,IAAI,KAAM,QAAO;AACrB,YAAQ;AACR,aAAS,IAAI;AAAA,EACf;AACA,QAAM,IAAI,cAAc;AACxB,QAAM,QAAQ,cAAc,OAAO,CAAC,GAAG,MAAM,IAAI,GAAG,CAAC,IAAI;AACzD,QAAM,WAAW,cAAc,OAAO,CAAC,GAAG,MAAM,KAAK,IAAI,UAAU,GAAG,CAAC,IAAI,KAAK,IAAI,GAAG,IAAI,CAAC;AAC5F,QAAM,OAAO,SAAS,IAAI,IAAK,OAAO,OAAQ;AAC9C,SAAO;AAAA,IACL;AAAA,IACA,eAAe,KAAK,KAAK,WAAW,CAAC;AAAA,IACrC,qBAAqB;AAAA,IACrB;AAAA,IACA,qBAAqB;AAAA,IACrB;AAAA,EACF;AACF;AAOO,SAAS,qBACd,cACA,OAAyB,CAAC,GACmD;AAC7E,SAAO;AAAA,IACL,KAAK,4BAA4B,cAAc,IAAI;AAAA,IACnD,OAAO,kCAAkC,cAAc,IAAI;AAAA,IAC3D,IAAI,aAAa,cAAc,IAAI;AAAA,EACrC;AACF;AAIA,SAAS,eAAkC;AACzC,SAAO,EAAE,OAAO,GAAG,eAAe,GAAG,qBAAqB,GAAG,GAAG,GAAG,qBAAqB,EAAE;AAC5F;AAEA,SAAS,MAAM,GAAW,IAAY,IAAoB;AACxD,MAAI,CAAC,OAAO,SAAS,CAAC,EAAG,QAAO;AAChC,SAAO,KAAK,IAAI,IAAI,KAAK,IAAI,IAAI,CAAC,CAAC;AACrC;","names":[]}
|
|
@@ -3,13 +3,13 @@ import {
|
|
|
3
3
|
aggregateRunScore,
|
|
4
4
|
clamp01,
|
|
5
5
|
computeFindingId
|
|
6
|
-
} from "./chunk-
|
|
6
|
+
} from "./chunk-WXQTVEKM.js";
|
|
7
7
|
import {
|
|
8
8
|
callLlmJson,
|
|
9
9
|
costReceiptFromLlm,
|
|
10
10
|
costReceiptFromLlmError,
|
|
11
11
|
maximumChargeForLlmRequest
|
|
12
|
-
} from "./chunk-
|
|
12
|
+
} from "./chunk-J7S4YM27.js";
|
|
13
13
|
import {
|
|
14
14
|
CostLedger
|
|
15
15
|
} from "./chunk-BGVTIE2C.js";
|
|
@@ -747,4 +747,4 @@ export {
|
|
|
747
747
|
runSemanticConceptJudge,
|
|
748
748
|
createSemanticConceptJudge
|
|
749
749
|
};
|
|
750
|
-
//# sourceMappingURL=chunk-
|
|
750
|
+
//# sourceMappingURL=chunk-WMJR67FX.js.map
|
|
@@ -3,7 +3,7 @@ import {
|
|
|
3
3
|
costReceiptFromLlm,
|
|
4
4
|
costReceiptFromLlmError,
|
|
5
5
|
maximumChargeForLlmRequest
|
|
6
|
-
} from "./chunk-
|
|
6
|
+
} from "./chunk-J7S4YM27.js";
|
|
7
7
|
import {
|
|
8
8
|
CostLedger
|
|
9
9
|
} from "./chunk-BGVTIE2C.js";
|
|
@@ -13,7 +13,7 @@ import {
|
|
|
13
13
|
} from "./chunk-IR3KBHOY.js";
|
|
14
14
|
import {
|
|
15
15
|
validateAgentProfileCell
|
|
16
|
-
} from "./chunk-
|
|
16
|
+
} from "./chunk-GC4ATIKK.js";
|
|
17
17
|
import {
|
|
18
18
|
canonicalize
|
|
19
19
|
} from "./chunk-VSMTAMNK.js";
|
|
@@ -2738,4 +2738,4 @@ export {
|
|
|
2738
2738
|
aggregateRunScore,
|
|
2739
2739
|
clamp012 as clamp01
|
|
2740
2740
|
};
|
|
2741
|
-
//# sourceMappingURL=chunk-
|
|
2741
|
+
//# sourceMappingURL=chunk-WXQTVEKM.js.map
|
package/dist/cli.js
CHANGED
|
@@ -4,9 +4,9 @@ import {
|
|
|
4
4
|
handleVersion,
|
|
5
5
|
runRpcBatch,
|
|
6
6
|
runRpcOnce,
|
|
7
|
-
|
|
8
|
-
} from "./chunk-
|
|
9
|
-
import "./chunk-
|
|
7
|
+
startServerAsync
|
|
8
|
+
} from "./chunk-EBDOTTZJ.js";
|
|
9
|
+
import "./chunk-J7S4YM27.js";
|
|
10
10
|
import "./chunk-BGVTIE2C.js";
|
|
11
11
|
import "./chunk-VI2UW6B6.js";
|
|
12
12
|
import "./chunk-PC4UYEBM.js";
|
|
@@ -15,6 +15,25 @@ import "./chunk-PZ5AY32C.js";
|
|
|
15
15
|
|
|
16
16
|
// src/cli.ts
|
|
17
17
|
import { writeFileSync } from "fs";
|
|
18
|
+
|
|
19
|
+
// src/cli-config.ts
|
|
20
|
+
function resolveCliLlmConfig(env = process.env) {
|
|
21
|
+
const explicitBaseUrl = nonEmpty(env.AGENT_EVAL_LLM_BASE_URL);
|
|
22
|
+
const explicitApiKey = nonEmpty(env.AGENT_EVAL_LLM_API_KEY);
|
|
23
|
+
const openAiApiKey = nonEmpty(env.OPENAI_API_KEY);
|
|
24
|
+
const tangleApiKey = nonEmpty(env.TANGLE_API_KEY);
|
|
25
|
+
const baseUrl = explicitBaseUrl ?? nonEmpty(env.OPENAI_BASE_URL) ?? nonEmpty(env.TANGLE_ROUTER_URL) ?? (openAiApiKey ? "https://api.openai.com/v1" : void 0) ?? (tangleApiKey ? "https://router.tangle.tools/v1" : void 0);
|
|
26
|
+
const apiKey = explicitApiKey ?? openAiApiKey ?? tangleApiKey;
|
|
27
|
+
const model = nonEmpty(env.AGENT_EVAL_LLM_MODEL) ?? nonEmpty(env.OPENAI_MODEL) ?? nonEmpty(env.TANGLE_MODEL);
|
|
28
|
+
const client = baseUrl || apiKey ? { ...baseUrl ? { baseUrl } : {}, ...apiKey ? { apiKey } : {} } : void 0;
|
|
29
|
+
return { ...client ? { client } : {}, ...model ? { model } : {} };
|
|
30
|
+
}
|
|
31
|
+
function nonEmpty(value) {
|
|
32
|
+
const trimmed = value?.trim();
|
|
33
|
+
return trimmed ? trimmed : void 0;
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
// src/cli.ts
|
|
18
37
|
function parseArgs(argv) {
|
|
19
38
|
const [command, ...rest] = argv;
|
|
20
39
|
const positional = [];
|
|
@@ -22,7 +41,17 @@ function parseArgs(argv) {
|
|
|
22
41
|
for (let i = 0; i < rest.length; i++) {
|
|
23
42
|
const tok = rest[i];
|
|
24
43
|
if (tok.startsWith("--")) {
|
|
25
|
-
const
|
|
44
|
+
const raw = tok.slice(2);
|
|
45
|
+
const equalsAt = raw.indexOf("=");
|
|
46
|
+
if (equalsAt >= 0) {
|
|
47
|
+
flags[raw.slice(0, equalsAt)] = raw.slice(equalsAt + 1);
|
|
48
|
+
continue;
|
|
49
|
+
}
|
|
50
|
+
const key = raw;
|
|
51
|
+
if (key === "help") {
|
|
52
|
+
flags[key] = "true";
|
|
53
|
+
continue;
|
|
54
|
+
}
|
|
26
55
|
const next = rest[i + 1];
|
|
27
56
|
if (next != null && !next.startsWith("--")) {
|
|
28
57
|
flags[key] = next;
|
|
@@ -30,20 +59,22 @@ function parseArgs(argv) {
|
|
|
30
59
|
} else {
|
|
31
60
|
flags[key] = "true";
|
|
32
61
|
}
|
|
62
|
+
} else if (tok === "-h") {
|
|
63
|
+
flags.help = "true";
|
|
33
64
|
} else {
|
|
34
65
|
positional.push(tok);
|
|
35
66
|
}
|
|
36
67
|
}
|
|
37
68
|
return { command: command ?? "help", positional, flags };
|
|
38
69
|
}
|
|
39
|
-
var HELP = `agent-eval
|
|
70
|
+
var HELP = `agent-eval: evaluation RPC and HTTP server.
|
|
40
71
|
|
|
41
72
|
Commands:
|
|
42
73
|
serve [--port 5005] [--host 127.0.0.1]
|
|
43
74
|
Start the HTTP server. POST /v1/judge, GET /v1/rubrics, GET /v1/version, GET /openapi.json.
|
|
44
75
|
rpc <method>
|
|
45
76
|
Read one JSON object from stdin (the params for <method>), write one
|
|
46
|
-
JSON object to stdout.
|
|
77
|
+
JSON object to stdout. Methods: judge, listRubrics, version.
|
|
47
78
|
rpc-batch <method>
|
|
48
79
|
Like 'rpc' but JSONL in / JSONL out.
|
|
49
80
|
openapi [--out openapi.json]
|
|
@@ -51,14 +82,31 @@ Commands:
|
|
|
51
82
|
version
|
|
52
83
|
Print server + wire-protocol version JSON.
|
|
53
84
|
|
|
85
|
+
Judge provider:
|
|
86
|
+
Set AGENT_EVAL_LLM_BASE_URL, AGENT_EVAL_LLM_API_KEY, and AGENT_EVAL_LLM_MODEL.
|
|
87
|
+
OPENAI_* and TANGLE_* equivalents are also accepted.
|
|
88
|
+
|
|
54
89
|
Without arguments, prints this help.`;
|
|
55
90
|
async function main() {
|
|
56
91
|
const { command, positional, flags } = parseArgs(process.argv.slice(2));
|
|
92
|
+
assertKnownFlags(command, flags);
|
|
93
|
+
if (flags.help === "true") {
|
|
94
|
+
process.stdout.write(`${HELP}
|
|
95
|
+
`);
|
|
96
|
+
return 0;
|
|
97
|
+
}
|
|
57
98
|
switch (command) {
|
|
58
99
|
case "serve": {
|
|
59
|
-
const port =
|
|
100
|
+
const port = parsePort(flags.port ?? "5005");
|
|
60
101
|
const host = flags.host ?? "127.0.0.1";
|
|
61
|
-
const
|
|
102
|
+
const llm = resolveCliLlmConfig();
|
|
103
|
+
const { server } = await startServerAsync({
|
|
104
|
+
port,
|
|
105
|
+
host,
|
|
106
|
+
llm: llm.client,
|
|
107
|
+
judgeModel: llm.model,
|
|
108
|
+
llmRouteRequirements: { requireExplicitBaseUrl: true }
|
|
109
|
+
});
|
|
62
110
|
const shutdown = (sig) => {
|
|
63
111
|
console.log(`[agent-eval] received ${sig}, shutting down`);
|
|
64
112
|
server.close(() => process.exit(0));
|
|
@@ -72,11 +120,21 @@ async function main() {
|
|
|
72
120
|
}
|
|
73
121
|
case "rpc": {
|
|
74
122
|
const [method] = positional;
|
|
75
|
-
|
|
123
|
+
const llm = resolveCliLlmConfig();
|
|
124
|
+
return await runRpcOnce(method, {
|
|
125
|
+
llm: llm.client,
|
|
126
|
+
judgeModel: llm.model,
|
|
127
|
+
llmRouteRequirements: { requireExplicitBaseUrl: true }
|
|
128
|
+
});
|
|
76
129
|
}
|
|
77
130
|
case "rpc-batch": {
|
|
78
131
|
const [method] = positional;
|
|
79
|
-
|
|
132
|
+
const llm = resolveCliLlmConfig();
|
|
133
|
+
return await runRpcBatch(method, {
|
|
134
|
+
llm: llm.client,
|
|
135
|
+
judgeModel: llm.model,
|
|
136
|
+
llmRouteRequirements: { requireExplicitBaseUrl: true }
|
|
137
|
+
});
|
|
80
138
|
}
|
|
81
139
|
case "openapi": {
|
|
82
140
|
const out = flags.out ?? "openapi.json";
|
|
@@ -88,6 +146,11 @@ async function main() {
|
|
|
88
146
|
}
|
|
89
147
|
case "version": {
|
|
90
148
|
process.stdout.write(`${JSON.stringify(handleVersion(), null, 2)}
|
|
149
|
+
`);
|
|
150
|
+
return 0;
|
|
151
|
+
}
|
|
152
|
+
case "--version": {
|
|
153
|
+
process.stdout.write(`${handleVersion().version}
|
|
91
154
|
`);
|
|
92
155
|
return 0;
|
|
93
156
|
}
|
|
@@ -105,6 +168,33 @@ ${HELP}
|
|
|
105
168
|
return 1;
|
|
106
169
|
}
|
|
107
170
|
}
|
|
171
|
+
var FLAGS_BY_COMMAND = {
|
|
172
|
+
serve: /* @__PURE__ */ new Set(["help", "host", "port"]),
|
|
173
|
+
rpc: /* @__PURE__ */ new Set(["help"]),
|
|
174
|
+
"rpc-batch": /* @__PURE__ */ new Set(["help"]),
|
|
175
|
+
openapi: /* @__PURE__ */ new Set(["help", "out"]),
|
|
176
|
+
version: /* @__PURE__ */ new Set(["help"]),
|
|
177
|
+
help: /* @__PURE__ */ new Set(),
|
|
178
|
+
"--help": /* @__PURE__ */ new Set(),
|
|
179
|
+
"-h": /* @__PURE__ */ new Set(),
|
|
180
|
+
"--version": /* @__PURE__ */ new Set(),
|
|
181
|
+
"": /* @__PURE__ */ new Set()
|
|
182
|
+
};
|
|
183
|
+
function assertKnownFlags(command, flags) {
|
|
184
|
+
const allowed = FLAGS_BY_COMMAND[command];
|
|
185
|
+
if (!allowed) return;
|
|
186
|
+
const unknown = Object.keys(flags).filter((flag) => !allowed.has(flag));
|
|
187
|
+
if (unknown.length > 0) {
|
|
188
|
+
throw new Error(`unknown flag for ${command || "help"}: --${unknown[0]}`);
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
function parsePort(raw) {
|
|
192
|
+
const port = Number(raw);
|
|
193
|
+
if (!Number.isInteger(port) || port < 0 || port > 65535) {
|
|
194
|
+
throw new Error(`--port must be an integer from 0 to 65535; received ${JSON.stringify(raw)}`);
|
|
195
|
+
}
|
|
196
|
+
return port;
|
|
197
|
+
}
|
|
108
198
|
main().then((code) => process.exit(code)).catch((err) => {
|
|
109
199
|
console.error("[agent-eval] cli error:", err);
|
|
110
200
|
process.exit(1);
|
package/dist/cli.js.map
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
{"version":3,"sources":["../src/cli.ts"],"sourcesContent":["#!/usr/bin/env node\n/**\n * agent-eval CLI.\n *\n * agent-eval serve [--port 5005] [--host 127.0.0.1]\n * agent-eval rpc <method> # one request from stdin → one response on stdout\n * agent-eval rpc-batch <method> # JSONL stdin → JSONL stdout\n * agent-eval openapi [--out path] # write OpenAPI spec\n * agent-eval version\n *\n * <method> is one of: judge, listRubrics, version. When omitted, the\n * stdin payload must be a full {method, params} envelope.\n */\nimport { writeFileSync } from 'node:fs'\nimport { handleVersion } from './wire/handlers'\nimport { buildOpenApi } from './wire/openapi'\nimport { runRpcBatch, runRpcOnce } from './wire/rpc'\nimport { startServer } from './wire/server'\n\ninterface Args {\n command: string\n positional: string[]\n flags: Record<string, string>\n}\n\nfunction parseArgs(argv: string[]): Args {\n const [command, ...rest] = argv\n const positional: string[] = []\n const flags: Record<string, string> = {}\n for (let i = 0; i < rest.length; i++) {\n const tok = rest[i]!\n if (tok.startsWith('--')) {\n const key = tok.slice(2)\n const next = rest[i + 1]\n if (next != null && !next.startsWith('--')) {\n flags[key] = next\n i++\n } else {\n flags[key] = 'true'\n }\n } else {\n positional.push(tok)\n }\n }\n return { command: command ?? 'help', positional, flags }\n}\n\nconst HELP = `agent-eval — wire-protocol entry point.\n\nCommands:\n serve [--port 5005] [--host 127.0.0.1]\n Start the HTTP server. POST /v1/judge, GET /v1/rubrics, GET /v1/version, GET /openapi.json.\n rpc <method>\n Read one JSON object from stdin (the params for <method>), write one\n JSON object to stdout. Method ∈ {judge, listRubrics, version}.\n rpc-batch <method>\n Like 'rpc' but JSONL in / JSONL out.\n openapi [--out openapi.json]\n Write the OpenAPI 3.1 spec.\n version\n Print server + wire-protocol version JSON.\n\nWithout arguments, prints this help.`\n\nasync function main(): Promise<number> {\n const { command, positional, flags } = parseArgs(process.argv.slice(2))\n\n switch (command) {\n case 'serve': {\n const port = Number(flags.port ?? 5005)\n const host = flags.host ?? '127.0.0.1'\n const server = startServer({ port, host })\n // Keep process alive on SIGINT/SIGTERM\n const shutdown = (sig: string) => {\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] received ${sig}, shutting down`)\n server.close(() => process.exit(0))\n // Force exit after 5s if close hangs\n setTimeout(() => process.exit(1), 5000).unref()\n }\n process.on('SIGINT', () => shutdown('SIGINT'))\n process.on('SIGTERM', () => shutdown('SIGTERM'))\n // Block forever\n await new Promise(() => {})\n return 0\n }\n case 'rpc': {\n const [method] = positional\n return await runRpcOnce(method)\n }\n case 'rpc-batch': {\n const [method] = positional\n return await runRpcBatch(method)\n }\n case 'openapi': {\n const out = flags.out ?? 'openapi.json'\n const spec = buildOpenApi(handleVersion().version)\n writeFileSync(out, `${JSON.stringify(spec, null, 2)}\\n`, 'utf-8')\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] wrote OpenAPI 3.1 spec to ${out}`)\n return 0\n }\n case 'version': {\n process.stdout.write(`${JSON.stringify(handleVersion(), null, 2)}\\n`)\n return 0\n }\n case 'help':\n case '--help':\n case '-h':\n case '':\n process.stdout.write(`${HELP}\\n`)\n return 0\n default:\n process.stderr.write(`unknown command: ${command}\\n${HELP}\\n`)\n return 1\n }\n}\n\nmain()\n .then((code) => process.exit(code))\n .catch((err) => {\n // eslint-disable-next-line no-console\n console.error('[agent-eval] cli error:', err)\n process.exit(1)\n })\n"],"mappings":";;;;;;;;;;;;;;;;AAaA,SAAS,qBAAqB;AAY9B,SAAS,UAAU,MAAsB;AACvC,QAAM,CAAC,SAAS,GAAG,IAAI,IAAI;AAC3B,QAAM,aAAuB,CAAC;AAC9B,QAAM,QAAgC,CAAC;AACvC,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;AACpC,UAAM,MAAM,KAAK,CAAC;AAClB,QAAI,IAAI,WAAW,IAAI,GAAG;AACxB,YAAM,MAAM,IAAI,MAAM,CAAC;AACvB,YAAM,OAAO,KAAK,IAAI,CAAC;AACvB,UAAI,QAAQ,QAAQ,CAAC,KAAK,WAAW,IAAI,GAAG;AAC1C,cAAM,GAAG,IAAI;AACb;AAAA,MACF,OAAO;AACL,cAAM,GAAG,IAAI;AAAA,MACf;AAAA,IACF,OAAO;AACL,iBAAW,KAAK,GAAG;AAAA,IACrB;AAAA,EACF;AACA,SAAO,EAAE,SAAS,WAAW,QAAQ,YAAY,MAAM;AACzD;AAEA,IAAM,OAAO;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAiBb,eAAe,OAAwB;AACrC,QAAM,EAAE,SAAS,YAAY,MAAM,IAAI,UAAU,QAAQ,KAAK,MAAM,CAAC,CAAC;AAEtE,UAAQ,SAAS;AAAA,IACf,KAAK,SAAS;AACZ,YAAM,OAAO,OAAO,MAAM,QAAQ,IAAI;AACtC,YAAM,OAAO,MAAM,QAAQ;AAC3B,YAAM,SAAS,YAAY,EAAE,MAAM,KAAK,CAAC;AAEzC,YAAM,WAAW,CAAC,QAAgB;AAEhC,gBAAQ,IAAI,yBAAyB,GAAG,iBAAiB;AACzD,eAAO,MAAM,MAAM,QAAQ,KAAK,CAAC,CAAC;AAElC,mBAAW,MAAM,QAAQ,KAAK,CAAC,GAAG,GAAI,EAAE,MAAM;AAAA,MAChD;AACA,cAAQ,GAAG,UAAU,MAAM,SAAS,QAAQ,CAAC;AAC7C,cAAQ,GAAG,WAAW,MAAM,SAAS,SAAS,CAAC;AAE/C,YAAM,IAAI,QAAQ,MAAM;AAAA,MAAC,CAAC;AAC1B,aAAO;AAAA,IACT;AAAA,IACA,KAAK,OAAO;AACV,YAAM,CAAC,MAAM,IAAI;AACjB,aAAO,MAAM,WAAW,MAAM;AAAA,IAChC;AAAA,IACA,KAAK,aAAa;AAChB,YAAM,CAAC,MAAM,IAAI;AACjB,aAAO,MAAM,YAAY,MAAM;AAAA,IACjC;AAAA,IACA,KAAK,WAAW;AACd,YAAM,MAAM,MAAM,OAAO;AACzB,YAAM,OAAO,aAAa,cAAc,EAAE,OAAO;AACjD,oBAAc,KAAK,GAAG,KAAK,UAAU,MAAM,MAAM,CAAC,CAAC;AAAA,GAAM,OAAO;AAEhE,cAAQ,IAAI,0CAA0C,GAAG,EAAE;AAC3D,aAAO;AAAA,IACT;AAAA,IACA,KAAK,WAAW;AACd,cAAQ,OAAO,MAAM,GAAG,KAAK,UAAU,cAAc,GAAG,MAAM,CAAC,CAAC;AAAA,CAAI;AACpE,aAAO;AAAA,IACT;AAAA,IACA,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AACH,cAAQ,OAAO,MAAM,GAAG,IAAI;AAAA,CAAI;AAChC,aAAO;AAAA,IACT;AACE,cAAQ,OAAO,MAAM,oBAAoB,OAAO;AAAA,EAAK,IAAI;AAAA,CAAI;AAC7D,aAAO;AAAA,EACX;AACF;AAEA,KAAK,EACF,KAAK,CAAC,SAAS,QAAQ,KAAK,IAAI,CAAC,EACjC,MAAM,CAAC,QAAQ;AAEd,UAAQ,MAAM,2BAA2B,GAAG;AAC5C,UAAQ,KAAK,CAAC;AAChB,CAAC;","names":[]}
|
|
1
|
+
{"version":3,"sources":["../src/cli.ts","../src/cli-config.ts"],"sourcesContent":["#!/usr/bin/env node\n/**\n * agent-eval CLI.\n *\n * agent-eval serve [--port 5005] [--host 127.0.0.1]\n * agent-eval rpc <method> # one request from stdin → one response on stdout\n * agent-eval rpc-batch <method> # JSONL stdin → JSONL stdout\n * agent-eval openapi [--out path] # write OpenAPI spec\n * agent-eval version\n *\n * <method> is one of: judge, listRubrics, version. When omitted, the\n * stdin payload must be a full {method, params} envelope.\n */\nimport { writeFileSync } from 'node:fs'\nimport { resolveCliLlmConfig } from './cli-config'\nimport { handleVersion } from './wire/handlers'\nimport { buildOpenApi } from './wire/openapi'\nimport { runRpcBatch, runRpcOnce } from './wire/rpc'\nimport { startServerAsync } from './wire/server'\n\ninterface Args {\n command: string\n positional: string[]\n flags: Record<string, string>\n}\n\nfunction parseArgs(argv: string[]): Args {\n const [command, ...rest] = argv\n const positional: string[] = []\n const flags: Record<string, string> = {}\n for (let i = 0; i < rest.length; i++) {\n const tok = rest[i]!\n if (tok.startsWith('--')) {\n const raw = tok.slice(2)\n const equalsAt = raw.indexOf('=')\n if (equalsAt >= 0) {\n flags[raw.slice(0, equalsAt)] = raw.slice(equalsAt + 1)\n continue\n }\n const key = raw\n if (key === 'help') {\n flags[key] = 'true'\n continue\n }\n const next = rest[i + 1]\n if (next != null && !next.startsWith('--')) {\n flags[key] = next\n i++\n } else {\n flags[key] = 'true'\n }\n } else if (tok === '-h') {\n flags.help = 'true'\n } else {\n positional.push(tok)\n }\n }\n return { command: command ?? 'help', positional, flags }\n}\n\nconst HELP = `agent-eval: evaluation RPC and HTTP server.\n\nCommands:\n serve [--port 5005] [--host 127.0.0.1]\n Start the HTTP server. POST /v1/judge, GET /v1/rubrics, GET /v1/version, GET /openapi.json.\n rpc <method>\n Read one JSON object from stdin (the params for <method>), write one\n JSON object to stdout. Methods: judge, listRubrics, version.\n rpc-batch <method>\n Like 'rpc' but JSONL in / JSONL out.\n openapi [--out openapi.json]\n Write the OpenAPI 3.1 spec.\n version\n Print server + wire-protocol version JSON.\n\nJudge provider:\n Set AGENT_EVAL_LLM_BASE_URL, AGENT_EVAL_LLM_API_KEY, and AGENT_EVAL_LLM_MODEL.\n OPENAI_* and TANGLE_* equivalents are also accepted.\n\nWithout arguments, prints this help.`\n\nasync function main(): Promise<number> {\n const { command, positional, flags } = parseArgs(process.argv.slice(2))\n assertKnownFlags(command, flags)\n\n if (flags.help === 'true') {\n process.stdout.write(`${HELP}\\n`)\n return 0\n }\n\n switch (command) {\n case 'serve': {\n const port = parsePort(flags.port ?? '5005')\n const host = flags.host ?? '127.0.0.1'\n const llm = resolveCliLlmConfig()\n const { server } = await startServerAsync({\n port,\n host,\n llm: llm.client,\n judgeModel: llm.model,\n llmRouteRequirements: { requireExplicitBaseUrl: true },\n })\n // Keep process alive on SIGINT/SIGTERM\n const shutdown = (sig: string) => {\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] received ${sig}, shutting down`)\n server.close(() => process.exit(0))\n // Force exit after 5s if close hangs\n setTimeout(() => process.exit(1), 5000).unref()\n }\n process.on('SIGINT', () => shutdown('SIGINT'))\n process.on('SIGTERM', () => shutdown('SIGTERM'))\n // Block forever\n await new Promise(() => {})\n return 0\n }\n case 'rpc': {\n const [method] = positional\n const llm = resolveCliLlmConfig()\n return await runRpcOnce(method, {\n llm: llm.client,\n judgeModel: llm.model,\n llmRouteRequirements: { requireExplicitBaseUrl: true },\n })\n }\n case 'rpc-batch': {\n const [method] = positional\n const llm = resolveCliLlmConfig()\n return await runRpcBatch(method, {\n llm: llm.client,\n judgeModel: llm.model,\n llmRouteRequirements: { requireExplicitBaseUrl: true },\n })\n }\n case 'openapi': {\n const out = flags.out ?? 'openapi.json'\n const spec = buildOpenApi(handleVersion().version)\n writeFileSync(out, `${JSON.stringify(spec, null, 2)}\\n`, 'utf-8')\n // eslint-disable-next-line no-console\n console.log(`[agent-eval] wrote OpenAPI 3.1 spec to ${out}`)\n return 0\n }\n case 'version': {\n process.stdout.write(`${JSON.stringify(handleVersion(), null, 2)}\\n`)\n return 0\n }\n case '--version': {\n process.stdout.write(`${handleVersion().version}\\n`)\n return 0\n }\n case 'help':\n case '--help':\n case '-h':\n case '':\n process.stdout.write(`${HELP}\\n`)\n return 0\n default:\n process.stderr.write(`unknown command: ${command}\\n${HELP}\\n`)\n return 1\n }\n}\n\nconst FLAGS_BY_COMMAND: Record<string, ReadonlySet<string>> = {\n serve: new Set(['help', 'host', 'port']),\n rpc: new Set(['help']),\n 'rpc-batch': new Set(['help']),\n openapi: new Set(['help', 'out']),\n version: new Set(['help']),\n help: new Set(),\n '--help': new Set(),\n '-h': new Set(),\n '--version': new Set(),\n '': new Set(),\n}\n\nfunction assertKnownFlags(command: string, flags: Record<string, string>): void {\n const allowed = FLAGS_BY_COMMAND[command]\n if (!allowed) return\n const unknown = Object.keys(flags).filter((flag) => !allowed.has(flag))\n if (unknown.length > 0) {\n throw new Error(`unknown flag for ${command || 'help'}: --${unknown[0]}`)\n }\n}\n\nfunction parsePort(raw: string): number {\n const port = Number(raw)\n if (!Number.isInteger(port) || port < 0 || port > 65_535) {\n throw new Error(`--port must be an integer from 0 to 65535; received ${JSON.stringify(raw)}`)\n }\n return port\n}\n\nmain()\n .then((code) => process.exit(code))\n .catch((err) => {\n // eslint-disable-next-line no-console\n console.error('[agent-eval] cli error:', err)\n process.exit(1)\n })\n","import type { LlmClientOptions } from './llm-client'\n\nexport interface CliLlmConfig {\n client?: LlmClientOptions\n model?: string\n}\n\nexport function resolveCliLlmConfig(env: NodeJS.ProcessEnv = process.env): CliLlmConfig {\n const explicitBaseUrl = nonEmpty(env.AGENT_EVAL_LLM_BASE_URL)\n const explicitApiKey = nonEmpty(env.AGENT_EVAL_LLM_API_KEY)\n const openAiApiKey = nonEmpty(env.OPENAI_API_KEY)\n const tangleApiKey = nonEmpty(env.TANGLE_API_KEY)\n const baseUrl =\n explicitBaseUrl ??\n nonEmpty(env.OPENAI_BASE_URL) ??\n nonEmpty(env.TANGLE_ROUTER_URL) ??\n (openAiApiKey ? 'https://api.openai.com/v1' : undefined) ??\n (tangleApiKey ? 'https://router.tangle.tools/v1' : undefined)\n const apiKey = explicitApiKey ?? openAiApiKey ?? tangleApiKey\n const model =\n nonEmpty(env.AGENT_EVAL_LLM_MODEL) ?? nonEmpty(env.OPENAI_MODEL) ?? nonEmpty(env.TANGLE_MODEL)\n\n const client =\n baseUrl || apiKey\n ? { ...(baseUrl ? { baseUrl } : {}), ...(apiKey ? { apiKey } : {}) }\n : undefined\n return { ...(client ? { client } : {}), ...(model ? { model } : {}) }\n}\n\nfunction nonEmpty(value: string | undefined): string | undefined {\n const trimmed = value?.trim()\n return trimmed ? trimmed : undefined\n}\n"],"mappings":";;;;;;;;;;;;;;;;AAaA,SAAS,qBAAqB;;;ACNvB,SAAS,oBAAoB,MAAyB,QAAQ,KAAmB;AACtF,QAAM,kBAAkB,SAAS,IAAI,uBAAuB;AAC5D,QAAM,iBAAiB,SAAS,IAAI,sBAAsB;AAC1D,QAAM,eAAe,SAAS,IAAI,cAAc;AAChD,QAAM,eAAe,SAAS,IAAI,cAAc;AAChD,QAAM,UACJ,mBACA,SAAS,IAAI,eAAe,KAC5B,SAAS,IAAI,iBAAiB,MAC7B,eAAe,8BAA8B,YAC7C,eAAe,mCAAmC;AACrD,QAAM,SAAS,kBAAkB,gBAAgB;AACjD,QAAM,QACJ,SAAS,IAAI,oBAAoB,KAAK,SAAS,IAAI,YAAY,KAAK,SAAS,IAAI,YAAY;AAE/F,QAAM,SACJ,WAAW,SACP,EAAE,GAAI,UAAU,EAAE,QAAQ,IAAI,CAAC,GAAI,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC,EAAG,IACjE;AACN,SAAO,EAAE,GAAI,SAAS,EAAE,OAAO,IAAI,CAAC,GAAI,GAAI,QAAQ,EAAE,MAAM,IAAI,CAAC,EAAG;AACtE;AAEA,SAAS,SAAS,OAA+C;AAC/D,QAAM,UAAU,OAAO,KAAK;AAC5B,SAAO,UAAU,UAAU;AAC7B;;;ADNA,SAAS,UAAU,MAAsB;AACvC,QAAM,CAAC,SAAS,GAAG,IAAI,IAAI;AAC3B,QAAM,aAAuB,CAAC;AAC9B,QAAM,QAAgC,CAAC;AACvC,WAAS,IAAI,GAAG,IAAI,KAAK,QAAQ,KAAK;AACpC,UAAM,MAAM,KAAK,CAAC;AAClB,QAAI,IAAI,WAAW,IAAI,GAAG;AACxB,YAAM,MAAM,IAAI,MAAM,CAAC;AACvB,YAAM,WAAW,IAAI,QAAQ,GAAG;AAChC,UAAI,YAAY,GAAG;AACjB,cAAM,IAAI,MAAM,GAAG,QAAQ,CAAC,IAAI,IAAI,MAAM,WAAW,CAAC;AACtD;AAAA,MACF;AACA,YAAM,MAAM;AACZ,UAAI,QAAQ,QAAQ;AAClB,cAAM,GAAG,IAAI;AACb;AAAA,MACF;AACA,YAAM,OAAO,KAAK,IAAI,CAAC;AACvB,UAAI,QAAQ,QAAQ,CAAC,KAAK,WAAW,IAAI,GAAG;AAC1C,cAAM,GAAG,IAAI;AACb;AAAA,MACF,OAAO;AACL,cAAM,GAAG,IAAI;AAAA,MACf;AAAA,IACF,WAAW,QAAQ,MAAM;AACvB,YAAM,OAAO;AAAA,IACf,OAAO;AACL,iBAAW,KAAK,GAAG;AAAA,IACrB;AAAA,EACF;AACA,SAAO,EAAE,SAAS,WAAW,QAAQ,YAAY,MAAM;AACzD;AAEA,IAAM,OAAO;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAAA;AAqBb,eAAe,OAAwB;AACrC,QAAM,EAAE,SAAS,YAAY,MAAM,IAAI,UAAU,QAAQ,KAAK,MAAM,CAAC,CAAC;AACtE,mBAAiB,SAAS,KAAK;AAE/B,MAAI,MAAM,SAAS,QAAQ;AACzB,YAAQ,OAAO,MAAM,GAAG,IAAI;AAAA,CAAI;AAChC,WAAO;AAAA,EACT;AAEA,UAAQ,SAAS;AAAA,IACf,KAAK,SAAS;AACZ,YAAM,OAAO,UAAU,MAAM,QAAQ,MAAM;AAC3C,YAAM,OAAO,MAAM,QAAQ;AAC3B,YAAM,MAAM,oBAAoB;AAChC,YAAM,EAAE,OAAO,IAAI,MAAM,iBAAiB;AAAA,QACxC;AAAA,QACA;AAAA,QACA,KAAK,IAAI;AAAA,QACT,YAAY,IAAI;AAAA,QAChB,sBAAsB,EAAE,wBAAwB,KAAK;AAAA,MACvD,CAAC;AAED,YAAM,WAAW,CAAC,QAAgB;AAEhC,gBAAQ,IAAI,yBAAyB,GAAG,iBAAiB;AACzD,eAAO,MAAM,MAAM,QAAQ,KAAK,CAAC,CAAC;AAElC,mBAAW,MAAM,QAAQ,KAAK,CAAC,GAAG,GAAI,EAAE,MAAM;AAAA,MAChD;AACA,cAAQ,GAAG,UAAU,MAAM,SAAS,QAAQ,CAAC;AAC7C,cAAQ,GAAG,WAAW,MAAM,SAAS,SAAS,CAAC;AAE/C,YAAM,IAAI,QAAQ,MAAM;AAAA,MAAC,CAAC;AAC1B,aAAO;AAAA,IACT;AAAA,IACA,KAAK,OAAO;AACV,YAAM,CAAC,MAAM,IAAI;AACjB,YAAM,MAAM,oBAAoB;AAChC,aAAO,MAAM,WAAW,QAAQ;AAAA,QAC9B,KAAK,IAAI;AAAA,QACT,YAAY,IAAI;AAAA,QAChB,sBAAsB,EAAE,wBAAwB,KAAK;AAAA,MACvD,CAAC;AAAA,IACH;AAAA,IACA,KAAK,aAAa;AAChB,YAAM,CAAC,MAAM,IAAI;AACjB,YAAM,MAAM,oBAAoB;AAChC,aAAO,MAAM,YAAY,QAAQ;AAAA,QAC/B,KAAK,IAAI;AAAA,QACT,YAAY,IAAI;AAAA,QAChB,sBAAsB,EAAE,wBAAwB,KAAK;AAAA,MACvD,CAAC;AAAA,IACH;AAAA,IACA,KAAK,WAAW;AACd,YAAM,MAAM,MAAM,OAAO;AACzB,YAAM,OAAO,aAAa,cAAc,EAAE,OAAO;AACjD,oBAAc,KAAK,GAAG,KAAK,UAAU,MAAM,MAAM,CAAC,CAAC;AAAA,GAAM,OAAO;AAEhE,cAAQ,IAAI,0CAA0C,GAAG,EAAE;AAC3D,aAAO;AAAA,IACT;AAAA,IACA,KAAK,WAAW;AACd,cAAQ,OAAO,MAAM,GAAG,KAAK,UAAU,cAAc,GAAG,MAAM,CAAC,CAAC;AAAA,CAAI;AACpE,aAAO;AAAA,IACT;AAAA,IACA,KAAK,aAAa;AAChB,cAAQ,OAAO,MAAM,GAAG,cAAc,EAAE,OAAO;AAAA,CAAI;AACnD,aAAO;AAAA,IACT;AAAA,IACA,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AAAA,IACL,KAAK;AACH,cAAQ,OAAO,MAAM,GAAG,IAAI;AAAA,CAAI;AAChC,aAAO;AAAA,IACT;AACE,cAAQ,OAAO,MAAM,oBAAoB,OAAO;AAAA,EAAK,IAAI;AAAA,CAAI;AAC7D,aAAO;AAAA,EACX;AACF;AAEA,IAAM,mBAAwD;AAAA,EAC5D,OAAO,oBAAI,IAAI,CAAC,QAAQ,QAAQ,MAAM,CAAC;AAAA,EACvC,KAAK,oBAAI,IAAI,CAAC,MAAM,CAAC;AAAA,EACrB,aAAa,oBAAI,IAAI,CAAC,MAAM,CAAC;AAAA,EAC7B,SAAS,oBAAI,IAAI,CAAC,QAAQ,KAAK,CAAC;AAAA,EAChC,SAAS,oBAAI,IAAI,CAAC,MAAM,CAAC;AAAA,EACzB,MAAM,oBAAI,IAAI;AAAA,EACd,UAAU,oBAAI,IAAI;AAAA,EAClB,MAAM,oBAAI,IAAI;AAAA,EACd,aAAa,oBAAI,IAAI;AAAA,EACrB,IAAI,oBAAI,IAAI;AACd;AAEA,SAAS,iBAAiB,SAAiB,OAAqC;AAC9E,QAAM,UAAU,iBAAiB,OAAO;AACxC,MAAI,CAAC,QAAS;AACd,QAAM,UAAU,OAAO,KAAK,KAAK,EAAE,OAAO,CAAC,SAAS,CAAC,QAAQ,IAAI,IAAI,CAAC;AACtE,MAAI,QAAQ,SAAS,GAAG;AACtB,UAAM,IAAI,MAAM,oBAAoB,WAAW,MAAM,OAAO,QAAQ,CAAC,CAAC,EAAE;AAAA,EAC1E;AACF;AAEA,SAAS,UAAU,KAAqB;AACtC,QAAM,OAAO,OAAO,GAAG;AACvB,MAAI,CAAC,OAAO,UAAU,IAAI,KAAK,OAAO,KAAK,OAAO,OAAQ;AACxD,UAAM,IAAI,MAAM,uDAAuD,KAAK,UAAU,GAAG,CAAC,EAAE;AAAA,EAC9F;AACA,SAAO;AACT;AAEA,KAAK,EACF,KAAK,CAAC,SAAS,QAAQ,KAAK,IAAI,CAAC,EACjC,MAAM,CAAC,QAAQ;AAEd,UAAQ,MAAM,2BAA2B,GAAG;AAC5C,UAAQ,KAAK,CAAC;AAChB,CAAC;","names":[]}
|
package/dist/contract/index.d.ts
CHANGED
|
@@ -1,11 +1,12 @@
|
|
|
1
|
+
import { z } from 'zod';
|
|
1
2
|
import { AgentCandidateExperiment, AgentCandidateBundle, AgentCandidateBenchmarkTask, AgentCandidateBenchmarkCellRef, AgentCandidateExperimentMeasurement, AgentImprovementMeasuredComparison, CandidateExecutionEvidence, AgentCandidateBenchmarkSuiteInputs, AgentCandidateBenchmarkTaskMaterial, AgentCandidateExperimentMaterial } from '@tangle-network/agent-interface';
|
|
2
3
|
import { AxFunction, AxAIService } from '@ax-llm/ax';
|
|
3
|
-
import { z } from 'zod';
|
|
4
4
|
|
|
5
5
|
type AgentProfileCellSchemaVersion = 'agent-profile-cell/v1';
|
|
6
|
-
type
|
|
6
|
+
type AgentProfileJsonObject = {
|
|
7
7
|
[key: string]: AgentProfileJson;
|
|
8
8
|
};
|
|
9
|
+
type AgentProfileJson = string | number | boolean | null | AgentProfileJson[] | AgentProfileJsonObject;
|
|
9
10
|
type AgentProfileDimensionValue = string | number | boolean | null;
|
|
10
11
|
interface AgentProfileSource {
|
|
11
12
|
/** Runtime/profile contract being fingerprinted, e.g. `agent-interface-profile`. */
|
|
@@ -919,6 +920,13 @@ interface LlmClientOptions {
|
|
|
919
920
|
deadlineMs?: number;
|
|
920
921
|
/** Total provider attempts. Legacy option name; default 3 (1 initial + 2 retries). */
|
|
921
922
|
maxRetries?: number;
|
|
923
|
+
/**
|
|
924
|
+
* Transport for requests that declare `jsonSchema`. `native` sends
|
|
925
|
+
* `response_format: json_schema`; `json-object` sends the broadly supported
|
|
926
|
+
* JSON mode and relies on the caller to include the schema in model-visible
|
|
927
|
+
* instructions. Default: `native`.
|
|
928
|
+
*/
|
|
929
|
+
jsonSchemaTransport?: 'native' | 'json-object';
|
|
922
930
|
/** Fetch implementation — defaults to global `fetch`. Override for custom transport (e.g. tests). */
|
|
923
931
|
fetch?: typeof fetch;
|
|
924
932
|
/**
|
|
@@ -2438,6 +2446,84 @@ interface RunImprovementLoopResult<TArtifact, TScenario extends Scenario> extend
|
|
|
2438
2446
|
*/
|
|
2439
2447
|
declare function runImprovementLoop<TScenario extends Scenario, TArtifact>(opts: RunImprovementLoopOptions<TScenario, TArtifact>): Promise<RunImprovementLoopResult<TArtifact, TScenario>>;
|
|
2440
2448
|
|
|
2449
|
+
/**
|
|
2450
|
+
* `llmJudge` — the single-LLM-call bridge that turns a rubric prompt into a
|
|
2451
|
+
* canonical campaign `JudgeConfig`.
|
|
2452
|
+
*
|
|
2453
|
+
* The `JudgeConfig` contract (src/campaign/types.ts) is deliberately a
|
|
2454
|
+
* function, not a fixed LLM-prompt shape: real consumers judge with
|
|
2455
|
+
* ensembles, deterministic checks, or one LLM call. `ensembleJudge`
|
|
2456
|
+
* (src/judge-panel.ts) covers the multi-model case; `buildAgreementJudge`
|
|
2457
|
+
* (src/campaign/distillation) covers the pure-comparator case. `llmJudge`
|
|
2458
|
+
* covers the common single-call case the `JudgeConfig` doc-comment names:
|
|
2459
|
+
* one model call against `prompt`, parsed into the canonical `JudgeScore`
|
|
2460
|
+
* (`{ dimensions, composite, notes }`) on the campaign [0,1] scale.
|
|
2461
|
+
*
|
|
2462
|
+
* Transport is injected as a `ChatClient` (src/analyst/chat-client.ts) — the
|
|
2463
|
+
* substrate's transport-agnostic LLM seam — so the judge stays decoupled from
|
|
2464
|
+
* router-vs-sandbox-vs-cli-bridge and is unit-testable with the `mock`
|
|
2465
|
+
* transport. The composite is computed by `weightedComposite` (the same
|
|
2466
|
+
* sum-normalized weighting `ensembleJudge` uses), so a lift is attributable to
|
|
2467
|
+
* the dimension scores, not to a bespoke reducer.
|
|
2468
|
+
*
|
|
2469
|
+
* Fail-loud throughout: an unparseable model response throws `JudgeParseError`;
|
|
2470
|
+
* a response missing a declared dimension throws; an out-of-range score throws.
|
|
2471
|
+
* A thrown judge is recorded by the campaign engine as a failed cell, never
|
|
2472
|
+
* folded into a silent zero.
|
|
2473
|
+
*/
|
|
2474
|
+
|
|
2475
|
+
/** A rubric dimension as a bare key or the full `{ key, description }` shape. A
|
|
2476
|
+
* bare string uses the key as its own description. */
|
|
2477
|
+
type LlmJudgeDimension = string | JudgeDimension;
|
|
2478
|
+
interface LlmJudgeOptions<TArtifact, TScenario extends Scenario = Scenario> {
|
|
2479
|
+
/** The injected LLM transport. One `chat()` call per `score()`. Required —
|
|
2480
|
+
* there is no default route, so a misconfigured judge fails at construction,
|
|
2481
|
+
* never silently against the free-tier router. */
|
|
2482
|
+
chat: ChatClient;
|
|
2483
|
+
/** Rubric dimensions the model scores. Each becomes a `[0,1]` field of the
|
|
2484
|
+
* returned `JudgeScore.dimensions`. Defaults to a single `quality` dimension. */
|
|
2485
|
+
dimensions?: LlmJudgeDimension[];
|
|
2486
|
+
/** Model id. Falls back to `chat.defaultModel`; one of the two MUST resolve. */
|
|
2487
|
+
model?: string;
|
|
2488
|
+
/** Explicit scoring revision for opaque transport or renderer changes. */
|
|
2489
|
+
judgeVersion?: string;
|
|
2490
|
+
temperature?: number;
|
|
2491
|
+
maxTokens?: number;
|
|
2492
|
+
/** Composite weights forwarded to `weightedComposite`: a partial map selects
|
|
2493
|
+
* AND weights exactly the named dimensions. Omit for a uniform mean. */
|
|
2494
|
+
weights?: Record<string, number>;
|
|
2495
|
+
/** Scale the model is prompted to score on, normalized into `[0,1]`:
|
|
2496
|
+
* - `'unit'` (default): the model returns `[0,1]` directly.
|
|
2497
|
+
* - `'ten'`: the model returns `[0,10]`; divided by 10 here.
|
|
2498
|
+
* The prompt is annotated with the expected range either way. */
|
|
2499
|
+
scale?: 'unit' | 'ten';
|
|
2500
|
+
/** Run this judge only on matching scenarios (mirrors `JudgeConfig.appliesTo`). */
|
|
2501
|
+
appliesTo?: (scenario: TScenario) => boolean;
|
|
2502
|
+
/** Render the artifact + scenario into the user message. Default:
|
|
2503
|
+
* pretty-printed JSON of `{ scenario, artifact }`. */
|
|
2504
|
+
renderUser?: (input: {
|
|
2505
|
+
artifact: TArtifact;
|
|
2506
|
+
scenario: TScenario;
|
|
2507
|
+
}) => string;
|
|
2508
|
+
/** Strict runtime contract; its JSON Schema is sent to the provider. */
|
|
2509
|
+
costLedger?: CostLedgerHandle;
|
|
2510
|
+
responseSchema?: {
|
|
2511
|
+
name: string;
|
|
2512
|
+
schema: z.ZodObject;
|
|
2513
|
+
};
|
|
2514
|
+
}
|
|
2515
|
+
/**
|
|
2516
|
+
* Build a campaign-shaped `JudgeConfig` whose `score()` makes ONE LLM call
|
|
2517
|
+
* against `prompt` and reduces the model's per-dimension scores to a canonical
|
|
2518
|
+
* `JudgeScore` in `[0,1]`.
|
|
2519
|
+
*
|
|
2520
|
+
* The model is instructed to return JSON `{ "dimensions": { <key>: <number>, … },
|
|
2521
|
+
* "notes": "…" }`; the helper strips fenced JSON, validates every declared
|
|
2522
|
+
* dimension is present and in range, normalizes by `scale`, and composites via
|
|
2523
|
+
* `weightedComposite`.
|
|
2524
|
+
*/
|
|
2525
|
+
declare function llmJudge<TArtifact = unknown, TScenario extends Scenario = Scenario>(name: string, prompt: string, opts: LlmJudgeOptions<TArtifact, TScenario>): JudgeConfig<TArtifact, TScenario>;
|
|
2526
|
+
|
|
2441
2527
|
declare const REFERENCE_EQUIVALENCE_JUDGE_VERSION = "reference-equivalence-judge-v1-2026-07-13";
|
|
2442
2528
|
declare const REFERENCE_EQUIVALENCE_INPUT_LIMITS: {
|
|
2443
2529
|
readonly userRequest: 8000;
|
|
@@ -3268,9 +3354,15 @@ interface InterRaterInsight {
|
|
|
3268
3354
|
raters: number;
|
|
3269
3355
|
/** Number of runs every rater scored. */
|
|
3270
3356
|
jointlyRated: number;
|
|
3271
|
-
/**
|
|
3357
|
+
/** Multi-rater weighted kappa over the jointly rated runs. */
|
|
3272
3358
|
kappa: number;
|
|
3273
|
-
/**
|
|
3359
|
+
/** Absolute agreement across raters, using ICC(2,1). */
|
|
3360
|
+
icc: number;
|
|
3361
|
+
/** Mean pairwise Pearson correlation. Correlation is not agreement. */
|
|
3362
|
+
pearson: number;
|
|
3363
|
+
/** Mean pairwise Spearman rank correlation. */
|
|
3364
|
+
spearman: number;
|
|
3365
|
+
/** Pairwise weighted kappa per rater pair (key = `"raterA::raterB"`). */
|
|
3274
3366
|
perPair: Record<string, number>;
|
|
3275
3367
|
/** Run ids where raters disagree the most — the high-value triage list. */
|
|
3276
3368
|
disagreementCases: Array<{
|
|
@@ -5165,4 +5257,4 @@ interface FromOtelSpansOptions {
|
|
|
5165
5257
|
}
|
|
5166
5258
|
declare function fromOtelSpans(opts: FromOtelSpansOptions): RunRecord[];
|
|
5167
5259
|
|
|
5168
|
-
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type CandidateExperimentExecutionInput, type ChatClient, type CodeAgentSessionAction, type CodeAgentSessionActionKind, type CodeAgentSessionActionStatus, type CodeAgentSessionActionSurface, type CodeAgentSessionDiagnostic, type CodeAgentSessionExecutionReceipt, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionObservation, type CodeAgentSessionSource, type CodeAgentSessionTerminalStatus, type CodeSurface, type CompareCandidateExperimentOptions, type CostLedgerHandle, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvidenceVector, type EvolutionaryProposerOptions, type ExecutionInsight, type ExecutionReport, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateContext, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaProposerOptions, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type MutableSurface, type Mutator, type ObjectiveSource, type OptimizationProposer, type OptimizerConfig, type OutcomeCorrelationInsight, type OutcomeStore, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunCampaignOptions, type RunCandidateExperimentOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario, type SealCandidateBenchmarkSuiteOptions, type SelfImproveBudget, type SelfImproveLlm, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evolutionaryProposer, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaProposer, heldOutGate, inMemoryCampaignStorage, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, summarizeExecution, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|
|
5260
|
+
export { type AgentEvalAgent, type AgentEvalEvaluateOptions, type AgentEvalImproveOptions, type AgentTraceContributor, type AgentTraceContributorType, type AgentTraceConversation, type AgentTraceFile, type AgentTraceIndex, type AgentTraceRange, type AgentTraceRecord, type AnalystFinding, type AnalyzeRunsOptions, type AuthoringProvenance, type AxisEvidence, type AxisVerdict, type BuildEvidenceVectorOptions, type CampaignAggregates, type CampaignArtifactWriter, type CampaignCellResult, type CampaignCostMeter, type CampaignResult, type CampaignStorage, type CampaignTraceWriter, type CandidateExperimentExecutionInput, type ChatClient, type CodeAgentSessionAction, type CodeAgentSessionActionKind, type CodeAgentSessionActionStatus, type CodeAgentSessionActionSurface, type CodeAgentSessionDiagnostic, type CodeAgentSessionExecutionReceipt, type CodeAgentSessionIntakeOptions, type CodeAgentSessionIntakeResult, type CodeAgentSessionMetrics, type CodeAgentSessionObservation, type CodeAgentSessionSource, type CodeAgentSessionTerminalStatus, type CodeSurface, type CompareCandidateExperimentOptions, type CostLedgerHandle, type CostProvenanceSummary, type CreateChatClientOpts, type DefaultAnalystRegistryOptions, type DefaultProductionGateOptions, type DefineAgentEvalOptions, type DefinedAgentEval, type DeploymentOutcome, type DispatchFn as Dispatch, type DispatchContext, type EvalCellScoreDelta, type EvalDimensionDelta, type EvalGenerationDiff, type EvalReportingSuiteInput, type EvalReportingSuiteOptions, type EvalReportingSuiteResult, type EvalRunDiff, type EvidenceVector, type EvolutionaryProposerOptions, type ExecutionInsight, type ExecutionReport, type FailureClusterInsight, type FeedbackTableMeta, type FeedbackTableRow, FileSystemOutcomeStore, type FileSystemOutcomeStoreOptions, type FromFeedbackTableOptions, type FromFeedbackTableResult, type FromOtelSpansOptions, type FromRunRecordDirOptions, type FromRunRecordDirResult, type Gate, type GateContext, type GateDecision, type GateResult, type GenerationCandidate, type GenerationRecord, type GepaProposerOptions, type HeldOutGateOptions, type HostedTenant, InMemoryOutcomeStore, type InsightReport, type InterRaterInsight, type JudgeConfig, type JudgeDimension, type JudgeInsight, type JudgeScore, type LiftInsight, type LlmJudgeDimension, type LlmJudgeOptions, type MutableSurface, type Mutator, type ObjectiveSource, type OptimizationProposer, type OptimizerConfig, type OutcomeCorrelationInsight, type OutcomeStore, type ParetoSignificanceGateOptions, type ParsedCodeAgentJsonl, type PartitionByAuthoringModelResult, type PromotionObjective, type PromotionPolicy, REFERENCE_EQUIVALENCE_INPUT_LIMITS, REFERENCE_EQUIVALENCE_JUDGE_VERSION, type Recommendation, type ReferenceEquivalenceJudgeInput, type ReferenceEquivalenceJudgeOptions, type ReferenceEquivalenceJudgeResult, type ReferenceEquivalenceScenario, type ReleaseSummary, type RunCampaignOptions, type RunCandidateExperimentOptions, type RunEvalOptions, type RunImprovementLoopOptions, type RunImprovementLoopResult, type RunRecordRejection, type ScalarDistribution, type Scenario, type SealCandidateBenchmarkSuiteOptions, type SelfImproveBudget, type SelfImproveLlm, type SelfImproveOptions, type SelfImproveProgressEvent, type SelfImproveResult, SelfImproveRunError, type SessionScript, type SummarizeExecutionOptions, type SurfaceProposer, type TokenUsageInsight, analyzeRuns, buildDefaultAnalystRegistry, buildEvidenceVector, campaignSplitDigest, composeGate, createChatClient, createReferenceEquivalenceJudge, defaultProductionGate, defineAgentEval, diffGenerations, diffRunBaselineToWinner, diffRuns, evalReportingSuite, evolutionaryProposer, fromClaudeCodeSession, fromCodexSession, fromFeedbackTable, fromKimiCodeSession, fromOpenCodeSession, fromOtelSpans, fromPiSession, fromPigraphSession, fromRunRecordDir, fsCampaignStorage, gepaProposer, heldOutGate, inMemoryCampaignStorage, llmJudge, measuredComparisonFromCandidateExperiment, observeCodeAgentSession, paretoPolicy, paretoSignificanceGate, parseAgentTrace, parseCodeAgentJsonl, partitionRunsByAuthoringModel, runCampaign, runCandidateExperiment, runEval, runImprovementLoop, runReferenceEquivalenceJudge, sealCandidateBenchmarkSuite, sealCandidateBenchmarkTask, sealCandidateExperiment, selfImprove, summarizeExecution, verifyCandidateBenchmarkSuite, verifyCandidateBenchmarkSuiteInputs, verifyCandidateBenchmarkTask, verifyCandidateExperiment, verifyCandidateExperimentComparison };
|
package/dist/contract/index.js
CHANGED
|
@@ -14,7 +14,7 @@ import {
|
|
|
14
14
|
import {
|
|
15
15
|
analyzeRuns,
|
|
16
16
|
summarizeExecution
|
|
17
|
-
} from "../chunk-
|
|
17
|
+
} from "../chunk-E3HAD4A3.js";
|
|
18
18
|
import {
|
|
19
19
|
REFERENCE_EQUIVALENCE_INPUT_LIMITS,
|
|
20
20
|
REFERENCE_EQUIVALENCE_JUDGE_VERSION,
|
|
@@ -27,6 +27,7 @@ import {
|
|
|
27
27
|
gepaProposer,
|
|
28
28
|
heldOutGate,
|
|
29
29
|
heldoutSignificance,
|
|
30
|
+
llmJudge,
|
|
30
31
|
loopProvenanceArgsFromResult,
|
|
31
32
|
paretoPolicy,
|
|
32
33
|
paretoSignificanceGate,
|
|
@@ -36,7 +37,7 @@ import {
|
|
|
36
37
|
runReferenceEquivalenceJudge,
|
|
37
38
|
surfaceContentHash,
|
|
38
39
|
surfaceHash
|
|
39
|
-
} from "../chunk-
|
|
40
|
+
} from "../chunk-HQY7LBV2.js";
|
|
40
41
|
import {
|
|
41
42
|
campaignSplitDigest,
|
|
42
43
|
createRunCostLedger,
|
|
@@ -48,15 +49,15 @@ import {
|
|
|
48
49
|
import {
|
|
49
50
|
buildDefaultAnalystRegistry,
|
|
50
51
|
createChatClient
|
|
51
|
-
} from "../chunk-
|
|
52
|
+
} from "../chunk-5YMKIFYP.js";
|
|
52
53
|
import "../chunk-HHWE3POT.js";
|
|
53
|
-
import "../chunk-
|
|
54
|
+
import "../chunk-WXQTVEKM.js";
|
|
54
55
|
import {
|
|
55
56
|
FileSystemOutcomeStore,
|
|
56
57
|
InMemoryOutcomeStore
|
|
57
58
|
} from "../chunk-3RF76KTD.js";
|
|
58
59
|
import "../chunk-ARU2PZFM.js";
|
|
59
|
-
import "../chunk-
|
|
60
|
+
import "../chunk-J7S4YM27.js";
|
|
60
61
|
import "../chunk-DPZAEKA6.js";
|
|
61
62
|
import {
|
|
62
63
|
pairedBootstrap
|
|
@@ -74,9 +75,9 @@ import "../chunk-IR3KBHOY.js";
|
|
|
74
75
|
import "../chunk-PC4UYEBM.js";
|
|
75
76
|
import {
|
|
76
77
|
parseRunRecordSafe
|
|
77
|
-
} from "../chunk-
|
|
78
|
+
} from "../chunk-LOW3U7JZ.js";
|
|
78
79
|
import "../chunk-MA6HLL3S.js";
|
|
79
|
-
import "../chunk-
|
|
80
|
+
import "../chunk-GC4ATIKK.js";
|
|
80
81
|
import "../chunk-VSMTAMNK.js";
|
|
81
82
|
import "../chunk-ONWEPEDO.js";
|
|
82
83
|
import {
|
|
@@ -1703,6 +1704,7 @@ export {
|
|
|
1703
1704
|
gepaProposer,
|
|
1704
1705
|
heldOutGate,
|
|
1705
1706
|
inMemoryCampaignStorage,
|
|
1707
|
+
llmJudge,
|
|
1706
1708
|
measuredComparisonFromCandidateExperiment,
|
|
1707
1709
|
observeCodeAgentSession,
|
|
1708
1710
|
paretoPolicy,
|