blume 1.6.5 → 1.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +22 -0
- package/bin/blume.mjs +3 -2
- package/dist/cli/chunk-0ewz4trd.js +679 -0
- package/dist/cli/chunk-0ewz4trd.js.map +15 -0
- package/dist/cli/chunk-27gtm2ym.js +69 -0
- package/dist/cli/chunk-27gtm2ym.js.map +11 -0
- package/dist/cli/chunk-2aj8ddew.js +72 -0
- package/dist/cli/chunk-2aj8ddew.js.map +10 -0
- package/dist/cli/chunk-3k0kzs6d.js +69 -0
- package/dist/cli/chunk-3k0kzs6d.js.map +11 -0
- package/dist/cli/chunk-3r94j3tc.js +221 -0
- package/dist/cli/chunk-3r94j3tc.js.map +10 -0
- package/dist/cli/chunk-4trphnvy.js +102 -0
- package/dist/cli/chunk-4trphnvy.js.map +11 -0
- package/dist/cli/chunk-4xyggvgf.js +21 -0
- package/dist/cli/chunk-4xyggvgf.js.map +10 -0
- package/dist/cli/chunk-5hs6gb7n.js +32 -0
- package/dist/cli/chunk-5hs6gb7n.js.map +10 -0
- package/dist/cli/chunk-5yvt556e.js +185 -0
- package/dist/cli/chunk-5yvt556e.js.map +11 -0
- package/dist/cli/chunk-62qsssnh.js +3808 -0
- package/dist/cli/chunk-62qsssnh.js.map +36 -0
- package/dist/cli/chunk-6kzzpsx8.js +26 -0
- package/dist/cli/chunk-6kzzpsx8.js.map +10 -0
- package/dist/cli/chunk-8gnpdsn1.js +952 -0
- package/dist/cli/chunk-8gnpdsn1.js.map +12 -0
- package/dist/cli/chunk-9sh49q0h.js +30 -0
- package/dist/cli/chunk-9sh49q0h.js.map +10 -0
- package/dist/cli/chunk-aerwpe14.js +2370 -0
- package/dist/cli/chunk-aerwpe14.js.map +15 -0
- package/dist/cli/chunk-ag1zyr5x.js +176 -0
- package/dist/cli/chunk-ag1zyr5x.js.map +10 -0
- package/dist/cli/chunk-bawgnt8x.js +277 -0
- package/dist/cli/chunk-bawgnt8x.js.map +11 -0
- package/dist/cli/chunk-bcy492zc.js +16 -0
- package/dist/cli/chunk-bcy492zc.js.map +10 -0
- package/dist/cli/chunk-btfr9yvw.js +41 -0
- package/dist/cli/chunk-btfr9yvw.js.map +10 -0
- package/dist/cli/chunk-cbjnx4s8.js +73 -0
- package/dist/cli/chunk-cbjnx4s8.js.map +10 -0
- package/dist/cli/chunk-cnvm6k3e.js +96 -0
- package/dist/cli/chunk-cnvm6k3e.js.map +10 -0
- package/dist/cli/chunk-etsqspj6.js +5170 -0
- package/dist/cli/chunk-etsqspj6.js.map +47 -0
- package/dist/cli/chunk-ev67ycx0.js +15 -0
- package/dist/cli/chunk-ev67ycx0.js.map +10 -0
- package/dist/cli/chunk-ey89bjj1.js +209 -0
- package/dist/cli/chunk-ey89bjj1.js.map +11 -0
- package/dist/cli/chunk-f75cqye8.js +76 -0
- package/dist/cli/chunk-f75cqye8.js.map +10 -0
- package/dist/cli/chunk-j00ezcg5.js +259 -0
- package/dist/cli/chunk-j00ezcg5.js.map +11 -0
- package/dist/cli/chunk-jtb45atp.js +467 -0
- package/dist/cli/chunk-jtb45atp.js.map +14 -0
- package/dist/cli/chunk-m3p3wahd.js +117 -0
- package/dist/cli/chunk-m3p3wahd.js.map +10 -0
- package/dist/cli/chunk-n0y172hf.js +387 -0
- package/dist/cli/chunk-n0y172hf.js.map +12 -0
- package/dist/cli/chunk-n4qjabmt.js +1062 -0
- package/dist/cli/chunk-n4qjabmt.js.map +25 -0
- package/dist/cli/chunk-nyqzjdhj.js +111 -0
- package/dist/cli/chunk-nyqzjdhj.js.map +11 -0
- package/dist/cli/chunk-pxj10x8y.js +35 -0
- package/dist/cli/chunk-pxj10x8y.js.map +10 -0
- package/dist/cli/chunk-s4jn7f1q.js +54 -0
- package/dist/cli/chunk-s4jn7f1q.js.map +10 -0
- package/dist/cli/chunk-s4k1pnvf.js +81 -0
- package/dist/cli/chunk-s4k1pnvf.js.map +10 -0
- package/dist/cli/chunk-s5e5jt53.js +227 -0
- package/dist/cli/chunk-s5e5jt53.js.map +11 -0
- package/dist/cli/chunk-sbdqrjbb.js +81 -0
- package/dist/cli/chunk-sbdqrjbb.js.map +10 -0
- package/dist/cli/chunk-tc89yh2r.js +136 -0
- package/dist/cli/chunk-tc89yh2r.js.map +10 -0
- package/dist/cli/chunk-vt8fgygt.js +23 -0
- package/dist/cli/chunk-vt8fgygt.js.map +10 -0
- package/dist/cli/chunk-vv237fp3.js +1002 -0
- package/dist/cli/chunk-vv237fp3.js.map +13 -0
- package/dist/cli/chunk-vv3f8mb6.js +5314 -0
- package/dist/cli/chunk-vv3f8mb6.js.map +58 -0
- package/dist/cli/chunk-vxv4x1n8.js +17 -0
- package/dist/cli/chunk-vxv4x1n8.js.map +10 -0
- package/dist/cli/chunk-wb067mv3.js +758 -0
- package/dist/cli/chunk-wb067mv3.js.map +13 -0
- package/dist/cli/chunk-wd27zjcz.js +60 -0
- package/dist/cli/chunk-wd27zjcz.js.map +10 -0
- package/dist/cli/chunk-wkq5tbtq.js +1141 -0
- package/dist/cli/chunk-wkq5tbtq.js.map +19 -0
- package/dist/cli/chunk-x1vrdjyk.js +1967 -0
- package/dist/cli/chunk-x1vrdjyk.js.map +34 -0
- package/dist/cli/chunk-x66c5yjn.js +23 -0
- package/dist/cli/chunk-x66c5yjn.js.map +10 -0
- package/dist/cli/index.js +55 -27597
- package/dist/cli/index.js.map +5 -243
- package/dist/types/ai/ask-context.d.ts +26 -0
- package/dist/types/core/code-fences.d.ts +11 -0
- package/dist/types/core/package-root.d.ts +1 -1
- package/dist/types/core/schema.d.ts +70 -0
- package/docs/02-deployment.mdx +1 -1
- package/docs/configuration/ask-ai.mdx +1 -1
- package/docs/configuration/customization.mdx +2 -9
- package/docs/reference/cli.mdx +1 -1
- package/package.json +4 -2
- package/src/ai/api/handlers.ts +4 -7
- package/src/ai/api/paths.ts +8 -0
- package/src/ai/api/spec.ts +2 -1
- package/src/ai/ask-context.ts +378 -22
- package/src/astro/generate.ts +20 -23
- package/src/astro/include-hmr.ts +10 -13
- package/src/astro/include-refresh.ts +0 -0
- package/src/astro/index.ts +6 -1
- package/src/astro/integration.ts +269 -53
- package/src/astro/module-types.ts +74 -0
- package/src/astro/templates.ts +85 -97
- package/src/audit/image-size.ts +10 -8
- package/src/cli/command-meta.ts +77 -0
- package/src/cli/commands/add.ts +2 -4
- package/src/cli/commands/audit.ts +2 -4
- package/src/cli/commands/build.ts +42 -346
- package/src/cli/commands/check.ts +2 -4
- package/src/cli/commands/dev.ts +31 -42
- package/src/cli/commands/doctor.ts +2 -4
- package/src/cli/commands/eject.ts +3 -41
- package/src/cli/commands/eval.ts +2 -5
- package/src/cli/commands/init.ts +2 -4
- package/src/cli/commands/mcp-stdio.ts +2 -5
- package/src/cli/commands/preview.ts +3 -5
- package/src/cli/commands/sync.ts +2 -4
- package/src/cli/commands/translate.ts +2 -5
- package/src/cli/commands/validate.ts +2 -4
- package/src/cli/commands/version.ts +2 -4
- package/src/cli/eject-scripts.ts +0 -45
- package/src/cli/host-args.ts +16 -0
- package/src/cli/index.ts +84 -35
- package/src/cli/lazy-command.ts +47 -0
- package/src/components/content/GithubInfo.astro +4 -1
- package/src/components/layout/PageLayout.astro +14 -3
- package/src/components/layout/ReferenceLayout.astro +14 -4
- package/src/components/layout/RootLayout.astro +14 -4
- package/src/components/layout/page-locale.ts +29 -0
- package/src/core/api-name.ts +18 -0
- package/src/core/code-fences.ts +48 -0
- package/src/core/content-assets.ts +3 -7
- package/src/core/includes.ts +3 -7
- package/src/core/package-root.ts +1 -1
- package/src/core/schema.ts +19 -0
- package/src/core/sources/normalize.ts +2 -37
- package/src/core/sources/obsidian.ts +3 -2
- package/src/core/svg-dimensions.ts +97 -0
- package/src/core/version-cut.ts +2 -2
- package/src/deploy/artifacts.ts +370 -0
- package/src/deploy/cloudflare-negotiation.ts +97 -32
- package/src/deploy/function-bundle.ts +66 -20
- package/src/deploy/sitemap.ts +6 -0
- package/src/deploy/vercel-negotiation.ts +8 -30
- package/src/og/card.ts +6 -12
- package/src/openapi/render-mdx.ts +9 -5
- package/src/registry/eject.ts +0 -2
|
@@ -0,0 +1,679 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
import {
|
|
3
|
+
agentArgs,
|
|
4
|
+
duration,
|
|
5
|
+
money,
|
|
6
|
+
parseVerdict,
|
|
7
|
+
readAgentOutput,
|
|
8
|
+
runAgentHeadless,
|
|
9
|
+
seconds,
|
|
10
|
+
writeMcpConfig
|
|
11
|
+
} from "./chunk-s5e5jt53.js";
|
|
12
|
+
import {
|
|
13
|
+
AGENTS,
|
|
14
|
+
launchAgent
|
|
15
|
+
} from "./chunk-8gnpdsn1.js";
|
|
16
|
+
import {
|
|
17
|
+
reportInternalError
|
|
18
|
+
} from "./chunk-btfr9yvw.js";
|
|
19
|
+
import {
|
|
20
|
+
buildMcpData
|
|
21
|
+
} from "./chunk-n4qjabmt.js";
|
|
22
|
+
import"./chunk-5yvt556e.js";
|
|
23
|
+
import"./chunk-3k0kzs6d.js";
|
|
24
|
+
import"./chunk-vxv4x1n8.js";
|
|
25
|
+
import {
|
|
26
|
+
YAML_SCHEMA,
|
|
27
|
+
scanProject
|
|
28
|
+
} from "./chunk-etsqspj6.js";
|
|
29
|
+
import"./chunk-vv3f8mb6.js";
|
|
30
|
+
import"./chunk-27gtm2ym.js";
|
|
31
|
+
import"./chunk-4xyggvgf.js";
|
|
32
|
+
import"./chunk-6kzzpsx8.js";
|
|
33
|
+
import {
|
|
34
|
+
BlumeError,
|
|
35
|
+
countBySeverity,
|
|
36
|
+
flushStdout,
|
|
37
|
+
logger
|
|
38
|
+
} from "./chunk-ey89bjj1.js";
|
|
39
|
+
import {
|
|
40
|
+
commandMeta
|
|
41
|
+
} from "./chunk-2aj8ddew.js";
|
|
42
|
+
|
|
43
|
+
// src/cli/commands/eval.ts
|
|
44
|
+
import { existsSync } from "node:fs";
|
|
45
|
+
import { defineCommand } from "citty";
|
|
46
|
+
import { join as join3 } from "pathe";
|
|
47
|
+
|
|
48
|
+
// src/eval/prompts.ts
|
|
49
|
+
var readerPrompt = (question) => `You are evaluating whether a product's documentation can answer a user's question.
|
|
50
|
+
|
|
51
|
+
Answer the question below using ONLY the connected documentation tools (search_docs, get_page, list_pages, get_navigation). Rules:
|
|
52
|
+
- Do not use prior knowledge about the product. Do not guess.
|
|
53
|
+
- Do not read files, run commands, or access the network.
|
|
54
|
+
- Search first, then read the most relevant pages with get_page.
|
|
55
|
+
- If the documentation does not contain the answer, say exactly what information is missing instead of inventing one.
|
|
56
|
+
|
|
57
|
+
Question: ${question.question}
|
|
58
|
+
|
|
59
|
+
Reply with a concise answer containing the specific facts the documentation provides. Plain text only.`;
|
|
60
|
+
var judgePrompt = (question, answer) => {
|
|
61
|
+
const facts = question.expected.map((fact) => `- ${fact}`).join(`
|
|
62
|
+
`);
|
|
63
|
+
return `You are grading an answer against expected facts. Do not use any tools.
|
|
64
|
+
|
|
65
|
+
Question: ${question.question}
|
|
66
|
+
|
|
67
|
+
Expected facts — each must be present in substance (paraphrase is fine, contradiction is not):
|
|
68
|
+
${facts}
|
|
69
|
+
|
|
70
|
+
Answer to grade:
|
|
71
|
+
"""
|
|
72
|
+
${answer}
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
An answer that states the documentation lacks the information FAILS.
|
|
76
|
+
|
|
77
|
+
Reply with ONLY this JSON object on a single line, no markdown fences:
|
|
78
|
+
{"pass": true|false, "score": 0.0-1.0, "missing": ["expected facts absent or contradicted"], "notes": "one sentence"}`;
|
|
79
|
+
};
|
|
80
|
+
var evalFixPrompt = (reportPath) => `Fix the documentation gaps found by \`blume eval\` in this project.
|
|
81
|
+
|
|
82
|
+
The full report is at ${reportPath}. It is JSON: each entry in \`eval.results\` with status "fail" is one question the documentation could not answer. Each carries the \`question\`, the \`expected\` facts, the judge's \`missing\` facts, and the reader agent's \`answer\` (what the docs currently convey). The matching \`diagnostics\` entry names the source \`file\` of the page that should answer it.
|
|
83
|
+
|
|
84
|
+
Work through every failed question:
|
|
85
|
+
1. Read the page named in the finding (or choose the best page when none is named).
|
|
86
|
+
2. Edit the documentation so it states the missing facts explicitly. Add prose, not filler; keep the page's voice.
|
|
87
|
+
3. Never delete questions from the evals file or weaken expected facts.
|
|
88
|
+
|
|
89
|
+
When you are done, run \`blume eval\` to verify, and repeat until every question passes.`;
|
|
90
|
+
var initPrompt = (evalsPath) => `Draft a starter evals file for \`blume eval\` in this documentation project.
|
|
91
|
+
|
|
92
|
+
Read the documentation source pages in this project and write ${evalsPath} with about 10 high-value questions a real user would ask — installation, configuration, deployment, and the project's headline features. For each question, list the expected facts a correct answer must state, grounded in what the documentation actually promises (never invent facts the docs don't state).
|
|
93
|
+
|
|
94
|
+
The file format is YAML:
|
|
95
|
+
|
|
96
|
+
questions:
|
|
97
|
+
- id: kebab-case-slug
|
|
98
|
+
question: One user question?
|
|
99
|
+
expected:
|
|
100
|
+
- a fact the answer must contain
|
|
101
|
+
- another required fact
|
|
102
|
+
routes:
|
|
103
|
+
- /route/of/the/page/that/answers/it
|
|
104
|
+
|
|
105
|
+
Rules:
|
|
106
|
+
- Every \`expected\` fact must be verifiable in the docs today.
|
|
107
|
+
- Prefer questions whose answers live on one page; set \`routes\` to that page.
|
|
108
|
+
- Keep ids unique and questions short.
|
|
109
|
+
|
|
110
|
+
When you are done, print the file and suggest running \`blume eval\` to try it.`;
|
|
111
|
+
|
|
112
|
+
// src/eval/report.ts
|
|
113
|
+
import { mkdtemp, writeFile } from "node:fs/promises";
|
|
114
|
+
import { tmpdir } from "node:os";
|
|
115
|
+
import { colors } from "consola/utils";
|
|
116
|
+
import { join, relative } from "pathe";
|
|
117
|
+
var GLYPH = {
|
|
118
|
+
error: "!",
|
|
119
|
+
fail: "✖",
|
|
120
|
+
pass: "✔",
|
|
121
|
+
skip: "⊘"
|
|
122
|
+
};
|
|
123
|
+
var STATUS_COLOR = {
|
|
124
|
+
error: colors.yellow,
|
|
125
|
+
fail: colors.red,
|
|
126
|
+
pass: colors.green,
|
|
127
|
+
skip: colors.dim
|
|
128
|
+
};
|
|
129
|
+
var ID_PAD = 28;
|
|
130
|
+
var questionLine = (result) => {
|
|
131
|
+
const color = STATUS_COLOR[result.status];
|
|
132
|
+
const glyph = color(GLYPH[result.status]);
|
|
133
|
+
const id = result.id.padEnd(ID_PAD);
|
|
134
|
+
if (result.status === "skip") {
|
|
135
|
+
return ` ${glyph} ${id} ${colors.dim("skipped")}`;
|
|
136
|
+
}
|
|
137
|
+
const score = result.score === undefined ? "" : result.score.toFixed(2);
|
|
138
|
+
const cost = money(result.costUsd);
|
|
139
|
+
const cells = [
|
|
140
|
+
color(result.status),
|
|
141
|
+
score,
|
|
142
|
+
colors.dim(seconds(result.durationMs)),
|
|
143
|
+
cost === "" ? "" : colors.dim(cost)
|
|
144
|
+
].filter((cell) => cell !== "").join(" ");
|
|
145
|
+
return ` ${glyph} ${id} ${cells}`;
|
|
146
|
+
};
|
|
147
|
+
var questionDetails = (result, verbose) => {
|
|
148
|
+
const lines = [];
|
|
149
|
+
if (result.status === "fail") {
|
|
150
|
+
for (const fact of result.missing) {
|
|
151
|
+
lines.push(` ${colors.dim(`missing: ${fact}`)}`);
|
|
152
|
+
}
|
|
153
|
+
}
|
|
154
|
+
if (result.status === "error" && result.detail) {
|
|
155
|
+
lines.push(` ${colors.dim(result.detail)}`);
|
|
156
|
+
}
|
|
157
|
+
if (verbose && result.answer && result.status !== "pass") {
|
|
158
|
+
lines.push(...result.answer.split(`
|
|
159
|
+
`).map((line) => ` ${colors.dim(`> ${line}`)}`));
|
|
160
|
+
}
|
|
161
|
+
return lines;
|
|
162
|
+
};
|
|
163
|
+
var summaryLine = (result) => {
|
|
164
|
+
const { counts } = result;
|
|
165
|
+
const parts = [
|
|
166
|
+
`${counts.pass} passed`,
|
|
167
|
+
counts.fail > 0 ? `${counts.fail} failed` : "",
|
|
168
|
+
counts.error > 0 ? `${counts.error} errored` : "",
|
|
169
|
+
counts.skip > 0 ? `${counts.skip} skipped` : "",
|
|
170
|
+
duration(result.durationMs),
|
|
171
|
+
money(result.costUsd)
|
|
172
|
+
].filter((part) => part !== "");
|
|
173
|
+
return parts.join(" · ");
|
|
174
|
+
};
|
|
175
|
+
var headerLine = (total, agent) => `${colors.bold("blume eval")} ${total} question(s) · ${AGENTS[agent].name}`;
|
|
176
|
+
var startLine = (id, index, total) => ` ${colors.dim(`▸ ${id} (${index + 1}/${total})`)}`;
|
|
177
|
+
var fixLines = (result, root) => result.diagnostics.filter((diagnostic) => diagnostic.code !== "BLUME_EVAL_ROUTE_UNKNOWN").map((finding) => {
|
|
178
|
+
const site = finding.file ? `${relative(root, finding.file)}${finding.line ? `:${finding.line}` : ""}` : "";
|
|
179
|
+
return ` ${colors.cyan("fix:")} ${site} ${colors.dim(finding.message)}`;
|
|
180
|
+
});
|
|
181
|
+
var warningLines = (result, root) => result.diagnostics.filter((diagnostic) => diagnostic.code === "BLUME_EVAL_ROUTE_UNKNOWN").map((finding) => {
|
|
182
|
+
const site = finding.file ? ` ${relative(root, finding.file)}${finding.line ? `:${finding.line}` : ""}` : "";
|
|
183
|
+
return ` ${colors.yellow("⚠")}${site} ${colors.dim(finding.message)}`;
|
|
184
|
+
});
|
|
185
|
+
var evalReportJson = (result, root, threshold) => {
|
|
186
|
+
const diagnostics = result.diagnostics.map((diagnostic) => diagnostic.file ? { ...diagnostic, file: relative(root, diagnostic.file) } : diagnostic);
|
|
187
|
+
return `${JSON.stringify({
|
|
188
|
+
diagnostics,
|
|
189
|
+
eval: {
|
|
190
|
+
agent: result.agent,
|
|
191
|
+
costUsd: result.costUsd,
|
|
192
|
+
counts: result.counts,
|
|
193
|
+
durationMs: result.durationMs,
|
|
194
|
+
results: result.results,
|
|
195
|
+
threshold
|
|
196
|
+
},
|
|
197
|
+
summary: countBySeverity(result.diagnostics)
|
|
198
|
+
}, null, 2)}
|
|
199
|
+
`;
|
|
200
|
+
};
|
|
201
|
+
var writeEvalReport = async (result, root, threshold) => {
|
|
202
|
+
const dir = await mkdtemp(join(tmpdir(), "blume-eval-"));
|
|
203
|
+
const path = join(dir, "report.json");
|
|
204
|
+
await writeFile(path, evalReportJson(result, root, threshold));
|
|
205
|
+
return path;
|
|
206
|
+
};
|
|
207
|
+
|
|
208
|
+
// src/eval/run.ts
|
|
209
|
+
import { mkdir, mkdtemp as mkdtemp2, writeFile as writeFile2 } from "node:fs/promises";
|
|
210
|
+
import { tmpdir as tmpdir2 } from "node:os";
|
|
211
|
+
import { join as join2 } from "pathe";
|
|
212
|
+
|
|
213
|
+
// src/eval/schema.ts
|
|
214
|
+
import { readFile } from "node:fs/promises";
|
|
215
|
+
import { load } from "js-yaml";
|
|
216
|
+
import { z } from "zod";
|
|
217
|
+
var ID_PATTERN = /^[a-z0-9][a-z0-9-]*$/u;
|
|
218
|
+
var questionSchema = z.strictObject({
|
|
219
|
+
expected: z.array(z.string().min(1)).min(1, "expected must list at least one fact"),
|
|
220
|
+
id: z.string().regex(ID_PATTERN, "id must be a kebab-case slug (a-z, 0-9, dashes)"),
|
|
221
|
+
question: z.string().min(1),
|
|
222
|
+
routes: z.union([z.string(), z.array(z.string())]).default([]).transform((value) => Array.isArray(value) ? value : [value]),
|
|
223
|
+
severity: z.enum(["error", "warning"]).default("error"),
|
|
224
|
+
skip: z.boolean().default(false)
|
|
225
|
+
});
|
|
226
|
+
var fullSchema = z.strictObject({
|
|
227
|
+
questions: z.array(questionSchema).min(1),
|
|
228
|
+
version: z.literal(1).default(1)
|
|
229
|
+
});
|
|
230
|
+
var evalsFileSchema = fullSchema.superRefine((value, context) => {
|
|
231
|
+
const seen = new Set;
|
|
232
|
+
for (const question of value.questions) {
|
|
233
|
+
if (seen.has(question.id)) {
|
|
234
|
+
context.addIssue({
|
|
235
|
+
code: z.ZodIssueCode.custom,
|
|
236
|
+
message: `duplicate question id "${question.id}"`,
|
|
237
|
+
path: ["questions"]
|
|
238
|
+
});
|
|
239
|
+
}
|
|
240
|
+
seen.add(question.id);
|
|
241
|
+
}
|
|
242
|
+
});
|
|
243
|
+
|
|
244
|
+
class EvalsFileError extends Error {
|
|
245
|
+
path;
|
|
246
|
+
constructor(path, message) {
|
|
247
|
+
super(message);
|
|
248
|
+
this.name = "EvalsFileError";
|
|
249
|
+
this.path = path;
|
|
250
|
+
}
|
|
251
|
+
}
|
|
252
|
+
var describeIssues = (error) => error.issues.map((issue) => {
|
|
253
|
+
const at = issue.path.length > 0 ? ` at ${issue.path.join(".")}` : "";
|
|
254
|
+
return `${issue.message}${at}`;
|
|
255
|
+
}).join("; ");
|
|
256
|
+
var loadEvalsFile = async (path) => {
|
|
257
|
+
let raw;
|
|
258
|
+
try {
|
|
259
|
+
raw = await readFile(path, "utf-8");
|
|
260
|
+
} catch {
|
|
261
|
+
throw new EvalsFileError(path, `No evals file found at ${path}. Run \`blume eval init\` to draft one.`);
|
|
262
|
+
}
|
|
263
|
+
let parsed;
|
|
264
|
+
try {
|
|
265
|
+
parsed = load(raw, { schema: YAML_SCHEMA });
|
|
266
|
+
} catch (error) {
|
|
267
|
+
const detail = error instanceof Error ? error.message : String(error);
|
|
268
|
+
throw new EvalsFileError(path, `Invalid YAML in ${path}: ${detail}`);
|
|
269
|
+
}
|
|
270
|
+
const candidate = Array.isArray(parsed) ? { questions: parsed } : parsed;
|
|
271
|
+
const result = evalsFileSchema.safeParse(candidate);
|
|
272
|
+
if (!result.success) {
|
|
273
|
+
throw new EvalsFileError(path, `Invalid evals file at ${path}: ${describeIssues(result.error)}`);
|
|
274
|
+
}
|
|
275
|
+
return { evals: result.data, raw };
|
|
276
|
+
};
|
|
277
|
+
var locateQuestion = (raw, id) => {
|
|
278
|
+
const pattern = new RegExp(`^\\s*-?\\s*id:\\s*["']?${id}["']?\\s*$`, "u");
|
|
279
|
+
const lines = raw.split(`
|
|
280
|
+
`);
|
|
281
|
+
for (const [index, line] of lines.entries()) {
|
|
282
|
+
if (pattern.test(line)) {
|
|
283
|
+
return index + 1;
|
|
284
|
+
}
|
|
285
|
+
}
|
|
286
|
+
return;
|
|
287
|
+
};
|
|
288
|
+
|
|
289
|
+
// src/eval/findings.ts
|
|
290
|
+
var DOCS_URL = "https://useblume.dev/docs/reference/eval";
|
|
291
|
+
var hintedRoute = (question, project) => {
|
|
292
|
+
for (const hint of question.routes) {
|
|
293
|
+
const route = project.manifest.routes.find((candidate) => candidate.path === hint);
|
|
294
|
+
if (route) {
|
|
295
|
+
return route;
|
|
296
|
+
}
|
|
297
|
+
}
|
|
298
|
+
};
|
|
299
|
+
var routeFindings = (question, project, anchor) => {
|
|
300
|
+
const known = new Set(project.manifest.routes.map((route) => route.path));
|
|
301
|
+
return question.routes.filter((hint) => !known.has(hint)).map((hint) => ({
|
|
302
|
+
code: "BLUME_EVAL_ROUTE_UNKNOWN",
|
|
303
|
+
docsUrl: DOCS_URL,
|
|
304
|
+
file: anchor.path,
|
|
305
|
+
line: locateQuestion(anchor.raw, question.id),
|
|
306
|
+
message: `Question "${question.id}" hints at route "${hint}", which matches no page.`,
|
|
307
|
+
severity: "warning",
|
|
308
|
+
suggestion: "Update the question's `routes` to the page's current route, or remove the hint."
|
|
309
|
+
}));
|
|
310
|
+
};
|
|
311
|
+
var questionFinding = (question, outcome, project, anchor) => {
|
|
312
|
+
const route = hintedRoute(question, project);
|
|
313
|
+
const site = route ? { file: route.sourcePath, url: route.path } : { file: anchor.path, line: locateQuestion(anchor.raw, question.id) };
|
|
314
|
+
if (outcome.status === "error") {
|
|
315
|
+
return {
|
|
316
|
+
code: "BLUME_EVAL_QUESTION_ERROR",
|
|
317
|
+
docsUrl: DOCS_URL,
|
|
318
|
+
message: `Eval run failed for "${question.question}"${outcome.detail ? ` — ${outcome.detail}` : ""}.`,
|
|
319
|
+
severity: question.severity,
|
|
320
|
+
suggestion: "Rerun `blume eval`; if it persists, check the agent CLI installation and the failure detail.",
|
|
321
|
+
...site
|
|
322
|
+
};
|
|
323
|
+
}
|
|
324
|
+
const missing = outcome.missing.length > 0 ? ` — missing: ${outcome.missing.join("; ")}` : "";
|
|
325
|
+
return {
|
|
326
|
+
code: "BLUME_EVAL_QUESTION_FAILED",
|
|
327
|
+
docsUrl: DOCS_URL,
|
|
328
|
+
message: `Docs could not answer: "${question.question}"${missing}`,
|
|
329
|
+
severity: question.severity,
|
|
330
|
+
suggestion: "State the missing facts on this page, then rerun `blume eval`.",
|
|
331
|
+
...site
|
|
332
|
+
};
|
|
333
|
+
};
|
|
334
|
+
|
|
335
|
+
// src/eval/run.ts
|
|
336
|
+
var DEFAULT_READER_TIMEOUT_MS = 180000;
|
|
337
|
+
var DEFAULT_JUDGE_TIMEOUT_MS = 60000;
|
|
338
|
+
var errored = (question, detail, durationMs) => ({
|
|
339
|
+
detail,
|
|
340
|
+
durationMs,
|
|
341
|
+
expected: question.expected,
|
|
342
|
+
id: question.id,
|
|
343
|
+
missing: [],
|
|
344
|
+
question: question.question,
|
|
345
|
+
routes: question.routes,
|
|
346
|
+
status: "error"
|
|
347
|
+
});
|
|
348
|
+
var runQuestion = async (question, index, context) => {
|
|
349
|
+
const started = performance.now();
|
|
350
|
+
const elapsed = () => Math.round(performance.now() - started);
|
|
351
|
+
const workDir = join2(context.dir, `work-${index}`);
|
|
352
|
+
await mkdir(workDir, { recursive: true });
|
|
353
|
+
const answerPath = join2(workDir, "answer.txt");
|
|
354
|
+
const reader = await context.run(context.bin, agentArgs(context.kind, { lastMessagePath: answerPath, mcp: context.mcp }), {
|
|
355
|
+
cwd: workDir,
|
|
356
|
+
prompt: readerPrompt(question),
|
|
357
|
+
timeoutMs: context.readerTimeoutMs
|
|
358
|
+
});
|
|
359
|
+
const answer = await readAgentOutput(context.kind, reader, answerPath);
|
|
360
|
+
if (answer.isError) {
|
|
361
|
+
return {
|
|
362
|
+
...errored(question, `reader ${answer.detail ?? "failed"}`, elapsed()),
|
|
363
|
+
costUsd: answer.costUsd
|
|
364
|
+
};
|
|
365
|
+
}
|
|
366
|
+
const verdictPath = join2(workDir, "verdict.txt");
|
|
367
|
+
const judge = await context.run(context.bin, agentArgs(context.kind, { lastMessagePath: verdictPath }), {
|
|
368
|
+
cwd: workDir,
|
|
369
|
+
prompt: judgePrompt(question, answer.text),
|
|
370
|
+
timeoutMs: context.judgeTimeoutMs
|
|
371
|
+
});
|
|
372
|
+
const graded = await readAgentOutput(context.kind, judge, verdictPath);
|
|
373
|
+
const costUsd = answer.costUsd === undefined && graded.costUsd === undefined ? undefined : (answer.costUsd ?? 0) + (graded.costUsd ?? 0);
|
|
374
|
+
if (graded.isError) {
|
|
375
|
+
return {
|
|
376
|
+
...errored(question, `judge ${graded.detail ?? "failed"}`, elapsed()),
|
|
377
|
+
answer: answer.text,
|
|
378
|
+
costUsd
|
|
379
|
+
};
|
|
380
|
+
}
|
|
381
|
+
const verdict = parseVerdict(graded.text);
|
|
382
|
+
if (!verdict) {
|
|
383
|
+
return {
|
|
384
|
+
...errored(question, "judge returned no parseable verdict", elapsed()),
|
|
385
|
+
answer: answer.text,
|
|
386
|
+
costUsd
|
|
387
|
+
};
|
|
388
|
+
}
|
|
389
|
+
return {
|
|
390
|
+
answer: answer.text,
|
|
391
|
+
costUsd,
|
|
392
|
+
durationMs: elapsed(),
|
|
393
|
+
expected: question.expected,
|
|
394
|
+
id: question.id,
|
|
395
|
+
missing: verdict.missing,
|
|
396
|
+
notes: verdict.notes || undefined,
|
|
397
|
+
question: question.question,
|
|
398
|
+
routes: question.routes,
|
|
399
|
+
score: verdict.score,
|
|
400
|
+
status: verdict.pass ? "pass" : "fail"
|
|
401
|
+
};
|
|
402
|
+
};
|
|
403
|
+
var runEval = async (options) => {
|
|
404
|
+
const started = performance.now();
|
|
405
|
+
const kind = options.agent;
|
|
406
|
+
const run = options.run ?? runAgentHeadless;
|
|
407
|
+
const anchor = { path: options.evalsPath, raw: options.rawEvals };
|
|
408
|
+
const dir = await mkdtemp2(join2(tmpdir2(), "blume-eval-"));
|
|
409
|
+
const snapshotPath = join2(dir, "mcp-data.json");
|
|
410
|
+
await writeFile2(snapshotPath, JSON.stringify(await buildMcpData(options.project)));
|
|
411
|
+
const mcp = await writeMcpConfig(dir, snapshotPath);
|
|
412
|
+
const context = {
|
|
413
|
+
bin: AGENTS[kind].bin,
|
|
414
|
+
dir,
|
|
415
|
+
judgeTimeoutMs: options.judgeTimeoutMs ?? DEFAULT_JUDGE_TIMEOUT_MS,
|
|
416
|
+
kind,
|
|
417
|
+
mcp,
|
|
418
|
+
readerTimeoutMs: options.readerTimeoutMs ?? DEFAULT_READER_TIMEOUT_MS,
|
|
419
|
+
run
|
|
420
|
+
};
|
|
421
|
+
const diagnostics = [];
|
|
422
|
+
const results = [];
|
|
423
|
+
const { questions } = options.evals;
|
|
424
|
+
for (const [index, question] of questions.entries()) {
|
|
425
|
+
diagnostics.push(...routeFindings(question, options.project, anchor));
|
|
426
|
+
if (question.skip) {
|
|
427
|
+
results.push({
|
|
428
|
+
durationMs: 0,
|
|
429
|
+
expected: question.expected,
|
|
430
|
+
id: question.id,
|
|
431
|
+
missing: [],
|
|
432
|
+
question: question.question,
|
|
433
|
+
routes: question.routes,
|
|
434
|
+
status: "skip"
|
|
435
|
+
});
|
|
436
|
+
continue;
|
|
437
|
+
}
|
|
438
|
+
options.onProgress?.({
|
|
439
|
+
id: question.id,
|
|
440
|
+
index,
|
|
441
|
+
kind: "question-start",
|
|
442
|
+
total: questions.length
|
|
443
|
+
});
|
|
444
|
+
const result = await runQuestion(question, index, context);
|
|
445
|
+
results.push(result);
|
|
446
|
+
if (result.status === "fail" || result.status === "error") {
|
|
447
|
+
diagnostics.push(questionFinding(question, {
|
|
448
|
+
detail: result.detail,
|
|
449
|
+
missing: result.missing,
|
|
450
|
+
status: result.status
|
|
451
|
+
}, options.project, anchor));
|
|
452
|
+
}
|
|
453
|
+
options.onProgress?.({
|
|
454
|
+
index,
|
|
455
|
+
kind: "question-end",
|
|
456
|
+
result,
|
|
457
|
+
total: questions.length
|
|
458
|
+
});
|
|
459
|
+
}
|
|
460
|
+
const counts = {
|
|
461
|
+
error: 0,
|
|
462
|
+
fail: 0,
|
|
463
|
+
pass: 0,
|
|
464
|
+
skip: 0
|
|
465
|
+
};
|
|
466
|
+
for (const result of results) {
|
|
467
|
+
counts[result.status] += 1;
|
|
468
|
+
}
|
|
469
|
+
const costs = results.flatMap((result) => result.costUsd === undefined ? [] : [result.costUsd]);
|
|
470
|
+
return {
|
|
471
|
+
agent: kind,
|
|
472
|
+
costUsd: costs.length > 0 ? costs.reduce((total, cost) => total + cost, 0) : undefined,
|
|
473
|
+
counts,
|
|
474
|
+
diagnostics,
|
|
475
|
+
durationMs: Math.round(performance.now() - started),
|
|
476
|
+
results
|
|
477
|
+
};
|
|
478
|
+
};
|
|
479
|
+
|
|
480
|
+
// src/cli/commands/eval.ts
|
|
481
|
+
var DEFAULT_FILE = "evals.yaml";
|
|
482
|
+
var DEFAULT_TIMEOUT_S = 180;
|
|
483
|
+
var isAgentKind = (value) => (value in AGENTS);
|
|
484
|
+
var notInstalled = (agent) => {
|
|
485
|
+
const cli = AGENTS[agent];
|
|
486
|
+
logger.error(`${cli.name} (\`${cli.bin}\`) was not found on PATH. Install it with \`${cli.install}\`.`);
|
|
487
|
+
return process.exit(1);
|
|
488
|
+
};
|
|
489
|
+
var launchAgentCode = async (agent, prompt) => {
|
|
490
|
+
try {
|
|
491
|
+
return await launchAgent(AGENTS[agent].bin, prompt);
|
|
492
|
+
} catch (error) {
|
|
493
|
+
if (error?.code !== "ENOENT") {
|
|
494
|
+
throw error;
|
|
495
|
+
}
|
|
496
|
+
return notInstalled(agent);
|
|
497
|
+
}
|
|
498
|
+
};
|
|
499
|
+
var passFraction = (result) => {
|
|
500
|
+
const ran = result.results.length - result.counts.skip;
|
|
501
|
+
return ran === 0 ? 1 : result.counts.pass / ran;
|
|
502
|
+
};
|
|
503
|
+
var parseFlags = (args) => {
|
|
504
|
+
if (!isAgentKind(args.agent)) {
|
|
505
|
+
logger.error(`Invalid --agent "${args.agent}" (use claude | codex).`);
|
|
506
|
+
process.exit(1);
|
|
507
|
+
}
|
|
508
|
+
if (args.action !== undefined && args.action !== "init") {
|
|
509
|
+
logger.error(`Unknown action "${args.action}" (did you mean "init"?).`);
|
|
510
|
+
process.exit(1);
|
|
511
|
+
}
|
|
512
|
+
if (args.json && args.fix) {
|
|
513
|
+
logger.error("--json and --fix are mutually exclusive.");
|
|
514
|
+
process.exit(1);
|
|
515
|
+
}
|
|
516
|
+
const threshold = args.threshold === undefined ? 1 : Number(args.threshold);
|
|
517
|
+
if (!Number.isFinite(threshold) || threshold < 0 || threshold > 1) {
|
|
518
|
+
logger.error(`Invalid --threshold "${args.threshold}" (use 0..1).`);
|
|
519
|
+
process.exit(1);
|
|
520
|
+
}
|
|
521
|
+
const timeoutS = args.timeout === undefined ? DEFAULT_TIMEOUT_S : Number(args.timeout);
|
|
522
|
+
if (!Number.isInteger(timeoutS) || timeoutS <= 0) {
|
|
523
|
+
logger.error(`Invalid --timeout "${args.timeout}" (whole seconds).`);
|
|
524
|
+
process.exit(1);
|
|
525
|
+
}
|
|
526
|
+
return { agent: args.agent, threshold, timeoutS };
|
|
527
|
+
};
|
|
528
|
+
var runFixHandoff = async (agent, result, root, threshold) => {
|
|
529
|
+
const count = result.counts.fail + result.counts.error;
|
|
530
|
+
if (count === 0) {
|
|
531
|
+
return;
|
|
532
|
+
}
|
|
533
|
+
const cli = AGENTS[agent];
|
|
534
|
+
const report = await writeEvalReport(result, root, threshold);
|
|
535
|
+
process.stderr.write(` Handing ${count} failed question${count === 1 ? "" : "s"} to ${cli.name}…
|
|
536
|
+
|
|
537
|
+
`);
|
|
538
|
+
const code = await launchAgentCode(agent, evalFixPrompt(report));
|
|
539
|
+
if (code !== 0) {
|
|
540
|
+
process.exit(code);
|
|
541
|
+
}
|
|
542
|
+
};
|
|
543
|
+
var runInit = async (agent, file) => {
|
|
544
|
+
const path = join3(process.cwd(), file);
|
|
545
|
+
if (existsSync(path)) {
|
|
546
|
+
logger.error(`${file} already exists — edit it directly, or pass --file to draft elsewhere.`);
|
|
547
|
+
process.exit(1);
|
|
548
|
+
}
|
|
549
|
+
const code = await launchAgentCode(agent, initPrompt(file));
|
|
550
|
+
if (code !== 0) {
|
|
551
|
+
process.exit(code);
|
|
552
|
+
}
|
|
553
|
+
};
|
|
554
|
+
var evalCommand = defineCommand({
|
|
555
|
+
args: {
|
|
556
|
+
action: {
|
|
557
|
+
description: 'Optional action: "init" drafts a starter evals file.',
|
|
558
|
+
required: false,
|
|
559
|
+
type: "positional"
|
|
560
|
+
},
|
|
561
|
+
agent: {
|
|
562
|
+
default: "claude",
|
|
563
|
+
description: "Agent CLI that reads and grades the docs: claude | codex.",
|
|
564
|
+
type: "string"
|
|
565
|
+
},
|
|
566
|
+
file: {
|
|
567
|
+
default: DEFAULT_FILE,
|
|
568
|
+
description: "The evals file to run.",
|
|
569
|
+
type: "string"
|
|
570
|
+
},
|
|
571
|
+
fix: {
|
|
572
|
+
description: "After a failing run, hand the report to the agent to fix the docs interactively.",
|
|
573
|
+
type: "boolean"
|
|
574
|
+
},
|
|
575
|
+
json: {
|
|
576
|
+
description: "Emit the report as JSON on stdout (for CI/editors).",
|
|
577
|
+
type: "boolean"
|
|
578
|
+
},
|
|
579
|
+
threshold: {
|
|
580
|
+
description: "Minimum passing fraction (0..1) before the run exits non-zero. Defaults to 1.",
|
|
581
|
+
type: "string"
|
|
582
|
+
},
|
|
583
|
+
timeout: {
|
|
584
|
+
description: `Reader time limit per question, in seconds. Defaults to ${DEFAULT_TIMEOUT_S}.`,
|
|
585
|
+
type: "string"
|
|
586
|
+
},
|
|
587
|
+
verbose: {
|
|
588
|
+
description: "Include the reader's full answer under each failure.",
|
|
589
|
+
type: "boolean"
|
|
590
|
+
}
|
|
591
|
+
},
|
|
592
|
+
meta: commandMeta.eval,
|
|
593
|
+
async run({ args }) {
|
|
594
|
+
const root = process.cwd();
|
|
595
|
+
const { agent, threshold, timeoutS } = parseFlags(args);
|
|
596
|
+
if (args.action === "init") {
|
|
597
|
+
await runInit(agent, args.file);
|
|
598
|
+
return;
|
|
599
|
+
}
|
|
600
|
+
let result;
|
|
601
|
+
try {
|
|
602
|
+
const project = await scanProject(root, { mode: "build" });
|
|
603
|
+
const evalsPath = join3(root, args.file);
|
|
604
|
+
const { evals, raw } = await loadEvalsFile(evalsPath);
|
|
605
|
+
process.stderr.write(`${headerLine(evals.questions.length, agent)}
|
|
606
|
+
|
|
607
|
+
`);
|
|
608
|
+
result = await runEval({
|
|
609
|
+
agent,
|
|
610
|
+
evals,
|
|
611
|
+
evalsPath,
|
|
612
|
+
onProgress: (event) => {
|
|
613
|
+
if (event.kind === "question-start") {
|
|
614
|
+
process.stderr.write(`${startLine(event.id, event.index, event.total)}
|
|
615
|
+
`);
|
|
616
|
+
return;
|
|
617
|
+
}
|
|
618
|
+
const lines = [
|
|
619
|
+
questionLine(event.result),
|
|
620
|
+
...questionDetails(event.result, Boolean(args.verbose))
|
|
621
|
+
];
|
|
622
|
+
process.stderr.write(`${lines.join(`
|
|
623
|
+
`)}
|
|
624
|
+
`);
|
|
625
|
+
},
|
|
626
|
+
project,
|
|
627
|
+
rawEvals: raw,
|
|
628
|
+
readerTimeoutMs: timeoutS * 1000
|
|
629
|
+
});
|
|
630
|
+
} catch (error) {
|
|
631
|
+
if (error instanceof EvalsFileError) {
|
|
632
|
+
logger.error(error.message);
|
|
633
|
+
process.exit(1);
|
|
634
|
+
}
|
|
635
|
+
if (error instanceof BlumeError) {
|
|
636
|
+
logger.error(error.diagnostic.message);
|
|
637
|
+
process.exit(1);
|
|
638
|
+
}
|
|
639
|
+
if (error?.code === "ENOENT") {
|
|
640
|
+
notInstalled(agent);
|
|
641
|
+
}
|
|
642
|
+
reportInternalError(error);
|
|
643
|
+
process.exit(1);
|
|
644
|
+
}
|
|
645
|
+
const tail = [
|
|
646
|
+
"",
|
|
647
|
+
...warningLines(result, root),
|
|
648
|
+
...fixLines(result, root),
|
|
649
|
+
"",
|
|
650
|
+
` ${summaryLine(result)}`,
|
|
651
|
+
""
|
|
652
|
+
];
|
|
653
|
+
process.stderr.write(tail.join(`
|
|
654
|
+
`));
|
|
655
|
+
const failed = passFraction(result) < threshold;
|
|
656
|
+
if (args.fix) {
|
|
657
|
+
await runFixHandoff(agent, result, root, threshold);
|
|
658
|
+
return;
|
|
659
|
+
}
|
|
660
|
+
if (args.json) {
|
|
661
|
+
process.stdout.write(evalReportJson(result, root, threshold));
|
|
662
|
+
if (failed) {
|
|
663
|
+
await flushStdout();
|
|
664
|
+
process.exit(1);
|
|
665
|
+
}
|
|
666
|
+
return;
|
|
667
|
+
}
|
|
668
|
+
if (failed) {
|
|
669
|
+
process.exit(1);
|
|
670
|
+
}
|
|
671
|
+
}
|
|
672
|
+
});
|
|
673
|
+
export {
|
|
674
|
+
evalCommand,
|
|
675
|
+
passFraction
|
|
676
|
+
};
|
|
677
|
+
|
|
678
|
+
//# debugId=D2F08D9122CA47D464756E2164756E21
|
|
679
|
+
//# sourceMappingURL=chunk-0ewz4trd.js.map
|