blume 1.6.4 → 1.6.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +41 -0
- package/bin/blume.mjs +3 -2
- package/dist/cli/chunk-0ewz4trd.js +679 -0
- package/dist/cli/chunk-0ewz4trd.js.map +15 -0
- package/dist/cli/chunk-27gtm2ym.js +69 -0
- package/dist/cli/chunk-27gtm2ym.js.map +11 -0
- package/dist/cli/chunk-2aj8ddew.js +72 -0
- package/dist/cli/chunk-2aj8ddew.js.map +10 -0
- package/dist/cli/chunk-3k0kzs6d.js +69 -0
- package/dist/cli/chunk-3k0kzs6d.js.map +11 -0
- package/dist/cli/chunk-3r94j3tc.js +221 -0
- package/dist/cli/chunk-3r94j3tc.js.map +10 -0
- package/dist/cli/chunk-4trphnvy.js +102 -0
- package/dist/cli/chunk-4trphnvy.js.map +11 -0
- package/dist/cli/chunk-4xyggvgf.js +21 -0
- package/dist/cli/chunk-4xyggvgf.js.map +10 -0
- package/dist/cli/chunk-5hs6gb7n.js +32 -0
- package/dist/cli/chunk-5hs6gb7n.js.map +10 -0
- package/dist/cli/chunk-5yvt556e.js +185 -0
- package/dist/cli/chunk-5yvt556e.js.map +11 -0
- package/dist/cli/chunk-62qsssnh.js +3808 -0
- package/dist/cli/chunk-62qsssnh.js.map +36 -0
- package/dist/cli/chunk-6kzzpsx8.js +26 -0
- package/dist/cli/chunk-6kzzpsx8.js.map +10 -0
- package/dist/cli/chunk-8gnpdsn1.js +952 -0
- package/dist/cli/chunk-8gnpdsn1.js.map +12 -0
- package/dist/cli/chunk-9sh49q0h.js +30 -0
- package/dist/cli/chunk-9sh49q0h.js.map +10 -0
- package/dist/cli/chunk-aerwpe14.js +2370 -0
- package/dist/cli/chunk-aerwpe14.js.map +15 -0
- package/dist/cli/chunk-ag1zyr5x.js +176 -0
- package/dist/cli/chunk-ag1zyr5x.js.map +10 -0
- package/dist/cli/chunk-bawgnt8x.js +277 -0
- package/dist/cli/chunk-bawgnt8x.js.map +11 -0
- package/dist/cli/chunk-bcy492zc.js +16 -0
- package/dist/cli/chunk-bcy492zc.js.map +10 -0
- package/dist/cli/chunk-btfr9yvw.js +41 -0
- package/dist/cli/chunk-btfr9yvw.js.map +10 -0
- package/dist/cli/chunk-cbjnx4s8.js +73 -0
- package/dist/cli/chunk-cbjnx4s8.js.map +10 -0
- package/dist/cli/chunk-cnvm6k3e.js +96 -0
- package/dist/cli/chunk-cnvm6k3e.js.map +10 -0
- package/dist/cli/chunk-etsqspj6.js +5170 -0
- package/dist/cli/chunk-etsqspj6.js.map +47 -0
- package/dist/cli/chunk-ev67ycx0.js +15 -0
- package/dist/cli/chunk-ev67ycx0.js.map +10 -0
- package/dist/cli/chunk-ey89bjj1.js +209 -0
- package/dist/cli/chunk-ey89bjj1.js.map +11 -0
- package/dist/cli/chunk-f75cqye8.js +76 -0
- package/dist/cli/chunk-f75cqye8.js.map +10 -0
- package/dist/cli/chunk-j00ezcg5.js +259 -0
- package/dist/cli/chunk-j00ezcg5.js.map +11 -0
- package/dist/cli/chunk-jtb45atp.js +467 -0
- package/dist/cli/chunk-jtb45atp.js.map +14 -0
- package/dist/cli/chunk-m3p3wahd.js +117 -0
- package/dist/cli/chunk-m3p3wahd.js.map +10 -0
- package/dist/cli/chunk-n0y172hf.js +387 -0
- package/dist/cli/chunk-n0y172hf.js.map +12 -0
- package/dist/cli/chunk-n4qjabmt.js +1062 -0
- package/dist/cli/chunk-n4qjabmt.js.map +25 -0
- package/dist/cli/chunk-nyqzjdhj.js +111 -0
- package/dist/cli/chunk-nyqzjdhj.js.map +11 -0
- package/dist/cli/chunk-pxj10x8y.js +35 -0
- package/dist/cli/chunk-pxj10x8y.js.map +10 -0
- package/dist/cli/chunk-s4jn7f1q.js +54 -0
- package/dist/cli/chunk-s4jn7f1q.js.map +10 -0
- package/dist/cli/chunk-s4k1pnvf.js +81 -0
- package/dist/cli/chunk-s4k1pnvf.js.map +10 -0
- package/dist/cli/chunk-s5e5jt53.js +227 -0
- package/dist/cli/chunk-s5e5jt53.js.map +11 -0
- package/dist/cli/chunk-sbdqrjbb.js +81 -0
- package/dist/cli/chunk-sbdqrjbb.js.map +10 -0
- package/dist/cli/chunk-tc89yh2r.js +136 -0
- package/dist/cli/chunk-tc89yh2r.js.map +10 -0
- package/dist/cli/chunk-vt8fgygt.js +23 -0
- package/dist/cli/chunk-vt8fgygt.js.map +10 -0
- package/dist/cli/chunk-vv237fp3.js +1002 -0
- package/dist/cli/chunk-vv237fp3.js.map +13 -0
- package/dist/cli/chunk-vv3f8mb6.js +5314 -0
- package/dist/cli/chunk-vv3f8mb6.js.map +58 -0
- package/dist/cli/chunk-vxv4x1n8.js +17 -0
- package/dist/cli/chunk-vxv4x1n8.js.map +10 -0
- package/dist/cli/chunk-wb067mv3.js +758 -0
- package/dist/cli/chunk-wb067mv3.js.map +13 -0
- package/dist/cli/chunk-wd27zjcz.js +60 -0
- package/dist/cli/chunk-wd27zjcz.js.map +10 -0
- package/dist/cli/chunk-wkq5tbtq.js +1141 -0
- package/dist/cli/chunk-wkq5tbtq.js.map +19 -0
- package/dist/cli/chunk-x1vrdjyk.js +1967 -0
- package/dist/cli/chunk-x1vrdjyk.js.map +34 -0
- package/dist/cli/chunk-x66c5yjn.js +23 -0
- package/dist/cli/chunk-x66c5yjn.js.map +10 -0
- package/dist/cli/index.js +55 -27587
- package/dist/cli/index.js.map +5 -243
- package/dist/types/ai/ask-context.d.ts +26 -0
- package/dist/types/core/code-fences.d.ts +11 -0
- package/dist/types/core/config-input.d.ts +10 -0
- package/dist/types/core/package-root.d.ts +1 -1
- package/dist/types/core/schema.d.ts +74 -1
- package/docs/02-deployment.mdx +1 -1
- package/docs/configuration/analytics.mdx +21 -2
- package/docs/configuration/ask-ai.mdx +1 -1
- package/docs/configuration/customization.mdx +2 -9
- package/docs/content/syntax.mdx +1 -1
- package/docs/reference/cli.mdx +1 -1
- package/package.json +16 -14
- package/src/ai/api/handlers.ts +4 -7
- package/src/ai/api/paths.ts +8 -0
- package/src/ai/api/spec.ts +2 -1
- package/src/ai/ask-context.ts +378 -22
- package/src/astro/generate.ts +25 -29
- package/src/astro/include-hmr.ts +10 -13
- package/src/astro/include-refresh.ts +0 -0
- package/src/astro/index.ts +6 -1
- package/src/astro/integration.ts +269 -53
- package/src/astro/module-types.ts +74 -0
- package/src/astro/templates.ts +85 -97
- package/src/audit/image-size.ts +10 -8
- package/src/cli/command-meta.ts +77 -0
- package/src/cli/commands/add.ts +2 -4
- package/src/cli/commands/audit.ts +2 -4
- package/src/cli/commands/build.ts +42 -346
- package/src/cli/commands/check.ts +2 -4
- package/src/cli/commands/dev.ts +31 -42
- package/src/cli/commands/doctor.ts +2 -4
- package/src/cli/commands/eject.ts +3 -41
- package/src/cli/commands/eval.ts +2 -5
- package/src/cli/commands/init.ts +2 -4
- package/src/cli/commands/mcp-stdio.ts +2 -5
- package/src/cli/commands/preview.ts +3 -5
- package/src/cli/commands/sync.ts +2 -4
- package/src/cli/commands/translate.ts +2 -5
- package/src/cli/commands/validate.ts +2 -4
- package/src/cli/commands/version.ts +2 -4
- package/src/cli/eject-scripts.ts +0 -45
- package/src/cli/host-args.ts +16 -0
- package/src/cli/index.ts +84 -35
- package/src/cli/lazy-command.ts +47 -0
- package/src/components/content/GithubInfo.astro +4 -1
- package/src/components/content/mermaid-element.ts +8 -0
- package/src/components/layout/Analytics.astro +20 -1
- package/src/components/layout/PageLayout.astro +14 -3
- package/src/components/layout/ReferenceLayout.astro +15 -4
- package/src/components/layout/RootLayout.astro +15 -4
- package/src/components/layout/analytics-client.ts +2 -1
- package/src/components/layout/page-locale.ts +29 -0
- package/src/components/openapi/AsyncApiOperation.astro +5 -3
- package/src/components/openapi/Authorization.astro +4 -6
- package/src/components/openapi/Bindings.astro +2 -2
- package/src/components/openapi/Description.astro +109 -0
- package/src/components/openapi/GraphqlFieldsTable.astro +5 -7
- package/src/components/openapi/GraphqlOperation.astro +4 -3
- package/src/components/openapi/GraphqlType.astro +4 -6
- package/src/components/openapi/ParametersTable.astro +5 -7
- package/src/components/openapi/RequestBody.astro +2 -4
- package/src/components/openapi/Responses.astro +4 -3
- package/src/components/openapi/SchemaProperty.astro +11 -6
- package/src/components/openapi/description.ts +91 -0
- package/src/core/api-name.ts +18 -0
- package/src/core/code-fences.ts +48 -0
- package/src/core/config-input.ts +10 -0
- package/src/core/content-assets.ts +3 -7
- package/src/core/includes.ts +3 -7
- package/src/core/package-root.ts +1 -1
- package/src/core/schema.ts +27 -0
- package/src/core/sources/normalize.ts +2 -37
- package/src/core/sources/obsidian.ts +3 -2
- package/src/core/svg-dimensions.ts +97 -0
- package/src/core/version-cut.ts +2 -2
- package/src/deploy/artifacts.ts +370 -0
- package/src/deploy/cloudflare-negotiation.ts +97 -32
- package/src/deploy/function-bundle.ts +66 -20
- package/src/deploy/sitemap.ts +6 -0
- package/src/deploy/vercel-negotiation.ts +8 -30
- package/src/og/card.ts +6 -12
- package/src/openapi/render-mdx.ts +9 -5
- package/src/registry/eject.ts +0 -2
- package/src/theme/entry.ts +9 -2
package/src/ai/ask-context.ts
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
import { normalizeRoute } from "../core/base-path.ts";
|
|
2
|
+
import { nextFenceState } from "../core/code-fences.ts";
|
|
3
|
+
import type { FenceState } from "../core/code-fences.ts";
|
|
2
4
|
import { buildOramaIndex, queryOramaIndex } from "../search/orama-index.ts";
|
|
3
5
|
import type { OramaDoc } from "../search/orama-index.ts";
|
|
4
6
|
|
|
@@ -68,45 +70,94 @@ export interface AskRetrievalOptions {
|
|
|
68
70
|
const EXCERPT_LEAD = 160;
|
|
69
71
|
|
|
70
72
|
/**
|
|
71
|
-
*
|
|
72
|
-
*
|
|
73
|
-
* window toward incidental matches instead of the meaningful terms.
|
|
73
|
+
* Remove English filler from retrieval and excerpt queries. Keep content verbs
|
|
74
|
+
* such as "sign", "file" and "close", which can name documentation topics.
|
|
74
75
|
*/
|
|
75
76
|
const STOPWORDS = new Set([
|
|
77
|
+
"a",
|
|
76
78
|
"about",
|
|
79
|
+
"again",
|
|
80
|
+
"against",
|
|
81
|
+
"all",
|
|
82
|
+
"am",
|
|
83
|
+
"an",
|
|
77
84
|
"and",
|
|
85
|
+
"any",
|
|
78
86
|
"are",
|
|
79
87
|
"as",
|
|
80
88
|
"at",
|
|
81
89
|
"be",
|
|
90
|
+
"been",
|
|
91
|
+
"before",
|
|
92
|
+
"being",
|
|
93
|
+
"both",
|
|
82
94
|
"but",
|
|
83
95
|
"by",
|
|
84
96
|
"can",
|
|
97
|
+
"did",
|
|
85
98
|
"do",
|
|
86
99
|
"does",
|
|
100
|
+
"done",
|
|
101
|
+
"each",
|
|
102
|
+
"every",
|
|
87
103
|
"for",
|
|
88
104
|
"from",
|
|
105
|
+
"had",
|
|
106
|
+
"has",
|
|
107
|
+
"have",
|
|
108
|
+
"having",
|
|
109
|
+
"he",
|
|
110
|
+
"her",
|
|
111
|
+
"hers",
|
|
112
|
+
"him",
|
|
113
|
+
"his",
|
|
89
114
|
"how",
|
|
115
|
+
"i",
|
|
116
|
+
"if",
|
|
90
117
|
"in",
|
|
91
118
|
"into",
|
|
92
119
|
"is",
|
|
93
120
|
"it",
|
|
94
121
|
"its",
|
|
122
|
+
"like",
|
|
123
|
+
"look",
|
|
124
|
+
"may",
|
|
125
|
+
"me",
|
|
126
|
+
"might",
|
|
127
|
+
"mine",
|
|
128
|
+
"must",
|
|
95
129
|
"my",
|
|
130
|
+
"nor",
|
|
96
131
|
"of",
|
|
97
132
|
"on",
|
|
98
133
|
"or",
|
|
134
|
+
"other",
|
|
99
135
|
"our",
|
|
136
|
+
"ours",
|
|
137
|
+
"shall",
|
|
138
|
+
"she",
|
|
139
|
+
"should",
|
|
140
|
+
"so",
|
|
141
|
+
"some",
|
|
142
|
+
"than",
|
|
100
143
|
"that",
|
|
101
144
|
"the",
|
|
145
|
+
"their",
|
|
146
|
+
"theirs",
|
|
147
|
+
"them",
|
|
148
|
+
"then",
|
|
149
|
+
"there",
|
|
102
150
|
"these",
|
|
151
|
+
"they",
|
|
103
152
|
"this",
|
|
104
153
|
"those",
|
|
105
154
|
"to",
|
|
155
|
+
"us",
|
|
106
156
|
"use",
|
|
107
157
|
"used",
|
|
108
158
|
"using",
|
|
109
159
|
"was",
|
|
160
|
+
"we",
|
|
110
161
|
"were",
|
|
111
162
|
"what",
|
|
112
163
|
"when",
|
|
@@ -114,9 +165,12 @@ const STOPWORDS = new Set([
|
|
|
114
165
|
"which",
|
|
115
166
|
"who",
|
|
116
167
|
"why",
|
|
168
|
+
"will",
|
|
117
169
|
"with",
|
|
170
|
+
"would",
|
|
118
171
|
"you",
|
|
119
172
|
"your",
|
|
173
|
+
"yours",
|
|
120
174
|
]);
|
|
121
175
|
|
|
122
176
|
/** A run of letters, combining marks and digits inside a word-like segment. */
|
|
@@ -149,13 +203,15 @@ const segmentQuery = (query: string): string[] => {
|
|
|
149
203
|
return pieces;
|
|
150
204
|
};
|
|
151
205
|
|
|
206
|
+
/** Lowercase word tokens of `text`, cut at the same boundaries as a query. */
|
|
207
|
+
const tokenize = (text: string): string[] =>
|
|
208
|
+
segmentQuery(text).flatMap((piece) => piece.match(TERM) ?? []);
|
|
209
|
+
|
|
152
210
|
/** Distinct, meaningful lowercase terms from a query (drops stopwords). */
|
|
153
|
-
const queryTerms = (query: string): string[] =>
|
|
154
|
-
|
|
155
|
-
return [...new Set(terms)].filter(
|
|
211
|
+
const queryTerms = (query: string): string[] =>
|
|
212
|
+
[...new Set(tokenize(query))].filter(
|
|
156
213
|
(term) => term.length >= 2 && !STOPWORDS.has(term)
|
|
157
214
|
);
|
|
158
|
-
};
|
|
159
215
|
|
|
160
216
|
/**
|
|
161
217
|
* The grounding preamble. The model is told to answer strictly from the injected
|
|
@@ -165,15 +221,87 @@ const queryTerms = (query: string): string[] => {
|
|
|
165
221
|
const BASE_INSTRUCTION =
|
|
166
222
|
"You are a helpful documentation assistant for this project. Answer the user's question using ONLY the documentation excerpts below. Each excerpt is headed by its page as `## Page Title (/route)`. If the answer is not covered by the excerpts, say you don't know and suggest where in the docs to look — do not invent details. Always cite the pages you drew from, and write every citation as a Markdown link to that page using its route, e.g. [Page Title](/route).";
|
|
167
223
|
|
|
168
|
-
/** The
|
|
169
|
-
const
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
224
|
+
/** The non-empty user turns, oldest first. Assistant turns never seed retrieval. */
|
|
225
|
+
const userTurns = (messages: AskMessage[]): string[] =>
|
|
226
|
+
messages
|
|
227
|
+
.filter((message) => message?.role === "user" && message.content?.trim())
|
|
228
|
+
.map((message) => message.content.trim());
|
|
229
|
+
|
|
230
|
+
/** Meaningful terms below which a follow-up cannot stand as a query on its own. */
|
|
231
|
+
const MIN_QUERY_TERMS = 3;
|
|
232
|
+
|
|
233
|
+
const FOLLOW_UP_OPENERS = new Set(["also", "and", "but", "then"]);
|
|
234
|
+
const ANAPHORA = new Set([
|
|
235
|
+
"it",
|
|
236
|
+
"its",
|
|
237
|
+
"that",
|
|
238
|
+
"them",
|
|
239
|
+
"these",
|
|
240
|
+
"they",
|
|
241
|
+
"this",
|
|
242
|
+
"those",
|
|
243
|
+
]);
|
|
244
|
+
|
|
245
|
+
const isFollowUp = (message: string): boolean => {
|
|
246
|
+
const words = segmentQuery(message);
|
|
247
|
+
const [first] = words;
|
|
248
|
+
return (
|
|
249
|
+
(first !== undefined && FOLLOW_UP_OPENERS.has(first)) ||
|
|
250
|
+
words.some((word) => ANAPHORA.has(word))
|
|
251
|
+
);
|
|
252
|
+
};
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* The texts that retrieve for the latest question, in rank order.
|
|
256
|
+
*
|
|
257
|
+
* The question itself always leads, verbatim: Orama's own tokenizer and BM25
|
|
258
|
+
* weighting see the whole sentence (version numbers, single-character CJK
|
|
259
|
+
* words, `--flags`, and the bigrams a ja/zh index depends on all survive), and
|
|
260
|
+
* a short question that names its subject ("Does it support i18n?") is never
|
|
261
|
+
* outvoted by whatever the reader asked before. Only when it reads like a
|
|
262
|
+
* follow-up — opener-led, pronoun-bearing, or nothing but filler ("Why?") —
|
|
263
|
+
* and is too short to stand alone does the nearest earlier user turn with
|
|
264
|
+
* content terms join as a second query, ranked behind the first so the earlier
|
|
265
|
+
* subject stays in view without displacing the current one. Assistant turns
|
|
266
|
+
* are excluded, so an incorrect answer cannot reinforce its own retrieval.
|
|
267
|
+
*/
|
|
268
|
+
const retrievalQueries = (turns: string[]): string[] => {
|
|
269
|
+
const [latest = "", ...earlier] = turns.toReversed();
|
|
270
|
+
const terms = queryTerms(latest);
|
|
271
|
+
if (terms.length >= MIN_QUERY_TERMS) {
|
|
272
|
+
return [latest];
|
|
273
|
+
}
|
|
274
|
+
if (terms.length > 0 && !isFollowUp(latest)) {
|
|
275
|
+
return [latest];
|
|
276
|
+
}
|
|
277
|
+
const context = earlier.find((turn) => queryTerms(turn).length > 0);
|
|
278
|
+
if (context === undefined) {
|
|
279
|
+
return [latest];
|
|
280
|
+
}
|
|
281
|
+
return terms.length === 0 ? [context] : [latest, context];
|
|
282
|
+
};
|
|
283
|
+
|
|
284
|
+
/**
|
|
285
|
+
* Merge ranked result lists round-robin — the first list's top hit, then the
|
|
286
|
+
* second's, and so on — dropping duplicate routes and stopping at `limit`.
|
|
287
|
+
*/
|
|
288
|
+
const interleave = (lists: OramaDoc[][], limit: number): OramaDoc[] => {
|
|
289
|
+
const merged: OramaDoc[] = [];
|
|
290
|
+
const seen = new Set<string>();
|
|
291
|
+
const depth = Math.max(...lists.map((list) => list.length));
|
|
292
|
+
for (let rank = 0; rank < depth; rank += 1) {
|
|
293
|
+
for (const list of lists) {
|
|
294
|
+
const doc = list[rank];
|
|
295
|
+
if (doc && !seen.has(doc.route)) {
|
|
296
|
+
seen.add(doc.route);
|
|
297
|
+
merged.push(doc);
|
|
298
|
+
}
|
|
299
|
+
if (merged.length >= limit) {
|
|
300
|
+
return merged;
|
|
301
|
+
}
|
|
174
302
|
}
|
|
175
303
|
}
|
|
176
|
-
return
|
|
304
|
+
return merged;
|
|
177
305
|
};
|
|
178
306
|
|
|
179
307
|
/**
|
|
@@ -248,6 +376,214 @@ export const relevantExcerpt = (
|
|
|
248
376
|
return withEllipsis(Math.max(0, best - lead));
|
|
249
377
|
};
|
|
250
378
|
|
|
379
|
+
/** A Markdown heading at level 2 or deeper — where a page divides itself. */
|
|
380
|
+
const SECTION_HEADING = /^ {0,3}#{2,6}[\t ]+.+$/u;
|
|
381
|
+
/** Any ATX heading, including the `#` title a lead-in may open with. */
|
|
382
|
+
const ANY_HEADING = /^ {0,3}#{1,6}[\t ]+.+$/u;
|
|
383
|
+
|
|
384
|
+
/** Offsets of the section headings in `text`, skipping fenced code. */
|
|
385
|
+
const sectionStarts = (text: string): number[] => {
|
|
386
|
+
const starts: number[] = [];
|
|
387
|
+
let fence: FenceState = null;
|
|
388
|
+
let offset = 0;
|
|
389
|
+
for (const line of text.split("\n")) {
|
|
390
|
+
const next = nextFenceState(line, fence);
|
|
391
|
+
if (fence === null && next === null && SECTION_HEADING.test(line)) {
|
|
392
|
+
starts.push(offset);
|
|
393
|
+
}
|
|
394
|
+
fence = next;
|
|
395
|
+
offset += line.length + 1;
|
|
396
|
+
}
|
|
397
|
+
return starts;
|
|
398
|
+
};
|
|
399
|
+
|
|
400
|
+
interface PageSection {
|
|
401
|
+
/** Word tokens of the section's heading line, or none when it has no heading. */
|
|
402
|
+
headingWords: string[];
|
|
403
|
+
/** Position in the page, for source-order output and omission markers. */
|
|
404
|
+
index: number;
|
|
405
|
+
text: string;
|
|
406
|
+
/** The section's word tokens, cut once so scoring is a prefix test. */
|
|
407
|
+
words: string[];
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
/** A page split into sections once, so per-request scoring never re-tokenizes. */
|
|
411
|
+
export interface ParsedPage {
|
|
412
|
+
sections: PageSection[];
|
|
413
|
+
/** NFC-normalized, LF-only, trimmed page text; excerpts slice from it. */
|
|
414
|
+
text: string;
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
/**
|
|
418
|
+
* Split a page at its `##`+ headings (outside code fences). The text above the
|
|
419
|
+
* first heading is the page's own lead-in and is a section like any other.
|
|
420
|
+
* Line endings are folded to LF first so a CRLF checkout splits and matches the
|
|
421
|
+
* same way as an LF one. Exported for testing; {@link createAskContext}
|
|
422
|
+
* parses each page once and caches it across requests.
|
|
423
|
+
*/
|
|
424
|
+
export const parsePage = (content: string): ParsedPage => {
|
|
425
|
+
const text = content.normalize("NFC").replaceAll("\r\n", "\n").trim();
|
|
426
|
+
const headings = sectionStarts(text);
|
|
427
|
+
if (headings.length === 0) {
|
|
428
|
+
return { sections: [], text };
|
|
429
|
+
}
|
|
430
|
+
const starts = headings[0] === 0 ? headings : [0, ...headings];
|
|
431
|
+
const sections = starts.map((start, index) => {
|
|
432
|
+
const section = text.slice(start, starts[index + 1]).trim();
|
|
433
|
+
const [firstLine = ""] = section.split("\n", 1);
|
|
434
|
+
const headingWords = ANY_HEADING.test(firstLine) ? tokenize(firstLine) : [];
|
|
435
|
+
return { headingWords, index, text: section, words: tokenize(section) };
|
|
436
|
+
});
|
|
437
|
+
return { sections, text };
|
|
438
|
+
};
|
|
439
|
+
|
|
440
|
+
interface ScoredSection extends PageSection {
|
|
441
|
+
/** How many distinct query terms the section mentions. */
|
|
442
|
+
coverage: number;
|
|
443
|
+
/** Term hits per word, so a long section can't win on bulk alone. */
|
|
444
|
+
density: number;
|
|
445
|
+
/** How many distinct query terms the section's heading names. */
|
|
446
|
+
titled: number;
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
interface TermMatch {
|
|
450
|
+
/** Every word that starts with a term counts once. */
|
|
451
|
+
hits: number;
|
|
452
|
+
/** Distinct terms some word starts with. */
|
|
453
|
+
matched: number;
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
/** How `words` match `terms` by prefix. */
|
|
457
|
+
const matchTerms = (words: string[], terms: string[]): TermMatch => {
|
|
458
|
+
const matched = new Set<string>();
|
|
459
|
+
let hits = 0;
|
|
460
|
+
for (const word of words) {
|
|
461
|
+
for (const term of terms) {
|
|
462
|
+
if (word.startsWith(term)) {
|
|
463
|
+
matched.add(term);
|
|
464
|
+
hits += 1;
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
return { hits, matched: matched.size };
|
|
469
|
+
};
|
|
470
|
+
|
|
471
|
+
/**
|
|
472
|
+
* Score a section by the query terms it covers, then by whether its heading
|
|
473
|
+
* names them, then by how densely it hits them. Raw hit counts would hand the
|
|
474
|
+
* excerpt to the longest section — a reference table that says "matter" once
|
|
475
|
+
* per row outscores the short "Closing a matter" section that actually answers
|
|
476
|
+
* "close matter" — and among sections covering the same terms, the one titled
|
|
477
|
+
* with a term is the one about it.
|
|
478
|
+
*/
|
|
479
|
+
const scoreSection = (section: PageSection, terms: string[]): ScoredSection => {
|
|
480
|
+
const body = matchTerms(section.words, terms);
|
|
481
|
+
return {
|
|
482
|
+
...section,
|
|
483
|
+
coverage: body.matched,
|
|
484
|
+
density: body.hits / Math.max(1, section.words.length),
|
|
485
|
+
titled: matchTerms(section.headingWords, terms).matched,
|
|
486
|
+
};
|
|
487
|
+
};
|
|
488
|
+
|
|
489
|
+
const excerptLongSection = (
|
|
490
|
+
section: string,
|
|
491
|
+
query: string,
|
|
492
|
+
max: number
|
|
493
|
+
): string => {
|
|
494
|
+
const [heading = "", ...rest] = section.split("\n");
|
|
495
|
+
const body = rest.join("\n").trim();
|
|
496
|
+
if (!SECTION_HEADING.test(heading) || body === "") {
|
|
497
|
+
return relevantExcerpt(section, query, max);
|
|
498
|
+
}
|
|
499
|
+
// The heading names what the model is reading, so keep it whenever it
|
|
500
|
+
// leaves at least half the budget for the body beneath it.
|
|
501
|
+
const room = max - heading.length - 1;
|
|
502
|
+
if (room < Math.floor(max / 2)) {
|
|
503
|
+
return relevantExcerpt(section, query, max);
|
|
504
|
+
}
|
|
505
|
+
return `${heading}\n${relevantExcerpt(body, query, room)}`;
|
|
506
|
+
};
|
|
507
|
+
|
|
508
|
+
/**
|
|
509
|
+
* Preserve headings and lists by selecting whole sections that cover the query
|
|
510
|
+
* best, then emitting them in document order. An oversized best section falls
|
|
511
|
+
* back to a relevant window under its heading; ellipses mark omitted content.
|
|
512
|
+
*/
|
|
513
|
+
const excerptPage = (page: ParsedPage, query: string, max: number): string => {
|
|
514
|
+
if (page.text.length <= max) {
|
|
515
|
+
return page.text;
|
|
516
|
+
}
|
|
517
|
+
const terms = queryTerms(query);
|
|
518
|
+
if (terms.length === 0 || page.sections.length === 0) {
|
|
519
|
+
return relevantExcerpt(page.text, query, max);
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
const ranked = page.sections
|
|
523
|
+
.map((section) => scoreSection(section, terms))
|
|
524
|
+
.filter((section) => section.coverage > 0)
|
|
525
|
+
.toSorted(
|
|
526
|
+
(a, b) =>
|
|
527
|
+
b.coverage - a.coverage ||
|
|
528
|
+
b.titled - a.titled ||
|
|
529
|
+
b.density - a.density ||
|
|
530
|
+
a.index - b.index
|
|
531
|
+
);
|
|
532
|
+
const [bestSection] = ranked;
|
|
533
|
+
if (!bestSection) {
|
|
534
|
+
return relevantExcerpt(page.text, query, max);
|
|
535
|
+
}
|
|
536
|
+
if (bestSection.text.length > max) {
|
|
537
|
+
// The window lands wherever the terms cluster, which is rarely the first
|
|
538
|
+
// line — so the heading that names what the model is reading would be the
|
|
539
|
+
// first thing cut. Hold it back and window only the body beneath it.
|
|
540
|
+
return excerptLongSection(bestSection.text, query, max);
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
const last = page.sections.length - 1;
|
|
544
|
+
const render = (selected: ScoredSection[]): string => {
|
|
545
|
+
const ordered = selected.toSorted((a, b) => a.index - b.index);
|
|
546
|
+
const parts: string[] = [];
|
|
547
|
+
let previous = -1;
|
|
548
|
+
for (const section of ordered) {
|
|
549
|
+
if (previous !== -1 && section.index !== previous + 1) {
|
|
550
|
+
parts.push("…");
|
|
551
|
+
}
|
|
552
|
+
parts.push(section.text);
|
|
553
|
+
previous = section.index;
|
|
554
|
+
}
|
|
555
|
+
const [first] = ordered;
|
|
556
|
+
if (first && first.index > 0) {
|
|
557
|
+
parts.unshift("…");
|
|
558
|
+
}
|
|
559
|
+
if (previous < last) {
|
|
560
|
+
parts.push("…");
|
|
561
|
+
}
|
|
562
|
+
return parts.join("\n\n");
|
|
563
|
+
};
|
|
564
|
+
|
|
565
|
+
// Like `relevantExcerpt`, the result may run two characters over `max` for
|
|
566
|
+
// the ellipses that mark omitted content.
|
|
567
|
+
const chosen: ScoredSection[] = [];
|
|
568
|
+
for (const section of ranked) {
|
|
569
|
+
const candidate = [...chosen, section];
|
|
570
|
+
if (render(candidate).length <= max + 2) {
|
|
571
|
+
chosen.push(section);
|
|
572
|
+
}
|
|
573
|
+
}
|
|
574
|
+
if (chosen.length === 0) {
|
|
575
|
+
return excerptLongSection(bestSection.text, query, max);
|
|
576
|
+
}
|
|
577
|
+
return render(chosen);
|
|
578
|
+
};
|
|
579
|
+
|
|
580
|
+
/** {@link excerptPage} over a page parsed on the spot. Exported for testing. */
|
|
581
|
+
export const sectionExcerpt = (
|
|
582
|
+
content: string,
|
|
583
|
+
query: string,
|
|
584
|
+
max: number
|
|
585
|
+
): string => excerptPage(parsePage(content), query, max);
|
|
586
|
+
|
|
251
587
|
/**
|
|
252
588
|
* Build the request-time grounding function for the Ask AI endpoint.
|
|
253
589
|
*
|
|
@@ -279,6 +615,17 @@ export const createAskContext = (
|
|
|
279
615
|
return dbPromise;
|
|
280
616
|
};
|
|
281
617
|
const byRoute = new Map(data.documents.map((doc) => [doc.route, doc]));
|
|
618
|
+
// Section splitting and tokenizing are per page, not per question, so each
|
|
619
|
+
// page is parsed on first use and reused for the life of the endpoint.
|
|
620
|
+
const parsed = new Map<string, ParsedPage>();
|
|
621
|
+
const pageOf = (doc: OramaDoc): ParsedPage => {
|
|
622
|
+
let page = parsed.get(doc.route);
|
|
623
|
+
if (page === undefined) {
|
|
624
|
+
page = parsePage(doc.content);
|
|
625
|
+
parsed.set(doc.route, page);
|
|
626
|
+
}
|
|
627
|
+
return page;
|
|
628
|
+
};
|
|
282
629
|
const instruction = options?.instructions
|
|
283
630
|
? `${BASE_INSTRUCTION}\n\n${options.instructions}`
|
|
284
631
|
: BASE_INSTRUCTION;
|
|
@@ -288,19 +635,27 @@ export const createAskContext = (
|
|
|
288
635
|
|
|
289
636
|
return async (messages, page) => {
|
|
290
637
|
const list = Array.isArray(messages) ? messages : [];
|
|
291
|
-
const
|
|
292
|
-
if (
|
|
638
|
+
const turns = userTurns(list);
|
|
639
|
+
if (turns.length === 0) {
|
|
293
640
|
return;
|
|
294
641
|
}
|
|
642
|
+
const queries = retrievalQueries(turns);
|
|
643
|
+
// The leading query is what the reader is asking about now; it also decides
|
|
644
|
+
// which part of each page is quoted.
|
|
645
|
+
const [query = ""] = queries;
|
|
295
646
|
|
|
296
647
|
// The current page anchors retrieval to its locale and is injected first.
|
|
297
648
|
const current = page?.path
|
|
298
649
|
? byRoute.get(normalizeRoute(page.path))
|
|
299
650
|
: undefined;
|
|
300
651
|
const db = await index();
|
|
301
|
-
const
|
|
302
|
-
|
|
303
|
-
|
|
652
|
+
const filters = { locale: current?.locale || undefined };
|
|
653
|
+
const hits = interleave(
|
|
654
|
+
await Promise.all(
|
|
655
|
+
queries.map((text) => queryOramaIndex(db, text, maxResults, filters))
|
|
656
|
+
),
|
|
657
|
+
maxResults
|
|
658
|
+
);
|
|
304
659
|
|
|
305
660
|
const seen = new Set<string>();
|
|
306
661
|
const sections: string[] = [];
|
|
@@ -309,14 +664,15 @@ export const createAskContext = (
|
|
|
309
664
|
if (seen.has(doc.route) || budget <= 0) {
|
|
310
665
|
return;
|
|
311
666
|
}
|
|
667
|
+
const parsedPage = pageOf(doc);
|
|
312
668
|
// Skip a page that would be cut to a junk fragment: its excerpt is only
|
|
313
669
|
// useful when it either fits whole or gets at least the minimum window.
|
|
314
|
-
if (budget < MIN_EXCERPT_CHARS &&
|
|
670
|
+
if (budget < MIN_EXCERPT_CHARS && parsedPage.text.length > budget) {
|
|
315
671
|
return;
|
|
316
672
|
}
|
|
317
673
|
seen.add(doc.route);
|
|
318
|
-
const body =
|
|
319
|
-
|
|
674
|
+
const body = excerptPage(
|
|
675
|
+
parsedPage,
|
|
320
676
|
query,
|
|
321
677
|
Math.min(excerptChars, budget)
|
|
322
678
|
);
|
package/src/astro/generate.ts
CHANGED
|
@@ -12,7 +12,6 @@ import {
|
|
|
12
12
|
import { createRequire } from "node:module";
|
|
13
13
|
import { pathToFileURL } from "node:url";
|
|
14
14
|
|
|
15
|
-
import { imageSize } from "image-size";
|
|
16
15
|
import pMap from "p-map";
|
|
17
16
|
import {
|
|
18
17
|
basename,
|
|
@@ -29,6 +28,7 @@ import { OPENAPI_PATH } from "../ai/api/paths.ts";
|
|
|
29
28
|
import { buildApiSpec } from "../ai/api/spec.ts";
|
|
30
29
|
import { buildAskData } from "../ai/ask-data.ts";
|
|
31
30
|
import { askBackendRuntimeDep, resolveAskBackend } from "../ai/ask.ts";
|
|
31
|
+
import { buildHomeLinkHeader } from "../ai/link-headers.ts";
|
|
32
32
|
import { buildRawMarkdown, markdownRoutePaths } from "../ai/markdown.ts";
|
|
33
33
|
import { buildMcpData } from "../ai/mcp/data.ts";
|
|
34
34
|
import type { McpData } from "../ai/mcp/data.ts";
|
|
@@ -64,6 +64,7 @@ import { packageRoot } from "../core/package-root.ts";
|
|
|
64
64
|
import type { BlumeProject } from "../core/project-graph.ts";
|
|
65
65
|
import type { ResolvedConfig } from "../core/schema.ts";
|
|
66
66
|
import { resolveDocsCollection } from "../core/sources/resolve.ts";
|
|
67
|
+
import { svgDimensions } from "../core/svg-dimensions.ts";
|
|
67
68
|
import { trimChar } from "../core/trim.ts";
|
|
68
69
|
import { resolveTsconfigAliases } from "../core/tsconfig-aliases.ts";
|
|
69
70
|
import type { Diagnostic, Navigation } from "../core/types.ts";
|
|
@@ -104,6 +105,7 @@ import {
|
|
|
104
105
|
exampleMarkdownLookup,
|
|
105
106
|
exampleScanRoots,
|
|
106
107
|
} from "./examples.ts";
|
|
108
|
+
import { publishDevNegotiation } from "./integration.ts";
|
|
107
109
|
import { discoverIslands } from "./islands.ts";
|
|
108
110
|
import {
|
|
109
111
|
customOgRoutes,
|
|
@@ -121,7 +123,6 @@ import {
|
|
|
121
123
|
changelogIndexTemplate,
|
|
122
124
|
contentAssetsEndpointTemplate,
|
|
123
125
|
contentConfigTemplate,
|
|
124
|
-
envTemplate,
|
|
125
126
|
exampleMapTemplate,
|
|
126
127
|
exampleWrapperTemplate,
|
|
127
128
|
examplesPageTemplate,
|
|
@@ -601,19 +602,18 @@ const islandFrameworkWarnings = (
|
|
|
601
602
|
* rather than let the build die with an opaque ERR_MODULE_NOT_FOUND from the
|
|
602
603
|
* hidden generated config. Availability mirrors the search-provider check: a
|
|
603
604
|
* dep resolves from the project root or from the Blume package itself.
|
|
605
|
+
* `pkgDir` is injectable for testing.
|
|
604
606
|
*/
|
|
605
|
-
const deploymentAdapterWarnings = (
|
|
607
|
+
export const deploymentAdapterWarnings = (
|
|
606
608
|
deployment: ResolvedConfig["deployment"],
|
|
607
|
-
root: string
|
|
609
|
+
root: string,
|
|
610
|
+
pkgDir: string = packageRoot()
|
|
608
611
|
): string[] => {
|
|
609
612
|
const dep =
|
|
610
613
|
deployment.output === "server" && deployment.adapter
|
|
611
614
|
? DEPLOYMENT_ADAPTER_DEPS.get(deployment.adapter)
|
|
612
615
|
: undefined;
|
|
613
|
-
if (
|
|
614
|
-
dep &&
|
|
615
|
-
!(canResolveFrom(root, dep) || canResolveFrom(packageRoot(), dep))
|
|
616
|
-
) {
|
|
616
|
+
if (dep && !(canResolveFrom(root, dep) || canResolveFrom(pkgDir, dep))) {
|
|
617
617
|
return [
|
|
618
618
|
`Deployment adapter "${deployment.adapter}" needs "${dep}", which isn't installed. Run \`npm install ${dep}\` (or your package manager's equivalent).`,
|
|
619
619
|
];
|
|
@@ -907,24 +907,13 @@ interface LogoDimensions {
|
|
|
907
907
|
}
|
|
908
908
|
|
|
909
909
|
/**
|
|
910
|
-
* Read dimensions from an SVG's explicit size or its view box
|
|
911
|
-
*
|
|
912
|
-
*
|
|
913
|
-
*
|
|
914
|
-
* `>` inside another attribute). An SVG with no usable size returns partial
|
|
915
|
-
* dimensions or throws; both collapse to undefined.
|
|
910
|
+
* Read dimensions from an SVG's explicit size or its view box, with the same
|
|
911
|
+
* root-tag parser og/card.ts uses for the OG brand mark so the header and the
|
|
912
|
+
* card can't disagree about one logo. An SVG with no usable size collapses to
|
|
913
|
+
* undefined.
|
|
916
914
|
*/
|
|
917
|
-
const
|
|
918
|
-
|
|
919
|
-
return;
|
|
920
|
-
}
|
|
921
|
-
try {
|
|
922
|
-
const { height, width } = imageSize(Buffer.from(svg));
|
|
923
|
-
return height && width ? { height, width } : undefined;
|
|
924
|
-
} catch {
|
|
925
|
-
return undefined;
|
|
926
|
-
}
|
|
927
|
-
};
|
|
915
|
+
const logoDimensions = (svg: string | undefined): LogoDimensions | undefined =>
|
|
916
|
+
svg ? (svgDimensions(svg) ?? undefined) : undefined;
|
|
928
917
|
|
|
929
918
|
/** Read a local SVG logo from the project root or public directory. */
|
|
930
919
|
const readLogoSvg = (
|
|
@@ -989,8 +978,8 @@ const resolveLogo = (project: BlumeProject): BlumeLogo | null => {
|
|
|
989
978
|
return { alt, href: brandHref, svg: lightSvg, text };
|
|
990
979
|
}
|
|
991
980
|
|
|
992
|
-
const lightDimensions =
|
|
993
|
-
const darkDimensions =
|
|
981
|
+
const lightDimensions = logoDimensions(lightSvg);
|
|
982
|
+
const darkDimensions = logoDimensions(darkSvg);
|
|
994
983
|
const dimensions =
|
|
995
984
|
lightDimensions || darkDimensions
|
|
996
985
|
? { dark: darkDimensions, light: lightDimensions }
|
|
@@ -1938,6 +1927,13 @@ export const generateRuntime = async (
|
|
|
1938
1927
|
|
|
1939
1928
|
const depsLinkWarning = await ensureDepsLink(out);
|
|
1940
1929
|
|
|
1930
|
+
// The dev negotiation inputs. Published in memory (below) rather than baked
|
|
1931
|
+
// into the generated config, so a content-route change never rewrites
|
|
1932
|
+
// `astro.config.mjs` — which would restart the dev server in place.
|
|
1933
|
+
const contentRoutes = markdownRoutePaths(project);
|
|
1934
|
+
const homeLinkHeader =
|
|
1935
|
+
buildHomeLinkHeader(config, contentRoutes) ?? undefined;
|
|
1936
|
+
|
|
1941
1937
|
const askEnabled = config.ai.ask?.enabled ?? false;
|
|
1942
1938
|
const exportPdf = config.export.pdf;
|
|
1943
1939
|
const exportEpub = config.export.epub;
|
|
@@ -2060,7 +2056,7 @@ export const generateRuntime = async (
|
|
|
2060
2056
|
askPath,
|
|
2061
2057
|
config,
|
|
2062
2058
|
contentRoot: docsCollection.base,
|
|
2063
|
-
contentRoutes
|
|
2059
|
+
contentRoutes,
|
|
2064
2060
|
context,
|
|
2065
2061
|
examplesPath,
|
|
2066
2062
|
examplesThemePath,
|
|
@@ -2081,7 +2077,6 @@ export const generateRuntime = async (
|
|
|
2081
2077
|
)
|
|
2082
2078
|
),
|
|
2083
2079
|
write(join(out, "tsconfig.json"), runtimeTsconfigTemplate()),
|
|
2084
|
-
write(join(srcDir, "env.d.ts"), envTemplate()),
|
|
2085
2080
|
write(
|
|
2086
2081
|
join(srcDir, "content.config.ts"),
|
|
2087
2082
|
contentConfigTemplate({
|
|
@@ -2392,6 +2387,7 @@ export const generateRuntime = async (
|
|
|
2392
2387
|
// Publish last, once every page that imports a module is on disk: a live
|
|
2393
2388
|
// dev server invalidates the changed modules and reloads the browser against
|
|
2394
2389
|
// the finished tree, never a half-written one.
|
|
2390
|
+
publishDevNegotiation({ contentRoutes, homeLinkHeader });
|
|
2395
2391
|
publishRuntimeModules(modules);
|
|
2396
2392
|
|
|
2397
2393
|
return { structuralChange: structural.some(Boolean), warnings };
|
package/src/astro/include-hmr.ts
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
|
-
import { readFile
|
|
1
|
+
import { readFile } from "node:fs/promises";
|
|
2
|
+
|
|
3
|
+
import { refreshBlumeContent } from "./integration.ts";
|
|
2
4
|
|
|
3
5
|
/**
|
|
4
6
|
* Dev-server invalidation for `<include>` partials. A partial is not an Astro
|
|
@@ -54,25 +56,20 @@ export const includeHmrPlugin = (graphPath: string): IncludeHmrPlugin => ({
|
|
|
54
56
|
return;
|
|
55
57
|
}
|
|
56
58
|
const { moduleGraph, ws } = ctx.server;
|
|
57
|
-
const now = new Date();
|
|
58
59
|
for (const includer of includers) {
|
|
59
60
|
for (const mod of moduleGraph.getModulesByFile(includer) ?? []) {
|
|
60
61
|
// SAFETY: the module came out of this module graph; `never` only
|
|
61
62
|
// reflects that the structural slice doesn't model the node type.
|
|
62
63
|
moduleGraph.invalidateModule(mod as never);
|
|
63
64
|
}
|
|
64
|
-
// Plain `.md` pages have no Vite module: their HTML lives in the
|
|
65
|
-
// content-layer store, rendered at sync time. Bump the page's mtime so
|
|
66
|
-
// Astro's content watcher re-syncs it — the include-aware digest
|
|
67
|
-
// (`withIncludeRefresh`) then forces a fresh render that re-reads the
|
|
68
|
-
// edited partial.
|
|
69
|
-
try {
|
|
70
|
-
// oxlint-disable-next-line no-await-in-loop -- ordered per-page touch
|
|
71
|
-
await utimes(includer, now, now);
|
|
72
|
-
} catch {
|
|
73
|
-
// The page may have been deleted since the graph was written.
|
|
74
|
-
}
|
|
75
65
|
}
|
|
66
|
+
// Plain `.md` pages have no Vite module: their HTML lives in the
|
|
67
|
+
// content-layer store, rendered at sync time. Ask Astro to re-run the
|
|
68
|
+
// loaders — `withIncludeRefresh` then evicts the includers whose partials
|
|
69
|
+
// changed, so the glob loader renders them afresh. `false` only before
|
|
70
|
+
// the server's `astro:server:setup` has run, when there is no store to go
|
|
71
|
+
// stale yet.
|
|
72
|
+
await refreshBlumeContent();
|
|
76
73
|
ws.send({ type: "full-reload" });
|
|
77
74
|
// The partial itself is not a module; suppress Vite's default handling.
|
|
78
75
|
return [];
|