blume 1.6.5 → 1.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +39 -0
- package/bin/blume.mjs +3 -2
- package/dist/cli/chunk-0qhq7b8q.js +111 -0
- package/dist/cli/chunk-0qhq7b8q.js.map +11 -0
- package/dist/cli/chunk-18tjv4f7.js +96 -0
- package/dist/cli/chunk-18tjv4f7.js.map +10 -0
- package/dist/cli/chunk-27gtm2ym.js +69 -0
- package/dist/cli/chunk-27gtm2ym.js.map +11 -0
- package/dist/cli/chunk-2aj8ddew.js +72 -0
- package/dist/cli/chunk-2aj8ddew.js.map +10 -0
- package/dist/cli/chunk-3r94j3tc.js +221 -0
- package/dist/cli/chunk-3r94j3tc.js.map +10 -0
- package/dist/cli/chunk-4trphnvy.js +102 -0
- package/dist/cli/chunk-4trphnvy.js.map +11 -0
- package/dist/cli/chunk-4xyggvgf.js +21 -0
- package/dist/cli/chunk-4xyggvgf.js.map +10 -0
- package/dist/cli/chunk-5d4q7121.js +4064 -0
- package/dist/cli/chunk-5d4q7121.js.map +40 -0
- package/dist/cli/chunk-5hs6gb7n.js +32 -0
- package/dist/cli/chunk-5hs6gb7n.js.map +10 -0
- package/dist/cli/chunk-6kzzpsx8.js +26 -0
- package/dist/cli/chunk-6kzzpsx8.js.map +10 -0
- package/dist/cli/chunk-8gnpdsn1.js +952 -0
- package/dist/cli/chunk-8gnpdsn1.js.map +12 -0
- package/dist/cli/chunk-9qs6acpw.js +176 -0
- package/dist/cli/chunk-9qs6acpw.js.map +10 -0
- package/dist/cli/chunk-agy5rzxy.js +2453 -0
- package/dist/cli/chunk-agy5rzxy.js.map +15 -0
- package/dist/cli/chunk-bcy492zc.js +16 -0
- package/dist/cli/chunk-bcy492zc.js.map +10 -0
- package/dist/cli/chunk-btfr9yvw.js +41 -0
- package/dist/cli/chunk-btfr9yvw.js.map +10 -0
- package/dist/cli/chunk-cbjnx4s8.js +73 -0
- package/dist/cli/chunk-cbjnx4s8.js.map +10 -0
- package/dist/cli/chunk-cfw6x4rm.js +1967 -0
- package/dist/cli/chunk-cfw6x4rm.js.map +34 -0
- package/dist/cli/chunk-ckh3a410.js +277 -0
- package/dist/cli/chunk-ckh3a410.js.map +11 -0
- package/dist/cli/chunk-drke6t0h.js +259 -0
- package/dist/cli/chunk-drke6t0h.js.map +11 -0
- package/dist/cli/chunk-ev67ycx0.js +15 -0
- package/dist/cli/chunk-ev67ycx0.js.map +10 -0
- package/dist/cli/chunk-ey89bjj1.js +209 -0
- package/dist/cli/chunk-ey89bjj1.js.map +11 -0
- package/dist/cli/chunk-j6pxe0dt.js +69 -0
- package/dist/cli/chunk-j6pxe0dt.js.map +11 -0
- package/dist/cli/chunk-jk1zwka1.js +387 -0
- package/dist/cli/chunk-jk1zwka1.js.map +12 -0
- package/dist/cli/chunk-jtb45atp.js +467 -0
- package/dist/cli/chunk-jtb45atp.js.map +14 -0
- package/dist/cli/chunk-jxkxjsc1.js +76 -0
- package/dist/cli/chunk-jxkxjsc1.js.map +10 -0
- package/dist/cli/chunk-kwx90v78.js +81 -0
- package/dist/cli/chunk-kwx90v78.js.map +10 -0
- package/dist/cli/chunk-n0nyat6g.js +30 -0
- package/dist/cli/chunk-n0nyat6g.js.map +10 -0
- package/dist/cli/chunk-pxj10x8y.js +35 -0
- package/dist/cli/chunk-pxj10x8y.js.map +10 -0
- package/dist/cli/chunk-qq9nm3qd.js +1141 -0
- package/dist/cli/chunk-qq9nm3qd.js.map +19 -0
- package/dist/cli/chunk-s102bysw.js +5170 -0
- package/dist/cli/chunk-s102bysw.js.map +47 -0
- package/dist/cli/chunk-s5dsk8bj.js +769 -0
- package/dist/cli/chunk-s5dsk8bj.js.map +13 -0
- package/dist/cli/chunk-s5e5jt53.js +227 -0
- package/dist/cli/chunk-s5e5jt53.js.map +11 -0
- package/dist/cli/chunk-sbdqrjbb.js +81 -0
- package/dist/cli/chunk-sbdqrjbb.js.map +10 -0
- package/dist/cli/chunk-tnskyrej.js +117 -0
- package/dist/cli/chunk-tnskyrej.js.map +10 -0
- package/dist/cli/chunk-v2ymm99c.js +1016 -0
- package/dist/cli/chunk-v2ymm99c.js.map +13 -0
- package/dist/cli/chunk-v5mm027v.js +185 -0
- package/dist/cli/chunk-v5mm027v.js.map +11 -0
- package/dist/cli/chunk-vt8fgygt.js +23 -0
- package/dist/cli/chunk-vt8fgygt.js.map +10 -0
- package/dist/cli/chunk-vxv4x1n8.js +17 -0
- package/dist/cli/chunk-vxv4x1n8.js.map +10 -0
- package/dist/cli/chunk-wd27zjcz.js +60 -0
- package/dist/cli/chunk-wd27zjcz.js.map +10 -0
- package/dist/cli/chunk-x66c5yjn.js +23 -0
- package/dist/cli/chunk-x66c5yjn.js.map +10 -0
- package/dist/cli/chunk-xv91q4nm.js +5314 -0
- package/dist/cli/chunk-xv91q4nm.js.map +58 -0
- package/dist/cli/chunk-y3g15rvv.js +679 -0
- package/dist/cli/chunk-y3g15rvv.js.map +15 -0
- package/dist/cli/chunk-ye9zdkgv.js +136 -0
- package/dist/cli/chunk-ye9zdkgv.js.map +10 -0
- package/dist/cli/chunk-ynacq3ev.js +1062 -0
- package/dist/cli/chunk-ynacq3ev.js.map +25 -0
- package/dist/cli/chunk-zr3ygrq3.js +54 -0
- package/dist/cli/chunk-zr3ygrq3.js.map +10 -0
- package/dist/cli/index.js +55 -27597
- package/dist/cli/index.js.map +5 -243
- package/dist/types/ai/ask-context.d.ts +26 -0
- package/dist/types/components/layout/nav-utils.d.ts +33 -1
- package/dist/types/core/code-fences.d.ts +11 -0
- package/dist/types/core/package-root.d.ts +1 -1
- package/dist/types/core/schema.d.ts +70 -0
- package/dist/types/theme/fonts.d.ts +22 -22
- package/docs/02-deployment.mdx +22 -1
- package/docs/configuration/ask-ai.mdx +1 -1
- package/docs/configuration/customization.mdx +2 -9
- package/docs/content/navigation.mdx +2 -0
- package/docs/content/syntax.mdx +1 -1
- package/docs/discoverability/open-graph.mdx +4 -0
- package/docs/reference/cli.mdx +1 -1
- package/package.json +4 -2
- package/src/ai/api/handlers.ts +4 -7
- package/src/ai/api/paths.ts +8 -0
- package/src/ai/api/spec.ts +2 -1
- package/src/ai/ask-context.ts +378 -22
- package/src/astro/generate.ts +161 -28
- package/src/astro/include-hmr.ts +10 -13
- package/src/astro/include-refresh.ts +0 -0
- package/src/astro/index.ts +6 -1
- package/src/astro/integration.ts +280 -53
- package/src/astro/module-types.ts +83 -0
- package/src/astro/templates.ts +256 -108
- package/src/audit/image-size.ts +10 -8
- package/src/cli/command-meta.ts +77 -0
- package/src/cli/commands/add.ts +2 -4
- package/src/cli/commands/audit.ts +2 -4
- package/src/cli/commands/build.ts +70 -346
- package/src/cli/commands/check.ts +2 -4
- package/src/cli/commands/dev.ts +31 -42
- package/src/cli/commands/doctor.ts +2 -4
- package/src/cli/commands/eject.ts +3 -41
- package/src/cli/commands/eval.ts +2 -5
- package/src/cli/commands/init.ts +2 -4
- package/src/cli/commands/mcp-stdio.ts +2 -5
- package/src/cli/commands/preview.ts +3 -5
- package/src/cli/commands/sync.ts +2 -4
- package/src/cli/commands/translate.ts +2 -5
- package/src/cli/commands/validate.ts +2 -4
- package/src/cli/commands/version.ts +2 -4
- package/src/cli/eject-scripts.ts +0 -45
- package/src/cli/host-args.ts +16 -0
- package/src/cli/index.ts +84 -35
- package/src/cli/lazy-command.ts +47 -0
- package/src/components/Icon.astro +24 -0
- package/src/components/content/GithubInfo.astro +4 -1
- package/src/components/icon-sprite-middleware.ts +41 -0
- package/src/components/icon-sprite.ts +93 -0
- package/src/components/layout/IconSprite.astro +11 -0
- package/src/components/layout/NavTree.astro +156 -188
- package/src/components/layout/NavTreeCache.astro +45 -0
- package/src/components/layout/NavTreeScript.astro +256 -0
- package/src/components/layout/PageActions.astro +11 -5
- package/src/components/layout/PageLayout.astro +21 -3
- package/src/components/layout/ReferenceLayout.astro +21 -4
- package/src/components/layout/RootLayout.astro +44 -6
- package/src/components/layout/nav-cache.ts +49 -0
- package/src/components/layout/nav-utils.ts +69 -1
- package/src/components/layout/page-locale.ts +29 -0
- package/src/core/api-name.ts +18 -0
- package/src/core/code-fences.ts +48 -0
- package/src/core/content-assets.ts +3 -7
- package/src/core/includes.ts +3 -7
- package/src/core/package-root.ts +1 -1
- package/src/core/schema.ts +19 -0
- package/src/core/sources/normalize.ts +2 -37
- package/src/core/sources/obsidian.ts +3 -2
- package/src/core/svg-dimensions.ts +97 -0
- package/src/core/version-cut.ts +2 -2
- package/src/deploy/artifacts.ts +370 -0
- package/src/deploy/cloudflare-negotiation.ts +97 -32
- package/src/deploy/function-bundle.ts +66 -20
- package/src/deploy/sitemap.ts +6 -0
- package/src/deploy/vercel-negotiation.ts +8 -30
- package/src/markdown/language-icon.ts +64 -20
- package/src/markdown/mermaid.ts +11 -0
- package/src/og/cache.ts +236 -0
- package/src/og/card.ts +18 -16
- package/src/og/index.ts +8 -1
- package/src/openapi/render-mdx.ts +9 -5
- package/src/registry/eject.ts +23 -10
- package/src/theme/entry.ts +41 -7
- package/src/theme/fonts.ts +30 -23
package/src/ai/ask-context.ts
CHANGED
|
@@ -1,4 +1,6 @@
|
|
|
1
1
|
import { normalizeRoute } from "../core/base-path.ts";
|
|
2
|
+
import { nextFenceState } from "../core/code-fences.ts";
|
|
3
|
+
import type { FenceState } from "../core/code-fences.ts";
|
|
2
4
|
import { buildOramaIndex, queryOramaIndex } from "../search/orama-index.ts";
|
|
3
5
|
import type { OramaDoc } from "../search/orama-index.ts";
|
|
4
6
|
|
|
@@ -68,45 +70,94 @@ export interface AskRetrievalOptions {
|
|
|
68
70
|
const EXCERPT_LEAD = 160;
|
|
69
71
|
|
|
70
72
|
/**
|
|
71
|
-
*
|
|
72
|
-
*
|
|
73
|
-
* window toward incidental matches instead of the meaningful terms.
|
|
73
|
+
* Remove English filler from retrieval and excerpt queries. Keep content verbs
|
|
74
|
+
* such as "sign", "file" and "close", which can name documentation topics.
|
|
74
75
|
*/
|
|
75
76
|
const STOPWORDS = new Set([
|
|
77
|
+
"a",
|
|
76
78
|
"about",
|
|
79
|
+
"again",
|
|
80
|
+
"against",
|
|
81
|
+
"all",
|
|
82
|
+
"am",
|
|
83
|
+
"an",
|
|
77
84
|
"and",
|
|
85
|
+
"any",
|
|
78
86
|
"are",
|
|
79
87
|
"as",
|
|
80
88
|
"at",
|
|
81
89
|
"be",
|
|
90
|
+
"been",
|
|
91
|
+
"before",
|
|
92
|
+
"being",
|
|
93
|
+
"both",
|
|
82
94
|
"but",
|
|
83
95
|
"by",
|
|
84
96
|
"can",
|
|
97
|
+
"did",
|
|
85
98
|
"do",
|
|
86
99
|
"does",
|
|
100
|
+
"done",
|
|
101
|
+
"each",
|
|
102
|
+
"every",
|
|
87
103
|
"for",
|
|
88
104
|
"from",
|
|
105
|
+
"had",
|
|
106
|
+
"has",
|
|
107
|
+
"have",
|
|
108
|
+
"having",
|
|
109
|
+
"he",
|
|
110
|
+
"her",
|
|
111
|
+
"hers",
|
|
112
|
+
"him",
|
|
113
|
+
"his",
|
|
89
114
|
"how",
|
|
115
|
+
"i",
|
|
116
|
+
"if",
|
|
90
117
|
"in",
|
|
91
118
|
"into",
|
|
92
119
|
"is",
|
|
93
120
|
"it",
|
|
94
121
|
"its",
|
|
122
|
+
"like",
|
|
123
|
+
"look",
|
|
124
|
+
"may",
|
|
125
|
+
"me",
|
|
126
|
+
"might",
|
|
127
|
+
"mine",
|
|
128
|
+
"must",
|
|
95
129
|
"my",
|
|
130
|
+
"nor",
|
|
96
131
|
"of",
|
|
97
132
|
"on",
|
|
98
133
|
"or",
|
|
134
|
+
"other",
|
|
99
135
|
"our",
|
|
136
|
+
"ours",
|
|
137
|
+
"shall",
|
|
138
|
+
"she",
|
|
139
|
+
"should",
|
|
140
|
+
"so",
|
|
141
|
+
"some",
|
|
142
|
+
"than",
|
|
100
143
|
"that",
|
|
101
144
|
"the",
|
|
145
|
+
"their",
|
|
146
|
+
"theirs",
|
|
147
|
+
"them",
|
|
148
|
+
"then",
|
|
149
|
+
"there",
|
|
102
150
|
"these",
|
|
151
|
+
"they",
|
|
103
152
|
"this",
|
|
104
153
|
"those",
|
|
105
154
|
"to",
|
|
155
|
+
"us",
|
|
106
156
|
"use",
|
|
107
157
|
"used",
|
|
108
158
|
"using",
|
|
109
159
|
"was",
|
|
160
|
+
"we",
|
|
110
161
|
"were",
|
|
111
162
|
"what",
|
|
112
163
|
"when",
|
|
@@ -114,9 +165,12 @@ const STOPWORDS = new Set([
|
|
|
114
165
|
"which",
|
|
115
166
|
"who",
|
|
116
167
|
"why",
|
|
168
|
+
"will",
|
|
117
169
|
"with",
|
|
170
|
+
"would",
|
|
118
171
|
"you",
|
|
119
172
|
"your",
|
|
173
|
+
"yours",
|
|
120
174
|
]);
|
|
121
175
|
|
|
122
176
|
/** A run of letters, combining marks and digits inside a word-like segment. */
|
|
@@ -149,13 +203,15 @@ const segmentQuery = (query: string): string[] => {
|
|
|
149
203
|
return pieces;
|
|
150
204
|
};
|
|
151
205
|
|
|
206
|
+
/** Lowercase word tokens of `text`, cut at the same boundaries as a query. */
|
|
207
|
+
const tokenize = (text: string): string[] =>
|
|
208
|
+
segmentQuery(text).flatMap((piece) => piece.match(TERM) ?? []);
|
|
209
|
+
|
|
152
210
|
/** Distinct, meaningful lowercase terms from a query (drops stopwords). */
|
|
153
|
-
const queryTerms = (query: string): string[] =>
|
|
154
|
-
|
|
155
|
-
return [...new Set(terms)].filter(
|
|
211
|
+
const queryTerms = (query: string): string[] =>
|
|
212
|
+
[...new Set(tokenize(query))].filter(
|
|
156
213
|
(term) => term.length >= 2 && !STOPWORDS.has(term)
|
|
157
214
|
);
|
|
158
|
-
};
|
|
159
215
|
|
|
160
216
|
/**
|
|
161
217
|
* The grounding preamble. The model is told to answer strictly from the injected
|
|
@@ -165,15 +221,87 @@ const queryTerms = (query: string): string[] => {
|
|
|
165
221
|
const BASE_INSTRUCTION =
|
|
166
222
|
"You are a helpful documentation assistant for this project. Answer the user's question using ONLY the documentation excerpts below. Each excerpt is headed by its page as `## Page Title (/route)`. If the answer is not covered by the excerpts, say you don't know and suggest where in the docs to look — do not invent details. Always cite the pages you drew from, and write every citation as a Markdown link to that page using its route, e.g. [Page Title](/route).";
|
|
167
223
|
|
|
168
|
-
/** The
|
|
169
|
-
const
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
224
|
+
/** The non-empty user turns, oldest first. Assistant turns never seed retrieval. */
|
|
225
|
+
const userTurns = (messages: AskMessage[]): string[] =>
|
|
226
|
+
messages
|
|
227
|
+
.filter((message) => message?.role === "user" && message.content?.trim())
|
|
228
|
+
.map((message) => message.content.trim());
|
|
229
|
+
|
|
230
|
+
/** Meaningful terms below which a follow-up cannot stand as a query on its own. */
|
|
231
|
+
const MIN_QUERY_TERMS = 3;
|
|
232
|
+
|
|
233
|
+
const FOLLOW_UP_OPENERS = new Set(["also", "and", "but", "then"]);
|
|
234
|
+
const ANAPHORA = new Set([
|
|
235
|
+
"it",
|
|
236
|
+
"its",
|
|
237
|
+
"that",
|
|
238
|
+
"them",
|
|
239
|
+
"these",
|
|
240
|
+
"they",
|
|
241
|
+
"this",
|
|
242
|
+
"those",
|
|
243
|
+
]);
|
|
244
|
+
|
|
245
|
+
const isFollowUp = (message: string): boolean => {
|
|
246
|
+
const words = segmentQuery(message);
|
|
247
|
+
const [first] = words;
|
|
248
|
+
return (
|
|
249
|
+
(first !== undefined && FOLLOW_UP_OPENERS.has(first)) ||
|
|
250
|
+
words.some((word) => ANAPHORA.has(word))
|
|
251
|
+
);
|
|
252
|
+
};
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* The texts that retrieve for the latest question, in rank order.
|
|
256
|
+
*
|
|
257
|
+
* The question itself always leads, verbatim: Orama's own tokenizer and BM25
|
|
258
|
+
* weighting see the whole sentence (version numbers, single-character CJK
|
|
259
|
+
* words, `--flags`, and the bigrams a ja/zh index depends on all survive), and
|
|
260
|
+
* a short question that names its subject ("Does it support i18n?") is never
|
|
261
|
+
* outvoted by whatever the reader asked before. Only when it reads like a
|
|
262
|
+
* follow-up — opener-led, pronoun-bearing, or nothing but filler ("Why?") —
|
|
263
|
+
* and is too short to stand alone does the nearest earlier user turn with
|
|
264
|
+
* content terms join as a second query, ranked behind the first so the earlier
|
|
265
|
+
* subject stays in view without displacing the current one. Assistant turns
|
|
266
|
+
* are excluded, so an incorrect answer cannot reinforce its own retrieval.
|
|
267
|
+
*/
|
|
268
|
+
const retrievalQueries = (turns: string[]): string[] => {
|
|
269
|
+
const [latest = "", ...earlier] = turns.toReversed();
|
|
270
|
+
const terms = queryTerms(latest);
|
|
271
|
+
if (terms.length >= MIN_QUERY_TERMS) {
|
|
272
|
+
return [latest];
|
|
273
|
+
}
|
|
274
|
+
if (terms.length > 0 && !isFollowUp(latest)) {
|
|
275
|
+
return [latest];
|
|
276
|
+
}
|
|
277
|
+
const context = earlier.find((turn) => queryTerms(turn).length > 0);
|
|
278
|
+
if (context === undefined) {
|
|
279
|
+
return [latest];
|
|
280
|
+
}
|
|
281
|
+
return terms.length === 0 ? [context] : [latest, context];
|
|
282
|
+
};
|
|
283
|
+
|
|
284
|
+
/**
|
|
285
|
+
* Merge ranked result lists round-robin — the first list's top hit, then the
|
|
286
|
+
* second's, and so on — dropping duplicate routes and stopping at `limit`.
|
|
287
|
+
*/
|
|
288
|
+
const interleave = (lists: OramaDoc[][], limit: number): OramaDoc[] => {
|
|
289
|
+
const merged: OramaDoc[] = [];
|
|
290
|
+
const seen = new Set<string>();
|
|
291
|
+
const depth = Math.max(...lists.map((list) => list.length));
|
|
292
|
+
for (let rank = 0; rank < depth; rank += 1) {
|
|
293
|
+
for (const list of lists) {
|
|
294
|
+
const doc = list[rank];
|
|
295
|
+
if (doc && !seen.has(doc.route)) {
|
|
296
|
+
seen.add(doc.route);
|
|
297
|
+
merged.push(doc);
|
|
298
|
+
}
|
|
299
|
+
if (merged.length >= limit) {
|
|
300
|
+
return merged;
|
|
301
|
+
}
|
|
174
302
|
}
|
|
175
303
|
}
|
|
176
|
-
return
|
|
304
|
+
return merged;
|
|
177
305
|
};
|
|
178
306
|
|
|
179
307
|
/**
|
|
@@ -248,6 +376,214 @@ export const relevantExcerpt = (
|
|
|
248
376
|
return withEllipsis(Math.max(0, best - lead));
|
|
249
377
|
};
|
|
250
378
|
|
|
379
|
+
/** A Markdown heading at level 2 or deeper — where a page divides itself. */
|
|
380
|
+
const SECTION_HEADING = /^ {0,3}#{2,6}[\t ]+.+$/u;
|
|
381
|
+
/** Any ATX heading, including the `#` title a lead-in may open with. */
|
|
382
|
+
const ANY_HEADING = /^ {0,3}#{1,6}[\t ]+.+$/u;
|
|
383
|
+
|
|
384
|
+
/** Offsets of the section headings in `text`, skipping fenced code. */
|
|
385
|
+
const sectionStarts = (text: string): number[] => {
|
|
386
|
+
const starts: number[] = [];
|
|
387
|
+
let fence: FenceState = null;
|
|
388
|
+
let offset = 0;
|
|
389
|
+
for (const line of text.split("\n")) {
|
|
390
|
+
const next = nextFenceState(line, fence);
|
|
391
|
+
if (fence === null && next === null && SECTION_HEADING.test(line)) {
|
|
392
|
+
starts.push(offset);
|
|
393
|
+
}
|
|
394
|
+
fence = next;
|
|
395
|
+
offset += line.length + 1;
|
|
396
|
+
}
|
|
397
|
+
return starts;
|
|
398
|
+
};
|
|
399
|
+
|
|
400
|
+
interface PageSection {
|
|
401
|
+
/** Word tokens of the section's heading line, or none when it has no heading. */
|
|
402
|
+
headingWords: string[];
|
|
403
|
+
/** Position in the page, for source-order output and omission markers. */
|
|
404
|
+
index: number;
|
|
405
|
+
text: string;
|
|
406
|
+
/** The section's word tokens, cut once so scoring is a prefix test. */
|
|
407
|
+
words: string[];
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
/** A page split into sections once, so per-request scoring never re-tokenizes. */
|
|
411
|
+
export interface ParsedPage {
|
|
412
|
+
sections: PageSection[];
|
|
413
|
+
/** NFC-normalized, LF-only, trimmed page text; excerpts slice from it. */
|
|
414
|
+
text: string;
|
|
415
|
+
}
|
|
416
|
+
|
|
417
|
+
/**
|
|
418
|
+
* Split a page at its `##`+ headings (outside code fences). The text above the
|
|
419
|
+
* first heading is the page's own lead-in and is a section like any other.
|
|
420
|
+
* Line endings are folded to LF first so a CRLF checkout splits and matches the
|
|
421
|
+
* same way as an LF one. Exported for testing; {@link createAskContext}
|
|
422
|
+
* parses each page once and caches it across requests.
|
|
423
|
+
*/
|
|
424
|
+
export const parsePage = (content: string): ParsedPage => {
|
|
425
|
+
const text = content.normalize("NFC").replaceAll("\r\n", "\n").trim();
|
|
426
|
+
const headings = sectionStarts(text);
|
|
427
|
+
if (headings.length === 0) {
|
|
428
|
+
return { sections: [], text };
|
|
429
|
+
}
|
|
430
|
+
const starts = headings[0] === 0 ? headings : [0, ...headings];
|
|
431
|
+
const sections = starts.map((start, index) => {
|
|
432
|
+
const section = text.slice(start, starts[index + 1]).trim();
|
|
433
|
+
const [firstLine = ""] = section.split("\n", 1);
|
|
434
|
+
const headingWords = ANY_HEADING.test(firstLine) ? tokenize(firstLine) : [];
|
|
435
|
+
return { headingWords, index, text: section, words: tokenize(section) };
|
|
436
|
+
});
|
|
437
|
+
return { sections, text };
|
|
438
|
+
};
|
|
439
|
+
|
|
440
|
+
interface ScoredSection extends PageSection {
|
|
441
|
+
/** How many distinct query terms the section mentions. */
|
|
442
|
+
coverage: number;
|
|
443
|
+
/** Term hits per word, so a long section can't win on bulk alone. */
|
|
444
|
+
density: number;
|
|
445
|
+
/** How many distinct query terms the section's heading names. */
|
|
446
|
+
titled: number;
|
|
447
|
+
}
|
|
448
|
+
|
|
449
|
+
interface TermMatch {
|
|
450
|
+
/** Every word that starts with a term counts once. */
|
|
451
|
+
hits: number;
|
|
452
|
+
/** Distinct terms some word starts with. */
|
|
453
|
+
matched: number;
|
|
454
|
+
}
|
|
455
|
+
|
|
456
|
+
/** How `words` match `terms` by prefix. */
|
|
457
|
+
const matchTerms = (words: string[], terms: string[]): TermMatch => {
|
|
458
|
+
const matched = new Set<string>();
|
|
459
|
+
let hits = 0;
|
|
460
|
+
for (const word of words) {
|
|
461
|
+
for (const term of terms) {
|
|
462
|
+
if (word.startsWith(term)) {
|
|
463
|
+
matched.add(term);
|
|
464
|
+
hits += 1;
|
|
465
|
+
}
|
|
466
|
+
}
|
|
467
|
+
}
|
|
468
|
+
return { hits, matched: matched.size };
|
|
469
|
+
};
|
|
470
|
+
|
|
471
|
+
/**
|
|
472
|
+
* Score a section by the query terms it covers, then by whether its heading
|
|
473
|
+
* names them, then by how densely it hits them. Raw hit counts would hand the
|
|
474
|
+
* excerpt to the longest section — a reference table that says "matter" once
|
|
475
|
+
* per row outscores the short "Closing a matter" section that actually answers
|
|
476
|
+
* "close matter" — and among sections covering the same terms, the one titled
|
|
477
|
+
* with a term is the one about it.
|
|
478
|
+
*/
|
|
479
|
+
const scoreSection = (section: PageSection, terms: string[]): ScoredSection => {
|
|
480
|
+
const body = matchTerms(section.words, terms);
|
|
481
|
+
return {
|
|
482
|
+
...section,
|
|
483
|
+
coverage: body.matched,
|
|
484
|
+
density: body.hits / Math.max(1, section.words.length),
|
|
485
|
+
titled: matchTerms(section.headingWords, terms).matched,
|
|
486
|
+
};
|
|
487
|
+
};
|
|
488
|
+
|
|
489
|
+
const excerptLongSection = (
|
|
490
|
+
section: string,
|
|
491
|
+
query: string,
|
|
492
|
+
max: number
|
|
493
|
+
): string => {
|
|
494
|
+
const [heading = "", ...rest] = section.split("\n");
|
|
495
|
+
const body = rest.join("\n").trim();
|
|
496
|
+
if (!SECTION_HEADING.test(heading) || body === "") {
|
|
497
|
+
return relevantExcerpt(section, query, max);
|
|
498
|
+
}
|
|
499
|
+
// The heading names what the model is reading, so keep it whenever it
|
|
500
|
+
// leaves at least half the budget for the body beneath it.
|
|
501
|
+
const room = max - heading.length - 1;
|
|
502
|
+
if (room < Math.floor(max / 2)) {
|
|
503
|
+
return relevantExcerpt(section, query, max);
|
|
504
|
+
}
|
|
505
|
+
return `${heading}\n${relevantExcerpt(body, query, room)}`;
|
|
506
|
+
};
|
|
507
|
+
|
|
508
|
+
/**
|
|
509
|
+
* Preserve headings and lists by selecting whole sections that cover the query
|
|
510
|
+
* best, then emitting them in document order. An oversized best section falls
|
|
511
|
+
* back to a relevant window under its heading; ellipses mark omitted content.
|
|
512
|
+
*/
|
|
513
|
+
const excerptPage = (page: ParsedPage, query: string, max: number): string => {
|
|
514
|
+
if (page.text.length <= max) {
|
|
515
|
+
return page.text;
|
|
516
|
+
}
|
|
517
|
+
const terms = queryTerms(query);
|
|
518
|
+
if (terms.length === 0 || page.sections.length === 0) {
|
|
519
|
+
return relevantExcerpt(page.text, query, max);
|
|
520
|
+
}
|
|
521
|
+
|
|
522
|
+
const ranked = page.sections
|
|
523
|
+
.map((section) => scoreSection(section, terms))
|
|
524
|
+
.filter((section) => section.coverage > 0)
|
|
525
|
+
.toSorted(
|
|
526
|
+
(a, b) =>
|
|
527
|
+
b.coverage - a.coverage ||
|
|
528
|
+
b.titled - a.titled ||
|
|
529
|
+
b.density - a.density ||
|
|
530
|
+
a.index - b.index
|
|
531
|
+
);
|
|
532
|
+
const [bestSection] = ranked;
|
|
533
|
+
if (!bestSection) {
|
|
534
|
+
return relevantExcerpt(page.text, query, max);
|
|
535
|
+
}
|
|
536
|
+
if (bestSection.text.length > max) {
|
|
537
|
+
// The window lands wherever the terms cluster, which is rarely the first
|
|
538
|
+
// line — so the heading that names what the model is reading would be the
|
|
539
|
+
// first thing cut. Hold it back and window only the body beneath it.
|
|
540
|
+
return excerptLongSection(bestSection.text, query, max);
|
|
541
|
+
}
|
|
542
|
+
|
|
543
|
+
const last = page.sections.length - 1;
|
|
544
|
+
const render = (selected: ScoredSection[]): string => {
|
|
545
|
+
const ordered = selected.toSorted((a, b) => a.index - b.index);
|
|
546
|
+
const parts: string[] = [];
|
|
547
|
+
let previous = -1;
|
|
548
|
+
for (const section of ordered) {
|
|
549
|
+
if (previous !== -1 && section.index !== previous + 1) {
|
|
550
|
+
parts.push("…");
|
|
551
|
+
}
|
|
552
|
+
parts.push(section.text);
|
|
553
|
+
previous = section.index;
|
|
554
|
+
}
|
|
555
|
+
const [first] = ordered;
|
|
556
|
+
if (first && first.index > 0) {
|
|
557
|
+
parts.unshift("…");
|
|
558
|
+
}
|
|
559
|
+
if (previous < last) {
|
|
560
|
+
parts.push("…");
|
|
561
|
+
}
|
|
562
|
+
return parts.join("\n\n");
|
|
563
|
+
};
|
|
564
|
+
|
|
565
|
+
// Like `relevantExcerpt`, the result may run two characters over `max` for
|
|
566
|
+
// the ellipses that mark omitted content.
|
|
567
|
+
const chosen: ScoredSection[] = [];
|
|
568
|
+
for (const section of ranked) {
|
|
569
|
+
const candidate = [...chosen, section];
|
|
570
|
+
if (render(candidate).length <= max + 2) {
|
|
571
|
+
chosen.push(section);
|
|
572
|
+
}
|
|
573
|
+
}
|
|
574
|
+
if (chosen.length === 0) {
|
|
575
|
+
return excerptLongSection(bestSection.text, query, max);
|
|
576
|
+
}
|
|
577
|
+
return render(chosen);
|
|
578
|
+
};
|
|
579
|
+
|
|
580
|
+
/** {@link excerptPage} over a page parsed on the spot. Exported for testing. */
|
|
581
|
+
export const sectionExcerpt = (
|
|
582
|
+
content: string,
|
|
583
|
+
query: string,
|
|
584
|
+
max: number
|
|
585
|
+
): string => excerptPage(parsePage(content), query, max);
|
|
586
|
+
|
|
251
587
|
/**
|
|
252
588
|
* Build the request-time grounding function for the Ask AI endpoint.
|
|
253
589
|
*
|
|
@@ -279,6 +615,17 @@ export const createAskContext = (
|
|
|
279
615
|
return dbPromise;
|
|
280
616
|
};
|
|
281
617
|
const byRoute = new Map(data.documents.map((doc) => [doc.route, doc]));
|
|
618
|
+
// Section splitting and tokenizing are per page, not per question, so each
|
|
619
|
+
// page is parsed on first use and reused for the life of the endpoint.
|
|
620
|
+
const parsed = new Map<string, ParsedPage>();
|
|
621
|
+
const pageOf = (doc: OramaDoc): ParsedPage => {
|
|
622
|
+
let page = parsed.get(doc.route);
|
|
623
|
+
if (page === undefined) {
|
|
624
|
+
page = parsePage(doc.content);
|
|
625
|
+
parsed.set(doc.route, page);
|
|
626
|
+
}
|
|
627
|
+
return page;
|
|
628
|
+
};
|
|
282
629
|
const instruction = options?.instructions
|
|
283
630
|
? `${BASE_INSTRUCTION}\n\n${options.instructions}`
|
|
284
631
|
: BASE_INSTRUCTION;
|
|
@@ -288,19 +635,27 @@ export const createAskContext = (
|
|
|
288
635
|
|
|
289
636
|
return async (messages, page) => {
|
|
290
637
|
const list = Array.isArray(messages) ? messages : [];
|
|
291
|
-
const
|
|
292
|
-
if (
|
|
638
|
+
const turns = userTurns(list);
|
|
639
|
+
if (turns.length === 0) {
|
|
293
640
|
return;
|
|
294
641
|
}
|
|
642
|
+
const queries = retrievalQueries(turns);
|
|
643
|
+
// The leading query is what the reader is asking about now; it also decides
|
|
644
|
+
// which part of each page is quoted.
|
|
645
|
+
const [query = ""] = queries;
|
|
295
646
|
|
|
296
647
|
// The current page anchors retrieval to its locale and is injected first.
|
|
297
648
|
const current = page?.path
|
|
298
649
|
? byRoute.get(normalizeRoute(page.path))
|
|
299
650
|
: undefined;
|
|
300
651
|
const db = await index();
|
|
301
|
-
const
|
|
302
|
-
|
|
303
|
-
|
|
652
|
+
const filters = { locale: current?.locale || undefined };
|
|
653
|
+
const hits = interleave(
|
|
654
|
+
await Promise.all(
|
|
655
|
+
queries.map((text) => queryOramaIndex(db, text, maxResults, filters))
|
|
656
|
+
),
|
|
657
|
+
maxResults
|
|
658
|
+
);
|
|
304
659
|
|
|
305
660
|
const seen = new Set<string>();
|
|
306
661
|
const sections: string[] = [];
|
|
@@ -309,14 +664,15 @@ export const createAskContext = (
|
|
|
309
664
|
if (seen.has(doc.route) || budget <= 0) {
|
|
310
665
|
return;
|
|
311
666
|
}
|
|
667
|
+
const parsedPage = pageOf(doc);
|
|
312
668
|
// Skip a page that would be cut to a junk fragment: its excerpt is only
|
|
313
669
|
// useful when it either fits whole or gets at least the minimum window.
|
|
314
|
-
if (budget < MIN_EXCERPT_CHARS &&
|
|
670
|
+
if (budget < MIN_EXCERPT_CHARS && parsedPage.text.length > budget) {
|
|
315
671
|
return;
|
|
316
672
|
}
|
|
317
673
|
seen.add(doc.route);
|
|
318
|
-
const body =
|
|
319
|
-
|
|
674
|
+
const body = excerptPage(
|
|
675
|
+
parsedPage,
|
|
320
676
|
query,
|
|
321
677
|
Math.min(excerptChars, budget)
|
|
322
678
|
);
|