eddie-jekyll 0.4.0 → 0.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,846 @@
1
+ // SPDX-License-Identifier: GPL-3.0-only
2
+ // Generated by widget/build.sh from widget/src/eddie-sw.js and widget/src/lib/*.js; edit those instead.
3
+ "use strict";
4
+ const EDDIE_ASSET_VERSION = "c0fdc503e01a";
5
+ import * as webllm from "https://cdn.jsdelivr.net/npm/@mlc-ai/web-llm@0.2.84/+esm";
6
+ const EDDIE_SW_TIER = "agent";
7
+ const initLiteWasm = null, liteWasmApi = null, EDDIE_LITE_WASM = null,
8
+ initDenseWasm = null, denseWasmApi = null, EDDIE_DENSE_WASM = null, transformers = null;
9
+ const EddieLib = {};
10
+ // SPDX-License-Identifier: GPL-3.0-only
11
+
12
+ // Agent helpers the widget shows in its UI: model selection, evidence
13
+ // assembly, stream display and the FAQ gate. Pure functions; no WebLLM, no
14
+ // DOM. The prompts and answer post-processing that run beside the model
15
+ // live in agent-llm.js (bundled only into the agent engine's hosts).
16
+
17
+ (function (factory) {
18
+ const api = factory();
19
+ if (typeof module === "object" && module && module.exports) {
20
+ module.exports = api;
21
+ } else if (typeof EddieLib === "object" && EddieLib) {
22
+ Object.assign(EddieLib, api);
23
+ } else {
24
+ globalThis.EddieLib = Object.assign(globalThis.EddieLib || {}, api);
25
+ }
26
+ })(function () {
27
+ "use strict";
28
+
29
+ const NOHIT = "The site doesn't cover that.";
30
+
31
+ const AGENT_MODEL_SIZES = {
32
+ "Qwen3.5-0.8B": 0.4e9,
33
+ "Qwen3.5-2B": 1.2e9,
34
+ "Qwen3.5-4B": 2.3e9,
35
+ };
36
+ const TWO_GIB = 2 * 1024 * 1024 * 1024;
37
+
38
+ /** Strip the WebLLM variant suffix to get the family name shown to visitors. */
39
+ function baseModelId(id) {
40
+ return String(id).replace(/-q\d+f(16|32)_\d+-MLC$/i, "");
41
+ }
42
+
43
+ function agentModelBytes(id) {
44
+ const base = baseModelId(id);
45
+ return Object.prototype.hasOwnProperty.call(AGENT_MODEL_SIZES, base) ? AGENT_MODEL_SIZES[base] : null;
46
+ }
47
+
48
+ /**
49
+ * Choose the WebLLM model id.
50
+ * opts: { mode: "auto"|"light"|"quality"|<id>, maxBufferSize, isMobile, hasF16 }
51
+ *
52
+ * "light" and "quality" are the two sizes the settings panel offers by
53
+ * name; "auto" picks between them from the adapter's buffer limit.
54
+ */
55
+ function selectAgentModel(opts) {
56
+ const o = opts || {};
57
+ const mode = (o.mode || "auto").trim();
58
+ const suffix = o.hasF16 ? "-q4f16_1-MLC" : "-q4f32_1-MLC";
59
+ let base;
60
+ if (mode === "auto") {
61
+ const big = Number(o.maxBufferSize) >= TWO_GIB && !o.isMobile;
62
+ base = big ? "Qwen3.5-2B" : "Qwen3.5-0.8B";
63
+ } else if (mode === "quality") {
64
+ base = "Qwen3.5-2B";
65
+ } else if (mode === "light") {
66
+ base = "Qwen3.5-0.8B";
67
+ } else {
68
+ return { id: mode, base: baseModelId(mode), sizeBytes: agentModelBytes(mode), explicit: true };
69
+ }
70
+ const id = base + suffix;
71
+ return { id, base, sizeBytes: AGENT_MODEL_SIZES[base], explicit: false };
72
+ }
73
+
74
+ function isMobileDevice(nav) {
75
+ const n = nav || {};
76
+ if (n.userAgentData && typeof n.userAgentData.mobile === "boolean") {
77
+ return n.userAgentData.mobile;
78
+ }
79
+ return /Mobi|Android|iPhone|iPad|iPod|Windows Phone/i.test(n.userAgent || "");
80
+ }
81
+
82
+ /**
83
+ * Display text for a partial stream: complete think blocks removed, and
84
+ * anything after an unclosed <think> hidden until it closes.
85
+ */
86
+ function visibleStreamText(partial) {
87
+ if (!partial) return "";
88
+ let out = String(partial).replace(/<think>[\s\S]*?<\/think>/g, "");
89
+ const open = out.indexOf("<think>");
90
+ if (open >= 0) out = out.substring(0, open);
91
+ return out.replace(/^\s+/, "");
92
+ }
93
+
94
+ function urlKey(url) {
95
+ return String(url || "").replace(/#.*$/, "").replace(/\/+$/, "").toLowerCase();
96
+ }
97
+
98
+ /**
99
+ * Round-robin merge of several result lists, deduplicated by URL, at most
100
+ * `max` items. Each result keeps its own fields.
101
+ */
102
+ function mergeEvidence(lists, max) {
103
+ const limit = max == null ? 6 : max;
104
+ const seen = new Set();
105
+ const out = [];
106
+ const arrays = (lists || []).map((l) => (Array.isArray(l) ? l : []));
107
+ const longest = arrays.reduce((n, l) => Math.max(n, l.length), 0);
108
+ for (let i = 0; i < longest && out.length < limit; i++) {
109
+ for (const list of arrays) {
110
+ if (out.length >= limit) break;
111
+ const r = list[i];
112
+ if (!r || !r.url) continue;
113
+ const key = urlKey(r.url);
114
+ if (seen.has(key)) continue;
115
+ seen.add(key);
116
+ out.push(r);
117
+ }
118
+ }
119
+ return out;
120
+ }
121
+
122
+ // FAQ card gate: prefer the WASM's fused `confident` flag (qa_lookup v0.4.1+);
123
+ // older indexes only carry a dense score, so fall back to a plain cutoff.
124
+ function faqPasses(hit, qaMode) {
125
+ if (!hit || typeof hit !== "object") return false;
126
+ if (qaMode === "off") return false;
127
+ if (qaMode === "always") return true;
128
+ if (typeof hit.confident === "boolean") return hit.confident;
129
+ return typeof hit.score === "number" && hit.score >= 0.5;
130
+ }
131
+
132
+ // Turn confident QA hits into agent evidence items ("Q: … A: …") so the
133
+ // answer model sees the FAQ lane, not only chunk text.
134
+ function qaEvidence(hits, max) {
135
+ const limit = max == null ? 2 : max;
136
+ const out = [];
137
+ for (const h of Array.isArray(hits) ? hits : []) {
138
+ if (out.length >= limit) break;
139
+ if (!faqPasses(h, "auto")) continue;
140
+ const q = String(h.question || "").trim();
141
+ const a = String(h.answer || "").trim();
142
+ if (!q || !a) continue;
143
+ out.push({ title: "FAQ: " + q, url: h.source_url || "", text: "Q: " + q + "\nA: " + a, faq: true });
144
+ }
145
+ return out;
146
+ }
147
+
148
+ return {
149
+ faqPasses,
150
+ qaEvidence,
151
+ NOHIT,
152
+ AGENT_MODEL_SIZES,
153
+ baseModelId,
154
+ agentModelBytes,
155
+ selectAgentModel,
156
+ isMobileDevice,
157
+ visibleStreamText,
158
+ mergeEvidence,
159
+ };
160
+ });
161
+
162
+ // SPDX-License-Identifier: GPL-3.0-only
163
+
164
+ // Agent LLM helpers used only where the model runs (agent-engine.js hosts:
165
+ // the agent service worker and the page-side agent worker): prompts, plan
166
+ // parsing and answer post-processing. Split out of agent.js so the widget
167
+ // bundle carries only the model-selection and evidence helpers it shows in
168
+ // the UI. Bundles that include this file always include agent.js first.
169
+
170
+ (function (factory) {
171
+ const api = factory();
172
+ if (typeof module === "object" && module && module.exports) {
173
+ module.exports = api;
174
+ } else if (typeof EddieLib === "object" && EddieLib) {
175
+ Object.assign(EddieLib, api);
176
+ } else {
177
+ globalThis.EddieLib = Object.assign(globalThis.EddieLib || {}, api);
178
+ }
179
+ })(function () {
180
+ "use strict";
181
+
182
+ // The fallback sentence lives in agent.js (the widget shows it too); the
183
+ // bundles concatenate agent.js ahead of this file, node requires it.
184
+ const NOHIT = (typeof module === "object" && module && module.exports ? require("./agent.js") : EddieLib).NOHIT;
185
+ const NOHIT_RE = /\bthe site (?:doesn['’]t|does not|didn['’]t|did not) cover (?:that|this|it)\.?/gi;
186
+
187
+ const PLAN_SCHEMA = {
188
+ type: "object",
189
+ properties: {
190
+ queries: { type: "array", items: { type: "string" }, minItems: 1, maxItems: 3 },
191
+ },
192
+ required: ["queries"],
193
+ };
194
+
195
+ function planPrompt(site) {
196
+ return `You write search queries for a site search engine. The site is ${site}. Reply with JSON only: {"queries": ["..."]}. Give 1 to 3 different short keyword queries (2 to 5 words each, no punctuation) that a site search engine would match against page text. Each query must be different. Do not answer the question.`;
197
+ }
198
+
199
+ function answerPrompt(site) {
200
+ return `You answer visitor questions about ${site} using only the numbered sources below the question. Answer the question directly in the first sentence. Write 1 to 3 sentences in your own words; never repeat a source's wording. End each sentence with the numbers of the sources it comes from, like [2] or [1][3]. Never cite a number that is not in the list. Do not add calculations or inferences that are not in the sources. If no source answers the question, your entire reply is: ${NOHIT} Never mix that sentence with an answer. Never use outside knowledge.`;
201
+ }
202
+
203
+ /** Remove <think>…</think> blocks; a dangling <think> loses only the tag. */
204
+ function stripThink(text) {
205
+ if (!text) return "";
206
+ let out = String(text).replace(/<think>[\s\S]*?<\/think>/g, "");
207
+ out = out.replace(/<think>/g, "").replace(/<\/think>/g, "");
208
+ return out.trim();
209
+ }
210
+
211
+ function extractJsonObject(text) {
212
+ const s = String(text);
213
+ const start = s.indexOf("{");
214
+ const end = s.lastIndexOf("}");
215
+ if (start < 0 || end <= start) return null;
216
+ try {
217
+ return JSON.parse(s.substring(start, end + 1));
218
+ } catch (_) {
219
+ return null;
220
+ }
221
+ }
222
+
223
+ function cleanQuery(q) {
224
+ return String(q)
225
+ .replace(/\/?no_think/gi, "")
226
+ .replace(/[\s"'`*]+$/g, "")
227
+ .replace(/^[\s"'`*\-\d.)]+/g, "")
228
+ .replace(/\s+/g, " ")
229
+ .trim();
230
+ }
231
+
232
+ /** Parse the planner reply into 1..3 distinct queries; falls back to the question. */
233
+ function parsePlan(text, question) {
234
+ const cleaned = stripThink(text);
235
+ const obj = extractJsonObject(cleaned);
236
+ const raw = obj && Array.isArray(obj.queries) ? obj.queries : [];
237
+ const seen = new Set();
238
+ const out = [];
239
+ for (const item of raw) {
240
+ if (typeof item !== "string") continue;
241
+ const q = cleanQuery(item);
242
+ if (q.length < 2 || q.length > 80) continue;
243
+ const key = q.toLowerCase();
244
+ if (seen.has(key)) continue;
245
+ seen.add(key);
246
+ out.push(q);
247
+ if (out.length === 3) break;
248
+ }
249
+ if (out.length === 0 && question && question.trim()) {
250
+ out.push(question.trim());
251
+ }
252
+ return out;
253
+ }
254
+
255
+ /** Cut to `max` characters at a word boundary, with an ellipsis. */
256
+ function truncateText(text, max) {
257
+ const limit = max == null ? 700 : max;
258
+ const s = String(text || "").replace(/\s+/g, " ").trim();
259
+ if (s.length <= limit) return s;
260
+ const cut = s.lastIndexOf(" ", limit - 1);
261
+ return s.substring(0, cut > limit * 0.6 ? cut : limit - 1).trim() + "…";
262
+ }
263
+
264
+ /** `[n] title (url)\ntext` blocks joined by blank lines. */
265
+ function formatEvidence(items, maxChars) {
266
+ return (items || [])
267
+ .map((e, i) => `[${i + 1}] ${e.title || e.url} (${e.url})\n${truncateText(e.text || e.snippet || "", maxChars)}`)
268
+ .join("\n\n");
269
+ }
270
+
271
+ function sourcesPrompt(items, question, maxChars) {
272
+ return `Sources:\n\n${formatEvidence(items, maxChars)}\n\nQuestion: ${question}`;
273
+ }
274
+
275
+ /**
276
+ * Post-process a raw model answer against the evidence list.
277
+ * Returns { answer, citations: [{n, url, title}], nohit }.
278
+ */
279
+ function postProcessAnswer(raw, evidence) {
280
+ const ev = Array.isArray(evidence) ? evidence : [];
281
+ let text = stripThink(raw);
282
+ // **[1]** -> [1]
283
+ text = text.replace(/\*\*\s*((?:\[\s*\d+(?:\s*,\s*\d+)*\s*\]\s*)+)\*\*/g, "$1");
284
+ // Lines that are only citation markers belong to the previous line.
285
+ const lines = text.split(/\r?\n/);
286
+ const merged = [];
287
+ for (const line of lines) {
288
+ const t = line.trim();
289
+ if (t && /^(?:\[\s*\d+(?:\s*,\s*\d+)*\s*\]\s*)+$/.test(t)) {
290
+ let j = merged.length - 1;
291
+ while (j >= 0 && merged[j].trim() === "") j--;
292
+ if (j >= 0) {
293
+ // Append only markers the previous line does not already carry.
294
+ const prev = merged[j];
295
+ const fresh = (t.match(/\[\s*\d+(?:\s*,\s*\d+)*\s*\]/g) || []).filter((m) => !prev.includes(m.replace(/\s+/g, "")));
296
+ merged.length = j + 1;
297
+ if (fresh.length) merged[j] = prev.replace(/\s+$/, "") + " " + fresh.join("");
298
+ continue;
299
+ }
300
+ }
301
+ merged.push(line);
302
+ }
303
+ text = merged.join("\n");
304
+
305
+ // Drop the fallback sentence when anything else remains.
306
+ const withoutFallback = text.replace(NOHIT_RE, "").replace(/[ \t]+\n/g, "\n").trim();
307
+ const residue = withoutFallback
308
+ .replace(/\[\s*\d+(?:\s*,\s*\d+)*\s*\]/g, "")
309
+ .replace(/^\s*(?:yes|no)\b/i, "")
310
+ .replace(/[\s.,;:!?'"*-]+/g, "");
311
+ const onlyFallback = residue === "";
312
+ if (onlyFallback) {
313
+ return { answer: NOHIT, citations: [], nohit: true };
314
+ }
315
+ text = withoutFallback;
316
+
317
+ // Map [n] citations to evidence; drop out-of-range ones.
318
+ const cited = [];
319
+ const seen = new Set();
320
+ text = text.replace(/\[\s*(\d+(?:\s*,\s*\d+)*)\s*\]/g, (_, body) => {
321
+ const nums = body.split(",").map((x) => Number(x.trim()));
322
+ let out = "";
323
+ for (const n of nums) {
324
+ if (n >= 1 && n <= ev.length) {
325
+ out += `[${n}]`;
326
+ if (!seen.has(n)) {
327
+ seen.add(n);
328
+ cited.push(n);
329
+ }
330
+ }
331
+ }
332
+ return out;
333
+ });
334
+ // Collapse duplicate adjacent markers ("[1][1]") and tidy whitespace.
335
+ text = text.replace(/(\[\d+\])(?:\s*\1)+/g, "$1");
336
+ text = text.replace(/[ \t]{2,}/g, " ").replace(/ +([.,;:!?])/g, "$1").replace(/\n{3,}/g, "\n\n").trim();
337
+ const citations = cited
338
+ .sort((a, b) => a - b)
339
+ .map((n) => ({ n, url: ev[n - 1].url, title: ev[n - 1].title || ev[n - 1].url }));
340
+ return { answer: text, citations, nohit: text === "" };
341
+ }
342
+
343
+ return {
344
+ PLAN_SCHEMA,
345
+ planPrompt,
346
+ answerPrompt,
347
+ sourcesPrompt,
348
+ stripThink,
349
+ parsePlan,
350
+ truncateText,
351
+ formatEvidence,
352
+ postProcessAnswer,
353
+ };
354
+ });
355
+
356
+ // SPDX-License-Identifier: GPL-3.0-only
357
+
358
+ // Eddie agent engine, host-independent.
359
+ //
360
+ // Everything the agent worker does (WebLLM load, plan, ask/stream, abort with
361
+ // the drained-stream lock rule) behind an `env` the host supplies:
362
+ //
363
+ // createAgentEngine({
364
+ // post(message) broadcast sink (unused today; progress goes to the loaders)
365
+ // loadWebLLM() -> Promise of the WebLLM module
366
+ // now() optional clock (tests)
367
+ // })
368
+ //
369
+ // `engine.handle(msg, reply)` dispatches one protocol message; `reply` is
370
+ // the sink for that message's answers (progress/loaded for `load`,
371
+ // plan_result, token/done/aborted for `ask`, error). Message shapes are in
372
+ // widget/README.md ("Worker protocol", agent section).
373
+
374
+ (function (factory) {
375
+ const api = factory();
376
+ if (typeof module === "object" && module && module.exports) {
377
+ module.exports = api;
378
+ } else if (typeof EddieLib === "object" && EddieLib) {
379
+ Object.assign(EddieLib, api);
380
+ } else {
381
+ globalThis.EddieLib = Object.assign(globalThis.EddieLib || {}, api);
382
+ }
383
+ })(function () {
384
+ "use strict";
385
+
386
+ const EVIDENCE_CHARS = 700;
387
+ const NOT_LOADED = "model not loaded";
388
+
389
+ function createAgentEngine(env) {
390
+ const lib = typeof EddieLib === "object" && EddieLib ? EddieLib : env.lib;
391
+ const now = env.now || (() => (typeof performance === "object" && performance.now ? performance.now() : Date.now()));
392
+
393
+ let webllm = null;
394
+ let engine = null;
395
+ let modelId = null;
396
+ let loading = null; // { model, promise, waiters: [reply] }
397
+ let active = null; // { requestId, aborted, reply }
398
+ let queue = Promise.resolve();
399
+
400
+ function handle(msg, reply) {
401
+ const m = msg || {};
402
+ const out = reply || env.post;
403
+ switch (m.type) {
404
+ case "load":
405
+ return load(m, out);
406
+ case "plan":
407
+ return enqueue(() => plan(m, out), m.requestId, out);
408
+ case "ask":
409
+ return enqueue(() => ask(m, out), m.requestId, out);
410
+ case "abort":
411
+ abort(m);
412
+ return Promise.resolve();
413
+ default:
414
+ postError(out, m.requestId, `unknown message type ${String(m.type)}`);
415
+ return Promise.resolve();
416
+ }
417
+ }
418
+
419
+ /** Snapshot for the service worker's `state` reply. */
420
+ function snapshot() {
421
+ return {
422
+ model: modelId,
423
+ loaded: !!engine,
424
+ loading: loading ? loading.model : null,
425
+ active: active ? active.requestId : null,
426
+ };
427
+ }
428
+
429
+ function enqueue(fn, requestId, reply) {
430
+ queue = queue
431
+ .then(() => {
432
+ console.debug("eddie agent engine: start", requestId);
433
+ return fn();
434
+ })
435
+ .catch((err) => postError(reply, requestId, describe(err)))
436
+ .then(() => console.debug("eddie agent engine: end", requestId));
437
+ return queue;
438
+ }
439
+
440
+ async function load(msg, reply) {
441
+ const model = String(msg.model || "");
442
+ if (!model) {
443
+ postError(reply, undefined, "load: model is required");
444
+ return;
445
+ }
446
+ if (engine && modelId === model) {
447
+ reply({ type: "loaded", model, loadMs: 0, cached: true });
448
+ return;
449
+ }
450
+ if (loading) {
451
+ // Another page (or an earlier message) is loading. Same model: join
452
+ // it, the fan-out delivers progress and loaded/error to us too.
453
+ // Different model: wait for it to settle, then load ours.
454
+ const join = loading.model === model;
455
+ if (join) loading.waiters.push(reply);
456
+ try {
457
+ await loading.promise;
458
+ } catch (_) {
459
+ // the fan-out already reported the error to the joined waiters
460
+ }
461
+ if (join) return;
462
+ if (engine && modelId === model) {
463
+ reply({ type: "loaded", model, loadMs: 0, cached: true });
464
+ return;
465
+ }
466
+ }
467
+ const job = { model, waiters: [reply], promise: null };
468
+ const fanout = (message) => {
469
+ for (const w of job.waiters) w(message);
470
+ };
471
+ job.promise = (async () => {
472
+ const t0 = now();
473
+ if (!webllm) {
474
+ fanout({ type: "progress", text: "Loading the WebLLM runtime…", progress: 0 });
475
+ webllm = await env.loadWebLLM();
476
+ }
477
+ if (engine) {
478
+ try {
479
+ await engine.unload();
480
+ } catch (_) {
481
+ // ignore
482
+ }
483
+ engine = null;
484
+ modelId = null;
485
+ }
486
+ const created = await webllm.CreateMLCEngine(model, {
487
+ initProgressCallback: (p) => {
488
+ fanout({
489
+ type: "progress",
490
+ text: p && p.text ? p.text : "Loading model…",
491
+ progress: p && typeof p.progress === "number" ? p.progress : null,
492
+ });
493
+ },
494
+ });
495
+ engine = created;
496
+ modelId = model;
497
+ fanout({ type: "loaded", model, loadMs: Math.round(now() - t0) });
498
+ })();
499
+ loading = job;
500
+ try {
501
+ await job.promise;
502
+ } catch (err) {
503
+ engine = null;
504
+ modelId = null;
505
+ fanout({ type: "error", requestId: undefined, message: describe(err) });
506
+ } finally {
507
+ if (loading === job) loading = null;
508
+ }
509
+ }
510
+
511
+ function requireEngine() {
512
+ if (!engine) throw new Error(NOT_LOADED);
513
+ }
514
+
515
+ async function plan(msg, reply) {
516
+ requireEngine();
517
+ const question = String(msg.question || "").trim();
518
+ const site = String(msg.site || "this website");
519
+ const t0 = now();
520
+ const replyMsg = await engine.chat.completions.create({
521
+ messages: [
522
+ { role: "system", content: lib.planPrompt(site) },
523
+ { role: "user", content: question },
524
+ ],
525
+ temperature: 0,
526
+ max_tokens: 100,
527
+ response_format: { type: "json_object", schema: JSON.stringify(lib.PLAN_SCHEMA) },
528
+ extra_body: { enable_thinking: false },
529
+ });
530
+ const content = replyMsg && replyMsg.choices && replyMsg.choices[0] && replyMsg.choices[0].message ? replyMsg.choices[0].message.content : "";
531
+ const queries = lib.parsePlan(content, question);
532
+ reply({ type: "plan_result", requestId: msg.requestId, queries, ms: Math.round(now() - t0) });
533
+ }
534
+
535
+ async function ask(msg, reply) {
536
+ requireEngine();
537
+ const requestId = msg.requestId;
538
+ const question = String(msg.question || "").trim();
539
+ const site = String(msg.site || "this website");
540
+ const evidence = Array.isArray(msg.evidence) ? msg.evidence.filter((e) => e && e.url) : [];
541
+ if (evidence.length === 0) {
542
+ reply({
543
+ type: "done",
544
+ requestId,
545
+ answer: lib.NOHIT,
546
+ citations: [],
547
+ nohit: true,
548
+ raw: "",
549
+ usage: { ttftMs: 0, totalMs: 0, tps: null, completionTokens: 0 },
550
+ });
551
+ return;
552
+ }
553
+ active = { requestId, aborted: false, reply };
554
+ const t0 = now();
555
+ let first = 0;
556
+ let text = "";
557
+ let usage = null;
558
+ try {
559
+ const stream = await engine.chat.completions.create({
560
+ messages: [
561
+ { role: "system", content: lib.answerPrompt(site) },
562
+ { role: "user", content: lib.sourcesPrompt(evidence, question, EVIDENCE_CHARS) },
563
+ ],
564
+ stream: true,
565
+ stream_options: { include_usage: true },
566
+ temperature: 0,
567
+ frequency_penalty: 0.5,
568
+ presence_penalty: 0,
569
+ max_tokens: 220,
570
+ extra_body: { enable_thinking: false },
571
+ });
572
+ // Never break out of this loop: WebLLM releases its generation lock at
573
+ // the end of the async generator, and an early exit skips that release,
574
+ // hanging every later completion. After interruptGenerate() the stream
575
+ // ends by itself within one decode step; drop the tokens until then.
576
+ for await (const chunk of stream) {
577
+ if (active.aborted) continue;
578
+ const delta = chunk && chunk.choices && chunk.choices[0] && chunk.choices[0].delta ? chunk.choices[0].delta.content : null;
579
+ if (delta) {
580
+ if (!first) first = now();
581
+ text += delta;
582
+ reply({ type: "token", requestId, text: delta });
583
+ }
584
+ if (chunk && chunk.usage) usage = chunk.usage;
585
+ }
586
+ } finally {
587
+ const wasAborted = active && active.aborted;
588
+ active = null;
589
+ if (wasAborted) {
590
+ reply({ type: "aborted", requestId });
591
+ return;
592
+ }
593
+ }
594
+ const processed = lib.postProcessAnswer(text, evidence);
595
+ const totalMs = Math.round(now() - t0);
596
+ reply({
597
+ type: "done",
598
+ requestId,
599
+ answer: processed.answer,
600
+ citations: processed.citations,
601
+ nohit: processed.nohit,
602
+ raw: text,
603
+ usage: {
604
+ ttftMs: first ? Math.round(first - t0) : totalMs,
605
+ totalMs,
606
+ tps: usage && usage.extra && typeof usage.extra.decode_tokens_per_s === "number" ? Math.round(usage.extra.decode_tokens_per_s) : null,
607
+ completionTokens: usage ? usage.completion_tokens : null,
608
+ },
609
+ });
610
+ }
611
+
612
+ function abort(msg) {
613
+ console.debug("eddie agent engine: abort", msg.requestId, active ? active.requestId : null);
614
+ if (!active) return;
615
+ if (msg.requestId != null && msg.requestId !== active.requestId) return;
616
+ active.aborted = true;
617
+ try {
618
+ if (engine) engine.interruptGenerate();
619
+ } catch (err) {
620
+ console.warn("eddie agent: interrupt failed", err);
621
+ }
622
+ }
623
+
624
+ /** Abort the active run if it belongs to a page that went away. */
625
+ function abortIfOwner(reply) {
626
+ if (active && active.reply === reply) abort({ requestId: active.requestId });
627
+ }
628
+
629
+ function postError(reply, requestId, message) {
630
+ reply({ type: "error", requestId: requestId == null ? undefined : requestId, message });
631
+ }
632
+
633
+ return { handle, state: snapshot, abortIfOwner };
634
+ }
635
+
636
+ function describe(err) {
637
+ if (err == null) return "unknown error";
638
+ if (typeof err === "string") return err;
639
+ return err.message || String(err);
640
+ }
641
+
642
+ /** True for the engine's "not loaded" replies (the client re-runs load). */
643
+ function isModelNotLoadedMessage(message) {
644
+ return typeof message === "string" && message.indexOf(NOT_LOADED) === 0;
645
+ }
646
+
647
+ return { createAgentEngine, AGENT_NOT_LOADED: NOT_LOADED, isModelNotLoadedMessage };
648
+ });
649
+
650
+ // SPDX-License-Identifier: GPL-3.0-only
651
+
652
+ // Eddie service worker: a persistent host for the search engine and the
653
+ // agent, so a navigation within the site does not throw away the loaded
654
+ // index, the dense model or the WebLLM engine.
655
+ //
656
+ // One source, four builds (widget/build.sh), because a service worker may
657
+ // not import() anything: every dependency is a static import that build.sh
658
+ // prepends per tier, along with `EDDIE_SW_TIER`:
659
+ //
660
+ // eddie-sw-lite.js initLiteWasm / liteWasmApi (eddie-lite-esm.js)
661
+ // eddie-sw-dense.js lite + initDenseWasm / denseWasmApi (eddie-dense-esm.js)
662
+ // eddie-sw-gpu.js lite + `transformers` (eddie-transformers-sw.js, the
663
+ // copy whose onnxruntime-web imports point at the ORT
664
+ // bundle build); search only, never WebLLM
665
+ // eddie-sw-agent.js `webllm` (jsDelivr; the esm.run alias redirects, and
666
+ // service worker script fetches reject redirects) and
667
+ // the agent engine only; never a search host
668
+ //
669
+ // Each tier is registered by the widget as a *module* service worker in
670
+ // its own scope under the asset directory (`/eddie/sw/<tier>/`), so a
671
+ // visitor who never accepts a model never installs the gpu tier's imports
672
+ // and one who never asks the agent a question never installs WebLLM.
673
+ // The worker never handles `fetch`, so the browser does not start it for
674
+ // navigations; pages reach it through
675
+ // `registration.active.postMessage({type: "connect"}, [port])`, one
676
+ // MessageChannel per page and engine ("search" or "agent"), and then speak
677
+ // exactly the dedicated-worker protocols over that port. Three extra
678
+ // messages exist on every port: `hello` (answered with the host's tier and
679
+ // capabilities and both engines' state), `ping` -> `pong` (keepalive: Chrome
680
+ // stops an idle service worker after ~30 s) and `state`.
681
+ //
682
+ // widget/build.sh concatenates widget/src/lib/*.js ahead of this file.
683
+
684
+ "use strict";
685
+
686
+ const lib = EddieLib;
687
+
688
+ const SEARCH_TYPES = new Set(["init", "cache_check", "cache_clear", "search", "page", "chunk", "qa"]);
689
+ const AGENT_TYPES = new Set(["load", "plan", "ask", "abort"]);
690
+ const startedAt = Date.now();
691
+ const hasGpu = !!(self.navigator && navigator.gpu && typeof navigator.gpu.requestAdapter === "function");
692
+ const canRunOnnx = EDDIE_SW_TIER === "gpu" && hasGpu;
693
+
694
+ /** A rejection that names the tier able to host what this one cannot; the widget moves the search there. */
695
+ function tierError(tier, what) {
696
+ const e = new Error(`the ${EDDIE_SW_TIER} service worker has no ${what}; the ${tier} tier hosts it`);
697
+ e.eddieTier = tier;
698
+ return e;
699
+ }
700
+
701
+ const connections = new Set(); // { port, kind, reply }
702
+
703
+ function broadcast(kind, message) {
704
+ for (const c of connections) {
705
+ if (c.kind === kind) c.reply(message);
706
+ }
707
+ }
708
+
709
+ /** ORT must use the factory embedded in its bundle build: no import() here. */
710
+ function configureTransformers(tf) {
711
+ const onnx = tf.env && tf.env.backends && tf.env.backends.onnx;
712
+ if (!onnx || !onnx.wasm) return;
713
+ tf.env.useWasmCache = false;
714
+ onnx.wasm.numThreads = 1;
715
+ const ver = onnx.versions && onnx.versions.web;
716
+ if (ver && !onnx.wasm.wasmPaths) {
717
+ onnx.wasm.wasmPaths = { wasm: `https://cdn.jsdelivr.net/npm/onnxruntime-web@${ver}/dist/ort-wasm-simd-threaded.asyncify.wasm` };
718
+ }
719
+ }
720
+
721
+ const wasmInits = {}; // variant -> Promise
722
+ function loadWasm(baseUrl, variant) {
723
+ const v = variant || "lite";
724
+ if (v === "lite") {
725
+ if (!wasmInits.lite) wasmInits.lite = initLiteWasm({ module_or_path: lib.assetUrl(baseUrl, EDDIE_LITE_WASM, lib.ASSET_VERSION) }).then(() => liteWasmApi);
726
+ return wasmInits.lite;
727
+ }
728
+ if (v === "dense") {
729
+ if (!initDenseWasm) return Promise.reject(tierError("dense", "CPU embedder"));
730
+ if (!wasmInits.dense) wasmInits.dense = initDenseWasm({ module_or_path: lib.assetUrl(baseUrl, EDDIE_DENSE_WASM, lib.ASSET_VERSION) }).then(() => denseWasmApi);
731
+ return wasmInits.dense;
732
+ }
733
+ return Promise.reject(new Error(`unknown wasm variant ${String(variant)}`));
734
+ }
735
+
736
+ // The agent tier bundles no search engine (widget/src/lib/search-engine.js).
737
+ const searchEngine = typeof lib.createSearchEngine === "function"
738
+ ? lib.createSearchEngine({
739
+ post: (message) => broadcast("search", message),
740
+ loadWasm,
741
+ loadTransformers: async () => {
742
+ if (!transformers) throw tierError("gpu", "transformers.js");
743
+ return transformers;
744
+ },
745
+ configureTransformers,
746
+ // Lane *choice* follows the adapter, whatever the tier: a lite worker asks
747
+ // consent for the webgpu lane, and the widget moves the search to the gpu
748
+ // tier on accept (or on a tier_required status if the lane is cached).
749
+ canRunWebGpuLane: hasGpu,
750
+ })
751
+ : null;
752
+
753
+ // Only the agent tier bundles the agent (widget/src/lib/agent*.js) at all.
754
+ const agentEngine = webllm && typeof lib.createAgentEngine === "function"
755
+ ? lib.createAgentEngine({
756
+ post: (message) => broadcast("agent", message),
757
+ loadWebLLM: async () => webllm,
758
+ })
759
+ : null;
760
+
761
+ function capabilities() {
762
+ return {
763
+ ok: true,
764
+ tier: EDDIE_SW_TIER,
765
+ // The WebGPU runtime this tier hosts (transformers.js or WebLLM) can run.
766
+ gpu: hasGpu && !!(transformers || webllm),
767
+ onnx: canRunOnnx,
768
+ denseWasm: !!initDenseWasm,
769
+ startedAt,
770
+ search: searchEngine ? searchEngine.state() : null,
771
+ agent: agentEngine ? agentEngine.state() : null,
772
+ };
773
+ }
774
+
775
+ function attach(port, kind) {
776
+ const conn = {
777
+ port,
778
+ kind: kind === "agent" ? "agent" : "search",
779
+ reply: (message) => {
780
+ try {
781
+ port.postMessage(message);
782
+ } catch (err) {
783
+ console.warn("eddie sw: reply failed", err);
784
+ }
785
+ },
786
+ };
787
+ connections.add(conn);
788
+ port.onmessage = (e) => route(conn, e.data || {});
789
+ // Chrome 132+ fires close when the page that owns the other end goes away.
790
+ port.addEventListener("close", () => detach(conn));
791
+ port.start();
792
+ }
793
+
794
+ function detach(conn) {
795
+ if (!connections.has(conn)) return;
796
+ connections.delete(conn);
797
+ if (agentEngine) agentEngine.abortIfOwner(conn.reply);
798
+ try {
799
+ conn.port.close();
800
+ } catch (_) {
801
+ // ignore
802
+ }
803
+ }
804
+
805
+ function route(conn, msg) {
806
+ switch (msg.type) {
807
+ case "hello":
808
+ conn.reply(Object.assign({ type: "hello", requestId: msg.requestId }, capabilities()));
809
+ return;
810
+ case "ping":
811
+ conn.reply({ type: "pong", requestId: msg.requestId });
812
+ return;
813
+ case "state":
814
+ conn.reply(Object.assign({ type: "state", requestId: msg.requestId }, capabilities()));
815
+ return;
816
+ case "disconnect":
817
+ detach(conn);
818
+ return;
819
+ default:
820
+ break;
821
+ }
822
+ if (SEARCH_TYPES.has(msg.type)) {
823
+ if (searchEngine) searchEngine.handle(msg, conn.reply);
824
+ else conn.reply({ type: "error", requestId: msg.requestId, message: `the ${EDDIE_SW_TIER} service worker has no search engine; the lite tier hosts it` });
825
+ } else if (AGENT_TYPES.has(msg.type)) {
826
+ if (agentEngine) agentEngine.handle(msg, conn.reply);
827
+ else conn.reply({ type: "error", requestId: msg.requestId, message: `the ${EDDIE_SW_TIER} service worker has no agent; the agent tier hosts it` });
828
+ } else {
829
+ conn.reply({ type: "error", requestId: msg.requestId, message: `unknown message type ${String(msg.type)}` });
830
+ }
831
+ }
832
+
833
+ self.addEventListener("install", () => {
834
+ self.skipWaiting();
835
+ });
836
+
837
+ self.addEventListener("activate", (e) => {
838
+ e.waitUntil(self.clients.claim());
839
+ });
840
+
841
+ self.addEventListener("message", (e) => {
842
+ const msg = e.data || {};
843
+ if (msg.type === "connect" && e.ports && e.ports[0]) {
844
+ attach(e.ports[0], msg.kind);
845
+ }
846
+ });