eddie-jekyll 0.4.0 → 0.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,12 +1,14 @@
1
1
  // SPDX-License-Identifier: GPL-3.0-only
2
2
  // Generated by widget/build.sh from widget/src/eddie-agent-worker.js and widget/src/lib/*.js; edit those instead.
3
3
  "use strict";
4
+ const EDDIE_ASSET_VERSION = "c0fdc503e01a";
4
5
  const EddieLib = {};
5
6
  // SPDX-License-Identifier: GPL-3.0-only
6
7
 
7
- // Agent helpers shared by the widget and the agent worker: model selection,
8
- // prompts, plan parsing, evidence assembly and answer post-processing.
9
- // Pure functions; no WebLLM, no DOM.
8
+ // Agent helpers the widget shows in its UI: model selection, evidence
9
+ // assembly, stream display and the FAQ gate. Pure functions; no WebLLM, no
10
+ // DOM. The prompts and answer post-processing that run beside the model
11
+ // live in agent-llm.js (bundled only into the agent engine's hosts).
10
12
 
11
13
  (function (factory) {
12
14
  const api = factory();
@@ -21,7 +23,6 @@ const EddieLib = {};
21
23
  "use strict";
22
24
 
23
25
  const NOHIT = "The site doesn't cover that.";
24
- const NOHIT_RE = /\bthe site (?:doesn['’]t|does not|didn['’]t|did not) cover (?:that|this|it)\.?/gi;
25
26
 
26
27
  const AGENT_MODEL_SIZES = {
27
28
  "Qwen3.5-0.8B": 0.4e9,
@@ -30,22 +31,6 @@ const EddieLib = {};
30
31
  };
31
32
  const TWO_GIB = 2 * 1024 * 1024 * 1024;
32
33
 
33
- const PLAN_SCHEMA = {
34
- type: "object",
35
- properties: {
36
- queries: { type: "array", items: { type: "string" }, minItems: 1, maxItems: 3 },
37
- },
38
- required: ["queries"],
39
- };
40
-
41
- function planPrompt(site) {
42
- return `You write search queries for a site search engine. The site is ${site}. Reply with JSON only: {"queries": ["..."]}. Give 1 to 3 different short keyword queries (2 to 5 words each, no punctuation) that a site search engine would match against page text. Each query must be different. Do not answer the question.`;
43
- }
44
-
45
- function answerPrompt(site) {
46
- return `You answer visitor questions about ${site} using only the numbered sources below the question. Answer the question directly in the first sentence. Write 1 to 3 sentences in your own words; never repeat a source's wording. End each sentence with the numbers of the sources it comes from, like [2] or [1][3]. Never cite a number that is not in the list. Do not add calculations or inferences that are not in the sources. If no source answers the question, your entire reply is: ${NOHIT} Never mix that sentence with an answer. Never use outside knowledge.`;
47
- }
48
-
49
34
  /** Strip the WebLLM variant suffix to get the family name shown to visitors. */
50
35
  function baseModelId(id) {
51
36
  return String(id).replace(/-q\d+f(16|32)_\d+-MLC$/i, "");
@@ -58,7 +43,10 @@ const EddieLib = {};
58
43
 
59
44
  /**
60
45
  * Choose the WebLLM model id.
61
- * opts: { mode: "auto"|"quality"|<id>, maxBufferSize, isMobile, hasF16 }
46
+ * opts: { mode: "auto"|"light"|"quality"|<id>, maxBufferSize, isMobile, hasF16 }
47
+ *
48
+ * "light" and "quality" are the two sizes the settings panel offers by
49
+ * name; "auto" picks between them from the adapter's buffer limit.
62
50
  */
63
51
  function selectAgentModel(opts) {
64
52
  const o = opts || {};
@@ -70,6 +58,8 @@ const EddieLib = {};
70
58
  base = big ? "Qwen3.5-2B" : "Qwen3.5-0.8B";
71
59
  } else if (mode === "quality") {
72
60
  base = "Qwen3.5-2B";
61
+ } else if (mode === "light") {
62
+ base = "Qwen3.5-0.8B";
73
63
  } else {
74
64
  return { id: mode, base: baseModelId(mode), sizeBytes: agentModelBytes(mode), explicit: true };
75
65
  }
@@ -85,14 +75,6 @@ const EddieLib = {};
85
75
  return /Mobi|Android|iPhone|iPad|iPod|Windows Phone/i.test(n.userAgent || "");
86
76
  }
87
77
 
88
- /** Remove <think>…</think> blocks; a dangling <think> loses only the tag. */
89
- function stripThink(text) {
90
- if (!text) return "";
91
- let out = String(text).replace(/<think>[\s\S]*?<\/think>/g, "");
92
- out = out.replace(/<think>/g, "").replace(/<\/think>/g, "");
93
- return out.trim();
94
- }
95
-
96
78
  /**
97
79
  * Display text for a partial stream: complete think blocks removed, and
98
80
  * anything after an unclosed <think> hidden until it closes.
@@ -105,6 +87,123 @@ const EddieLib = {};
105
87
  return out.replace(/^\s+/, "");
106
88
  }
107
89
 
90
+ function urlKey(url) {
91
+ return String(url || "").replace(/#.*$/, "").replace(/\/+$/, "").toLowerCase();
92
+ }
93
+
94
+ /**
95
+ * Round-robin merge of several result lists, deduplicated by URL, at most
96
+ * `max` items. Each result keeps its own fields.
97
+ */
98
+ function mergeEvidence(lists, max) {
99
+ const limit = max == null ? 6 : max;
100
+ const seen = new Set();
101
+ const out = [];
102
+ const arrays = (lists || []).map((l) => (Array.isArray(l) ? l : []));
103
+ const longest = arrays.reduce((n, l) => Math.max(n, l.length), 0);
104
+ for (let i = 0; i < longest && out.length < limit; i++) {
105
+ for (const list of arrays) {
106
+ if (out.length >= limit) break;
107
+ const r = list[i];
108
+ if (!r || !r.url) continue;
109
+ const key = urlKey(r.url);
110
+ if (seen.has(key)) continue;
111
+ seen.add(key);
112
+ out.push(r);
113
+ }
114
+ }
115
+ return out;
116
+ }
117
+
118
+ // FAQ card gate: prefer the WASM's fused `confident` flag (qa_lookup v0.4.1+);
119
+ // older indexes only carry a dense score, so fall back to a plain cutoff.
120
+ function faqPasses(hit, qaMode) {
121
+ if (!hit || typeof hit !== "object") return false;
122
+ if (qaMode === "off") return false;
123
+ if (qaMode === "always") return true;
124
+ if (typeof hit.confident === "boolean") return hit.confident;
125
+ return typeof hit.score === "number" && hit.score >= 0.5;
126
+ }
127
+
128
+ // Turn confident QA hits into agent evidence items ("Q: … A: …") so the
129
+ // answer model sees the FAQ lane, not only chunk text.
130
+ function qaEvidence(hits, max) {
131
+ const limit = max == null ? 2 : max;
132
+ const out = [];
133
+ for (const h of Array.isArray(hits) ? hits : []) {
134
+ if (out.length >= limit) break;
135
+ if (!faqPasses(h, "auto")) continue;
136
+ const q = String(h.question || "").trim();
137
+ const a = String(h.answer || "").trim();
138
+ if (!q || !a) continue;
139
+ out.push({ title: "FAQ: " + q, url: h.source_url || "", text: "Q: " + q + "\nA: " + a, faq: true });
140
+ }
141
+ return out;
142
+ }
143
+
144
+ return {
145
+ faqPasses,
146
+ qaEvidence,
147
+ NOHIT,
148
+ AGENT_MODEL_SIZES,
149
+ baseModelId,
150
+ agentModelBytes,
151
+ selectAgentModel,
152
+ isMobileDevice,
153
+ visibleStreamText,
154
+ mergeEvidence,
155
+ };
156
+ });
157
+
158
+ // SPDX-License-Identifier: GPL-3.0-only
159
+
160
+ // Agent LLM helpers used only where the model runs (agent-engine.js hosts:
161
+ // the agent service worker and the page-side agent worker): prompts, plan
162
+ // parsing and answer post-processing. Split out of agent.js so the widget
163
+ // bundle carries only the model-selection and evidence helpers it shows in
164
+ // the UI. Bundles that include this file always include agent.js first.
165
+
166
+ (function (factory) {
167
+ const api = factory();
168
+ if (typeof module === "object" && module && module.exports) {
169
+ module.exports = api;
170
+ } else if (typeof EddieLib === "object" && EddieLib) {
171
+ Object.assign(EddieLib, api);
172
+ } else {
173
+ globalThis.EddieLib = Object.assign(globalThis.EddieLib || {}, api);
174
+ }
175
+ })(function () {
176
+ "use strict";
177
+
178
+ // The fallback sentence lives in agent.js (the widget shows it too); the
179
+ // bundles concatenate agent.js ahead of this file, node requires it.
180
+ const NOHIT = (typeof module === "object" && module && module.exports ? require("./agent.js") : EddieLib).NOHIT;
181
+ const NOHIT_RE = /\bthe site (?:doesn['’]t|does not|didn['’]t|did not) cover (?:that|this|it)\.?/gi;
182
+
183
+ const PLAN_SCHEMA = {
184
+ type: "object",
185
+ properties: {
186
+ queries: { type: "array", items: { type: "string" }, minItems: 1, maxItems: 3 },
187
+ },
188
+ required: ["queries"],
189
+ };
190
+
191
+ function planPrompt(site) {
192
+ return `You write search queries for a site search engine. The site is ${site}. Reply with JSON only: {"queries": ["..."]}. Give 1 to 3 different short keyword queries (2 to 5 words each, no punctuation) that a site search engine would match against page text. Each query must be different. Do not answer the question.`;
193
+ }
194
+
195
+ function answerPrompt(site) {
196
+ return `You answer visitor questions about ${site} using only the numbered sources below the question. Answer the question directly in the first sentence. Write 1 to 3 sentences in your own words; never repeat a source's wording. End each sentence with the numbers of the sources it comes from, like [2] or [1][3]. Never cite a number that is not in the list. Do not add calculations or inferences that are not in the sources. If no source answers the question, your entire reply is: ${NOHIT} Never mix that sentence with an answer. Never use outside knowledge.`;
197
+ }
198
+
199
+ /** Remove <think>…</think> blocks; a dangling <think> loses only the tag. */
200
+ function stripThink(text) {
201
+ if (!text) return "";
202
+ let out = String(text).replace(/<think>[\s\S]*?<\/think>/g, "");
203
+ out = out.replace(/<think>/g, "").replace(/<\/think>/g, "");
204
+ return out.trim();
205
+ }
206
+
108
207
  function extractJsonObject(text) {
109
208
  const s = String(text);
110
209
  const start = s.indexOf("{");
@@ -149,34 +248,6 @@ const EddieLib = {};
149
248
  return out;
150
249
  }
151
250
 
152
- function urlKey(url) {
153
- return String(url || "").replace(/#.*$/, "").replace(/\/+$/, "").toLowerCase();
154
- }
155
-
156
- /**
157
- * Round-robin merge of several result lists, deduplicated by URL, at most
158
- * `max` items. Each result keeps its own fields.
159
- */
160
- function mergeEvidence(lists, max) {
161
- const limit = max == null ? 6 : max;
162
- const seen = new Set();
163
- const out = [];
164
- const arrays = (lists || []).map((l) => (Array.isArray(l) ? l : []));
165
- const longest = arrays.reduce((n, l) => Math.max(n, l.length), 0);
166
- for (let i = 0; i < longest && out.length < limit; i++) {
167
- for (const list of arrays) {
168
- if (out.length >= limit) break;
169
- const r = list[i];
170
- if (!r || !r.url) continue;
171
- const key = urlKey(r.url);
172
- if (seen.has(key)) continue;
173
- seen.add(key);
174
- out.push(r);
175
- }
176
- }
177
- return out;
178
- }
179
-
180
251
  /** Cut to `max` characters at a word boundary, with an ellipsis. */
181
252
  function truncateText(text, max) {
182
253
  const limit = max == null ? 700 : max;
@@ -266,20 +337,12 @@ const EddieLib = {};
266
337
  }
267
338
 
268
339
  return {
269
- NOHIT,
270
- AGENT_MODEL_SIZES,
271
340
  PLAN_SCHEMA,
272
341
  planPrompt,
273
342
  answerPrompt,
274
343
  sourcesPrompt,
275
- baseModelId,
276
- agentModelBytes,
277
- selectAgentModel,
278
- isMobileDevice,
279
344
  stripThink,
280
- visibleStreamText,
281
345
  parsePlan,
282
- mergeEvidence,
283
346
  truncateText,
284
347
  formatEvidence,
285
348
  postProcessAnswer,
@@ -288,249 +351,320 @@ const EddieLib = {};
288
351
 
289
352
  // SPDX-License-Identifier: GPL-3.0-only
290
353
 
291
- // Eddie agent worker (module worker, created on the first "Ask").
354
+ // Eddie agent engine, host-independent.
292
355
  //
293
- // Runs WebLLM in the worker; retrieval stays in the widget, which owns the
294
- // search worker and passes evidence in. widget/build.sh concatenates
295
- // widget/src/lib/agent.js ahead of this file (EddieLib).
356
+ // Everything the agent worker does (WebLLM load, plan, ask/stream, abort with
357
+ // the drained-stream lock rule) behind an `env` the host supplies:
296
358
  //
297
- // Protocol (main thread -> worker):
298
- // load {model}
299
- // plan {requestId, question, site}
300
- // ask {requestId, question, site, evidence: [{title, url, text}]}
301
- // abort {requestId?}
302
- // (worker -> main thread):
303
- // progress {text, progress}
304
- // loaded {model, loadMs}
305
- // plan_result {requestId, queries, ms}
306
- // token {requestId, text}
307
- // done {requestId, answer, citations: [{n, url, title}], nohit, usage}
308
- // aborted {requestId}
309
- // error {requestId?, message}
310
-
311
- "use strict";
359
+ // createAgentEngine({
360
+ // post(message) broadcast sink (unused today; progress goes to the loaders)
361
+ // loadWebLLM() -> Promise of the WebLLM module
362
+ // now() optional clock (tests)
363
+ // })
364
+ //
365
+ // `engine.handle(msg, reply)` dispatches one protocol message; `reply` is
366
+ // the sink for that message's answers (progress/loaded for `load`,
367
+ // plan_result, token/done/aborted for `ask`, error). Message shapes are in
368
+ // widget/README.md ("Worker protocol", agent section).
312
369
 
313
- const WEBLLM_URL = "https://esm.run/@mlc-ai/web-llm@0.2.84";
314
- const EVIDENCE_CHARS = 700;
370
+ (function (factory) {
371
+ const api = factory();
372
+ if (typeof module === "object" && module && module.exports) {
373
+ module.exports = api;
374
+ } else if (typeof EddieLib === "object" && EddieLib) {
375
+ Object.assign(EddieLib, api);
376
+ } else {
377
+ globalThis.EddieLib = Object.assign(globalThis.EddieLib || {}, api);
378
+ }
379
+ })(function () {
380
+ "use strict";
315
381
 
316
- const lib = EddieLib;
382
+ const EVIDENCE_CHARS = 700;
383
+ const NOT_LOADED = "model not loaded";
384
+
385
+ function createAgentEngine(env) {
386
+ const lib = typeof EddieLib === "object" && EddieLib ? EddieLib : env.lib;
387
+ const now = env.now || (() => (typeof performance === "object" && performance.now ? performance.now() : Date.now()));
388
+
389
+ let webllm = null;
390
+ let engine = null;
391
+ let modelId = null;
392
+ let loading = null; // { model, promise, waiters: [reply] }
393
+ let active = null; // { requestId, aborted, reply }
394
+ let queue = Promise.resolve();
395
+
396
+ function handle(msg, reply) {
397
+ const m = msg || {};
398
+ const out = reply || env.post;
399
+ switch (m.type) {
400
+ case "load":
401
+ return load(m, out);
402
+ case "plan":
403
+ return enqueue(() => plan(m, out), m.requestId, out);
404
+ case "ask":
405
+ return enqueue(() => ask(m, out), m.requestId, out);
406
+ case "abort":
407
+ abort(m);
408
+ return Promise.resolve();
409
+ default:
410
+ postError(out, m.requestId, `unknown message type ${String(m.type)}`);
411
+ return Promise.resolve();
412
+ }
413
+ }
317
414
 
318
- let webllm = null;
319
- let engine = null;
320
- let modelId = null;
321
- let loading = null;
322
- let active = null; // { requestId, aborted }
323
- let queue = Promise.resolve();
415
+ /** Snapshot for the service worker's `state` reply. */
416
+ function snapshot() {
417
+ return {
418
+ model: modelId,
419
+ loaded: !!engine,
420
+ loading: loading ? loading.model : null,
421
+ active: active ? active.requestId : null,
422
+ };
423
+ }
324
424
 
325
- self.onmessage = function (e) {
326
- const msg = e.data || {};
327
- switch (msg.type) {
328
- case "load":
329
- load(msg);
330
- break;
331
- case "plan":
332
- enqueue(() => plan(msg), msg.requestId);
333
- break;
334
- case "ask":
335
- enqueue(() => ask(msg), msg.requestId);
336
- break;
337
- case "abort":
338
- abort(msg);
339
- break;
340
- default:
341
- postError(msg.requestId, `unknown message type ${String(msg.type)}`);
342
- }
343
- };
425
+ function enqueue(fn, requestId, reply) {
426
+ queue = queue
427
+ .then(() => {
428
+ console.debug("eddie agent engine: start", requestId);
429
+ return fn();
430
+ })
431
+ .catch((err) => postError(reply, requestId, describe(err)))
432
+ .then(() => console.debug("eddie agent engine: end", requestId));
433
+ return queue;
434
+ }
344
435
 
345
- function enqueue(fn, requestId) {
346
- queue = queue
347
- .then(() => {
348
- console.debug("eddie agent worker: start", requestId);
349
- return fn();
350
- })
351
- .catch((err) => postError(requestId, describe(err)))
352
- .then(() => console.debug("eddie agent worker: end", requestId));
353
- }
354
-
355
- async function load(msg) {
356
- const model = String(msg.model || "");
357
- if (!model) {
358
- postError(undefined, "load: model is required");
359
- return;
360
- }
361
- if (engine && modelId === model) {
362
- self.postMessage({ type: "loaded", model, loadMs: 0, cached: true });
363
- return;
364
- }
365
- if (loading) {
366
- try {
367
- await loading;
368
- } catch (_) {
369
- // fall through and try again
436
+ async function load(msg, reply) {
437
+ const model = String(msg.model || "");
438
+ if (!model) {
439
+ postError(reply, undefined, "load: model is required");
440
+ return;
441
+ }
442
+ if (engine && modelId === model) {
443
+ reply({ type: "loaded", model, loadMs: 0, cached: true });
444
+ return;
445
+ }
446
+ if (loading) {
447
+ // Another page (or an earlier message) is loading. Same model: join
448
+ // it, the fan-out delivers progress and loaded/error to us too.
449
+ // Different model: wait for it to settle, then load ours.
450
+ const join = loading.model === model;
451
+ if (join) loading.waiters.push(reply);
452
+ try {
453
+ await loading.promise;
454
+ } catch (_) {
455
+ // the fan-out already reported the error to the joined waiters
456
+ }
457
+ if (join) return;
458
+ if (engine && modelId === model) {
459
+ reply({ type: "loaded", model, loadMs: 0, cached: true });
460
+ return;
461
+ }
462
+ }
463
+ const job = { model, waiters: [reply], promise: null };
464
+ const fanout = (message) => {
465
+ for (const w of job.waiters) w(message);
466
+ };
467
+ job.promise = (async () => {
468
+ const t0 = now();
469
+ if (!webllm) {
470
+ fanout({ type: "progress", text: "Loading the WebLLM runtime…", progress: 0 });
471
+ webllm = await env.loadWebLLM();
472
+ }
473
+ if (engine) {
474
+ try {
475
+ await engine.unload();
476
+ } catch (_) {
477
+ // ignore
478
+ }
479
+ engine = null;
480
+ modelId = null;
481
+ }
482
+ const created = await webllm.CreateMLCEngine(model, {
483
+ initProgressCallback: (p) => {
484
+ fanout({
485
+ type: "progress",
486
+ text: p && p.text ? p.text : "Loading model…",
487
+ progress: p && typeof p.progress === "number" ? p.progress : null,
488
+ });
489
+ },
490
+ });
491
+ engine = created;
492
+ modelId = model;
493
+ fanout({ type: "loaded", model, loadMs: Math.round(now() - t0) });
494
+ })();
495
+ loading = job;
496
+ try {
497
+ await job.promise;
498
+ } catch (err) {
499
+ engine = null;
500
+ modelId = null;
501
+ fanout({ type: "error", requestId: undefined, message: describe(err) });
502
+ } finally {
503
+ if (loading === job) loading = null;
504
+ }
370
505
  }
371
- if (engine && modelId === model) {
372
- self.postMessage({ type: "loaded", model, loadMs: 0, cached: true });
373
- return;
506
+
507
+ function requireEngine() {
508
+ if (!engine) throw new Error(NOT_LOADED);
374
509
  }
375
- }
376
- loading = (async () => {
377
- const t0 = performance.now();
378
- if (!webllm) {
379
- self.postMessage({ type: "progress", text: "Loading the WebLLM runtime…", progress: 0 });
380
- webllm = await import(WEBLLM_URL);
510
+
511
+ async function plan(msg, reply) {
512
+ requireEngine();
513
+ const question = String(msg.question || "").trim();
514
+ const site = String(msg.site || "this website");
515
+ const t0 = now();
516
+ const replyMsg = await engine.chat.completions.create({
517
+ messages: [
518
+ { role: "system", content: lib.planPrompt(site) },
519
+ { role: "user", content: question },
520
+ ],
521
+ temperature: 0,
522
+ max_tokens: 100,
523
+ response_format: { type: "json_object", schema: JSON.stringify(lib.PLAN_SCHEMA) },
524
+ extra_body: { enable_thinking: false },
525
+ });
526
+ const content = replyMsg && replyMsg.choices && replyMsg.choices[0] && replyMsg.choices[0].message ? replyMsg.choices[0].message.content : "";
527
+ const queries = lib.parsePlan(content, question);
528
+ reply({ type: "plan_result", requestId: msg.requestId, queries, ms: Math.round(now() - t0) });
381
529
  }
382
- if (engine) {
530
+
531
+ async function ask(msg, reply) {
532
+ requireEngine();
533
+ const requestId = msg.requestId;
534
+ const question = String(msg.question || "").trim();
535
+ const site = String(msg.site || "this website");
536
+ const evidence = Array.isArray(msg.evidence) ? msg.evidence.filter((e) => e && e.url) : [];
537
+ if (evidence.length === 0) {
538
+ reply({
539
+ type: "done",
540
+ requestId,
541
+ answer: lib.NOHIT,
542
+ citations: [],
543
+ nohit: true,
544
+ raw: "",
545
+ usage: { ttftMs: 0, totalMs: 0, tps: null, completionTokens: 0 },
546
+ });
547
+ return;
548
+ }
549
+ active = { requestId, aborted: false, reply };
550
+ const t0 = now();
551
+ let first = 0;
552
+ let text = "";
553
+ let usage = null;
383
554
  try {
384
- await engine.unload();
385
- } catch (_) {
386
- // ignore
555
+ const stream = await engine.chat.completions.create({
556
+ messages: [
557
+ { role: "system", content: lib.answerPrompt(site) },
558
+ { role: "user", content: lib.sourcesPrompt(evidence, question, EVIDENCE_CHARS) },
559
+ ],
560
+ stream: true,
561
+ stream_options: { include_usage: true },
562
+ temperature: 0,
563
+ frequency_penalty: 0.5,
564
+ presence_penalty: 0,
565
+ max_tokens: 220,
566
+ extra_body: { enable_thinking: false },
567
+ });
568
+ // Never break out of this loop: WebLLM releases its generation lock at
569
+ // the end of the async generator, and an early exit skips that release,
570
+ // hanging every later completion. After interruptGenerate() the stream
571
+ // ends by itself within one decode step; drop the tokens until then.
572
+ for await (const chunk of stream) {
573
+ if (active.aborted) continue;
574
+ const delta = chunk && chunk.choices && chunk.choices[0] && chunk.choices[0].delta ? chunk.choices[0].delta.content : null;
575
+ if (delta) {
576
+ if (!first) first = now();
577
+ text += delta;
578
+ reply({ type: "token", requestId, text: delta });
579
+ }
580
+ if (chunk && chunk.usage) usage = chunk.usage;
581
+ }
582
+ } finally {
583
+ const wasAborted = active && active.aborted;
584
+ active = null;
585
+ if (wasAborted) {
586
+ reply({ type: "aborted", requestId });
587
+ return;
588
+ }
387
589
  }
388
- engine = null;
389
- modelId = null;
590
+ const processed = lib.postProcessAnswer(text, evidence);
591
+ const totalMs = Math.round(now() - t0);
592
+ reply({
593
+ type: "done",
594
+ requestId,
595
+ answer: processed.answer,
596
+ citations: processed.citations,
597
+ nohit: processed.nohit,
598
+ raw: text,
599
+ usage: {
600
+ ttftMs: first ? Math.round(first - t0) : totalMs,
601
+ totalMs,
602
+ tps: usage && usage.extra && typeof usage.extra.decode_tokens_per_s === "number" ? Math.round(usage.extra.decode_tokens_per_s) : null,
603
+ completionTokens: usage ? usage.completion_tokens : null,
604
+ },
605
+ });
390
606
  }
391
- const created = await webllm.CreateMLCEngine(model, {
392
- initProgressCallback: (p) => {
393
- self.postMessage({
394
- type: "progress",
395
- text: p && p.text ? p.text : "Loading model…",
396
- progress: p && typeof p.progress === "number" ? p.progress : null,
397
- });
398
- },
399
- });
400
- engine = created;
401
- modelId = model;
402
- self.postMessage({ type: "loaded", model, loadMs: Math.round(performance.now() - t0) });
403
- })();
404
- try {
405
- await loading;
406
- } catch (err) {
407
- engine = null;
408
- modelId = null;
409
- postError(undefined, describe(err));
410
- } finally {
411
- loading = null;
412
- }
413
- }
414
-
415
- function requireEngine() {
416
- if (!engine) throw new Error("model not loaded");
417
- }
418
-
419
- async function plan(msg) {
420
- requireEngine();
421
- const question = String(msg.question || "").trim();
422
- const site = String(msg.site || "this website");
423
- const t0 = performance.now();
424
- const reply = await engine.chat.completions.create({
425
- messages: [
426
- { role: "system", content: lib.planPrompt(site) },
427
- { role: "user", content: question },
428
- ],
429
- temperature: 0,
430
- max_tokens: 100,
431
- response_format: { type: "json_object", schema: JSON.stringify(lib.PLAN_SCHEMA) },
432
- extra_body: { enable_thinking: false },
433
- });
434
- const content = reply && reply.choices && reply.choices[0] && reply.choices[0].message ? reply.choices[0].message.content : "";
435
- const queries = lib.parsePlan(content, question);
436
- self.postMessage({ type: "plan_result", requestId: msg.requestId, queries, ms: Math.round(performance.now() - t0) });
437
- }
438
-
439
- async function ask(msg) {
440
- requireEngine();
441
- const requestId = msg.requestId;
442
- const question = String(msg.question || "").trim();
443
- const site = String(msg.site || "this website");
444
- const evidence = Array.isArray(msg.evidence) ? msg.evidence.filter((e) => e && e.url) : [];
445
- if (evidence.length === 0) {
446
- self.postMessage({
447
- type: "done",
448
- requestId,
449
- answer: lib.NOHIT,
450
- citations: [],
451
- nohit: true,
452
- raw: "",
453
- usage: { ttftMs: 0, totalMs: 0, tps: null, completionTokens: 0 },
454
- });
455
- return;
456
- }
457
- active = { requestId, aborted: false };
458
- const t0 = performance.now();
459
- let first = 0;
460
- let text = "";
461
- let usage = null;
462
- try {
463
- const stream = await engine.chat.completions.create({
464
- messages: [
465
- { role: "system", content: lib.answerPrompt(site) },
466
- { role: "user", content: lib.sourcesPrompt(evidence, question, EVIDENCE_CHARS) },
467
- ],
468
- stream: true,
469
- stream_options: { include_usage: true },
470
- temperature: 0,
471
- frequency_penalty: 0.5,
472
- presence_penalty: 0,
473
- max_tokens: 220,
474
- extra_body: { enable_thinking: false },
475
- });
476
- // Never break out of this loop: WebLLM releases its generation lock at
477
- // the end of the async generator, and an early exit skips that release,
478
- // hanging every later completion. After interruptGenerate() the stream
479
- // ends by itself within one decode step; drop the tokens until then.
480
- for await (const chunk of stream) {
481
- if (active.aborted) continue;
482
- const delta = chunk && chunk.choices && chunk.choices[0] && chunk.choices[0].delta ? chunk.choices[0].delta.content : null;
483
- if (delta) {
484
- if (!first) first = performance.now();
485
- text += delta;
486
- self.postMessage({ type: "token", requestId, text: delta });
607
+
608
+ function abort(msg) {
609
+ console.debug("eddie agent engine: abort", msg.requestId, active ? active.requestId : null);
610
+ if (!active) return;
611
+ if (msg.requestId != null && msg.requestId !== active.requestId) return;
612
+ active.aborted = true;
613
+ try {
614
+ if (engine) engine.interruptGenerate();
615
+ } catch (err) {
616
+ console.warn("eddie agent: interrupt failed", err);
487
617
  }
488
- if (chunk && chunk.usage) usage = chunk.usage;
489
618
  }
490
- } finally {
491
- const wasAborted = active && active.aborted;
492
- active = null;
493
- if (wasAborted) {
494
- self.postMessage({ type: "aborted", requestId });
495
- return;
619
+
620
+ /** Abort the active run if it belongs to a page that went away. */
621
+ function abortIfOwner(reply) {
622
+ if (active && active.reply === reply) abort({ requestId: active.requestId });
623
+ }
624
+
625
+ function postError(reply, requestId, message) {
626
+ reply({ type: "error", requestId: requestId == null ? undefined : requestId, message });
496
627
  }
628
+
629
+ return { handle, state: snapshot, abortIfOwner };
497
630
  }
498
- const processed = lib.postProcessAnswer(text, evidence);
499
- const totalMs = Math.round(performance.now() - t0);
500
- self.postMessage({
501
- type: "done",
502
- requestId,
503
- answer: processed.answer,
504
- citations: processed.citations,
505
- nohit: processed.nohit,
506
- raw: text,
507
- usage: {
508
- ttftMs: first ? Math.round(first - t0) : totalMs,
509
- totalMs,
510
- tps: usage && usage.extra && typeof usage.extra.decode_tokens_per_s === "number" ? Math.round(usage.extra.decode_tokens_per_s) : null,
511
- completionTokens: usage ? usage.completion_tokens : null,
512
- },
513
- });
514
- }
515
-
516
- function abort(msg) {
517
- console.debug("eddie agent worker: abort", msg.requestId, active ? active.requestId : null);
518
- if (!active) return;
519
- if (msg.requestId != null && msg.requestId !== active.requestId) return;
520
- active.aborted = true;
521
- try {
522
- if (engine) engine.interruptGenerate();
523
- } catch (err) {
524
- console.warn("eddie agent: interrupt failed", err);
631
+
632
+ function describe(err) {
633
+ if (err == null) return "unknown error";
634
+ if (typeof err === "string") return err;
635
+ return err.message || String(err);
525
636
  }
526
- }
527
637
 
528
- function postError(requestId, message) {
529
- self.postMessage({ type: "error", requestId: requestId == null ? undefined : requestId, message });
530
- }
638
+ /** True for the engine's "not loaded" replies (the client re-runs load). */
639
+ function isModelNotLoadedMessage(message) {
640
+ return typeof message === "string" && message.indexOf(NOT_LOADED) === 0;
641
+ }
642
+
643
+ return { createAgentEngine, AGENT_NOT_LOADED: NOT_LOADED, isModelNotLoadedMessage };
644
+ });
645
+
646
+ // SPDX-License-Identifier: GPL-3.0-only
647
+
648
+ // Eddie agent worker (module worker, created on the first "Ask"): the
649
+ // fallback host when the service worker (eddie-sw.js) is unavailable, same
650
+ // protocol either way. The engine is widget/src/lib/agent-engine.js; this
651
+ // file binds it to a dedicated module worker and loads WebLLM with a dynamic
652
+ // import(). widget/build.sh concatenates widget/src/lib/agent.js and
653
+ // agent-engine.js ahead of this file (EddieLib).
654
+ //
655
+ // Protocol: see widget/README.md ("Worker protocol", agent section).
656
+
657
+ "use strict";
658
+
659
+ const WEBLLM_URL = "https://esm.run/@mlc-ai/web-llm@0.2.84";
660
+
661
+ const lib = EddieLib;
531
662
 
532
- function describe(err) {
533
- if (err == null) return "unknown error";
534
- if (typeof err === "string") return err;
535
- return err.message || String(err);
536
- }
663
+ const engine = lib.createAgentEngine({
664
+ post: (message) => self.postMessage(message),
665
+ loadWebLLM: () => import(WEBLLM_URL),
666
+ });
667
+
668
+ self.onmessage = function (e) {
669
+ engine.handle(e.data || {}, (message) => self.postMessage(message));
670
+ };