eddie-jekyll 0.4.0 → 0.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/assets/assets.list +46 -0
- data/assets/eddie-agent-worker.js +425 -291
- data/assets/eddie-boot.js +234 -0
- data/assets/eddie-dense-esm.js +624 -0
- data/assets/eddie-dense.js +632 -0
- data/assets/eddie-dense.wasm +0 -0
- data/assets/eddie-lite-esm.js +578 -0
- data/assets/{eddie-wasm.js → eddie-lite.js} +105 -61
- data/assets/eddie-lite.wasm +0 -0
- data/assets/eddie-sw-agent.js +846 -0
- data/assets/eddie-sw-dense.js +1873 -0
- data/assets/eddie-sw-gpu.js +1873 -0
- data/assets/eddie-sw-lite.js +1872 -0
- data/assets/eddie-transformers-sw.js +30926 -0
- data/assets/eddie-widget.js +1908 -449
- data/assets/eddie-worker.js +1205 -652
- data/scripts/install.sh +17 -9
- metadata +16 -5
- data/assets/eddie.wasm +0 -0
|
@@ -1,12 +1,14 @@
|
|
|
1
1
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
2
2
|
// Generated by widget/build.sh from widget/src/eddie-agent-worker.js and widget/src/lib/*.js; edit those instead.
|
|
3
3
|
"use strict";
|
|
4
|
+
const EDDIE_ASSET_VERSION = "c0fdc503e01a";
|
|
4
5
|
const EddieLib = {};
|
|
5
6
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
6
7
|
|
|
7
|
-
// Agent helpers
|
|
8
|
-
//
|
|
9
|
-
//
|
|
8
|
+
// Agent helpers the widget shows in its UI: model selection, evidence
|
|
9
|
+
// assembly, stream display and the FAQ gate. Pure functions; no WebLLM, no
|
|
10
|
+
// DOM. The prompts and answer post-processing that run beside the model
|
|
11
|
+
// live in agent-llm.js (bundled only into the agent engine's hosts).
|
|
10
12
|
|
|
11
13
|
(function (factory) {
|
|
12
14
|
const api = factory();
|
|
@@ -21,7 +23,6 @@ const EddieLib = {};
|
|
|
21
23
|
"use strict";
|
|
22
24
|
|
|
23
25
|
const NOHIT = "The site doesn't cover that.";
|
|
24
|
-
const NOHIT_RE = /\bthe site (?:doesn['’]t|does not|didn['’]t|did not) cover (?:that|this|it)\.?/gi;
|
|
25
26
|
|
|
26
27
|
const AGENT_MODEL_SIZES = {
|
|
27
28
|
"Qwen3.5-0.8B": 0.4e9,
|
|
@@ -30,22 +31,6 @@ const EddieLib = {};
|
|
|
30
31
|
};
|
|
31
32
|
const TWO_GIB = 2 * 1024 * 1024 * 1024;
|
|
32
33
|
|
|
33
|
-
const PLAN_SCHEMA = {
|
|
34
|
-
type: "object",
|
|
35
|
-
properties: {
|
|
36
|
-
queries: { type: "array", items: { type: "string" }, minItems: 1, maxItems: 3 },
|
|
37
|
-
},
|
|
38
|
-
required: ["queries"],
|
|
39
|
-
};
|
|
40
|
-
|
|
41
|
-
function planPrompt(site) {
|
|
42
|
-
return `You write search queries for a site search engine. The site is ${site}. Reply with JSON only: {"queries": ["..."]}. Give 1 to 3 different short keyword queries (2 to 5 words each, no punctuation) that a site search engine would match against page text. Each query must be different. Do not answer the question.`;
|
|
43
|
-
}
|
|
44
|
-
|
|
45
|
-
function answerPrompt(site) {
|
|
46
|
-
return `You answer visitor questions about ${site} using only the numbered sources below the question. Answer the question directly in the first sentence. Write 1 to 3 sentences in your own words; never repeat a source's wording. End each sentence with the numbers of the sources it comes from, like [2] or [1][3]. Never cite a number that is not in the list. Do not add calculations or inferences that are not in the sources. If no source answers the question, your entire reply is: ${NOHIT} Never mix that sentence with an answer. Never use outside knowledge.`;
|
|
47
|
-
}
|
|
48
|
-
|
|
49
34
|
/** Strip the WebLLM variant suffix to get the family name shown to visitors. */
|
|
50
35
|
function baseModelId(id) {
|
|
51
36
|
return String(id).replace(/-q\d+f(16|32)_\d+-MLC$/i, "");
|
|
@@ -58,7 +43,10 @@ const EddieLib = {};
|
|
|
58
43
|
|
|
59
44
|
/**
|
|
60
45
|
* Choose the WebLLM model id.
|
|
61
|
-
* opts: { mode: "auto"|"quality"|<id>, maxBufferSize, isMobile, hasF16 }
|
|
46
|
+
* opts: { mode: "auto"|"light"|"quality"|<id>, maxBufferSize, isMobile, hasF16 }
|
|
47
|
+
*
|
|
48
|
+
* "light" and "quality" are the two sizes the settings panel offers by
|
|
49
|
+
* name; "auto" picks between them from the adapter's buffer limit.
|
|
62
50
|
*/
|
|
63
51
|
function selectAgentModel(opts) {
|
|
64
52
|
const o = opts || {};
|
|
@@ -70,6 +58,8 @@ const EddieLib = {};
|
|
|
70
58
|
base = big ? "Qwen3.5-2B" : "Qwen3.5-0.8B";
|
|
71
59
|
} else if (mode === "quality") {
|
|
72
60
|
base = "Qwen3.5-2B";
|
|
61
|
+
} else if (mode === "light") {
|
|
62
|
+
base = "Qwen3.5-0.8B";
|
|
73
63
|
} else {
|
|
74
64
|
return { id: mode, base: baseModelId(mode), sizeBytes: agentModelBytes(mode), explicit: true };
|
|
75
65
|
}
|
|
@@ -85,14 +75,6 @@ const EddieLib = {};
|
|
|
85
75
|
return /Mobi|Android|iPhone|iPad|iPod|Windows Phone/i.test(n.userAgent || "");
|
|
86
76
|
}
|
|
87
77
|
|
|
88
|
-
/** Remove <think>…</think> blocks; a dangling <think> loses only the tag. */
|
|
89
|
-
function stripThink(text) {
|
|
90
|
-
if (!text) return "";
|
|
91
|
-
let out = String(text).replace(/<think>[\s\S]*?<\/think>/g, "");
|
|
92
|
-
out = out.replace(/<think>/g, "").replace(/<\/think>/g, "");
|
|
93
|
-
return out.trim();
|
|
94
|
-
}
|
|
95
|
-
|
|
96
78
|
/**
|
|
97
79
|
* Display text for a partial stream: complete think blocks removed, and
|
|
98
80
|
* anything after an unclosed <think> hidden until it closes.
|
|
@@ -105,6 +87,123 @@ const EddieLib = {};
|
|
|
105
87
|
return out.replace(/^\s+/, "");
|
|
106
88
|
}
|
|
107
89
|
|
|
90
|
+
function urlKey(url) {
|
|
91
|
+
return String(url || "").replace(/#.*$/, "").replace(/\/+$/, "").toLowerCase();
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Round-robin merge of several result lists, deduplicated by URL, at most
|
|
96
|
+
* `max` items. Each result keeps its own fields.
|
|
97
|
+
*/
|
|
98
|
+
function mergeEvidence(lists, max) {
|
|
99
|
+
const limit = max == null ? 6 : max;
|
|
100
|
+
const seen = new Set();
|
|
101
|
+
const out = [];
|
|
102
|
+
const arrays = (lists || []).map((l) => (Array.isArray(l) ? l : []));
|
|
103
|
+
const longest = arrays.reduce((n, l) => Math.max(n, l.length), 0);
|
|
104
|
+
for (let i = 0; i < longest && out.length < limit; i++) {
|
|
105
|
+
for (const list of arrays) {
|
|
106
|
+
if (out.length >= limit) break;
|
|
107
|
+
const r = list[i];
|
|
108
|
+
if (!r || !r.url) continue;
|
|
109
|
+
const key = urlKey(r.url);
|
|
110
|
+
if (seen.has(key)) continue;
|
|
111
|
+
seen.add(key);
|
|
112
|
+
out.push(r);
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
return out;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
// FAQ card gate: prefer the WASM's fused `confident` flag (qa_lookup v0.4.1+);
|
|
119
|
+
// older indexes only carry a dense score, so fall back to a plain cutoff.
|
|
120
|
+
function faqPasses(hit, qaMode) {
|
|
121
|
+
if (!hit || typeof hit !== "object") return false;
|
|
122
|
+
if (qaMode === "off") return false;
|
|
123
|
+
if (qaMode === "always") return true;
|
|
124
|
+
if (typeof hit.confident === "boolean") return hit.confident;
|
|
125
|
+
return typeof hit.score === "number" && hit.score >= 0.5;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// Turn confident QA hits into agent evidence items ("Q: … A: …") so the
|
|
129
|
+
// answer model sees the FAQ lane, not only chunk text.
|
|
130
|
+
function qaEvidence(hits, max) {
|
|
131
|
+
const limit = max == null ? 2 : max;
|
|
132
|
+
const out = [];
|
|
133
|
+
for (const h of Array.isArray(hits) ? hits : []) {
|
|
134
|
+
if (out.length >= limit) break;
|
|
135
|
+
if (!faqPasses(h, "auto")) continue;
|
|
136
|
+
const q = String(h.question || "").trim();
|
|
137
|
+
const a = String(h.answer || "").trim();
|
|
138
|
+
if (!q || !a) continue;
|
|
139
|
+
out.push({ title: "FAQ: " + q, url: h.source_url || "", text: "Q: " + q + "\nA: " + a, faq: true });
|
|
140
|
+
}
|
|
141
|
+
return out;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
return {
|
|
145
|
+
faqPasses,
|
|
146
|
+
qaEvidence,
|
|
147
|
+
NOHIT,
|
|
148
|
+
AGENT_MODEL_SIZES,
|
|
149
|
+
baseModelId,
|
|
150
|
+
agentModelBytes,
|
|
151
|
+
selectAgentModel,
|
|
152
|
+
isMobileDevice,
|
|
153
|
+
visibleStreamText,
|
|
154
|
+
mergeEvidence,
|
|
155
|
+
};
|
|
156
|
+
});
|
|
157
|
+
|
|
158
|
+
// SPDX-License-Identifier: GPL-3.0-only
|
|
159
|
+
|
|
160
|
+
// Agent LLM helpers used only where the model runs (agent-engine.js hosts:
|
|
161
|
+
// the agent service worker and the page-side agent worker): prompts, plan
|
|
162
|
+
// parsing and answer post-processing. Split out of agent.js so the widget
|
|
163
|
+
// bundle carries only the model-selection and evidence helpers it shows in
|
|
164
|
+
// the UI. Bundles that include this file always include agent.js first.
|
|
165
|
+
|
|
166
|
+
(function (factory) {
|
|
167
|
+
const api = factory();
|
|
168
|
+
if (typeof module === "object" && module && module.exports) {
|
|
169
|
+
module.exports = api;
|
|
170
|
+
} else if (typeof EddieLib === "object" && EddieLib) {
|
|
171
|
+
Object.assign(EddieLib, api);
|
|
172
|
+
} else {
|
|
173
|
+
globalThis.EddieLib = Object.assign(globalThis.EddieLib || {}, api);
|
|
174
|
+
}
|
|
175
|
+
})(function () {
|
|
176
|
+
"use strict";
|
|
177
|
+
|
|
178
|
+
// The fallback sentence lives in agent.js (the widget shows it too); the
|
|
179
|
+
// bundles concatenate agent.js ahead of this file, node requires it.
|
|
180
|
+
const NOHIT = (typeof module === "object" && module && module.exports ? require("./agent.js") : EddieLib).NOHIT;
|
|
181
|
+
const NOHIT_RE = /\bthe site (?:doesn['’]t|does not|didn['’]t|did not) cover (?:that|this|it)\.?/gi;
|
|
182
|
+
|
|
183
|
+
const PLAN_SCHEMA = {
|
|
184
|
+
type: "object",
|
|
185
|
+
properties: {
|
|
186
|
+
queries: { type: "array", items: { type: "string" }, minItems: 1, maxItems: 3 },
|
|
187
|
+
},
|
|
188
|
+
required: ["queries"],
|
|
189
|
+
};
|
|
190
|
+
|
|
191
|
+
function planPrompt(site) {
|
|
192
|
+
return `You write search queries for a site search engine. The site is ${site}. Reply with JSON only: {"queries": ["..."]}. Give 1 to 3 different short keyword queries (2 to 5 words each, no punctuation) that a site search engine would match against page text. Each query must be different. Do not answer the question.`;
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
function answerPrompt(site) {
|
|
196
|
+
return `You answer visitor questions about ${site} using only the numbered sources below the question. Answer the question directly in the first sentence. Write 1 to 3 sentences in your own words; never repeat a source's wording. End each sentence with the numbers of the sources it comes from, like [2] or [1][3]. Never cite a number that is not in the list. Do not add calculations or inferences that are not in the sources. If no source answers the question, your entire reply is: ${NOHIT} Never mix that sentence with an answer. Never use outside knowledge.`;
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/** Remove <think>…</think> blocks; a dangling <think> loses only the tag. */
|
|
200
|
+
function stripThink(text) {
|
|
201
|
+
if (!text) return "";
|
|
202
|
+
let out = String(text).replace(/<think>[\s\S]*?<\/think>/g, "");
|
|
203
|
+
out = out.replace(/<think>/g, "").replace(/<\/think>/g, "");
|
|
204
|
+
return out.trim();
|
|
205
|
+
}
|
|
206
|
+
|
|
108
207
|
function extractJsonObject(text) {
|
|
109
208
|
const s = String(text);
|
|
110
209
|
const start = s.indexOf("{");
|
|
@@ -149,34 +248,6 @@ const EddieLib = {};
|
|
|
149
248
|
return out;
|
|
150
249
|
}
|
|
151
250
|
|
|
152
|
-
function urlKey(url) {
|
|
153
|
-
return String(url || "").replace(/#.*$/, "").replace(/\/+$/, "").toLowerCase();
|
|
154
|
-
}
|
|
155
|
-
|
|
156
|
-
/**
|
|
157
|
-
* Round-robin merge of several result lists, deduplicated by URL, at most
|
|
158
|
-
* `max` items. Each result keeps its own fields.
|
|
159
|
-
*/
|
|
160
|
-
function mergeEvidence(lists, max) {
|
|
161
|
-
const limit = max == null ? 6 : max;
|
|
162
|
-
const seen = new Set();
|
|
163
|
-
const out = [];
|
|
164
|
-
const arrays = (lists || []).map((l) => (Array.isArray(l) ? l : []));
|
|
165
|
-
const longest = arrays.reduce((n, l) => Math.max(n, l.length), 0);
|
|
166
|
-
for (let i = 0; i < longest && out.length < limit; i++) {
|
|
167
|
-
for (const list of arrays) {
|
|
168
|
-
if (out.length >= limit) break;
|
|
169
|
-
const r = list[i];
|
|
170
|
-
if (!r || !r.url) continue;
|
|
171
|
-
const key = urlKey(r.url);
|
|
172
|
-
if (seen.has(key)) continue;
|
|
173
|
-
seen.add(key);
|
|
174
|
-
out.push(r);
|
|
175
|
-
}
|
|
176
|
-
}
|
|
177
|
-
return out;
|
|
178
|
-
}
|
|
179
|
-
|
|
180
251
|
/** Cut to `max` characters at a word boundary, with an ellipsis. */
|
|
181
252
|
function truncateText(text, max) {
|
|
182
253
|
const limit = max == null ? 700 : max;
|
|
@@ -266,20 +337,12 @@ const EddieLib = {};
|
|
|
266
337
|
}
|
|
267
338
|
|
|
268
339
|
return {
|
|
269
|
-
NOHIT,
|
|
270
|
-
AGENT_MODEL_SIZES,
|
|
271
340
|
PLAN_SCHEMA,
|
|
272
341
|
planPrompt,
|
|
273
342
|
answerPrompt,
|
|
274
343
|
sourcesPrompt,
|
|
275
|
-
baseModelId,
|
|
276
|
-
agentModelBytes,
|
|
277
|
-
selectAgentModel,
|
|
278
|
-
isMobileDevice,
|
|
279
344
|
stripThink,
|
|
280
|
-
visibleStreamText,
|
|
281
345
|
parsePlan,
|
|
282
|
-
mergeEvidence,
|
|
283
346
|
truncateText,
|
|
284
347
|
formatEvidence,
|
|
285
348
|
postProcessAnswer,
|
|
@@ -288,249 +351,320 @@ const EddieLib = {};
|
|
|
288
351
|
|
|
289
352
|
// SPDX-License-Identifier: GPL-3.0-only
|
|
290
353
|
|
|
291
|
-
// Eddie agent
|
|
354
|
+
// Eddie agent engine, host-independent.
|
|
292
355
|
//
|
|
293
|
-
//
|
|
294
|
-
//
|
|
295
|
-
// widget/src/lib/agent.js ahead of this file (EddieLib).
|
|
356
|
+
// Everything the agent worker does (WebLLM load, plan, ask/stream, abort with
|
|
357
|
+
// the drained-stream lock rule) behind an `env` the host supplies:
|
|
296
358
|
//
|
|
297
|
-
//
|
|
298
|
-
//
|
|
299
|
-
//
|
|
300
|
-
//
|
|
301
|
-
//
|
|
302
|
-
//
|
|
303
|
-
//
|
|
304
|
-
//
|
|
305
|
-
//
|
|
306
|
-
//
|
|
307
|
-
// done {requestId, answer, citations: [{n, url, title}], nohit, usage}
|
|
308
|
-
// aborted {requestId}
|
|
309
|
-
// error {requestId?, message}
|
|
310
|
-
|
|
311
|
-
"use strict";
|
|
359
|
+
// createAgentEngine({
|
|
360
|
+
// post(message) broadcast sink (unused today; progress goes to the loaders)
|
|
361
|
+
// loadWebLLM() -> Promise of the WebLLM module
|
|
362
|
+
// now() optional clock (tests)
|
|
363
|
+
// })
|
|
364
|
+
//
|
|
365
|
+
// `engine.handle(msg, reply)` dispatches one protocol message; `reply` is
|
|
366
|
+
// the sink for that message's answers (progress/loaded for `load`,
|
|
367
|
+
// plan_result, token/done/aborted for `ask`, error). Message shapes are in
|
|
368
|
+
// widget/README.md ("Worker protocol", agent section).
|
|
312
369
|
|
|
313
|
-
|
|
314
|
-
const
|
|
370
|
+
(function (factory) {
|
|
371
|
+
const api = factory();
|
|
372
|
+
if (typeof module === "object" && module && module.exports) {
|
|
373
|
+
module.exports = api;
|
|
374
|
+
} else if (typeof EddieLib === "object" && EddieLib) {
|
|
375
|
+
Object.assign(EddieLib, api);
|
|
376
|
+
} else {
|
|
377
|
+
globalThis.EddieLib = Object.assign(globalThis.EddieLib || {}, api);
|
|
378
|
+
}
|
|
379
|
+
})(function () {
|
|
380
|
+
"use strict";
|
|
315
381
|
|
|
316
|
-
const
|
|
382
|
+
const EVIDENCE_CHARS = 700;
|
|
383
|
+
const NOT_LOADED = "model not loaded";
|
|
384
|
+
|
|
385
|
+
function createAgentEngine(env) {
|
|
386
|
+
const lib = typeof EddieLib === "object" && EddieLib ? EddieLib : env.lib;
|
|
387
|
+
const now = env.now || (() => (typeof performance === "object" && performance.now ? performance.now() : Date.now()));
|
|
388
|
+
|
|
389
|
+
let webllm = null;
|
|
390
|
+
let engine = null;
|
|
391
|
+
let modelId = null;
|
|
392
|
+
let loading = null; // { model, promise, waiters: [reply] }
|
|
393
|
+
let active = null; // { requestId, aborted, reply }
|
|
394
|
+
let queue = Promise.resolve();
|
|
395
|
+
|
|
396
|
+
function handle(msg, reply) {
|
|
397
|
+
const m = msg || {};
|
|
398
|
+
const out = reply || env.post;
|
|
399
|
+
switch (m.type) {
|
|
400
|
+
case "load":
|
|
401
|
+
return load(m, out);
|
|
402
|
+
case "plan":
|
|
403
|
+
return enqueue(() => plan(m, out), m.requestId, out);
|
|
404
|
+
case "ask":
|
|
405
|
+
return enqueue(() => ask(m, out), m.requestId, out);
|
|
406
|
+
case "abort":
|
|
407
|
+
abort(m);
|
|
408
|
+
return Promise.resolve();
|
|
409
|
+
default:
|
|
410
|
+
postError(out, m.requestId, `unknown message type ${String(m.type)}`);
|
|
411
|
+
return Promise.resolve();
|
|
412
|
+
}
|
|
413
|
+
}
|
|
317
414
|
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
|
|
322
|
-
|
|
323
|
-
|
|
415
|
+
/** Snapshot for the service worker's `state` reply. */
|
|
416
|
+
function snapshot() {
|
|
417
|
+
return {
|
|
418
|
+
model: modelId,
|
|
419
|
+
loaded: !!engine,
|
|
420
|
+
loading: loading ? loading.model : null,
|
|
421
|
+
active: active ? active.requestId : null,
|
|
422
|
+
};
|
|
423
|
+
}
|
|
324
424
|
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
|
|
335
|
-
enqueue(() => ask(msg), msg.requestId);
|
|
336
|
-
break;
|
|
337
|
-
case "abort":
|
|
338
|
-
abort(msg);
|
|
339
|
-
break;
|
|
340
|
-
default:
|
|
341
|
-
postError(msg.requestId, `unknown message type ${String(msg.type)}`);
|
|
342
|
-
}
|
|
343
|
-
};
|
|
425
|
+
function enqueue(fn, requestId, reply) {
|
|
426
|
+
queue = queue
|
|
427
|
+
.then(() => {
|
|
428
|
+
console.debug("eddie agent engine: start", requestId);
|
|
429
|
+
return fn();
|
|
430
|
+
})
|
|
431
|
+
.catch((err) => postError(reply, requestId, describe(err)))
|
|
432
|
+
.then(() => console.debug("eddie agent engine: end", requestId));
|
|
433
|
+
return queue;
|
|
434
|
+
}
|
|
344
435
|
|
|
345
|
-
function
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
436
|
+
async function load(msg, reply) {
|
|
437
|
+
const model = String(msg.model || "");
|
|
438
|
+
if (!model) {
|
|
439
|
+
postError(reply, undefined, "load: model is required");
|
|
440
|
+
return;
|
|
441
|
+
}
|
|
442
|
+
if (engine && modelId === model) {
|
|
443
|
+
reply({ type: "loaded", model, loadMs: 0, cached: true });
|
|
444
|
+
return;
|
|
445
|
+
}
|
|
446
|
+
if (loading) {
|
|
447
|
+
// Another page (or an earlier message) is loading. Same model: join
|
|
448
|
+
// it, the fan-out delivers progress and loaded/error to us too.
|
|
449
|
+
// Different model: wait for it to settle, then load ours.
|
|
450
|
+
const join = loading.model === model;
|
|
451
|
+
if (join) loading.waiters.push(reply);
|
|
452
|
+
try {
|
|
453
|
+
await loading.promise;
|
|
454
|
+
} catch (_) {
|
|
455
|
+
// the fan-out already reported the error to the joined waiters
|
|
456
|
+
}
|
|
457
|
+
if (join) return;
|
|
458
|
+
if (engine && modelId === model) {
|
|
459
|
+
reply({ type: "loaded", model, loadMs: 0, cached: true });
|
|
460
|
+
return;
|
|
461
|
+
}
|
|
462
|
+
}
|
|
463
|
+
const job = { model, waiters: [reply], promise: null };
|
|
464
|
+
const fanout = (message) => {
|
|
465
|
+
for (const w of job.waiters) w(message);
|
|
466
|
+
};
|
|
467
|
+
job.promise = (async () => {
|
|
468
|
+
const t0 = now();
|
|
469
|
+
if (!webllm) {
|
|
470
|
+
fanout({ type: "progress", text: "Loading the WebLLM runtime…", progress: 0 });
|
|
471
|
+
webllm = await env.loadWebLLM();
|
|
472
|
+
}
|
|
473
|
+
if (engine) {
|
|
474
|
+
try {
|
|
475
|
+
await engine.unload();
|
|
476
|
+
} catch (_) {
|
|
477
|
+
// ignore
|
|
478
|
+
}
|
|
479
|
+
engine = null;
|
|
480
|
+
modelId = null;
|
|
481
|
+
}
|
|
482
|
+
const created = await webllm.CreateMLCEngine(model, {
|
|
483
|
+
initProgressCallback: (p) => {
|
|
484
|
+
fanout({
|
|
485
|
+
type: "progress",
|
|
486
|
+
text: p && p.text ? p.text : "Loading model…",
|
|
487
|
+
progress: p && typeof p.progress === "number" ? p.progress : null,
|
|
488
|
+
});
|
|
489
|
+
},
|
|
490
|
+
});
|
|
491
|
+
engine = created;
|
|
492
|
+
modelId = model;
|
|
493
|
+
fanout({ type: "loaded", model, loadMs: Math.round(now() - t0) });
|
|
494
|
+
})();
|
|
495
|
+
loading = job;
|
|
496
|
+
try {
|
|
497
|
+
await job.promise;
|
|
498
|
+
} catch (err) {
|
|
499
|
+
engine = null;
|
|
500
|
+
modelId = null;
|
|
501
|
+
fanout({ type: "error", requestId: undefined, message: describe(err) });
|
|
502
|
+
} finally {
|
|
503
|
+
if (loading === job) loading = null;
|
|
504
|
+
}
|
|
370
505
|
}
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
506
|
+
|
|
507
|
+
function requireEngine() {
|
|
508
|
+
if (!engine) throw new Error(NOT_LOADED);
|
|
374
509
|
}
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
510
|
+
|
|
511
|
+
async function plan(msg, reply) {
|
|
512
|
+
requireEngine();
|
|
513
|
+
const question = String(msg.question || "").trim();
|
|
514
|
+
const site = String(msg.site || "this website");
|
|
515
|
+
const t0 = now();
|
|
516
|
+
const replyMsg = await engine.chat.completions.create({
|
|
517
|
+
messages: [
|
|
518
|
+
{ role: "system", content: lib.planPrompt(site) },
|
|
519
|
+
{ role: "user", content: question },
|
|
520
|
+
],
|
|
521
|
+
temperature: 0,
|
|
522
|
+
max_tokens: 100,
|
|
523
|
+
response_format: { type: "json_object", schema: JSON.stringify(lib.PLAN_SCHEMA) },
|
|
524
|
+
extra_body: { enable_thinking: false },
|
|
525
|
+
});
|
|
526
|
+
const content = replyMsg && replyMsg.choices && replyMsg.choices[0] && replyMsg.choices[0].message ? replyMsg.choices[0].message.content : "";
|
|
527
|
+
const queries = lib.parsePlan(content, question);
|
|
528
|
+
reply({ type: "plan_result", requestId: msg.requestId, queries, ms: Math.round(now() - t0) });
|
|
381
529
|
}
|
|
382
|
-
|
|
530
|
+
|
|
531
|
+
async function ask(msg, reply) {
|
|
532
|
+
requireEngine();
|
|
533
|
+
const requestId = msg.requestId;
|
|
534
|
+
const question = String(msg.question || "").trim();
|
|
535
|
+
const site = String(msg.site || "this website");
|
|
536
|
+
const evidence = Array.isArray(msg.evidence) ? msg.evidence.filter((e) => e && e.url) : [];
|
|
537
|
+
if (evidence.length === 0) {
|
|
538
|
+
reply({
|
|
539
|
+
type: "done",
|
|
540
|
+
requestId,
|
|
541
|
+
answer: lib.NOHIT,
|
|
542
|
+
citations: [],
|
|
543
|
+
nohit: true,
|
|
544
|
+
raw: "",
|
|
545
|
+
usage: { ttftMs: 0, totalMs: 0, tps: null, completionTokens: 0 },
|
|
546
|
+
});
|
|
547
|
+
return;
|
|
548
|
+
}
|
|
549
|
+
active = { requestId, aborted: false, reply };
|
|
550
|
+
const t0 = now();
|
|
551
|
+
let first = 0;
|
|
552
|
+
let text = "";
|
|
553
|
+
let usage = null;
|
|
383
554
|
try {
|
|
384
|
-
await engine.
|
|
385
|
-
|
|
386
|
-
|
|
555
|
+
const stream = await engine.chat.completions.create({
|
|
556
|
+
messages: [
|
|
557
|
+
{ role: "system", content: lib.answerPrompt(site) },
|
|
558
|
+
{ role: "user", content: lib.sourcesPrompt(evidence, question, EVIDENCE_CHARS) },
|
|
559
|
+
],
|
|
560
|
+
stream: true,
|
|
561
|
+
stream_options: { include_usage: true },
|
|
562
|
+
temperature: 0,
|
|
563
|
+
frequency_penalty: 0.5,
|
|
564
|
+
presence_penalty: 0,
|
|
565
|
+
max_tokens: 220,
|
|
566
|
+
extra_body: { enable_thinking: false },
|
|
567
|
+
});
|
|
568
|
+
// Never break out of this loop: WebLLM releases its generation lock at
|
|
569
|
+
// the end of the async generator, and an early exit skips that release,
|
|
570
|
+
// hanging every later completion. After interruptGenerate() the stream
|
|
571
|
+
// ends by itself within one decode step; drop the tokens until then.
|
|
572
|
+
for await (const chunk of stream) {
|
|
573
|
+
if (active.aborted) continue;
|
|
574
|
+
const delta = chunk && chunk.choices && chunk.choices[0] && chunk.choices[0].delta ? chunk.choices[0].delta.content : null;
|
|
575
|
+
if (delta) {
|
|
576
|
+
if (!first) first = now();
|
|
577
|
+
text += delta;
|
|
578
|
+
reply({ type: "token", requestId, text: delta });
|
|
579
|
+
}
|
|
580
|
+
if (chunk && chunk.usage) usage = chunk.usage;
|
|
581
|
+
}
|
|
582
|
+
} finally {
|
|
583
|
+
const wasAborted = active && active.aborted;
|
|
584
|
+
active = null;
|
|
585
|
+
if (wasAborted) {
|
|
586
|
+
reply({ type: "aborted", requestId });
|
|
587
|
+
return;
|
|
588
|
+
}
|
|
387
589
|
}
|
|
388
|
-
|
|
389
|
-
|
|
590
|
+
const processed = lib.postProcessAnswer(text, evidence);
|
|
591
|
+
const totalMs = Math.round(now() - t0);
|
|
592
|
+
reply({
|
|
593
|
+
type: "done",
|
|
594
|
+
requestId,
|
|
595
|
+
answer: processed.answer,
|
|
596
|
+
citations: processed.citations,
|
|
597
|
+
nohit: processed.nohit,
|
|
598
|
+
raw: text,
|
|
599
|
+
usage: {
|
|
600
|
+
ttftMs: first ? Math.round(first - t0) : totalMs,
|
|
601
|
+
totalMs,
|
|
602
|
+
tps: usage && usage.extra && typeof usage.extra.decode_tokens_per_s === "number" ? Math.round(usage.extra.decode_tokens_per_s) : null,
|
|
603
|
+
completionTokens: usage ? usage.completion_tokens : null,
|
|
604
|
+
},
|
|
605
|
+
});
|
|
390
606
|
}
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
modelId = model;
|
|
402
|
-
self.postMessage({ type: "loaded", model, loadMs: Math.round(performance.now() - t0) });
|
|
403
|
-
})();
|
|
404
|
-
try {
|
|
405
|
-
await loading;
|
|
406
|
-
} catch (err) {
|
|
407
|
-
engine = null;
|
|
408
|
-
modelId = null;
|
|
409
|
-
postError(undefined, describe(err));
|
|
410
|
-
} finally {
|
|
411
|
-
loading = null;
|
|
412
|
-
}
|
|
413
|
-
}
|
|
414
|
-
|
|
415
|
-
function requireEngine() {
|
|
416
|
-
if (!engine) throw new Error("model not loaded");
|
|
417
|
-
}
|
|
418
|
-
|
|
419
|
-
async function plan(msg) {
|
|
420
|
-
requireEngine();
|
|
421
|
-
const question = String(msg.question || "").trim();
|
|
422
|
-
const site = String(msg.site || "this website");
|
|
423
|
-
const t0 = performance.now();
|
|
424
|
-
const reply = await engine.chat.completions.create({
|
|
425
|
-
messages: [
|
|
426
|
-
{ role: "system", content: lib.planPrompt(site) },
|
|
427
|
-
{ role: "user", content: question },
|
|
428
|
-
],
|
|
429
|
-
temperature: 0,
|
|
430
|
-
max_tokens: 100,
|
|
431
|
-
response_format: { type: "json_object", schema: JSON.stringify(lib.PLAN_SCHEMA) },
|
|
432
|
-
extra_body: { enable_thinking: false },
|
|
433
|
-
});
|
|
434
|
-
const content = reply && reply.choices && reply.choices[0] && reply.choices[0].message ? reply.choices[0].message.content : "";
|
|
435
|
-
const queries = lib.parsePlan(content, question);
|
|
436
|
-
self.postMessage({ type: "plan_result", requestId: msg.requestId, queries, ms: Math.round(performance.now() - t0) });
|
|
437
|
-
}
|
|
438
|
-
|
|
439
|
-
async function ask(msg) {
|
|
440
|
-
requireEngine();
|
|
441
|
-
const requestId = msg.requestId;
|
|
442
|
-
const question = String(msg.question || "").trim();
|
|
443
|
-
const site = String(msg.site || "this website");
|
|
444
|
-
const evidence = Array.isArray(msg.evidence) ? msg.evidence.filter((e) => e && e.url) : [];
|
|
445
|
-
if (evidence.length === 0) {
|
|
446
|
-
self.postMessage({
|
|
447
|
-
type: "done",
|
|
448
|
-
requestId,
|
|
449
|
-
answer: lib.NOHIT,
|
|
450
|
-
citations: [],
|
|
451
|
-
nohit: true,
|
|
452
|
-
raw: "",
|
|
453
|
-
usage: { ttftMs: 0, totalMs: 0, tps: null, completionTokens: 0 },
|
|
454
|
-
});
|
|
455
|
-
return;
|
|
456
|
-
}
|
|
457
|
-
active = { requestId, aborted: false };
|
|
458
|
-
const t0 = performance.now();
|
|
459
|
-
let first = 0;
|
|
460
|
-
let text = "";
|
|
461
|
-
let usage = null;
|
|
462
|
-
try {
|
|
463
|
-
const stream = await engine.chat.completions.create({
|
|
464
|
-
messages: [
|
|
465
|
-
{ role: "system", content: lib.answerPrompt(site) },
|
|
466
|
-
{ role: "user", content: lib.sourcesPrompt(evidence, question, EVIDENCE_CHARS) },
|
|
467
|
-
],
|
|
468
|
-
stream: true,
|
|
469
|
-
stream_options: { include_usage: true },
|
|
470
|
-
temperature: 0,
|
|
471
|
-
frequency_penalty: 0.5,
|
|
472
|
-
presence_penalty: 0,
|
|
473
|
-
max_tokens: 220,
|
|
474
|
-
extra_body: { enable_thinking: false },
|
|
475
|
-
});
|
|
476
|
-
// Never break out of this loop: WebLLM releases its generation lock at
|
|
477
|
-
// the end of the async generator, and an early exit skips that release,
|
|
478
|
-
// hanging every later completion. After interruptGenerate() the stream
|
|
479
|
-
// ends by itself within one decode step; drop the tokens until then.
|
|
480
|
-
for await (const chunk of stream) {
|
|
481
|
-
if (active.aborted) continue;
|
|
482
|
-
const delta = chunk && chunk.choices && chunk.choices[0] && chunk.choices[0].delta ? chunk.choices[0].delta.content : null;
|
|
483
|
-
if (delta) {
|
|
484
|
-
if (!first) first = performance.now();
|
|
485
|
-
text += delta;
|
|
486
|
-
self.postMessage({ type: "token", requestId, text: delta });
|
|
607
|
+
|
|
608
|
+
function abort(msg) {
|
|
609
|
+
console.debug("eddie agent engine: abort", msg.requestId, active ? active.requestId : null);
|
|
610
|
+
if (!active) return;
|
|
611
|
+
if (msg.requestId != null && msg.requestId !== active.requestId) return;
|
|
612
|
+
active.aborted = true;
|
|
613
|
+
try {
|
|
614
|
+
if (engine) engine.interruptGenerate();
|
|
615
|
+
} catch (err) {
|
|
616
|
+
console.warn("eddie agent: interrupt failed", err);
|
|
487
617
|
}
|
|
488
|
-
if (chunk && chunk.usage) usage = chunk.usage;
|
|
489
618
|
}
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
619
|
+
|
|
620
|
+
/** Abort the active run if it belongs to a page that went away. */
|
|
621
|
+
function abortIfOwner(reply) {
|
|
622
|
+
if (active && active.reply === reply) abort({ requestId: active.requestId });
|
|
623
|
+
}
|
|
624
|
+
|
|
625
|
+
function postError(reply, requestId, message) {
|
|
626
|
+
reply({ type: "error", requestId: requestId == null ? undefined : requestId, message });
|
|
496
627
|
}
|
|
628
|
+
|
|
629
|
+
return { handle, state: snapshot, abortIfOwner };
|
|
497
630
|
}
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
answer: processed.answer,
|
|
504
|
-
citations: processed.citations,
|
|
505
|
-
nohit: processed.nohit,
|
|
506
|
-
raw: text,
|
|
507
|
-
usage: {
|
|
508
|
-
ttftMs: first ? Math.round(first - t0) : totalMs,
|
|
509
|
-
totalMs,
|
|
510
|
-
tps: usage && usage.extra && typeof usage.extra.decode_tokens_per_s === "number" ? Math.round(usage.extra.decode_tokens_per_s) : null,
|
|
511
|
-
completionTokens: usage ? usage.completion_tokens : null,
|
|
512
|
-
},
|
|
513
|
-
});
|
|
514
|
-
}
|
|
515
|
-
|
|
516
|
-
function abort(msg) {
|
|
517
|
-
console.debug("eddie agent worker: abort", msg.requestId, active ? active.requestId : null);
|
|
518
|
-
if (!active) return;
|
|
519
|
-
if (msg.requestId != null && msg.requestId !== active.requestId) return;
|
|
520
|
-
active.aborted = true;
|
|
521
|
-
try {
|
|
522
|
-
if (engine) engine.interruptGenerate();
|
|
523
|
-
} catch (err) {
|
|
524
|
-
console.warn("eddie agent: interrupt failed", err);
|
|
631
|
+
|
|
632
|
+
function describe(err) {
|
|
633
|
+
if (err == null) return "unknown error";
|
|
634
|
+
if (typeof err === "string") return err;
|
|
635
|
+
return err.message || String(err);
|
|
525
636
|
}
|
|
526
|
-
}
|
|
527
637
|
|
|
528
|
-
|
|
529
|
-
|
|
530
|
-
|
|
638
|
+
/** True for the engine's "not loaded" replies (the client re-runs load). */
|
|
639
|
+
function isModelNotLoadedMessage(message) {
|
|
640
|
+
return typeof message === "string" && message.indexOf(NOT_LOADED) === 0;
|
|
641
|
+
}
|
|
642
|
+
|
|
643
|
+
return { createAgentEngine, AGENT_NOT_LOADED: NOT_LOADED, isModelNotLoadedMessage };
|
|
644
|
+
});
|
|
645
|
+
|
|
646
|
+
// SPDX-License-Identifier: GPL-3.0-only
|
|
647
|
+
|
|
648
|
+
// Eddie agent worker (module worker, created on the first "Ask"): the
|
|
649
|
+
// fallback host when the service worker (eddie-sw.js) is unavailable, same
|
|
650
|
+
// protocol either way. The engine is widget/src/lib/agent-engine.js; this
|
|
651
|
+
// file binds it to a dedicated module worker and loads WebLLM with a dynamic
|
|
652
|
+
// import(). widget/build.sh concatenates widget/src/lib/agent.js and
|
|
653
|
+
// agent-engine.js ahead of this file (EddieLib).
|
|
654
|
+
//
|
|
655
|
+
// Protocol: see widget/README.md ("Worker protocol", agent section).
|
|
656
|
+
|
|
657
|
+
"use strict";
|
|
658
|
+
|
|
659
|
+
const WEBLLM_URL = "https://esm.run/@mlc-ai/web-llm@0.2.84";
|
|
660
|
+
|
|
661
|
+
const lib = EddieLib;
|
|
531
662
|
|
|
532
|
-
|
|
533
|
-
|
|
534
|
-
|
|
535
|
-
|
|
536
|
-
|
|
663
|
+
const engine = lib.createAgentEngine({
|
|
664
|
+
post: (message) => self.postMessage(message),
|
|
665
|
+
loadWebLLM: () => import(WEBLLM_URL),
|
|
666
|
+
});
|
|
667
|
+
|
|
668
|
+
self.onmessage = function (e) {
|
|
669
|
+
engine.handle(e.data || {}, (message) => self.postMessage(message));
|
|
670
|
+
};
|