@yeaft/webchat-agent 0.1.663 → 0.1.665

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@yeaft/webchat-agent",
3
- "version": "0.1.663",
3
+ "version": "0.1.665",
4
4
  "description": "Remote agent for Yeaft WebChat — connects worker machines to the central server",
5
5
  "main": "index.js",
6
6
  "type": "module",
package/unify/index.js CHANGED
@@ -26,7 +26,6 @@ export { MemoryStore, parseEntry, serializeEntry, MEMORY_KINDS } from './memory/
26
26
 
27
27
  // Phase 5: Advanced features
28
28
  export { KINDS, KIND_PRIORITY, KIND_DESCRIPTIONS, IMPORTANCE_LEVELS, validateEntry, parseScopePath, getAncestorScopes, areScopesRelated } from './memory/types.js';
29
- export { scanEntries, scoreEntry, findStaleEntries, findDuplicateGroups, summarizeScan } from './memory/scan.js';
30
29
  export { runStopHooks } from './stop-hooks.js';
31
30
  export { MCPManager, createMCPManager } from './mcp.js';
32
31
  export { SkillManager, createSkillManager, parseSkill, serializeSkill } from './skills.js';
package/unify/init.js CHANGED
@@ -79,7 +79,6 @@ const SUBDIRS = [
79
79
  'conversation/blobs',
80
80
  'memory/entries',
81
81
  'tasks',
82
- 'dream',
83
82
  'skills',
84
83
  ];
85
84
 
@@ -1,8 +1,24 @@
1
1
  /**
2
2
  * web-search.js — Web search tool.
3
3
  *
4
- * Delegates to an external search API or LLM-based web search.
5
- * Supports configurable search providers via Yeaft config.
4
+ * Strategy (in order):
5
+ * 1. Tavily API (default; configured via ~/.yeaft/config.json → search.tavilyApiKey)
6
+ * 2. Generic searchApiUrl (legacy; user-supplied JSON-returning endpoint)
7
+ * 3. HTML-scrape fallback: DuckDuckGo lite then Bing
8
+ * (works on residential IPs; cloud IPs are usually flagged as bots)
9
+ *
10
+ * Config shape in ~/.yeaft/config.json:
11
+ * {
12
+ * "search": {
13
+ * "tavilyApiKey": "tvly-...",
14
+ * "searchApiUrl": "https://...", // optional, alternative JSON endpoint
15
+ * "disableHtmlFallback": false // optional, opt-out of scraping
16
+ * }
17
+ * }
18
+ *
19
+ * The result is JSON-stringified so the LLM can parse it. We intentionally
20
+ * keep the output shape consistent across providers: { provider, query,
21
+ * answer?, results: [{title, url, snippet}] }.
6
22
  */
7
23
 
8
24
  import { defineTool } from './types.js';
@@ -36,44 +52,204 @@ Guidelines:
36
52
  isReadOnly: () => true,
37
53
  async execute(input, ctx) {
38
54
  const { query, limit = 5 } = input;
39
- if (!query) return JSON.stringify({ error: 'query is required' });
40
-
41
- try {
42
- // Check if adapter supports web search natively (some LLM providers have built-in search)
43
- const adapter = ctx?.adapter;
44
- if (adapter && typeof adapter.webSearch === 'function') {
45
- const results = await adapter.webSearch(query, limit);
46
- return JSON.stringify(results, null, 2);
47
- }
48
-
49
- // Check config for search API endpoint
50
- const searchUrl = ctx?.config?.searchApiUrl;
51
- if (searchUrl) {
52
- const url = new URL(searchUrl);
53
- url.searchParams.set('q', query);
54
- url.searchParams.set('limit', String(limit));
55
-
56
- const response = await fetch(url.toString(), {
57
- signal: ctx?.signal,
58
- headers: { 'User-Agent': 'Yeaft/1.0' },
59
- });
60
-
61
- if (!response.ok) {
62
- return JSON.stringify({ error: `Search API returned ${response.status}: ${response.statusText}` });
63
- }
64
-
65
- const data = await response.json();
66
- return JSON.stringify(data, null, 2);
67
- }
68
-
69
- // Fallback: no search provider configured
70
- return JSON.stringify({
71
- error: 'No web search provider configured.',
72
- hint: 'Configure searchApiUrl in ~/.yeaft/config.json or use an LLM provider with built-in search.',
73
- });
74
- } catch (err) {
75
- if (err.name === 'AbortError') return JSON.stringify({ error: 'Search cancelled' });
76
- return JSON.stringify({ error: `Web search failed: ${err.message}` });
55
+ if (!query || typeof query !== 'string') {
56
+ return JSON.stringify({ error: 'query is required' });
57
+ }
58
+
59
+ const search = ctx?.config?.search || {};
60
+ const signal = ctx?.signal;
61
+ const errors = [];
62
+
63
+ // 1. Tavily — default, fast, structured.
64
+ if (search.tavilyApiKey) {
65
+ const r = await tryTavily(query, limit, search.tavilyApiKey, signal);
66
+ if (r.ok) return JSON.stringify(r.data, null, 2);
67
+ errors.push(`tavily: ${r.error}`);
68
+ }
69
+
70
+ // 2. Generic JSON endpoint (legacy escape hatch — SearXNG, custom proxy, etc).
71
+ const genericUrl = search.searchApiUrl || ctx?.config?.searchApiUrl;
72
+ if (genericUrl) {
73
+ const r = await tryGenericApi(query, limit, genericUrl, signal);
74
+ if (r.ok) return JSON.stringify(r.data, null, 2);
75
+ errors.push(`searchApiUrl: ${r.error}`);
76
+ }
77
+
78
+ // 3. HTML-scrape fallback. Often blocked on cloud IPs; useful for
79
+ // self-hosted / residential setups with no API key.
80
+ if (!search.disableHtmlFallback) {
81
+ const r = await tryHtmlScrape(query, limit, signal);
82
+ if (r.ok) return JSON.stringify(r.data, null, 2);
83
+ errors.push(`html: ${r.error}`);
77
84
  }
85
+
86
+ return JSON.stringify({
87
+ error: 'No web search backend succeeded.',
88
+ attempted: errors,
89
+ hint: 'Set search.tavilyApiKey in ~/.yeaft/config.json (free tier: https://tavily.com).',
90
+ });
78
91
  },
79
92
  });
93
+
94
+ // ─── Backend implementations ────────────────────────────────────────
95
+
96
+ async function tryTavily(query, limit, apiKey, signal) {
97
+ try {
98
+ const res = await fetch('https://api.tavily.com/search', {
99
+ method: 'POST',
100
+ signal,
101
+ headers: { 'Content-Type': 'application/json' },
102
+ body: JSON.stringify({
103
+ api_key: apiKey,
104
+ query,
105
+ max_results: Math.max(1, Math.min(limit, 10)),
106
+ include_answer: true,
107
+ search_depth: 'basic',
108
+ }),
109
+ });
110
+ if (!res.ok) {
111
+ const text = await res.text().catch(() => '');
112
+ return { ok: false, error: `${res.status} ${res.statusText} ${text.slice(0, 200)}` };
113
+ }
114
+ const data = await res.json();
115
+ return {
116
+ ok: true,
117
+ data: {
118
+ provider: 'tavily',
119
+ query,
120
+ answer: data.answer || null,
121
+ results: (data.results || []).slice(0, limit).map((r) => ({
122
+ title: r.title,
123
+ url: r.url,
124
+ snippet: r.content,
125
+ score: r.score,
126
+ })),
127
+ },
128
+ };
129
+ } catch (err) {
130
+ if (err?.name === 'AbortError') return { ok: false, error: 'cancelled' };
131
+ return { ok: false, error: err.message || String(err) };
132
+ }
133
+ }
134
+
135
+ async function tryGenericApi(query, limit, urlStr, signal) {
136
+ try {
137
+ const url = new URL(urlStr);
138
+ url.searchParams.set('q', query);
139
+ url.searchParams.set('limit', String(limit));
140
+ const res = await fetch(url.toString(), {
141
+ signal,
142
+ headers: { 'User-Agent': 'Yeaft/1.0', Accept: 'application/json' },
143
+ });
144
+ if (!res.ok) return { ok: false, error: `${res.status} ${res.statusText}` };
145
+ const data = await res.json();
146
+ return { ok: true, data: { provider: 'generic', query, ...data } };
147
+ } catch (err) {
148
+ if (err?.name === 'AbortError') return { ok: false, error: 'cancelled' };
149
+ return { ok: false, error: err.message || String(err) };
150
+ }
151
+ }
152
+
153
+ /**
154
+ * HTML-scrape fallback. Tries DuckDuckGo's lite HTML endpoint first
155
+ * (smaller markup, but more aggressive bot detection on cloud IPs),
156
+ * then Bing. We intentionally keep the regex-based parsers minimal —
157
+ * they break less than full DOM selectors when sites tweak markup.
158
+ */
159
+ async function tryHtmlScrape(query, limit, signal) {
160
+ const ua = 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/120.0.0.0 Safari/537.36';
161
+ const headers = { 'User-Agent': ua, 'Accept-Language': 'en-US,en;q=0.9' };
162
+
163
+ try {
164
+ const ddg = await fetch(`https://html.duckduckgo.com/html/?q=${encodeURIComponent(query)}`, { signal, headers });
165
+ if (ddg.ok) {
166
+ const html = await ddg.text();
167
+ const results = parseDdgHtml(html, limit);
168
+ if (results.length) return { ok: true, data: { provider: 'duckduckgo-html', query, results } };
169
+ }
170
+ } catch (err) {
171
+ if (err?.name === 'AbortError') return { ok: false, error: 'cancelled' };
172
+ }
173
+
174
+ try {
175
+ const bing = await fetch(`https://www.bing.com/search?q=${encodeURIComponent(query)}`, { signal, headers });
176
+ if (bing.ok) {
177
+ const html = await bing.text();
178
+ const results = parseBingHtml(html, limit);
179
+ if (results.length) return { ok: true, data: { provider: 'bing-html', query, results } };
180
+ }
181
+ } catch (err) {
182
+ if (err?.name === 'AbortError') return { ok: false, error: 'cancelled' };
183
+ }
184
+
185
+ return { ok: false, error: 'all HTML scrape backends returned 0 results (likely bot-blocked)' };
186
+ }
187
+
188
+ /**
189
+ * Parse DDG lite HTML. Each result is wrapped in
190
+ * <a class="result__a" href="…">title</a>
191
+ * <a class="result__snippet">snippet</a>
192
+ * Hash classes are not used here, so plain regex is fine.
193
+ */
194
+ function parseDdgHtml(html, limit) {
195
+ const results = [];
196
+ const linkRe = /<a[^>]+class="result__a"[^>]+href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/g;
197
+ const snippetRe = /<a[^>]+class="result__snippet"[^>]*>([\s\S]*?)<\/a>/g;
198
+ const links = [...html.matchAll(linkRe)];
199
+ const snippets = [...html.matchAll(snippetRe)];
200
+ for (let i = 0; i < links.length && results.length < limit; i++) {
201
+ const url = decodeDdgUrl(links[i][1]);
202
+ const title = stripTags(links[i][2]).trim();
203
+ const snippet = snippets[i] ? stripTags(snippets[i][1]).trim() : '';
204
+ if (url && title) results.push({ title, url, snippet });
205
+ }
206
+ return results;
207
+ }
208
+
209
+ /**
210
+ * DDG often wraps outbound URLs in `/l/?uddg=…` redirects. Unwrap.
211
+ */
212
+ function decodeDdgUrl(href) {
213
+ try {
214
+ if (href.startsWith('//')) href = 'https:' + href;
215
+ const u = new URL(href, 'https://duckduckgo.com');
216
+ const target = u.searchParams.get('uddg');
217
+ return target ? decodeURIComponent(target) : u.toString();
218
+ } catch {
219
+ return href;
220
+ }
221
+ }
222
+
223
+ /**
224
+ * Parse Bing search HTML. Result blocks: <li class="b_algo"> with
225
+ * <h2><a href="…">title</a></h2> and <p>snippet</p>. The class names
226
+ * have been stable for years; if Bing rotates them this will fail
227
+ * gracefully (no results extracted) and we'll surface the error upstream.
228
+ */
229
+ function parseBingHtml(html, limit) {
230
+ const results = [];
231
+ const blockRe = /<li[^>]+class="[^"]*\bb_algo\b[^"]*"[^>]*>([\s\S]*?)<\/li>/g;
232
+ for (const m of html.matchAll(blockRe)) {
233
+ if (results.length >= limit) break;
234
+ const block = m[1];
235
+ const linkM = block.match(/<h2[^>]*>\s*<a[^>]+href="([^"]+)"[^>]*>([\s\S]*?)<\/a>/);
236
+ if (!linkM) continue;
237
+ const url = linkM[1];
238
+ const title = stripTags(linkM[2]).trim();
239
+ const pM = block.match(/<p[^>]*>([\s\S]*?)<\/p>/);
240
+ const snippet = pM ? stripTags(pM[1]).trim() : '';
241
+ if (url && title) results.push({ title, url, snippet });
242
+ }
243
+ return results;
244
+ }
245
+
246
+ function stripTags(s) {
247
+ return s
248
+ .replace(/<[^>]+>/g, '')
249
+ .replace(/&amp;/g, '&')
250
+ .replace(/&lt;/g, '<')
251
+ .replace(/&gt;/g, '>')
252
+ .replace(/&quot;/g, '"')
253
+ .replace(/&#39;/g, "'")
254
+ .replace(/&nbsp;/g, ' ');
255
+ }
@@ -1,273 +0,0 @@
1
- /**
2
- * scan.js — Memory header scanning and scope/tag matching
3
- *
4
- * Fast in-memory scanning of entry frontmatter for:
5
- * - Scope tree traversal
6
- * - Tag overlap scoring
7
- * - Kind-based filtering
8
- * - Stale entry detection (for Dream)
9
- *
10
- * Reference: yeaft-unify-core-systems.md §3.3, yeaft-unify-design.md §5.1
11
- */
12
-
13
- import { KINDS, KIND_PRIORITY, IMPORTANCE_WEIGHT, getAncestorScopes } from './types.js';
14
-
15
- // ─── Scan Results ──────────────────────────────────────────
16
-
17
- /**
18
- * @typedef {Object} ScanResult
19
- * @property {object[]} entries — all parsed entries
20
- * @property {Map<string, number>} scopeCount — scope → entry count
21
- * @property {Map<string, number>} kindCount — kind → entry count
22
- * @property {Map<string, Set<string>>} tagIndex — tag → set of entry names
23
- * @property {number} totalEntries — total count
24
- */
25
-
26
- /**
27
- * Scan all entries from a MemoryStore and build indexes.
28
- *
29
- * @param {import('./store.js').MemoryStore} memoryStore
30
- * @returns {ScanResult}
31
- */
32
- export function scanEntries(memoryStore) {
33
- const entries = memoryStore.listEntries();
34
-
35
- const scopeCount = new Map();
36
- const kindCount = new Map();
37
- const tagIndex = new Map();
38
-
39
- for (const entry of entries) {
40
- // Scope count
41
- const scope = entry.scope || 'global';
42
- scopeCount.set(scope, (scopeCount.get(scope) || 0) + 1);
43
-
44
- // Kind count
45
- const kind = entry.kind || 'fact';
46
- kindCount.set(kind, (kindCount.get(kind) || 0) + 1);
47
-
48
- // Tag index
49
- const tags = entry.tags || [];
50
- for (const tag of tags) {
51
- const lowerTag = tag.toLowerCase();
52
- if (!tagIndex.has(lowerTag)) tagIndex.set(lowerTag, new Set());
53
- tagIndex.get(lowerTag).add(entry.name);
54
- }
55
- }
56
-
57
- return {
58
- entries,
59
- scopeCount,
60
- kindCount,
61
- tagIndex,
62
- totalEntries: entries.length,
63
- };
64
- }
65
-
66
- // ─── Scoring Functions ─────────────────────────────────────
67
-
68
- /**
69
- * Score an entry for relevance to a query context.
70
- *
71
- * Scoring factors:
72
- * - Scope match: exact=5, parent/child=3, global=1
73
- * - Tag overlap: 2 per matching tag
74
- * - Kind priority: see KIND_PRIORITY
75
- * - Importance weight: see IMPORTANCE_WEIGHT
76
- * - Frequency bonus: log2(frequency)
77
- * - Recency bonus: entries updated in last 7 days get +2
78
- *
79
- * @param {object} entry — memory entry
80
- * @param {{ scope?: string, tags?: string[], preferKinds?: string[] }} context
81
- * @returns {number} — relevance score
82
- */
83
- export function scoreEntry(entry, context = {}) {
84
- let score = 0;
85
-
86
- // Scope match
87
- if (context.scope && entry.scope) {
88
- if (entry.scope === context.scope) {
89
- score += 5; // exact match
90
- } else {
91
- const ancestors = getAncestorScopes(context.scope);
92
- if (ancestors.includes(entry.scope)) {
93
- score += 3; // ancestor match
94
- } else if (entry.scope.startsWith(context.scope + '/')) {
95
- score += 3; // descendant match
96
- } else if (entry.scope === 'global') {
97
- score += 1; // global fallback
98
- }
99
- }
100
- }
101
-
102
- // Tag overlap
103
- if (context.tags && context.tags.length > 0 && entry.tags) {
104
- const entryTags = new Set(entry.tags.map(t => t.toLowerCase()));
105
- for (const tag of context.tags) {
106
- if (entryTags.has(tag.toLowerCase())) {
107
- score += 2;
108
- }
109
- }
110
- }
111
-
112
- // Kind priority
113
- const kindPriority = KIND_PRIORITY[entry.kind] || 0;
114
- score += kindPriority * 0.5;
115
-
116
- // Preferred kinds bonus
117
- if (context.preferKinds && context.preferKinds.includes(entry.kind)) {
118
- score += 2;
119
- }
120
-
121
- // Importance weight
122
- const impWeight = IMPORTANCE_WEIGHT[entry.importance] || IMPORTANCE_WEIGHT.normal;
123
- score += impWeight * 0.5;
124
-
125
- // Frequency bonus (logarithmic)
126
- const freq = entry.frequency || 1;
127
- score += Math.log2(Math.max(freq, 1));
128
-
129
- // Recency bonus
130
- if (entry.updated_at) {
131
- const daysSince = (Date.now() - new Date(entry.updated_at).getTime()) / (1000 * 60 * 60 * 24);
132
- if (daysSince <= 7) score += 2;
133
- else if (daysSince <= 30) score += 1;
134
- }
135
-
136
- return score;
137
- }
138
-
139
- // ─── Stale Detection (for Dream) ────────────────────────────
140
-
141
- /**
142
- * Find entries that are potentially stale.
143
- *
144
- * Stale criteria:
145
- * - context entries older than 30 days
146
- * - entries never recalled (frequency = 1) and older than 60 days
147
- * - relation entries older than 90 days
148
- *
149
- * @param {object[]} entries
150
- * @returns {object[]} — stale entries
151
- */
152
- export function findStaleEntries(entries) {
153
- const now = Date.now();
154
- const stale = [];
155
-
156
- for (const entry of entries) {
157
- const updatedAt = entry.updated_at ? new Date(entry.updated_at).getTime() : 0;
158
- const daysSince = (now - updatedAt) / (1000 * 60 * 60 * 24);
159
-
160
- let isStale = false;
161
-
162
- // Context entries become stale fast
163
- if (entry.kind === 'context' && daysSince > 30) {
164
- isStale = true;
165
- }
166
-
167
- // Entries never recalled and old
168
- if ((entry.frequency || 1) <= 1 && daysSince > 60) {
169
- isStale = true;
170
- }
171
-
172
- // Relations are volatile
173
- if (entry.kind === 'relation' && daysSince > 90) {
174
- isStale = true;
175
- }
176
-
177
- if (isStale) {
178
- stale.push({ ...entry, _daysSinceUpdate: Math.round(daysSince) });
179
- }
180
- }
181
-
182
- return stale;
183
- }
184
-
185
- // ─── Duplicate Detection (for Dream Merge) ──────────────────
186
-
187
- /**
188
- * Find groups of entries that are potentially duplicates.
189
- * Entries are grouped if they share ≥2 tags AND the same kind.
190
- *
191
- * @param {object[]} entries
192
- * @returns {object[][]} — groups of potentially duplicate entries
193
- */
194
- export function findDuplicateGroups(entries) {
195
- const groups = [];
196
- const visited = new Set();
197
-
198
- for (let i = 0; i < entries.length; i++) {
199
- if (visited.has(i)) continue;
200
-
201
- const group = [entries[i]];
202
- const eTags = new Set((entries[i].tags || []).map(t => t.toLowerCase()));
203
-
204
- for (let j = i + 1; j < entries.length; j++) {
205
- if (visited.has(j)) continue;
206
- if (entries[i].kind !== entries[j].kind) continue;
207
-
208
- const jTags = new Set((entries[j].tags || []).map(t => t.toLowerCase()));
209
- let overlap = 0;
210
- for (const tag of eTags) {
211
- if (jTags.has(tag)) overlap++;
212
- }
213
-
214
- if (overlap >= 2) {
215
- group.push(entries[j]);
216
- visited.add(j);
217
- }
218
- }
219
-
220
- if (group.length > 1) {
221
- visited.add(i);
222
- groups.push(group);
223
- }
224
- }
225
-
226
- return groups;
227
- }
228
-
229
- // ─── Stats Summary ──────────────────────────────────────────
230
-
231
- /**
232
- * Generate a text summary of memory state (for Dream prompts).
233
- *
234
- * @param {ScanResult} scan
235
- * @returns {string}
236
- */
237
- export function summarizeScan(scan) {
238
- const lines = [];
239
-
240
- lines.push(`Total entries: ${scan.totalEntries}`);
241
-
242
- // Kind breakdown
243
- const kindLines = [];
244
- for (const kind of KINDS) {
245
- const count = scan.kindCount.get(kind) || 0;
246
- if (count > 0) kindLines.push(`${kind}: ${count}`);
247
- }
248
- if (kindLines.length > 0) {
249
- lines.push(`Kinds: ${kindLines.join(', ')}`);
250
- }
251
-
252
- // Scope breakdown (top 10)
253
- const scopeEntries = [...scan.scopeCount.entries()]
254
- .sort((a, b) => b[1] - a[1])
255
- .slice(0, 10);
256
- if (scopeEntries.length > 0) {
257
- lines.push('Top scopes:');
258
- for (const [scope, count] of scopeEntries) {
259
- lines.push(` ${scope}: ${count}`);
260
- }
261
- }
262
-
263
- // Tag cloud (top 20)
264
- const tagEntries = [...scan.tagIndex.entries()]
265
- .map(([tag, names]) => [tag, names.size])
266
- .sort((a, b) => b[1] - a[1])
267
- .slice(0, 20);
268
- if (tagEntries.length > 0) {
269
- lines.push(`Top tags: ${tagEntries.map(([t, c]) => `${t}(${c})`).join(', ')}`);
270
- }
271
-
272
- return lines.join('\n');
273
- }