@vierratale/ai 0.1.0-beta.14 → 0.1.0-beta.16

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,208 +1,55 @@
1
- const GOOGLE_SEARCH = 'https://www.google.com/search';
2
- const STARTPAGE_SEARCH = 'https://www.startpage.com/sp/search';
3
1
  const DDG_SEARCH = 'https://html.duckduckgo.com/html/';
4
2
  const WIKI_SEARCH = 'https://en.wikipedia.org/w/api.php';
5
-
6
- const UA = 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36';
7
- const GOOGLE_HEADERS = {
8
- 'User-Agent': UA,
9
- Accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8',
10
- 'Accept-Language': 'en-US,en;q=0.9',
11
- 'Sec-Fetch-Mode': 'navigate',
12
- 'Sec-Fetch-Site': 'none',
13
- 'Upgrade-Insecure-Requests': '1',
14
- Cookie: 'CONSENT=YES+cb.20220419-07-p0.en+FX+700; SOCS=CAISEwgDEgk0NzU3NTA3MjQaBwgAARICIAA',
15
- };
16
- const STARTPAGE_HEADERS = {
17
- 'User-Agent': UA,
18
- Accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
19
- 'Accept-Language': 'en-US,en;q=0.9',
20
- };
21
-
22
- function isBlockedBody(html) {
23
- const s = String(html || '').slice(0, 5000).toLowerCase();
24
- return /enablejs|sorry\/index|unusual traffic|recaptcha|captcha|proof.?of.?work|just a moment|browser check/.test(s);
25
- }
26
-
27
- function isStartpageChallenge(body) {
28
- return /challenge|"difficulty"|"issuedAt"/.test(String(body || '').slice(0, 3000));
29
- }
30
-
31
- function stripTags(text) {
32
- const map = {
33
- '&amp;': '&', '&lt;': '<', '&gt;': '>', '&quot;': '"',
34
- '&#039;': "'", '&#39;': "'", '&apos;': "'", '&nbsp;': ' ',
35
- '&#x27;': "'", '&hellip;': '...', '&#8211;': '-', '&#8217;': "'",
36
- };
37
- return String(text)
38
- .replace(/<[^>]*>/g, '')
39
- .replace(/&[a-zA-Z0-9#]+;/g, (m) => map[m] ?? '')
40
- .replace(/\s{2,}/g, ' ')
41
- .trim();
42
- }
43
-
44
- function decodeGoogleUrl(url) {
45
- if (!url) return url;
46
- if (url.startsWith('/url?')) {
47
- const m = url.match(/[?&]q=([^&]+)/);
48
- return m ? safeDecode(m[1]) : url;
49
- }
50
- if (url.startsWith('/') ) return url;
51
- return safeDecode(url);
52
- }
53
-
54
- function safeDecode(s) {
55
- try {
56
- return decodeURIComponent(String(s).replace(/\+/g, ' '));
57
- } catch {
58
- return String(s);
59
- }
60
- }
3
+ const UA = 'Mozilla/5.0 (X11; Linux x86_64) AppleWebKit/537.36';
61
4
 
62
5
  export class WebSearch {
63
6
  static async search(query, maxResults = 5) {
64
- const results = await this._searchGoogle(query, maxResults);
65
- if (results.length > 0) return results;
66
-
67
- const startpage = await this._searchStartpage(query, maxResults);
68
- if (startpage.length > 0) return startpage.map((r) => ({ ...r, source: 'google' }));
69
-
70
- const ddg = await this._searchDuckDuckGo(query, maxResults);
71
- if (ddg.length > 0) return ddg;
72
-
73
- return this._searchWikipedia(query, maxResults);
74
- }
75
-
76
- static async _searchGoogle(query, maxResults) {
77
- const url = `${GOOGLE_SEARCH}?q=${encodeURIComponent(query)}&num=${Math.max(maxResults, 5)}&hl=en&gl=us&udm=14`;
78
- try {
79
- const resp = await fetch(url, {
80
- headers: GOOGLE_HEADERS,
81
- signal: AbortSignal.timeout(12000),
82
- redirect: 'follow',
83
- });
84
- if (!resp.ok) return [];
85
- const html = await resp.text();
86
- if (isBlockedBody(html) || isStartpageChallenge(html)) return [];
87
- return this._parseGoogle(html, maxResults);
88
- } catch {
89
- return [];
7
+ let results = await this._searchDuckDuckGo(query, maxResults);
8
+ if (results.length === 0) {
9
+ results = await this._searchWikipedia(query, maxResults);
90
10
  }
11
+ return results;
91
12
  }
92
13
 
93
- static async _searchStartpage(query, maxResults) {
14
+ static async _searchDuckDuckGo(query, maxResults) {
15
+ const url = `${DDG_SEARCH}?q=${encodeURIComponent(query)}`;
94
16
  try {
95
- const cookie = await this._startpageCookie();
96
- const resp = await fetch(STARTPAGE_SEARCH, {
97
- method: 'POST',
98
- headers: {
99
- ...STARTPAGE_HEADERS,
100
- 'Content-Type': 'application/x-www-form-urlencoded',
101
- Cookie: cookie,
102
- },
103
- body: `query=${encodeURIComponent(query)}&cat=web&language=english&privacy_preference=0`,
104
- signal: AbortSignal.timeout(12000),
105
- redirect: 'follow',
17
+ const resp = await fetch(url, {
18
+ headers: { 'User-Agent': UA },
19
+ signal: AbortSignal.timeout(10000),
106
20
  });
107
21
  if (!resp.ok) return [];
108
22
  const html = await resp.text();
109
- if (isBlockedBody(html) || isStartpageChallenge(html)) return [];
110
- return this._parseStartpage(html, maxResults);
23
+ return this._parseDuckDuckGo(html, maxResults);
111
24
  } catch {
112
25
  return [];
113
26
  }
114
27
  }
115
28
 
116
- static async _startpageCookie() {
117
- try {
118
- const home = await fetch('https://www.startpage.com/', {
119
- headers: STARTPAGE_HEADERS,
120
- signal: AbortSignal.timeout(8000),
121
- });
122
- const cookies = home.headers.getSetCookie ? home.headers.getSetCookie() : [home.headers.get('set-cookie')].filter(Boolean);
123
- return cookies.map((c) => c.split(';')[0]).join('; ');
124
- } catch {
125
- return '';
126
- }
127
- }
128
-
129
- static _parseGoogle(html, maxResults) {
130
- const out = [];
131
- const seen = new Set();
132
- const resultRegex = /<a[^>]+href=["']([^"']+)["'][^>]*>(?:(?!<\/a>).)*?<h3[^>]*>(.*?)<\/h3>(?:(?!<\/a>).)*?<\/a>/gis;
133
- let m;
134
- while ((m = resultRegex.exec(html)) !== null && out.length < maxResults) {
135
- let url = decodeGoogleUrl(m[1]);
136
- if (!url || !/^https?:/.test(url)) continue;
137
- const title = stripTags(m[2]);
138
- if (!title || seen.has(url)) continue;
139
- const after = html.slice(resultRegex.lastIndex, resultRegex.lastIndex + 2000);
140
- const snippet = this._snippetAfter(after);
141
- seen.add(url);
142
- out.push({ title, url, snippet, source: 'google' });
143
- }
144
- return out;
145
- }
146
-
147
- static _snippetAfter(after) {
148
- const vwi = after.match(/<div[^>]*class=["']VwiC3b[^"']*["'][^>]*>([\s\S]{0,500}?)<\/div>/);
149
- if (vwi) {
150
- const text = stripTags(vwi[1]);
151
- if (text) return text.slice(0, 300);
152
- }
153
- const span = after.match(/<span[^>]*>([\s\S]{0,500}?)<\/span>/);
154
- if (span) {
155
- const text = stripTags(span[1]);
156
- if (text) return text.slice(0, 300);
157
- }
158
- const byline = after.slice(0, 1200).replace(/<[^>]+>/g, ' ').replace(/\s+/g, ' ').trim();
159
- return byline.slice(0, 300);
160
- }
161
-
162
- static _parseStartpage(html, maxResults) {
163
- const out = [];
164
- const seen = new Set();
165
- const linkRegex = /<a[^>]+href="((?:https?:)?\/\/[^"]+)"[^>]*>(?:(?!<\/a>).)*?<h3[^>]*>(.*?)<\/h3>(?:(?!<\/a>).)*?<\/a>/gis;
166
- let m;
167
- while ((m = linkRegex.exec(html)) !== null && out.length < maxResults) {
168
- let url = m[1];
169
- if (url.startsWith('//')) url = 'https:' + url;
170
- if (!url.startsWith('http') || /startpage\.com|google\.com/.test(url)) continue;
171
- const title = stripTags(m[2]);
172
- if (!title || seen.has(url)) continue;
173
- const after = html.slice(linkRegex.lastIndex);
174
- const snippet = this._snippetAfter(after);
175
- seen.add(url);
176
- out.push({ title, url, snippet, source: 'google' });
177
- }
178
- if (out.length > 0) return out;
179
- return this._parseStartpageOlder(html, maxResults);
180
- }
181
-
182
- static _parseStartpageOlder(html, maxResults) {
183
- const out = [];
184
- const blocks = html.split(/<section class="w-gl__result"|class="result"/gi).slice(1);
185
- for (const block of blocks.slice(0, maxResults)) {
186
- const a = block.match(/href="((?:https?:)?\/\/[^"]+)"/);
187
- const t = block.match(/<h3[^>]*>(.*?)<\/h3>/s);
188
- if (!a || !t) continue;
189
- let url = a[1];
190
- if (url.startsWith('//')) url = 'https:' + url;
191
- const snippet = stripTags(block).slice(0, 300);
192
- out.push({ title: stripTags(t[1]), url, snippet, source: 'google' });
193
- }
194
- return out;
195
- }
196
-
197
- static async _searchDuckDuckGo(query, maxResults) {
198
- const url = `${DDG_SEARCH}?q=${encodeURIComponent(query)}`;
29
+ static async _searchWikipedia(query, maxResults) {
30
+ const params = new URLSearchParams({
31
+ action: 'query',
32
+ list: 'search',
33
+ srsearch: query,
34
+ srlimit: String(maxResults),
35
+ format: 'json',
36
+ utf8: '1',
37
+ });
38
+ const url = `${WIKI_SEARCH}?${params.toString()}`;
199
39
  try {
200
40
  const resp = await fetch(url, {
201
41
  headers: { 'User-Agent': UA },
202
42
  signal: AbortSignal.timeout(10000),
203
43
  });
204
44
  if (!resp.ok) return [];
205
- return this._parseDuckDuckGo(await resp.text(), maxResults);
45
+ const data = await resp.json();
46
+ const search = data?.query?.search || [];
47
+ return search.map((r) => ({
48
+ title: r.title,
49
+ url: `https://en.wikipedia.org/wiki/${encodeURIComponent(r.title.replace(/ /g, '_'))}`,
50
+ snippet: this._stripTags(r.snippet || ''),
51
+ source: 'wikipedia',
52
+ }));
206
53
  } catch {
207
54
  return [];
208
55
  }
@@ -215,8 +62,8 @@ export class WebSearch {
215
62
  let count = 0;
216
63
  while ((match = resultRegex.exec(html)) !== null && count < maxResults) {
217
64
  let url = match[1];
218
- const snippet = stripTags(match[3] || '');
219
- const title = stripTags(match[2] || '');
65
+ const snippet = this._stripTags(match[3] || '');
66
+ const title = this._stripTags(match[2] || '');
220
67
  url = this._decodeDdgUrl(url);
221
68
  if (url && !url.includes('duckduckgo.com/y.js')) {
222
69
  results.push({ title, url, snippet, source: 'duckduckgo' });
@@ -228,35 +75,19 @@ export class WebSearch {
228
75
 
229
76
  static _decodeDdgUrl(url) {
230
77
  const match = url.match(/uddg=([^&]+)/);
231
- return match ? safeDecode(match[1]) : url;
78
+ return match ? decodeURIComponent(match[1]) : url;
232
79
  }
233
80
 
234
- static async _searchWikipedia(query, maxResults) {
235
- const params = new URLSearchParams({
236
- action: 'query',
237
- list: 'search',
238
- srsearch: query,
239
- srlimit: String(maxResults),
240
- format: 'json',
241
- utf8: '1',
242
- });
243
- const url = `${WIKI_SEARCH}?${params.toString()}`;
244
- try {
245
- const resp = await fetch(url, {
246
- headers: { 'User-Agent': UA },
247
- signal: AbortSignal.timeout(10000),
248
- });
249
- if (!resp.ok) return [];
250
- const data = await resp.json();
251
- const search = data?.query?.search || [];
252
- return search.map((r) => ({
253
- title: r.title,
254
- url: `https://en.wikipedia.org/wiki/${encodeURIComponent(r.title.replace(/ /g, '_'))}`,
255
- snippet: stripTags(r.snippet || ''),
256
- source: 'wikipedia',
257
- }));
258
- } catch {
259
- return [];
260
- }
81
+ static _stripTags(text) {
82
+ const map = {
83
+ '&amp;': '&', '&lt;': '<', '&gt;': '>', '&quot;': '"',
84
+ '&#039;': "'", '&#39;': "'", '&apos;': "'", '&nbsp;': ' ',
85
+ '&#x27;': "'", '&hellip;': '...',
86
+ };
87
+ return String(text)
88
+ .replace(/<[^>]*>/g, '')
89
+ .replace(/&[a-zA-Z0-9#]+;/g, (m) => map[m] ?? '')
90
+ .replace(/\s{2,}/g, ' ')
91
+ .trim();
261
92
  }
262
- }
93
+ }
package/LICENSE DELETED
@@ -1,21 +0,0 @@
1
- MIT License
2
-
3
- Copyright (c) 2026 Vierratale
4
-
5
- Permission is hereby granted, free of charge, to any person obtaining a copy
6
- of this software and associated documentation files (the "Software"), to deal
7
- in the Software without restriction, including without limitation the rights
8
- to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
- copies of the Software, and to permit persons to whom the Software is
10
- furnished to do so, subject to the following conditions:
11
-
12
- The above copyright notice and this permission notice shall be included in all
13
- copies or substantial portions of the Software.
14
-
15
- THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
- IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
- FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
- AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
- LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
- OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
- SOFTWARE.