@vierratale/ai 0.1.0-beta.10 → 0.1.0-beta.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@vierratale/ai",
3
- "version": "0.1.0-beta.10",
3
+ "version": "0.1.0-beta.11",
4
4
  "description": "VierrataleAI - Intelligent terminal assistant",
5
5
  "type": "module",
6
6
  "files": [
@@ -1,6 +1,6 @@
1
1
  export const Branding = {
2
2
  APP_NAME: 'VierrataleAI',
3
- VERSION: '0.1.0-beta.10',
3
+ VERSION: '0.1.0-beta.11',
4
4
 
5
5
  colors: {
6
6
  primary: '\x1b[38;2;124;58;237m',
@@ -2,6 +2,10 @@ export const UA =
2
2
  'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 ' +
3
3
  '(KHTML, like Gecko) Chrome/126.0.0.0 Safari/537.36';
4
4
 
5
+ const ALT_UA =
6
+ 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 ' +
7
+ '(KHTML, like Gecko) Version/17.4 Safari/605.1.15';
8
+
5
9
  const MAX_BYTES = 200000; // 200KB cap
6
10
  const MAX_TEXT = 8000; // ~8k chars of readable text
7
11
 
@@ -16,10 +20,18 @@ const BLOCK_MARKERS = [
16
20
  'client challenge',
17
21
  'checking your browser',
18
22
  'enable javascript and cookies',
19
- 'cf-chl',
23
+ 'verify you are human',
20
24
  'challenge-platform',
21
25
  'attention required!',
22
- 'powershell install',
26
+ 'unusual traffic',
27
+ 'access denied',
28
+ 'too many requests',
29
+ 'cf-error-details',
30
+ 'atomic challenge',
31
+ 'incapsula',
32
+ 'rh-captcha',
33
+ 'privacy pass',
34
+ 'error code: 1020',
23
35
  ];
24
36
 
25
37
  // Browser-like headers so header-based bot checks see a real browser.
@@ -39,6 +51,12 @@ export function browserHeaders(accept = 'text/html,application/xhtml+xml,applica
39
51
  };
40
52
  }
41
53
 
54
+ function altHeaders() {
55
+ const h = browserHeaders();
56
+ h['User-Agent'] = ALT_UA;
57
+ return h;
58
+ }
59
+
42
60
  // File types the AI can download off a fetched page.
43
61
  const ASSET_TYPES = {
44
62
  pdf: 'pdf', zip: 'zip', tar: 'tar', gz: 'gz', '7z': '7z', rar: 'rar',
@@ -61,6 +79,9 @@ export function isBlockedBody(html) {
61
79
  return BLOCK_MARKERS.some((m) => s.includes(m));
62
80
  }
63
81
 
82
+ // Optional namespace prefix for XML tags (e.g. <sm:loc>).
83
+ const NSTAG = (name) => `(?:[a-z][\\w.-]*:)?${name}`;
84
+
64
85
  export class WebFetch {
65
86
  static normalizeUrl(input) {
66
87
  const trimmed = (input || '').trim();
@@ -81,52 +102,63 @@ export class WebFetch {
81
102
  }
82
103
  }
83
104
 
105
+ // Fetch any URL and return readable text regardless of content type.
84
106
  static async fetch(rawUrl, maxText = MAX_TEXT) {
85
107
  const url = this.normalizeUrl(rawUrl);
86
108
  if (!url) {
87
109
  throw new Error('Invalid URL. Provide a valid http(s) address.');
88
110
  }
89
111
 
90
- let finalUrl = url;
91
- let html = '';
92
- let blocked = false;
93
- try {
94
- const res = await this._fetchHtml(url);
95
- finalUrl = res.url;
96
- html = res.html;
97
- const status = res.status;
98
- blocked = status !== 200 || isBlockedBody(html);
99
- } catch (err) {
100
- blocked = true;
101
- // _fetchHtml errors are network/cert/PROTOCOL issues; a fallback may
102
- // still succeed (e.g. registry API is reachable even when the HTML
103
- // page's CDN is not).
112
+ let res = await this._get(url, browserHeaders());
113
+
114
+ // Transport-level failure (TLS, no https) → retry plain http.
115
+ if (!res.ok) {
116
+ const alt = url.replace(/^https:/i, 'http:');
117
+ if (alt !== url) res = await this._get(alt, browserHeaders());
104
118
  }
105
119
 
106
- // Bot-walled npm/pypi pages degrade to their open JSON API.
107
- if (blocked) {
108
- const registry = await this._registryFallback(finalUrl || url, maxText);
120
+ // Hard wall with a challenge challenge → one retry with a different UA.
121
+ if (res.ok && (res.status === 403 || res.status === 429) && isBlockedBody(res.text)) {
122
+ const retried = await this._get(url, altHeaders());
123
+ if (retried.ok && retried.status === 200 && !isBlockedBody(retried.text)) res = retried;
124
+ }
125
+
126
+ if (!res.ok) {
127
+ const registry = await this._registryFallback(res.url || url, maxText);
109
128
  if (registry) return registry;
129
+ throw new Error(`Could not reach ${res.url || url}`);
110
130
  }
111
131
 
132
+ const status = res.status;
133
+ const bodyText = res.text;
134
+ const blocked = status >= 400 || isBlockedBody(bodyText);
135
+ if (blocked) {
136
+ const registry = await this._registryFallback(res.url || url, maxText);
137
+ if (registry) return registry;
138
+ }
112
139
  if (blocked) {
113
- const status = 'blocked';
114
- throw new Error(`Request blocked (${status}) for ${finalUrl || url}`);
140
+ throw new Error(`Request blocked (HTTP ${status}) for ${res.url || url}`);
115
141
  }
116
142
 
117
- return {
118
- url: finalUrl,
119
- title: this._extractTitle(html),
120
- text: this._extractText(html).slice(0, maxText),
121
- assets: this._extractAssets(html, finalUrl),
122
- };
143
+ const kind = this._classify(res.url, res.contentType, bodyText);
144
+ return this._render(kind, res, bodyText, maxText);
145
+ }
146
+
147
+ static async _get(url, headers) {
148
+ try {
149
+ const r = await this._fetchBytes(url, headers);
150
+ const text = this._decodeText(r.body, r.contentType);
151
+ return { ok: true, url: r.url, status: r.status, contentType: r.contentType, body: r.body, text };
152
+ } catch {
153
+ return { ok: false, url };
154
+ }
123
155
  }
124
156
 
125
- static async _fetchHtml(url) {
157
+ static async _fetchBytes(url, headers) {
126
158
  let resp;
127
159
  try {
128
160
  resp = await fetch(url, {
129
- headers: browserHeaders(),
161
+ headers,
130
162
  redirect: 'follow',
131
163
  signal: AbortSignal.timeout(15000),
132
164
  });
@@ -134,20 +166,316 @@ export class WebFetch {
134
166
  throw new Error(`Could not reach ${url}`);
135
167
  }
136
168
 
137
- const reader = resp.body.getReader();
138
- const decoder = new TextDecoder('utf-8', { fatal: false });
139
- let html = '';
140
- let received = 0;
141
- while (received < MAX_BYTES) {
142
- const { done, value } = await reader.read();
143
- if (done) break;
144
- received += value.length;
145
- html += decoder.decode(value, { stream: true });
169
+ const chunks = [];
170
+ let total = 0;
171
+ const reader = resp.body && resp.body.getReader();
172
+ if (reader) {
173
+ while (total < MAX_BYTES) {
174
+ const { done, value } = await reader.read();
175
+ if (done) break;
176
+ total += value.byteLength;
177
+ chunks.push(Buffer.from(value));
178
+ }
179
+ reader.cancel();
146
180
  }
147
- reader.cancel();
148
- html += decoder.decode();
181
+ return {
182
+ url: resp.url || url,
183
+ status: resp.status,
184
+ contentType: String(resp.headers.get('content-type') || '').toLowerCase(),
185
+ body: Buffer.concat(chunks, total),
186
+ };
187
+ }
188
+
189
+ // --- charset-aware decoding ------------------------------------------- //
149
190
 
150
- return { url: resp.url || url, html, status: resp.status };
191
+ static _sniffCharset(contentType, bytes) {
192
+ const m = /charset=(["']?)([a-z0-9._-]+)\1/i.exec(contentType || '');
193
+ if (m) return m[2].toLowerCase();
194
+ const head = bytes.slice(0, 2048).toString('latin1').toLowerCase();
195
+ const meta = /<meta[^>]+charset=["']?\s*([a-z0-9._-]+)/i.exec(head);
196
+ if (meta) return meta[1].toLowerCase();
197
+ const httpEquiv = /<meta[^>]+http-equiv=["']content-type["'][^>]+content=["']text\/html; charset=([a-z0-9._-]+)/i.exec(head);
198
+ return httpEquiv ? httpEquiv[1].toLowerCase() : '';
199
+ }
200
+
201
+ static _decodeText(bytes, contentType) {
202
+ const charset = this._sniffCharset(contentType, bytes);
203
+ if (charset) {
204
+ try {
205
+ return new TextDecoder(charset, { fatal: false }).decode(bytes);
206
+ } catch {}
207
+ }
208
+ let text = new TextDecoder('utf-8', { fatal: false }).decode(bytes);
209
+ if (this._badDecode(text, bytes)) {
210
+ try {
211
+ text = new TextDecoder('windows-1252', { fatal: false }).decode(bytes);
212
+ } catch {}
213
+ }
214
+ return text;
215
+ }
216
+
217
+ static _badDecode(text, bytes) {
218
+ if (!text || !bytes || bytes.length === 0) return false;
219
+ let bad = 0;
220
+ for (let i = 0; i < text.length; i++) if (text.charCodeAt(i) === 0xfffd) bad++;
221
+ return bad / bytes.length > 0.001;
222
+ }
223
+
224
+ // --- content-type classification --------------------------------------- //
225
+
226
+ static _classify(url, contentType, text) {
227
+ const meta = ((contentType || '').split(';')[0] || '').trim().toLowerCase();
228
+ const head = (text || '').slice(0, 240).trimStart();
229
+ if (/\bjson\b/.test(meta) || /\+\w*json\b/.test(meta)) return 'json';
230
+ if (/\b(?:rss|atom|xml)\b/.test(meta) || meta === 'application/sitemap+xml') return 'xml';
231
+ if (/\/xhtml\+html$/.test(meta) || /html/.test(meta)) return 'html';
232
+ if (/markdown/.test(meta)) return 'markdown';
233
+ if (meta === 'application/pdf') return 'pdf';
234
+ if (/^application\/(?:json|x-ndjson|xml|rss|atom)/.test(meta)) return 'json';
235
+ if (/^text\//.test(meta) && !/html/.test(meta)) return 'text';
236
+ if (head.startsWith('{') || head.startsWith('[')) return 'json';
237
+ if (head.startsWith('<?xml') || /^<!DOCTYPE\s+(?!html)/i.test(head)) return 'xml';
238
+ if (/^<!doctype html/i.test(head) || /<html\b/i.test(head.slice(0, 400))) return 'html';
239
+ if (/^(image|video|audio|font)\//.test(meta)) return 'binary';
240
+ if (/^application\//.test(meta)) return 'binary';
241
+ if (/\/pdf$/.test(url.split('?')[0].toLowerCase())) return 'pdf';
242
+ return 'text';
243
+ }
244
+
245
+ static _render(kind, res, bodyText, maxText) {
246
+ const url = res.url;
247
+
248
+ if (kind === 'json') {
249
+ let flat = bodyText.trim();
250
+ try {
251
+ flat = this._flattenJson(JSON.parse(bodyText));
252
+ } catch {}
253
+ return { url, title: this._titleFromUrl(url), text: flat.slice(0, maxText), assets: [] };
254
+ }
255
+
256
+ if (kind === 'xml') {
257
+ return {
258
+ url,
259
+ title: this._titleFromUrl(url),
260
+ text: this._xmlToText(bodyText).slice(0, maxText),
261
+ assets: this._extractAssets(bodyText, url),
262
+ };
263
+ }
264
+
265
+ if (kind === 'text' || kind === 'markdown') {
266
+ return { url, title: this._titleFromUrl(url), text: bodyText.trim().slice(0, maxText), assets: [] };
267
+ }
268
+
269
+ if (kind === 'pdf' || kind === 'binary') {
270
+ let parsed;
271
+ try {
272
+ parsed = new URL(url);
273
+ } catch {
274
+ parsed = null;
275
+ }
276
+ let type = detectAssetType(url) || 'file';
277
+ if (kind === 'pdf') type = 'pdf';
278
+ return {
279
+ url,
280
+ title: this._titleFromUrl(url),
281
+ text: this._binaryNote(res, kind, type),
282
+ assets: [{ url, name: this._assetName(parsed, type), type }],
283
+ };
284
+ }
285
+
286
+ // HTML
287
+ const desc = this._extractDescription(bodyText);
288
+ let text = this._extractText(bodyText);
289
+ if (text.length < 2400 && text.length < maxText) {
290
+ const embedded = this._extractEmbeddedJson(bodyText);
291
+ if (embedded !== null && embedded !== undefined) {
292
+ const flat = this._flattenJson(embedded).trim();
293
+ if (flat) text = `${text}\n\n[embedded page data]\n${flat}`.trim();
294
+ }
295
+ const links = this._extractLinks(bodyText, url);
296
+ if (links.length >= 3 && text.length < 2400) {
297
+ text = `${text}\n\n[Links on page]\n${links.slice(0, 25).map((l) => `- ${l.text}: ${l.url}`).join('\n')}`.trim();
298
+ }
299
+ }
300
+ let out = text.slice(0, maxText);
301
+ if (desc && out.length < maxText - 600 && !out.includes(desc.slice(0, 40))) {
302
+ out = `${desc.slice(0, 400)}\n\n${out}`;
303
+ }
304
+ let title = this._extractTitle(bodyText);
305
+ if (!title) title = this._titleFromUrl(url);
306
+ return { url, title, text: out, assets: this._extractAssets(bodyText, url) };
307
+ }
308
+
309
+ static _binaryNote(res, kind, type) {
310
+ const ct = res.contentType || 'unknown content type';
311
+ const size = res.body ? `~${(res.body.length / 1024).toFixed(1)} KB` : '?';
312
+ return [
313
+ `This resource is a ${kind === 'pdf' ? 'PDF' : 'binary/non-text file'} (${ct}, ${size}) and cannot be read as text.`,
314
+ `Use the download_url tool to save it locally: ${res.url}${type !== 'file' ? ` (type: ${type})` : ''}`,
315
+ ].join('\n');
316
+ }
317
+
318
+ static _titleFromUrl(url) {
319
+ try {
320
+ const u = new URL(url);
321
+ const seg = decodeURIComponent(u.pathname.split('/').filter(Boolean).pop() || '');
322
+ const base = u.hostname.replace(/^www\./i, '');
323
+ if (seg && !/\.(html?|php|aspx?)$/i.test(seg)) return base;
324
+ const cleaned = seg.replace(/\.(html?|php|aspx?)$/i, '').replace(/[-_]+/g, ' ').trim();
325
+ return cleaned || base;
326
+ } catch {
327
+ return url;
328
+ }
329
+ }
330
+
331
+ static _flattenJson(data, maxLines = 600) {
332
+ const out = [];
333
+ const walk = (v, label) => {
334
+ if (out.length >= maxLines) return;
335
+ if (v === null || v === undefined) {
336
+ if (label) out.push(`${label}: null`);
337
+ return;
338
+ }
339
+ if (Array.isArray(v)) {
340
+ if (label) out.push(`${label} (${v.length})`);
341
+ for (let i = 0; i < v.length; i++) walk(v[i], label ? `${label}[${i}]` : `[${i}]`);
342
+ } else if (typeof v === 'object') {
343
+ if (label) out.push(`${label} {${Object.keys(v).length}}`);
344
+ for (const k of Object.keys(v)) walk(v[k], label ? `${label}.${k}` : k);
345
+ } else {
346
+ const s = typeof v === 'string' ? v.replace(/\s+/g, ' ').trim() : String(v);
347
+ out.push(`${label ? label + ': ' : ''}${s.slice(0, 400)}`);
348
+ }
349
+ };
350
+ walk(data, '');
351
+ return out.join('\n');
352
+ }
353
+
354
+ // --- XML / feeds / sitemaps -------------------------------------------- //
355
+
356
+ static _xmlToText(xml) {
357
+ const doc = (xml || '').trimStart();
358
+ if (/<sitemapindex\b/i.test(doc) || /<urlset\b/i.test(doc)) {
359
+ const locs = [];
360
+ const re = new RegExp(`<${NSTAG('loc')}(?:[^>]*)>([\\s\\S]*?)<\\/${NSTAG('loc')}>`, 'gi');
361
+ let m;
362
+ while ((m = re.exec(xml)) !== null) {
363
+ const u = this._stripEntities(m[1]).replace(/\s+/g, ' ').trim();
364
+ if (u && /^https?:\/\//i.test(u) && !locs.includes(u)) locs.push(u);
365
+ if (locs.length >= 100) break;
366
+ }
367
+ if (locs.length) {
368
+ return `Sitemap (${locs.length} URLs)\n${locs.slice(0, 60).map((u) => `* ${u}`).join('\n')}`;
369
+ }
370
+ }
371
+ if (/<(?:rss|feed)\b/i.test(doc)) return this._feedToText(xml);
372
+ return this._extractText(xml).slice(0, MAX_TEXT);
373
+ }
374
+
375
+ static _feedToText(xml) {
376
+ const lines = [];
377
+ const top = new RegExp(`<${NSTAG('channel')}[\\s\\S]*?<\\/${NSTAG('channel')}>|<${NSTAG('feed')}[\\s\\S]*?<\\/${NSTAG('feed')}>`, 'i');
378
+ const topMatch = top.exec(xml);
379
+ let title = '';
380
+ if (topMatch) {
381
+ const t = new RegExp(`<${NSTAG('title')}(?:[^>]*)>([\\s\\S]*?)<\\/${NSTAG('title')}>`, 'i').exec(topMatch[0]);
382
+ if (t) title = this._stripTags(t[1]);
383
+ }
384
+ if (title) lines.push(title);
385
+
386
+ const itemRe = new RegExp(`<${NSTAG('item')}[\\s\\S]*?<\\/${NSTAG('item')}>|<${NSTAG('entry')}[\\s\\S]*?<\\/${NSTAG('entry')}>`, 'gi');
387
+ let m;
388
+ let count = 0;
389
+ while ((m = itemRe.exec(xml)) !== null && count < 40) {
390
+ const b = m[0];
391
+ count++;
392
+ const it = new RegExp(`<${NSTAG('title')}(?:[^>]*)>([\\s\\S]*?)<\\/${NSTAG('title')}>`, 'i').exec(b);
393
+ const il = new RegExp(`<${NSTAG('link')}(?:[^>]*)>([\\s\\S]*?)<\\/${NSTAG('link')}>`, 'i').exec(b) ||
394
+ new RegExp(`<${NSTAG('link')}(?:[^>]*)\\bhref=["']([^"']+)["']`, 'i').exec(b);
395
+ const di = /<(?:[a-z][\w.-]*:)?(?:description|summary|subtitle)[^>]*>([\s\S]*?)<\/(?:[a-z][\w.-]*:)?(?:description|summary|subtitle)>/i.exec(b);
396
+ const item = [];
397
+ if (it) item.push(this._stripTags(it[1]));
398
+ if (il) item.push(` ${this._stripTags(il[1]).replace(/\s+/g, ' ').trim()}`);
399
+ if (di && di[1].trim()) {
400
+ const d = this._stripTags(di[1]);
401
+ if (d) item.push(` ${d}`);
402
+ }
403
+ if (item.length) lines.push(item.join('\n'));
404
+ }
405
+ if (lines.length === 0) return this._extractText(xml).slice(0, MAX_TEXT);
406
+ return lines.join('\n\n');
407
+ }
408
+
409
+ // --- SPA / embedded JSON ------------------------------------------------ //
410
+
411
+ static _extractEmbeddedJson(html) {
412
+ const next = /<script[^>]+id=["']__NEXT_DATA__["'][^>]*>([\s\S]*?)<\/script>/i.exec(html);
413
+ if (next && next[1].trim()) {
414
+ try {
415
+ return JSON.parse(next[1].trim());
416
+ } catch {}
417
+ }
418
+ const ld = /<script[^>]+type=["']application\/ld\+json["'][^>]*>([\s\S]*?)<\/script>/gi;
419
+ let c;
420
+ while ((c = ld.exec(html)) !== null) {
421
+ if (!c[1].trim()) continue;
422
+ try {
423
+ return JSON.parse(c[1].trim());
424
+ } catch {}
425
+ }
426
+ const st = /window\.__PRELOADED_STATE__\s*=\s*/i.exec(html);
427
+ if (st) {
428
+ let i = st.index + st[0].length;
429
+ let depth = 0;
430
+ const start = i;
431
+ while (i < html.length) {
432
+ const ch = html[i];
433
+ if (ch === '{') depth++;
434
+ else if (ch === '}') {
435
+ depth--;
436
+ if (depth === 0) break;
437
+ }
438
+ i++;
439
+ }
440
+ if (depth === 0) {
441
+ try {
442
+ return JSON.parse(html.slice(start, i + 1).replace(/;?\s*$/, '').trim());
443
+ } catch {}
444
+ }
445
+ }
446
+ return null;
447
+ }
448
+
449
+ static _extractDescription(html) {
450
+ const m1 = /<meta[^>]+name=["']description["'][^>]+content=["']([^"']*)["'][^>]*>/i.exec(html);
451
+ if (m1) return this._stripTags(m1[1]);
452
+ const m2 = /<meta[^>]+content=["']([^"']*)["'][^>]+name=["']description["'][^>]*>/i.exec(html);
453
+ return m2 ? this._stripTags(m2[1]) : '';
454
+ }
455
+
456
+ static _extractLinks(html, baseUrl) {
457
+ const links = [];
458
+ const seen = new Set();
459
+ const re = /<a\b[^>]*href=["']([^"']+)["'][^>]*>((?:[^<]|<(?!\/a\b)[^>]+>)*?)<\/a>/gi;
460
+ let m;
461
+ while ((m = re.exec(html)) !== null) {
462
+ const href = m[1].trim().split('#')[0];
463
+ if (!href || /^(javascript|mailto|tel|data):/i.test(href)) continue;
464
+ let resolved;
465
+ try {
466
+ const u = new URL(href, baseUrl);
467
+ if (u.protocol !== 'http:' && u.protocol !== 'https:') continue;
468
+ resolved = u.toString();
469
+ } catch {
470
+ continue;
471
+ }
472
+ if (seen.has(resolved)) continue;
473
+ seen.add(resolved);
474
+ const text = this._stripTags(m[2]).slice(0, 80);
475
+ links.push({ text: text || resolved, url: resolved });
476
+ if (links.length >= 30) break;
477
+ }
478
+ return links;
151
479
  }
152
480
 
153
481
  // --- Registry fallback (npm + PyPI) via their open JSON APIs ---------- //
@@ -280,6 +608,7 @@ export class WebFetch {
280
608
  }
281
609
 
282
610
  static _assetName(url, type) {
611
+ if (!url) return `download.${type}`;
283
612
  const path = url.pathname.split('/').filter(Boolean);
284
613
  let name = path.length ? decodeURIComponent(path[path.length - 1]) : '';
285
614
  if (!name || type === 'unknown') name = `${url.hostname.replace(/[^a-z0-9.-]/gi, '_')}.${type}`;
@@ -293,17 +622,19 @@ export class WebFetch {
293
622
  }
294
623
 
295
624
  static _extractText(html) {
296
- // Strip block tags' contents that add no readable value.
297
- let s = html.replace(/<(script|style|noscript|svg|head|iframe|form|nav|footer|aside)[^>]*>[\s\S]*?<\/\1>/gi, ' ');
625
+ // Drop comments and tag-blocks that add no readable value.
626
+ let s = String(html || '')
627
+ .replace(/<!--[\s\S]*?-->/g, ' ')
628
+ .replace(/<(script|style|noscript|svg|head|iframe|form|nav|footer|aside|template|object)[^>]*>[\s\S]*?<\/\1>/gi, ' ');
298
629
 
299
630
  // Force spacing around block-level elements so words don't merge.
300
- s = s.replace(/<\/(p|div|h[1-6]|li|tr|br|section|article)>/gi, '\n');
631
+ s = s.replace(/<\/(p|div|h[1-6]|li|tr|br|section|article|blockquote|pre|table)>/gi, '\n');
301
632
  s = s.replace(/<(br|li|tr)[^>]*>/gi, '\n');
302
633
 
303
634
  // Remove remaining tags.
304
635
  s = s.replace(/<[^>]+>/g, ' ');
305
636
 
306
- // College entities.
637
+ // Collapse entities.
307
638
  s = this._stripEntities(s);
308
639
 
309
640
  // Collapse whitespace and trim lines.
@@ -316,6 +647,10 @@ export class WebFetch {
316
647
  .trim();
317
648
  }
318
649
 
650
+ static _stripTags(text) {
651
+ return this._stripEntities(text);
652
+ }
653
+
319
654
  static _stripEntities(text) {
320
655
  const map = {
321
656
  '&amp;': '&', '&lt;': '<', '&gt;': '>', '&quot;': '"',