@wenathlan/saddle 1.8.4 → 1.8.6
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +10 -10
- package/browser/playwright.js +22 -0
- package/docs/.gitkeep +0 -0
- package/docs/actionsincident.md +29 -0
- package/docs/branchaudit.md +21 -0
- package/docs/ecosystemplan.md +3 -3
- package/docs/featureaudit.md +1 -1
- package/docs/gapmatrix.md +8 -7
- package/docs/libraryapi.md +1 -0
- package/docs/logs/.gitkeep +0 -0
- package/docs/packageaudit185.md +23 -0
- package/docs/plans/00.index.md +50 -0
- package/docs/plans/01.architecture.md +86 -0
- package/docs/plans/02.research.computer.use.md +58 -0
- package/docs/plans/03.research.captcha.bypass.md +68 -0
- package/docs/plans/04.research.sandbox.ai.md +52 -0
- package/docs/plans/05.capture.platform.md +57 -0
- package/docs/plans/06.dependencies.md +97 -0
- package/docs/plans/07.captcha.test.page.md +41 -0
- package/docs/plans/08.production.infra.md +70 -0
- package/docs/plans/09.database.schema.md +121 -0
- package/docs/plans/10.cloudinary.storage.md +57 -0
- package/docs/plans/11.movement.logs.json.md +72 -0
- package/docs/plans/12.research.atlas.agent.browser.md +79 -0
- package/docs/plans/13.research.anti.detection.md +898 -0
- package/docs/plans/14.research.proxy.md +1495 -0
- package/docs/plans/15.research.retry.rate.limit.md +1958 -0
- package/docs/plans/16.research.crawling.md +1417 -0
- package/docs/plans/17.research.caching.md +1610 -0
- package/docs/plans/18.research.content.extraction.md +1952 -0
- package/docs/plans/19.research.errors.events.md +1523 -0
- package/docs/plans/20.research.zod.validation.md +1350 -0
- package/docs/plans/21.research.batch.concurrency.md +1888 -0
- package/docs/plans/22.research.universal.runtime.md +944 -0
- package/docs/plans/23.research.ai.integration.md +1465 -0
- package/docs/plans/24.research.memory.persistence.md +1979 -0
- package/docs/plans/25.research.server.api.md +342 -0
- package/docs/plans/26.research.compilation.md +249 -0
- package/docs/plans/27.research.html.parsing.md +251 -0
- package/docs/plans/28.action.plan.md +50 -0
- package/docs/plans/29.api.reference.md +174 -0
- package/docs/plans/30.architecture.plan.md +94 -0
- package/docs/plans/31.auditoria.dados.md +163 -0
- package/docs/plans/32.bots.automacao.computacional.md +214 -0
- package/docs/plans/33.bots.codigo.revisao.md +220 -0
- package/docs/plans/34.bots.seguranca.cicd.md +366 -0
- package/docs/plans/35.comparativo.concorrencia.md +464 -0
- package/docs/plans/36.computational.memory.md +340 -0
- package/docs/plans/37.deploystrategy.md +394 -0
- package/docs/plans/38.flow.md +155 -0
- package/docs/plans/39.multi.platform.bot.md +252 -0
- package/docs/plans/40.npm.publish.md +250 -0
- package/docs/plans/41.o.que.falta.md +407 -0
- package/docs/plans/42.pesquisa.concorrencia.md +721 -0
- package/docs/plans/43.plan.universal.architecture.md +496 -0
- package/docs/plans/44.reference.md +100 -0
- package/docs/plans/45.robotarchitecture.md +237 -0
- package/docs/plans/46.scdnintegration.md +284 -0
- package/docs/plans/47.multiforge.readme.md +129 -0
- package/docs/plans/48.theory.v4.repo.os.md +152 -0
- package/docs/plans/49.third.party.infra.md +12 -0
- package/docs/plans/50.file.as.compute.md +39 -0
- package/docs/plans/51.architecture.virtual.processor.md +80 -0
- package/docs/plans/52.manifesto.v8.md +11 -0
- package/docs/plans/58.cdn.list.md +23 -0
- package/docs/plans/59.sql.frameworks.md +33 -0
- package/docs/plans/60.sql.thirdparty.md +26 -0
- package/docs/plans/61.objective.multiforge.md +63 -0
- package/docs/plans/62.huggingface.upload.md +26 -0
- package/docs/plans/63.kaggle.upload.md +24 -0
- package/docs/plans/64.npm.storage.md +30 -0
- package/docs/plans/65.rclone.terabox.md +32 -0
- package/docs/plans/66.buckets.and.models.todo.md +14 -0
- package/docs/plans/67.database.todo.md +13 -0
- package/docs/plans/68.deploy.packages.todo.md +12 -0
- package/docs/plans/69.report.human.operator.md +133 -0
- package/docs/plans/70.report.brain2qwerty.ems.md +135 -0
- package/docs/plans/71.report.hd.infinito.vram.md +155 -0
- package/docs/plans/72.plan.hd.infinito.node.md +146 -0
- package/docs/plans/73.plan.scifi.repos.md +125 -0
- package/docs/plans/74.000.manifesto.v8.flat.2..md +11 -0
- package/docs/plans/README.md +489 -0
- package/docs/plans/aggregate_platforms.mjs +146 -0
- package/docs/plans/examplesession.json +36 -0
- package/docs/plans/missing-facts.md +192 -0
- package/docs/plans/models.md +64 -0
- package/docs/plans/organize.cjs +270 -0
- package/docs/plans/platforms.md +2887 -0
- package/docs/plans/sites.md +31322 -0
- package/docs/platformpipelineaudit.md +6 -2
- package/docs/registryresearch.md +2 -0
- package/docs/release.md +4 -4
- package/docs/release184notes.md +2 -2
- package/docs/release185notes.md +7 -0
- package/docs/releaseassets.md +16 -0
- package/docs/sources/farm.py +117 -0
- package/docs/sources/html/saddle1.html +132 -0
- package/docs/sources/html/saddle2.html +157 -0
- package/docs/sources/html/saddle3.html +119 -0
- package/docs/sources/html/saddle4.html +144 -0
- package/docs/sources/html/saddle5.html +72 -0
- package/docs/sources/html/saddle6.html +171 -0
- package/docs/sources/html/saddle7.html +236 -0
- package/docs/sources/saddle.ts +74 -0
- package/docs/sources/schema.prisma +88 -0
- package/docs/sources/script.sh +64 -0
- package/docs/sources/workflows.yml +458 -0
- package/docs/talks1/_body.txt +14 -0
- package/docs/talks1/_index.md +15 -0
- package/docs/talks1/_screenshot.png +0 -0
- package/docs/talks1/assistant-01.md +5 -0
- package/docs/talks1/assistant-02.md +5 -0
- package/docs/talks1/assistant-03.md +531 -0
- package/docs/talks1/assistant-04.md +26 -0
- package/docs/talks1/assistant-05.md +774 -0
- package/docs/talks1/assistant-06.md +1718 -0
- package/docs/talks1/scrape-share.cjs +185 -0
- package/docs/talks1/scrape-share.ts +183 -0
- package/docs/talks1/user-01.md +3 -0
- package/docs/talks1/user-02.md +3 -0
- package/docs/talks1/user-03.md +88 -0
- package/docs/talks1/user-04.md +3 -0
- package/docs/talks1/user-05.md +3 -0
- package/docs/talks1/user-06.md +88 -0
- package/docs/talks1/user-07.md +88 -0
- package/docs/talks2/_body.txt +14 -0
- package/docs/talks2/_index.md +16 -0
- package/docs/talks2/_screenshot.png +0 -0
- package/docs/talks2/assistant-01.md +5 -0
- package/docs/talks2/assistant-02.md +5 -0
- package/docs/talks2/assistant-03.md +424 -0
- package/docs/talks2/assistant-04.md +598 -0
- package/docs/talks2/assistant-05.md +1280 -0
- package/docs/talks2/assistant-06.md +1227 -0
- package/docs/talks2/assistant-07.md +1252 -0
- package/docs/talks2/user-01.md +3 -0
- package/docs/talks2/user-02.md +3 -0
- package/docs/talks2/user-03.md +88 -0
- package/docs/talks2/user-04.md +88 -0
- package/docs/talks2/user-05.md +88 -0
- package/docs/talks2/user-06.md +88 -0
- package/docs/talks2/user-07.md +3 -0
- package/docs/talks3/_body.txt +467 -0
- package/docs/talks3/_index.md +10 -0
- package/docs/talks3/_screenshot.png +0 -0
- package/docs/talks3/assistant-01.md +417 -0
- package/docs/talks3/assistant-02.md +417 -0
- package/docs/talks3/assistant-03.md +29 -0
- package/docs/talks3/assistant-04.md +727 -0
- package/docs/talks3/user-01.md +88 -0
- package/docs/talks3/user-02.md +88 -0
- package/docs/talks3/user-03.md +3 -0
- package/docs/talks3/user-04.md +3 -0
- package/docs/talks4/_body.txt +14 -0
- package/docs/talks4/_index.md +12 -0
- package/docs/talks4/_screenshot.png +0 -0
- package/docs/talks4/assistant-01.md +5 -0
- package/docs/talks4/assistant-02.md +5 -0
- package/docs/talks4/assistant-03.md +35 -0
- package/docs/talks4/assistant-04.md +512 -0
- package/docs/talks4/assistant-05.md +599 -0
- package/docs/talks4/user-01.md +3 -0
- package/docs/talks4/user-02.md +3 -0
- package/docs/talks4/user-03.md +88 -0
- package/docs/talks4/user-04.md +88 -0
- package/docs/talks4/user-05.md +7 -0
- package/docs/talks5/_body.txt +14 -0
- package/docs/talks5/_index.md +13 -0
- package/docs/talks5/_screenshot.png +0 -0
- package/docs/talks5/assistant-01.md +5 -0
- package/docs/talks5/assistant-02.md +5 -0
- package/docs/talks5/assistant-03.md +690 -0
- package/docs/talks5/assistant-04.md +758 -0
- package/docs/talks5/assistant-05.md +974 -0
- package/docs/talks5/user-01.md +3 -0
- package/docs/talks5/user-02.md +3 -0
- package/docs/talks5/user-03.md +105 -0
- package/docs/talks5/user-04.md +105 -0
- package/docs/talks5/user-05.md +63 -0
- package/docs/talks5/user-06.md +105 -0
- package/docs/talks6/_body.txt +14 -0
- package/docs/talks6/_index.md +9 -0
- package/docs/talks6/_screenshot.png +0 -0
- package/docs/talks6/assistant-01.md +5 -0
- package/docs/talks6/assistant-02.md +5 -0
- package/docs/talks6/assistant-03.md +1499 -0
- package/docs/talks6/user-01.md +3 -0
- package/docs/talks6/user-02.md +3 -0
- package/docs/talks6/user-03.md +88 -0
- package/docs/talks6/user-04.md +88 -0
- package/docs/talks7/_body.txt +14 -0
- package/docs/talks7/_index.md +10 -0
- package/docs/talks7/_screenshot.png +0 -0
- package/docs/talks7/assistant-01.md +5 -0
- package/docs/talks7/assistant-02.md +5 -0
- package/docs/talks7/assistant-03.md +523 -0
- package/docs/talks7/assistant-04.md +617 -0
- package/docs/talks7/user-01.md +3 -0
- package/docs/talks7/user-02.md +3 -0
- package/docs/talks7/user-03.md +105 -0
- package/docs/talks7/user-04.md +67 -0
- package/docs/talks8/conversa1.txt +1322 -0
- package/docs/talks8/conversa2.txt +237 -0
- package/docs/talks9/Beyond the Obvious_ 50 Plataformas Auto-Hospedadas de Forja de C/303/263digo para Al/303/251m de Gitea e GitLab.md" +174 -0
- package/docs/talks9/De NPM a Multi-Linguagem_ Uma Arquitetura T/303/251cnica para a Execu/303/247/303/243o Integrada de C/303/263digo no Ecossistema Node.js.md" +59 -0
- package/docs/talks9/De NPM a VMs Virtuais_ Uma An/303/241lise Arquitet/303/264nica para a Realiza/303/247/303/243o do Ciclo de Vida do Projeto SADDLE.md" +91 -0
- package/docs/talks9/Mapeamento da Engrenagem Computacional_ Uma Arquitetura para Execu/303/247/303/243o Isolada e Persist/303/252ncia em Ambientes Distribu/303/255dos.md" +116 -0
- package/docs/talks9/O Cen/303/241rio Pr/303/241tico do SADDLE_ Uma An/303/241lise de Viabilidade e Modelo de Ciclo de Vida Integrado.md" +128 -0
- package/docs/talks9/README (2).md +489 -0
- package/docs/talks9/README.md +198 -0
- package/docs/talks9/Viabilidade do Saddle_ Uma An/303/241lise T/303/251cnica da Transforma/303/247/303/243o de Armazenamento Remoto em Mem/303/263ria Computacional.md" +80 -0
- package/docs/talks9/conversa.txt +544 -0
- package/docs/talks9/other (2).md +39 -0
- package/docs/talks9/other.md +57 -0
- package/docs/talks9/outro.txt +24 -0
- package/extension/README.md +2 -2
- package/extension/build.js +1 -1
- package/extension/content.js +46 -4
- package/extension/manifest.json +1 -1
- package/extension/pagebridge.js +41 -0
- package/extension/protocol.js +21 -2
- package/extension/serviceworker.js +12 -5
- package/package.json +165 -10
- package/release/assets.js +70 -0
- package/scrape/agent.ts +122 -0
- package/scrape/batch.ts +79 -0
- package/scrape/biome.json +76 -0
- package/scrape/browser.ts +222 -0
- package/scrape/cache.ts +84 -0
- package/scrape/chunking.ts +193 -0
- package/scrape/cli.ts +105 -0
- package/scrape/crawler.ts +115 -0
- package/scrape/dev-server.ts +94 -0
- package/scrape/errors.ts +132 -0
- package/scrape/events.ts +26 -0
- package/scrape/extract.ts +165 -0
- package/scrape/fetch.ts +105 -0
- package/scrape/formats.ts +85 -0
- package/scrape/headers.ts +71 -0
- package/scrape/index.ts +92 -0
- package/scrape/jsdom.d.ts +6 -0
- package/scrape/llms-txt.ts +84 -0
- package/scrape/middleware.ts +90 -0
- package/scrape/package-lock.json +9397 -0
- package/scrape/package.json +1420 -0
- package/scrape/pool.ts +95 -0
- package/scrape/port.ts +18 -0
- package/scrape/proxy.ts +103 -0
- package/scrape/rate-limiter.ts +95 -0
- package/scrape/renderer.ts +194 -0
- package/scrape/retry.ts +64 -0
- package/scrape/robots.ts +137 -0
- package/scrape/scrape.ts +123 -0
- package/scrape/serialize.ts +310 -0
- package/scrape/server.ts +137 -0
- package/scrape/session.ts +109 -0
- package/scrape/sitemap.ts +131 -0
- package/scrape/tokens.ts +45 -0
- package/scrape/tsconfig.json +28 -0
- package/scrape/types.ts +214 -0
- package/scrape/utils.ts +77 -0
- package/scrape/vite.config.ts +55 -0
- package/scrape/vitest.config.ts +17 -0
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
import * as cheerio from 'cheerio';
|
|
2
|
+
import type { ExtractOptions, LinkInfo, ImageInfo, TableInfo, PageMetadata } from './types.js';
|
|
3
|
+
import { normalizeUrl, isInternalUrl } from './utils.js';
|
|
4
|
+
|
|
5
|
+
export interface ExtractedContent {
|
|
6
|
+
content: string;
|
|
7
|
+
text: string;
|
|
8
|
+
links: LinkInfo[];
|
|
9
|
+
images: ImageInfo[];
|
|
10
|
+
tables: TableInfo[];
|
|
11
|
+
metadata: PageMetadata;
|
|
12
|
+
jsonLd: unknown[];
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
function extractReadableContent($: cheerio.CheerioAPI): string {
|
|
16
|
+
const articleSelectors = [
|
|
17
|
+
'article', '[role="main"]', 'main', '.post-content', '.article-content',
|
|
18
|
+
'.entry-content', '.content', '#content', '.prose', '.markdown-body',
|
|
19
|
+
];
|
|
20
|
+
for (const sel of articleSelectors) {
|
|
21
|
+
const el = $(sel).first();
|
|
22
|
+
if (el.length) return el.text().trim();
|
|
23
|
+
}
|
|
24
|
+
return $('body').text().trim();
|
|
25
|
+
}
|
|
26
|
+
|
|
27
|
+
function extractMetadata($: cheerio.CheerioAPI): PageMetadata {
|
|
28
|
+
const getMeta = (name: string): string | undefined => {
|
|
29
|
+
return $(`meta[name="${name}"], meta[property="${name}"]`).attr('content') || undefined;
|
|
30
|
+
};
|
|
31
|
+
return {
|
|
32
|
+
title: $('title').text() || getMeta('og:title') || '',
|
|
33
|
+
description: getMeta('description') || getMeta('og:description'),
|
|
34
|
+
favicon: $('link[rel="icon"]').attr('href') || $('link[rel="shortcut icon"]').attr('href'),
|
|
35
|
+
charset: $('meta[charset]').attr('charset') || $('meta[http-equiv="Content-Type"]').attr('content'),
|
|
36
|
+
language: $('html').attr('lang') || undefined,
|
|
37
|
+
author: getMeta('author'),
|
|
38
|
+
publishedDate: getMeta('article:published_time') || getMeta('date'),
|
|
39
|
+
ogImage: getMeta('og:image'),
|
|
40
|
+
ogType: getMeta('og:type'),
|
|
41
|
+
keywords: getMeta('keywords')?.split(',').map(k => k.trim()),
|
|
42
|
+
};
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
function extractJsonLd($: cheerio.CheerioAPI): unknown[] {
|
|
46
|
+
const results: unknown[] = [];
|
|
47
|
+
$('script[type="application/ld+json"]').each((_, el) => {
|
|
48
|
+
try {
|
|
49
|
+
const data = JSON.parse($(el).html() || '');
|
|
50
|
+
if (data['@graph']) {
|
|
51
|
+
results.push(...(Array.isArray(data['@graph']) ? data['@graph'] : [data['@graph']]));
|
|
52
|
+
} else {
|
|
53
|
+
results.push(data);
|
|
54
|
+
}
|
|
55
|
+
} catch {}
|
|
56
|
+
});
|
|
57
|
+
return results;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
function extractLinks($: cheerio.CheerioAPI, baseUrl: string): LinkInfo[] {
|
|
61
|
+
const links: LinkInfo[] = [];
|
|
62
|
+
$('a[href]').each((_, el) => {
|
|
63
|
+
const href = $(el).attr('href') || '';
|
|
64
|
+
const text = $(el).text().trim().slice(0, 200);
|
|
65
|
+
if (!href || href.startsWith('#') || href.startsWith('javascript:')) return;
|
|
66
|
+
const fullUrl = normalizeUrl(href, baseUrl);
|
|
67
|
+
links.push({
|
|
68
|
+
href: fullUrl,
|
|
69
|
+
text: text || fullUrl,
|
|
70
|
+
isInternal: isInternalUrl(fullUrl, baseUrl),
|
|
71
|
+
isExternal: !isInternalUrl(fullUrl, baseUrl),
|
|
72
|
+
});
|
|
73
|
+
});
|
|
74
|
+
return links;
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
function extractImages($: cheerio.CheerioAPI, baseUrl: string): ImageInfo[] {
|
|
78
|
+
const images: ImageInfo[] = [];
|
|
79
|
+
$('img[src]').each((_, el) => {
|
|
80
|
+
const src = normalizeUrl($(el).attr('src') || '', baseUrl);
|
|
81
|
+
const alt = $(el).attr('alt') || '';
|
|
82
|
+
const width = parseInt($(el).attr('width') || '') || undefined;
|
|
83
|
+
const height = parseInt($(el).attr('height') || '') || undefined;
|
|
84
|
+
if (src) images.push({ src, alt: alt.slice(0, 200), width, height });
|
|
85
|
+
});
|
|
86
|
+
return images;
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
function extractTables($: cheerio.CheerioAPI): TableInfo[] {
|
|
90
|
+
const tables: TableInfo[] = [];
|
|
91
|
+
$('table').each((_, table) => {
|
|
92
|
+
const headers: string[] = [];
|
|
93
|
+
const rows: string[][] = [];
|
|
94
|
+
$('thead tr th, thead tr td', table).each((_, th) => {
|
|
95
|
+
headers.push($(th).text().trim());
|
|
96
|
+
});
|
|
97
|
+
$('tbody tr, > tr', table).each((_, tr) => {
|
|
98
|
+
if ($(tr).parent().is('thead')) return;
|
|
99
|
+
const row: string[] = [];
|
|
100
|
+
$('td, th', tr).each((_, td) => {
|
|
101
|
+
row.push($(td).text().trim());
|
|
102
|
+
});
|
|
103
|
+
if (row.length) rows.push(row);
|
|
104
|
+
});
|
|
105
|
+
const caption = $('caption', table).text().trim() || undefined;
|
|
106
|
+
if (rows.length) tables.push({ headers, rows, caption });
|
|
107
|
+
});
|
|
108
|
+
return tables;
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
function htmlToText(html: string): string {
|
|
112
|
+
return html
|
|
113
|
+
.replace(/<br\s*\/?>/gi, '\n')
|
|
114
|
+
.replace(/<\/p>/gi, '\n\n')
|
|
115
|
+
.replace(/<\/div>/gi, '\n')
|
|
116
|
+
.replace(/<\/h[1-6]>/gi, '\n')
|
|
117
|
+
.replace(/<\/li>/gi, '\n')
|
|
118
|
+
.replace(/<[^>]+>/g, '')
|
|
119
|
+
.replace(/ /g, ' ')
|
|
120
|
+
.replace(/&/g, '&')
|
|
121
|
+
.replace(/</g, '<')
|
|
122
|
+
.replace(/>/g, '>')
|
|
123
|
+
.replace(/"/g, '"')
|
|
124
|
+
.replace(/'/g, "'")
|
|
125
|
+
.replace(/\n{3,}/g, '\n\n')
|
|
126
|
+
.trim();
|
|
127
|
+
}
|
|
128
|
+
|
|
129
|
+
export async function extractContent(html: string, options: ExtractOptions = {}): Promise<ExtractedContent> {
|
|
130
|
+
const $ = cheerio.load(html);
|
|
131
|
+
const baseUrl = $('base').attr('href') || '';
|
|
132
|
+
|
|
133
|
+
if (options.removeSelectors?.length) {
|
|
134
|
+
for (const sel of options.removeSelectors) {
|
|
135
|
+
$(sel).remove();
|
|
136
|
+
}
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
const metadata = extractMetadata($);
|
|
140
|
+
const jsonLd = extractJsonLd($);
|
|
141
|
+
const links = options.preserveLinks !== false ? extractLinks($, baseUrl) : [];
|
|
142
|
+
const images = options.preserveImages !== false ? extractImages($, baseUrl) : [];
|
|
143
|
+
const tables = options.preserveTables !== false ? extractTables($) : [];
|
|
144
|
+
|
|
145
|
+
const text = options.readable
|
|
146
|
+
? extractReadableContent($)
|
|
147
|
+
: $('body').text().trim();
|
|
148
|
+
|
|
149
|
+
const content = htmlToText(text);
|
|
150
|
+
|
|
151
|
+
return {
|
|
152
|
+
content: options.maxLength ? content.slice(0, options.maxLength) : content,
|
|
153
|
+
text,
|
|
154
|
+
links,
|
|
155
|
+
images,
|
|
156
|
+
tables,
|
|
157
|
+
metadata,
|
|
158
|
+
jsonLd,
|
|
159
|
+
};
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
export async function extractReadable(html: string): Promise<string> {
|
|
163
|
+
const result = await extractContent(html, { readable: true });
|
|
164
|
+
return result.content;
|
|
165
|
+
}
|
package/scrape/fetch.ts
ADDED
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
import * as cheerio from 'cheerio';
|
|
2
|
+
import type { ScrapeOptions } from './types.js';
|
|
3
|
+
import { getHeaders, getRandomProfile } from './headers.js';
|
|
4
|
+
import { WebScrapeError, ErrorCode } from './errors.js';
|
|
5
|
+
|
|
6
|
+
export interface FetchResult {
|
|
7
|
+
html: string;
|
|
8
|
+
url: string;
|
|
9
|
+
status: number;
|
|
10
|
+
headers: Record<string, string>;
|
|
11
|
+
duration: number;
|
|
12
|
+
}
|
|
13
|
+
|
|
14
|
+
export async function fetchHtml(url: string, options: {
|
|
15
|
+
timeout?: number;
|
|
16
|
+
headers?: Record<string, string>;
|
|
17
|
+
userAgent?: string;
|
|
18
|
+
proxy?: string;
|
|
19
|
+
signal?: AbortSignal;
|
|
20
|
+
} = {}): Promise<FetchResult> {
|
|
21
|
+
const startTime = Date.now();
|
|
22
|
+
const profile = getRandomProfile();
|
|
23
|
+
const baseHeaders = getHeaders(profile);
|
|
24
|
+
const headers = { ...baseHeaders, ...options.headers };
|
|
25
|
+
if (options.userAgent) headers['User-Agent'] = options.userAgent;
|
|
26
|
+
|
|
27
|
+
const controller = new AbortController();
|
|
28
|
+
const timeoutId = setTimeout(() => controller.abort(), options.timeout || 30000);
|
|
29
|
+
const signal = options.signal ? AbortSignal.any([controller.signal, options.signal]) : controller.signal;
|
|
30
|
+
|
|
31
|
+
try {
|
|
32
|
+
const response = await fetch(url, {
|
|
33
|
+
method: 'GET',
|
|
34
|
+
headers,
|
|
35
|
+
signal,
|
|
36
|
+
redirect: 'follow',
|
|
37
|
+
});
|
|
38
|
+
|
|
39
|
+
clearTimeout(timeoutId);
|
|
40
|
+
|
|
41
|
+
if (!response.ok) {
|
|
42
|
+
throw new WebScrapeError(
|
|
43
|
+
`HTTP ${response.status}: ${response.statusText}`,
|
|
44
|
+
response.status === 403 ? ErrorCode.BLOCKED :
|
|
45
|
+
response.status === 429 ? ErrorCode.RATE_LIMITED :
|
|
46
|
+
response.status >= 500 ? ErrorCode.NETWORK_ERROR :
|
|
47
|
+
ErrorCode.NETWORK_ERROR,
|
|
48
|
+
response.status,
|
|
49
|
+
[408, 429, 500, 502, 503, 504].includes(response.status),
|
|
50
|
+
{ url, status: response.status }
|
|
51
|
+
);
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
const html = await response.text();
|
|
55
|
+
const responseHeaders: Record<string, string> = {};
|
|
56
|
+
response.headers.forEach((value, key) => { responseHeaders[key] = value; });
|
|
57
|
+
|
|
58
|
+
return {
|
|
59
|
+
html,
|
|
60
|
+
url: response.url || url,
|
|
61
|
+
status: response.status,
|
|
62
|
+
headers: responseHeaders,
|
|
63
|
+
duration: Date.now() - startTime,
|
|
64
|
+
};
|
|
65
|
+
} catch (error) {
|
|
66
|
+
clearTimeout(timeoutId);
|
|
67
|
+
if (error instanceof WebScrapeError) throw error;
|
|
68
|
+
if ((error as Error).name === 'AbortError') {
|
|
69
|
+
throw new WebScrapeError(
|
|
70
|
+
`Request timeout after ${options.timeout || 30000}ms`,
|
|
71
|
+
ErrorCode.TIMEOUT,
|
|
72
|
+
504,
|
|
73
|
+
true,
|
|
74
|
+
{ url, timeout: options.timeout || 30000 }
|
|
75
|
+
);
|
|
76
|
+
}
|
|
77
|
+
throw new WebScrapeError(
|
|
78
|
+
`Network error: ${(error as Error).message}`,
|
|
79
|
+
ErrorCode.NETWORK_ERROR,
|
|
80
|
+
503,
|
|
81
|
+
true,
|
|
82
|
+
{ url },
|
|
83
|
+
error as Error
|
|
84
|
+
);
|
|
85
|
+
}
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
export async function fetchAndParse(url: string, options: Partial<ScrapeOptions> = {}): Promise<FetchResult> {
|
|
89
|
+
return fetchHtml(url, {
|
|
90
|
+
timeout: options.timeout,
|
|
91
|
+
headers: options.headers,
|
|
92
|
+
userAgent: options.userAgent,
|
|
93
|
+
proxy: options.proxy,
|
|
94
|
+
});
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
export function detectRenderingMode(html: string): 'static' | 'spa' | 'hydrated' {
|
|
98
|
+
if (/<div id=["'](?:root|app|__next|__nuxt)["']>\s*<\/div>/.test(html)) return 'spa';
|
|
99
|
+
if (html.includes('__NEXT_DATA__') || html.includes('__NUXT__') || html.includes('__VUE_SSR_DATA__')) return 'hydrated';
|
|
100
|
+
const $ = cheerio.load(html);
|
|
101
|
+
if ($('script[type="application/ld+json"]').length > 0) return 'static';
|
|
102
|
+
const textContent = $.text().replace(/\s+/g, ' ').trim();
|
|
103
|
+
if (textContent.length > 500) return 'static';
|
|
104
|
+
return 'spa';
|
|
105
|
+
}
|
|
@@ -0,0 +1,85 @@
|
|
|
1
|
+
import type { ScrapeFormat, SerializeTarget, SerializeOptions, ScrapeResult } from './types.js';
|
|
2
|
+
import { serializeResult as sr } from './serialize.js';
|
|
3
|
+
|
|
4
|
+
const FORMAT_ALIASES: Record<string, ScrapeFormat> = {
|
|
5
|
+
md: 'markdown',
|
|
6
|
+
txt: 'text',
|
|
7
|
+
plain: 'text',
|
|
8
|
+
xml: 'xml',
|
|
9
|
+
json: 'json',
|
|
10
|
+
redis: 'redis',
|
|
11
|
+
html: 'html',
|
|
12
|
+
h: 'html',
|
|
13
|
+
};
|
|
14
|
+
|
|
15
|
+
const FORMAT_EXTENSIONS: Record<string, ScrapeFormat> = {
|
|
16
|
+
'.md': 'markdown',
|
|
17
|
+
'.markdown': 'markdown',
|
|
18
|
+
'.html': 'html',
|
|
19
|
+
'.htm': 'html',
|
|
20
|
+
'.txt': 'text',
|
|
21
|
+
'.json': 'json',
|
|
22
|
+
'.xml': 'xml',
|
|
23
|
+
'.redis': 'redis',
|
|
24
|
+
};
|
|
25
|
+
|
|
26
|
+
export function resolveFormat(format: string): ScrapeFormat {
|
|
27
|
+
const lower = format.toLowerCase().trim();
|
|
28
|
+
return FORMAT_ALIASES[lower] || (lower as ScrapeFormat);
|
|
29
|
+
}
|
|
30
|
+
|
|
31
|
+
export function formatFromExtension(ext: string): ScrapeFormat | null {
|
|
32
|
+
return FORMAT_EXTENSIONS[ext.toLowerCase()] ?? null;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export function extensionForFormat(format: ScrapeFormat): string {
|
|
36
|
+
const map: Record<ScrapeFormat, string> = {
|
|
37
|
+
markdown: '.md',
|
|
38
|
+
html: '.html',
|
|
39
|
+
text: '.txt',
|
|
40
|
+
json: '.json',
|
|
41
|
+
xml: '.xml',
|
|
42
|
+
redis: '.redis',
|
|
43
|
+
};
|
|
44
|
+
return map[format] || '.md';
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
export function detectContentType(html: string): 'article' | 'list' | 'page' | 'other' {
|
|
48
|
+
if (html.includes('<article') || html.includes('entry-content') || html.includes('class="post-content"') || html.includes('class="article-content"') || html.includes('role="main"')) {
|
|
49
|
+
return 'article';
|
|
50
|
+
}
|
|
51
|
+
const listCount = (html.match(/<li>/gi) || []).length;
|
|
52
|
+
if (listCount > 20) return 'list';
|
|
53
|
+
if (html.includes('<article') || html.includes('entry-content')) return 'article';
|
|
54
|
+
return 'page';
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export function buildSerializeOptions(format: ScrapeFormat, pretty = true): SerializeOptions {
|
|
58
|
+
const targetMap: Record<ScrapeFormat, SerializeTarget> = {
|
|
59
|
+
markdown: 'markdown',
|
|
60
|
+
html: 'markdown',
|
|
61
|
+
text: 'text',
|
|
62
|
+
json: 'json',
|
|
63
|
+
xml: 'xml',
|
|
64
|
+
redis: 'redis',
|
|
65
|
+
};
|
|
66
|
+
return {
|
|
67
|
+
format: targetMap[format] || 'markdown',
|
|
68
|
+
pretty,
|
|
69
|
+
includeMetadata: true,
|
|
70
|
+
};
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
export async function convertResult(result: ScrapeResult, targetFormat: ScrapeFormat): Promise<ScrapeResult> {
|
|
74
|
+
if (result.format === targetFormat) return result;
|
|
75
|
+
|
|
76
|
+
const opts = buildSerializeOptions(targetFormat);
|
|
77
|
+
const serialized = sr(result, opts);
|
|
78
|
+
|
|
79
|
+
return {
|
|
80
|
+
...result,
|
|
81
|
+
content: serialized.content,
|
|
82
|
+
format: targetFormat,
|
|
83
|
+
size: serialized.size,
|
|
84
|
+
};
|
|
85
|
+
}
|
|
@@ -0,0 +1,71 @@
|
|
|
1
|
+
export interface HeaderProfile {
|
|
2
|
+
userAgent: string;
|
|
3
|
+
accept: string;
|
|
4
|
+
acceptLanguage: string;
|
|
5
|
+
secChUa: string;
|
|
6
|
+
secChUaPlatform: string;
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
const CHROME_PROFILES: HeaderProfile[] = [
|
|
10
|
+
{
|
|
11
|
+
userAgent: 'Mozilla/5.0 (Windows NT 10.0; Win64; x64) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
|
|
12
|
+
accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8',
|
|
13
|
+
acceptLanguage: 'en-US,en;q=0.9',
|
|
14
|
+
secChUa: '"Chromium";v="131", "Not_A Brand";v="24"',
|
|
15
|
+
secChUaPlatform: '"Windows"',
|
|
16
|
+
},
|
|
17
|
+
{
|
|
18
|
+
userAgent: 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 (KHTML, like Gecko) Chrome/131.0.0.0 Safari/537.36',
|
|
19
|
+
accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,image/apng,*/*;q=0.8',
|
|
20
|
+
acceptLanguage: 'en-US,en;q=0.9',
|
|
21
|
+
secChUa: '"Chromium";v="131", "Not_A Brand";v="24"',
|
|
22
|
+
secChUaPlatform: '"macOS"',
|
|
23
|
+
},
|
|
24
|
+
{
|
|
25
|
+
userAgent: 'Mozilla/5.0 (Windows NT 10.0; Win64; x64; rv:133.0) Gecko/20100101 Firefox/133.0',
|
|
26
|
+
accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,image/avif,image/webp,*/*;q=0.8',
|
|
27
|
+
acceptLanguage: 'en-US,en;q=0.5',
|
|
28
|
+
secChUa: '',
|
|
29
|
+
secChUaPlatform: '"Windows"',
|
|
30
|
+
},
|
|
31
|
+
{
|
|
32
|
+
userAgent: 'Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/605.1.15 (KHTML, like Gecko) Version/18.2 Safari/605.1.15',
|
|
33
|
+
accept: 'text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8',
|
|
34
|
+
acceptLanguage: 'en-US,en;q=0.9',
|
|
35
|
+
secChUa: '',
|
|
36
|
+
secChUaPlatform: '"macOS"',
|
|
37
|
+
},
|
|
38
|
+
];
|
|
39
|
+
|
|
40
|
+
export function getRandomProfile(): HeaderProfile {
|
|
41
|
+
return CHROME_PROFILES[Math.floor(Math.random() * CHROME_PROFILES.length)];
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
export function getHeaders(profile?: HeaderProfile): Record<string, string> {
|
|
45
|
+
const p = profile || getRandomProfile();
|
|
46
|
+
const headers: Record<string, string> = {
|
|
47
|
+
'User-Agent': p.userAgent,
|
|
48
|
+
'Accept': p.accept,
|
|
49
|
+
'Accept-Language': p.acceptLanguage,
|
|
50
|
+
'Accept-Encoding': 'gzip, deflate, br',
|
|
51
|
+
'Connection': 'keep-alive',
|
|
52
|
+
'Upgrade-Insecure-Requests': '1',
|
|
53
|
+
'Sec-Fetch-Dest': 'document',
|
|
54
|
+
'Sec-Fetch-Mode': 'navigate',
|
|
55
|
+
'Sec-Fetch-Site': 'none',
|
|
56
|
+
'Sec-Fetch-User': '?1',
|
|
57
|
+
};
|
|
58
|
+
if (p.secChUa) {
|
|
59
|
+
headers['sec-ch-ua'] = p.secChUa;
|
|
60
|
+
headers['sec-ch-ua-mobile'] = '?0';
|
|
61
|
+
headers['sec-ch-ua-platform'] = p.secChUaPlatform;
|
|
62
|
+
}
|
|
63
|
+
return headers;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export function mergeHeaders(
|
|
67
|
+
base: Record<string, string>,
|
|
68
|
+
override: Record<string, string>
|
|
69
|
+
): Record<string, string> {
|
|
70
|
+
return { ...base, ...override };
|
|
71
|
+
}
|
package/scrape/index.ts
ADDED
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
// ─── Core Scraping ───
|
|
2
|
+
export { AgentBrowser, createBrowser } from './browser.js';
|
|
3
|
+
export { scrapeUrl, scrapeHtml, scrapeWithBrowser } from './scrape.js';
|
|
4
|
+
export { extractContent, extractReadable } from './extract.js';
|
|
5
|
+
export type { ExtractedContent } from './extract.js';
|
|
6
|
+
export { fetchHtml, fetchAndParse, detectRenderingMode } from './fetch.js';
|
|
7
|
+
export type { FetchResult } from './fetch.js';
|
|
8
|
+
|
|
9
|
+
// ─── Serialization ───
|
|
10
|
+
export { serializeResult, serializeHtml } from './serialize.js';
|
|
11
|
+
export { resolveFormat, formatFromExtension, extensionForFormat, buildSerializeOptions, convertResult } from './formats.js';
|
|
12
|
+
|
|
13
|
+
// ─── Renderer ───
|
|
14
|
+
export { PygameRenderer, createRenderer } from './renderer.js';
|
|
15
|
+
|
|
16
|
+
// ─── AI/LLM ───
|
|
17
|
+
export { formatForAgent, buildContext } from './agent.js';
|
|
18
|
+
export { estimateTokens, countTokens, fitsInContext, truncateToTokens, tokenCost } from './tokens.js';
|
|
19
|
+
export type { ModelType } from './tokens.js';
|
|
20
|
+
export { chunkMarkdown, chunkText, formatChunksForRAG } from './chunking.js';
|
|
21
|
+
export type { Chunk, ChunkOptions } from './chunking.js';
|
|
22
|
+
export { generateLlmsTxt, generateLlmsFullTxt } from './llms-txt.js';
|
|
23
|
+
export type { LlmsTxtOptions } from './llms-txt.js';
|
|
24
|
+
|
|
25
|
+
// ─── Infrastructure ───
|
|
26
|
+
export { WebScrapeError, ValidationError, TimeoutError, BlockedError, RateLimitError, ProxyError, ParseError, AuthError, NetworkError, BrowserNotAvailableError } from './errors.js';
|
|
27
|
+
export type { ErrorCode } from './errors.js';
|
|
28
|
+
export { createEventEmitter, globalEmitter } from './events.js';
|
|
29
|
+
export type { WebScrapeEvents, EventEmitter } from './events.js';
|
|
30
|
+
export { withRetry, isRetryableError, AbortError } from './retry.js';
|
|
31
|
+
export { RateLimiter, createRateLimiter } from './rate-limiter.js';
|
|
32
|
+
export type { RateLimiterConfig } from './rate-limiter.js';
|
|
33
|
+
export { ProxyPool, createProxyPool } from './proxy.js';
|
|
34
|
+
export type { ProxyPoolConfig, ProxyRotationStrategy } from './proxy.js';
|
|
35
|
+
export { CookieJar, ScrapingSession, createSession } from './session.js';
|
|
36
|
+
export type { Cookie } from './session.js';
|
|
37
|
+
export { getRandomProfile, getHeaders, mergeHeaders } from './headers.js';
|
|
38
|
+
export type { HeaderProfile } from './headers.js';
|
|
39
|
+
export { WebScrapeCache, createCache } from './cache.js';
|
|
40
|
+
export type { CacheConfig, CacheEntry } from './cache.js';
|
|
41
|
+
export { MiddlewarePipeline, createPipeline, loggingMiddleware, timeoutMiddleware, retryMiddleware } from './middleware.js';
|
|
42
|
+
export type { Middleware, MiddlewareContext, MiddlewareNext } from './middleware.js';
|
|
43
|
+
|
|
44
|
+
// ─── Crawling ───
|
|
45
|
+
export { crawl } from './crawler.js';
|
|
46
|
+
export type { CrawlOptions } from './crawler.js';
|
|
47
|
+
export { fetchSitemap, discoverSitemaps, parseSitemap } from './sitemap.js';
|
|
48
|
+
export type { SitemapUrl, SitemapResult } from './sitemap.js';
|
|
49
|
+
export { fetchRobotsTxt, isAllowed, getCrawlDelay, getSitemaps } from './robots.js';
|
|
50
|
+
export type { RobotsTxt, RobotsDirective } from './robots.js';
|
|
51
|
+
|
|
52
|
+
// ─── Batch & Pool ───
|
|
53
|
+
export { batchScrape, batchScrapeSequential } from './batch.js';
|
|
54
|
+
export type { BatchResult } from './batch.js';
|
|
55
|
+
export { BrowserPool, createPool } from './pool.js';
|
|
56
|
+
export type { PoolConfig } from './pool.js';
|
|
57
|
+
|
|
58
|
+
// ─── Server ───
|
|
59
|
+
export { createServer, app } from './server.js';
|
|
60
|
+
export type { ServerConfig } from './server.js';
|
|
61
|
+
|
|
62
|
+
// ─── Utilities ───
|
|
63
|
+
export { ensureDir, writeOutput, slugify, truncate, chunkText as chunkTextUtil, delay, isValidUrl, normalizeUrl, isInternalUrl } from './utils.js';
|
|
64
|
+
export { randomPort, resetPort } from './utils/port.js';
|
|
65
|
+
|
|
66
|
+
// ─── Types ───
|
|
67
|
+
export {
|
|
68
|
+
ScrapeOptionsSchema,
|
|
69
|
+
BrowserAgentConfigSchema,
|
|
70
|
+
} from './types.js';
|
|
71
|
+
export type {
|
|
72
|
+
ScrapeFormat,
|
|
73
|
+
SerializeTarget,
|
|
74
|
+
ScrapeOptions,
|
|
75
|
+
ScrapeResult,
|
|
76
|
+
PageMetadata,
|
|
77
|
+
LinkInfo,
|
|
78
|
+
ImageInfo,
|
|
79
|
+
TableInfo,
|
|
80
|
+
BrowserAgentConfig,
|
|
81
|
+
ExtractOptions,
|
|
82
|
+
SerializeOptions,
|
|
83
|
+
SerializedOutput,
|
|
84
|
+
AgentOutput,
|
|
85
|
+
RenderOptions,
|
|
86
|
+
ChromeCommand,
|
|
87
|
+
CliOptions,
|
|
88
|
+
RetryConfig,
|
|
89
|
+
ProxyConfig,
|
|
90
|
+
CrawlStats,
|
|
91
|
+
BatchOptions,
|
|
92
|
+
} from './types.js';
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
import type { ScrapeResult } from './types.js';
|
|
2
|
+
|
|
3
|
+
export interface LlmsTxtOptions {
|
|
4
|
+
siteName?: string;
|
|
5
|
+
description?: string;
|
|
6
|
+
includeOptional?: boolean;
|
|
7
|
+
}
|
|
8
|
+
|
|
9
|
+
export function generateLlmsTxt(
|
|
10
|
+
results: ScrapeResult[],
|
|
11
|
+
options: LlmsTxtOptions = {}
|
|
12
|
+
): string {
|
|
13
|
+
const siteName = options.siteName || results[0]?.metadata?.title || 'Site';
|
|
14
|
+
const description = options.description || results[0]?.metadata?.description || '';
|
|
15
|
+
|
|
16
|
+
const lines: string[] = [];
|
|
17
|
+
lines.push(`# ${siteName}`);
|
|
18
|
+
lines.push('');
|
|
19
|
+
if (description) {
|
|
20
|
+
lines.push(`> ${description}`);
|
|
21
|
+
lines.push('');
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
const pages: ScrapeResult[] = [];
|
|
25
|
+
const docs: ScrapeResult[] = [];
|
|
26
|
+
|
|
27
|
+
for (const result of results) {
|
|
28
|
+
const url = result.url.toLowerCase();
|
|
29
|
+
if (url.includes('doc') || url.includes('guide') || url.includes('api')) {
|
|
30
|
+
docs.push(result);
|
|
31
|
+
} else {
|
|
32
|
+
pages.push(result);
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
if (pages.length > 0) {
|
|
37
|
+
lines.push('## Pages');
|
|
38
|
+
lines.push('');
|
|
39
|
+
for (const page of pages.slice(0, 20)) {
|
|
40
|
+
const title = page.metadata.title || page.title;
|
|
41
|
+
const desc = page.metadata.description || '';
|
|
42
|
+
lines.push(`- ${title}: ${desc || page.url}`);
|
|
43
|
+
}
|
|
44
|
+
lines.push('');
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
if (docs.length > 0) {
|
|
48
|
+
lines.push('## Documentation');
|
|
49
|
+
lines.push('');
|
|
50
|
+
for (const doc of docs.slice(0, 20)) {
|
|
51
|
+
const title = doc.metadata.title || doc.title;
|
|
52
|
+
const desc = doc.metadata.description || '';
|
|
53
|
+
lines.push(`- ${title}: ${desc || doc.url}`);
|
|
54
|
+
}
|
|
55
|
+
lines.push('');
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
if (options.includeOptional && results.length > 5) {
|
|
59
|
+
lines.push('## Optional');
|
|
60
|
+
lines.push('');
|
|
61
|
+
lines.push(`- Full content: ${results.length} pages scraped`);
|
|
62
|
+
lines.push(`- Last updated: ${new Date().toISOString().split('T')[0]}`);
|
|
63
|
+
lines.push('');
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
return lines.join('\n');
|
|
67
|
+
}
|
|
68
|
+
|
|
69
|
+
export function generateLlmsFullTxt(results: ScrapeResult[]): string {
|
|
70
|
+
const parts: string[] = [];
|
|
71
|
+
|
|
72
|
+
for (const result of results) {
|
|
73
|
+
parts.push(`# ${result.title || result.metadata.title || 'Untitled'}`);
|
|
74
|
+
parts.push('');
|
|
75
|
+
parts.push(`Source: ${result.url}`);
|
|
76
|
+
parts.push('');
|
|
77
|
+
parts.push(result.content);
|
|
78
|
+
parts.push('');
|
|
79
|
+
parts.push('---');
|
|
80
|
+
parts.push('');
|
|
81
|
+
}
|
|
82
|
+
|
|
83
|
+
return parts.join('\n');
|
|
84
|
+
}
|