@wenathlan/saddle 1.8.4 → 1.8.6

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (263) hide show
  1. package/README.md +10 -10
  2. package/browser/playwright.js +22 -0
  3. package/docs/.gitkeep +0 -0
  4. package/docs/actionsincident.md +29 -0
  5. package/docs/branchaudit.md +21 -0
  6. package/docs/ecosystemplan.md +3 -3
  7. package/docs/featureaudit.md +1 -1
  8. package/docs/gapmatrix.md +8 -7
  9. package/docs/libraryapi.md +1 -0
  10. package/docs/logs/.gitkeep +0 -0
  11. package/docs/packageaudit185.md +23 -0
  12. package/docs/plans/00.index.md +50 -0
  13. package/docs/plans/01.architecture.md +86 -0
  14. package/docs/plans/02.research.computer.use.md +58 -0
  15. package/docs/plans/03.research.captcha.bypass.md +68 -0
  16. package/docs/plans/04.research.sandbox.ai.md +52 -0
  17. package/docs/plans/05.capture.platform.md +57 -0
  18. package/docs/plans/06.dependencies.md +97 -0
  19. package/docs/plans/07.captcha.test.page.md +41 -0
  20. package/docs/plans/08.production.infra.md +70 -0
  21. package/docs/plans/09.database.schema.md +121 -0
  22. package/docs/plans/10.cloudinary.storage.md +57 -0
  23. package/docs/plans/11.movement.logs.json.md +72 -0
  24. package/docs/plans/12.research.atlas.agent.browser.md +79 -0
  25. package/docs/plans/13.research.anti.detection.md +898 -0
  26. package/docs/plans/14.research.proxy.md +1495 -0
  27. package/docs/plans/15.research.retry.rate.limit.md +1958 -0
  28. package/docs/plans/16.research.crawling.md +1417 -0
  29. package/docs/plans/17.research.caching.md +1610 -0
  30. package/docs/plans/18.research.content.extraction.md +1952 -0
  31. package/docs/plans/19.research.errors.events.md +1523 -0
  32. package/docs/plans/20.research.zod.validation.md +1350 -0
  33. package/docs/plans/21.research.batch.concurrency.md +1888 -0
  34. package/docs/plans/22.research.universal.runtime.md +944 -0
  35. package/docs/plans/23.research.ai.integration.md +1465 -0
  36. package/docs/plans/24.research.memory.persistence.md +1979 -0
  37. package/docs/plans/25.research.server.api.md +342 -0
  38. package/docs/plans/26.research.compilation.md +249 -0
  39. package/docs/plans/27.research.html.parsing.md +251 -0
  40. package/docs/plans/28.action.plan.md +50 -0
  41. package/docs/plans/29.api.reference.md +174 -0
  42. package/docs/plans/30.architecture.plan.md +94 -0
  43. package/docs/plans/31.auditoria.dados.md +163 -0
  44. package/docs/plans/32.bots.automacao.computacional.md +214 -0
  45. package/docs/plans/33.bots.codigo.revisao.md +220 -0
  46. package/docs/plans/34.bots.seguranca.cicd.md +366 -0
  47. package/docs/plans/35.comparativo.concorrencia.md +464 -0
  48. package/docs/plans/36.computational.memory.md +340 -0
  49. package/docs/plans/37.deploystrategy.md +394 -0
  50. package/docs/plans/38.flow.md +155 -0
  51. package/docs/plans/39.multi.platform.bot.md +252 -0
  52. package/docs/plans/40.npm.publish.md +250 -0
  53. package/docs/plans/41.o.que.falta.md +407 -0
  54. package/docs/plans/42.pesquisa.concorrencia.md +721 -0
  55. package/docs/plans/43.plan.universal.architecture.md +496 -0
  56. package/docs/plans/44.reference.md +100 -0
  57. package/docs/plans/45.robotarchitecture.md +237 -0
  58. package/docs/plans/46.scdnintegration.md +284 -0
  59. package/docs/plans/47.multiforge.readme.md +129 -0
  60. package/docs/plans/48.theory.v4.repo.os.md +152 -0
  61. package/docs/plans/49.third.party.infra.md +12 -0
  62. package/docs/plans/50.file.as.compute.md +39 -0
  63. package/docs/plans/51.architecture.virtual.processor.md +80 -0
  64. package/docs/plans/52.manifesto.v8.md +11 -0
  65. package/docs/plans/58.cdn.list.md +23 -0
  66. package/docs/plans/59.sql.frameworks.md +33 -0
  67. package/docs/plans/60.sql.thirdparty.md +26 -0
  68. package/docs/plans/61.objective.multiforge.md +63 -0
  69. package/docs/plans/62.huggingface.upload.md +26 -0
  70. package/docs/plans/63.kaggle.upload.md +24 -0
  71. package/docs/plans/64.npm.storage.md +30 -0
  72. package/docs/plans/65.rclone.terabox.md +32 -0
  73. package/docs/plans/66.buckets.and.models.todo.md +14 -0
  74. package/docs/plans/67.database.todo.md +13 -0
  75. package/docs/plans/68.deploy.packages.todo.md +12 -0
  76. package/docs/plans/69.report.human.operator.md +133 -0
  77. package/docs/plans/70.report.brain2qwerty.ems.md +135 -0
  78. package/docs/plans/71.report.hd.infinito.vram.md +155 -0
  79. package/docs/plans/72.plan.hd.infinito.node.md +146 -0
  80. package/docs/plans/73.plan.scifi.repos.md +125 -0
  81. package/docs/plans/74.000.manifesto.v8.flat.2..md +11 -0
  82. package/docs/plans/README.md +489 -0
  83. package/docs/plans/aggregate_platforms.mjs +146 -0
  84. package/docs/plans/examplesession.json +36 -0
  85. package/docs/plans/missing-facts.md +192 -0
  86. package/docs/plans/models.md +64 -0
  87. package/docs/plans/organize.cjs +270 -0
  88. package/docs/plans/platforms.md +2887 -0
  89. package/docs/plans/sites.md +31322 -0
  90. package/docs/platformpipelineaudit.md +6 -2
  91. package/docs/registryresearch.md +2 -0
  92. package/docs/release.md +4 -4
  93. package/docs/release184notes.md +2 -2
  94. package/docs/release185notes.md +7 -0
  95. package/docs/releaseassets.md +16 -0
  96. package/docs/sources/farm.py +117 -0
  97. package/docs/sources/html/saddle1.html +132 -0
  98. package/docs/sources/html/saddle2.html +157 -0
  99. package/docs/sources/html/saddle3.html +119 -0
  100. package/docs/sources/html/saddle4.html +144 -0
  101. package/docs/sources/html/saddle5.html +72 -0
  102. package/docs/sources/html/saddle6.html +171 -0
  103. package/docs/sources/html/saddle7.html +236 -0
  104. package/docs/sources/saddle.ts +74 -0
  105. package/docs/sources/schema.prisma +88 -0
  106. package/docs/sources/script.sh +64 -0
  107. package/docs/sources/workflows.yml +458 -0
  108. package/docs/talks1/_body.txt +14 -0
  109. package/docs/talks1/_index.md +15 -0
  110. package/docs/talks1/_screenshot.png +0 -0
  111. package/docs/talks1/assistant-01.md +5 -0
  112. package/docs/talks1/assistant-02.md +5 -0
  113. package/docs/talks1/assistant-03.md +531 -0
  114. package/docs/talks1/assistant-04.md +26 -0
  115. package/docs/talks1/assistant-05.md +774 -0
  116. package/docs/talks1/assistant-06.md +1718 -0
  117. package/docs/talks1/scrape-share.cjs +185 -0
  118. package/docs/talks1/scrape-share.ts +183 -0
  119. package/docs/talks1/user-01.md +3 -0
  120. package/docs/talks1/user-02.md +3 -0
  121. package/docs/talks1/user-03.md +88 -0
  122. package/docs/talks1/user-04.md +3 -0
  123. package/docs/talks1/user-05.md +3 -0
  124. package/docs/talks1/user-06.md +88 -0
  125. package/docs/talks1/user-07.md +88 -0
  126. package/docs/talks2/_body.txt +14 -0
  127. package/docs/talks2/_index.md +16 -0
  128. package/docs/talks2/_screenshot.png +0 -0
  129. package/docs/talks2/assistant-01.md +5 -0
  130. package/docs/talks2/assistant-02.md +5 -0
  131. package/docs/talks2/assistant-03.md +424 -0
  132. package/docs/talks2/assistant-04.md +598 -0
  133. package/docs/talks2/assistant-05.md +1280 -0
  134. package/docs/talks2/assistant-06.md +1227 -0
  135. package/docs/talks2/assistant-07.md +1252 -0
  136. package/docs/talks2/user-01.md +3 -0
  137. package/docs/talks2/user-02.md +3 -0
  138. package/docs/talks2/user-03.md +88 -0
  139. package/docs/talks2/user-04.md +88 -0
  140. package/docs/talks2/user-05.md +88 -0
  141. package/docs/talks2/user-06.md +88 -0
  142. package/docs/talks2/user-07.md +3 -0
  143. package/docs/talks3/_body.txt +467 -0
  144. package/docs/talks3/_index.md +10 -0
  145. package/docs/talks3/_screenshot.png +0 -0
  146. package/docs/talks3/assistant-01.md +417 -0
  147. package/docs/talks3/assistant-02.md +417 -0
  148. package/docs/talks3/assistant-03.md +29 -0
  149. package/docs/talks3/assistant-04.md +727 -0
  150. package/docs/talks3/user-01.md +88 -0
  151. package/docs/talks3/user-02.md +88 -0
  152. package/docs/talks3/user-03.md +3 -0
  153. package/docs/talks3/user-04.md +3 -0
  154. package/docs/talks4/_body.txt +14 -0
  155. package/docs/talks4/_index.md +12 -0
  156. package/docs/talks4/_screenshot.png +0 -0
  157. package/docs/talks4/assistant-01.md +5 -0
  158. package/docs/talks4/assistant-02.md +5 -0
  159. package/docs/talks4/assistant-03.md +35 -0
  160. package/docs/talks4/assistant-04.md +512 -0
  161. package/docs/talks4/assistant-05.md +599 -0
  162. package/docs/talks4/user-01.md +3 -0
  163. package/docs/talks4/user-02.md +3 -0
  164. package/docs/talks4/user-03.md +88 -0
  165. package/docs/talks4/user-04.md +88 -0
  166. package/docs/talks4/user-05.md +7 -0
  167. package/docs/talks5/_body.txt +14 -0
  168. package/docs/talks5/_index.md +13 -0
  169. package/docs/talks5/_screenshot.png +0 -0
  170. package/docs/talks5/assistant-01.md +5 -0
  171. package/docs/talks5/assistant-02.md +5 -0
  172. package/docs/talks5/assistant-03.md +690 -0
  173. package/docs/talks5/assistant-04.md +758 -0
  174. package/docs/talks5/assistant-05.md +974 -0
  175. package/docs/talks5/user-01.md +3 -0
  176. package/docs/talks5/user-02.md +3 -0
  177. package/docs/talks5/user-03.md +105 -0
  178. package/docs/talks5/user-04.md +105 -0
  179. package/docs/talks5/user-05.md +63 -0
  180. package/docs/talks5/user-06.md +105 -0
  181. package/docs/talks6/_body.txt +14 -0
  182. package/docs/talks6/_index.md +9 -0
  183. package/docs/talks6/_screenshot.png +0 -0
  184. package/docs/talks6/assistant-01.md +5 -0
  185. package/docs/talks6/assistant-02.md +5 -0
  186. package/docs/talks6/assistant-03.md +1499 -0
  187. package/docs/talks6/user-01.md +3 -0
  188. package/docs/talks6/user-02.md +3 -0
  189. package/docs/talks6/user-03.md +88 -0
  190. package/docs/talks6/user-04.md +88 -0
  191. package/docs/talks7/_body.txt +14 -0
  192. package/docs/talks7/_index.md +10 -0
  193. package/docs/talks7/_screenshot.png +0 -0
  194. package/docs/talks7/assistant-01.md +5 -0
  195. package/docs/talks7/assistant-02.md +5 -0
  196. package/docs/talks7/assistant-03.md +523 -0
  197. package/docs/talks7/assistant-04.md +617 -0
  198. package/docs/talks7/user-01.md +3 -0
  199. package/docs/talks7/user-02.md +3 -0
  200. package/docs/talks7/user-03.md +105 -0
  201. package/docs/talks7/user-04.md +67 -0
  202. package/docs/talks8/conversa1.txt +1322 -0
  203. package/docs/talks8/conversa2.txt +237 -0
  204. package/docs/talks9/Beyond the Obvious_ 50 Plataformas Auto-Hospedadas de Forja de C/303/263digo para Al/303/251m de Gitea e GitLab.md" +174 -0
  205. package/docs/talks9/De NPM a Multi-Linguagem_ Uma Arquitetura T/303/251cnica para a Execu/303/247/303/243o Integrada de C/303/263digo no Ecossistema Node.js.md" +59 -0
  206. package/docs/talks9/De NPM a VMs Virtuais_ Uma An/303/241lise Arquitet/303/264nica para a Realiza/303/247/303/243o do Ciclo de Vida do Projeto SADDLE.md" +91 -0
  207. package/docs/talks9/Mapeamento da Engrenagem Computacional_ Uma Arquitetura para Execu/303/247/303/243o Isolada e Persist/303/252ncia em Ambientes Distribu/303/255dos.md" +116 -0
  208. package/docs/talks9/O Cen/303/241rio Pr/303/241tico do SADDLE_ Uma An/303/241lise de Viabilidade e Modelo de Ciclo de Vida Integrado.md" +128 -0
  209. package/docs/talks9/README (2).md +489 -0
  210. package/docs/talks9/README.md +198 -0
  211. package/docs/talks9/Viabilidade do Saddle_ Uma An/303/241lise T/303/251cnica da Transforma/303/247/303/243o de Armazenamento Remoto em Mem/303/263ria Computacional.md" +80 -0
  212. package/docs/talks9/conversa.txt +544 -0
  213. package/docs/talks9/other (2).md +39 -0
  214. package/docs/talks9/other.md +57 -0
  215. package/docs/talks9/outro.txt +24 -0
  216. package/extension/README.md +2 -2
  217. package/extension/build.js +1 -1
  218. package/extension/content.js +46 -4
  219. package/extension/manifest.json +1 -1
  220. package/extension/pagebridge.js +41 -0
  221. package/extension/protocol.js +21 -2
  222. package/extension/serviceworker.js +12 -5
  223. package/package.json +165 -10
  224. package/release/assets.js +70 -0
  225. package/scrape/agent.ts +122 -0
  226. package/scrape/batch.ts +79 -0
  227. package/scrape/biome.json +76 -0
  228. package/scrape/browser.ts +222 -0
  229. package/scrape/cache.ts +84 -0
  230. package/scrape/chunking.ts +193 -0
  231. package/scrape/cli.ts +105 -0
  232. package/scrape/crawler.ts +115 -0
  233. package/scrape/dev-server.ts +94 -0
  234. package/scrape/errors.ts +132 -0
  235. package/scrape/events.ts +26 -0
  236. package/scrape/extract.ts +165 -0
  237. package/scrape/fetch.ts +105 -0
  238. package/scrape/formats.ts +85 -0
  239. package/scrape/headers.ts +71 -0
  240. package/scrape/index.ts +92 -0
  241. package/scrape/jsdom.d.ts +6 -0
  242. package/scrape/llms-txt.ts +84 -0
  243. package/scrape/middleware.ts +90 -0
  244. package/scrape/package-lock.json +9397 -0
  245. package/scrape/package.json +1420 -0
  246. package/scrape/pool.ts +95 -0
  247. package/scrape/port.ts +18 -0
  248. package/scrape/proxy.ts +103 -0
  249. package/scrape/rate-limiter.ts +95 -0
  250. package/scrape/renderer.ts +194 -0
  251. package/scrape/retry.ts +64 -0
  252. package/scrape/robots.ts +137 -0
  253. package/scrape/scrape.ts +123 -0
  254. package/scrape/serialize.ts +310 -0
  255. package/scrape/server.ts +137 -0
  256. package/scrape/session.ts +109 -0
  257. package/scrape/sitemap.ts +131 -0
  258. package/scrape/tokens.ts +45 -0
  259. package/scrape/tsconfig.json +28 -0
  260. package/scrape/types.ts +214 -0
  261. package/scrape/utils.ts +77 -0
  262. package/scrape/vite.config.ts +55 -0
  263. package/scrape/vitest.config.ts +17 -0
@@ -0,0 +1,193 @@
1
+ import { estimateTokens } from './tokens.js';
2
+
3
+ export interface ChunkOptions {
4
+ maxTokens?: number;
5
+ overlapTokens?: number;
6
+ preserveCodeBlocks?: boolean;
7
+ includeHeadingPath?: boolean;
8
+ }
9
+
10
+ export interface Chunk {
11
+ content: string;
12
+ headingPath: string[];
13
+ tokenCount: number;
14
+ index: number;
15
+ }
16
+
17
+ const DEFAULT_CHUNK_OPTIONS: Required<ChunkOptions> = {
18
+ maxTokens: 512,
19
+ overlapTokens: 50,
20
+ preserveCodeBlocks: true,
21
+ includeHeadingPath: true,
22
+ };
23
+
24
+ function splitByHeaders(markdown: string): { heading: string; level: number; content: string }[] {
25
+ const sections: { heading: string; level: number; content: string }[] = [];
26
+ const lines = markdown.split('\n');
27
+ let currentHeading = '';
28
+ let currentLevel = 0;
29
+ let currentContent: string[] = [];
30
+
31
+ for (const line of lines) {
32
+ const headerMatch = line.match(/^(#{1,6})\s+(.+)/);
33
+ if (headerMatch) {
34
+ if (currentContent.length > 0 || currentHeading) {
35
+ sections.push({
36
+ heading: currentHeading,
37
+ level: currentLevel,
38
+ content: currentContent.join('\n').trim(),
39
+ });
40
+ }
41
+ currentLevel = headerMatch[1].length;
42
+ currentHeading = headerMatch[2];
43
+ currentContent = [];
44
+ } else {
45
+ currentContent.push(line);
46
+ }
47
+ }
48
+
49
+ if (currentContent.length > 0 || currentHeading) {
50
+ sections.push({
51
+ heading: currentHeading,
52
+ level: currentLevel,
53
+ content: currentContent.join('\n').trim(),
54
+ });
55
+ }
56
+
57
+ return sections;
58
+ }
59
+
60
+ function splitByParagraphs(text: string): string[] {
61
+ return text.split(/\n\n+/).filter(p => p.trim().length > 0);
62
+ }
63
+
64
+ function splitBySentences(text: string): string[] {
65
+ return text.match(/[^.!?]+[.!?]+/g) || [text];
66
+ }
67
+
68
+ export function chunkMarkdown(markdown: string, options: ChunkOptions = {}): Chunk[] {
69
+ const opts = { ...DEFAULT_CHUNK_OPTIONS, ...options };
70
+ const chunks: Chunk[] = [];
71
+ const sections = splitByHeaders(markdown);
72
+ const headingPath: string[] = [];
73
+ let headingDepth = 0;
74
+
75
+ for (const section of sections) {
76
+ while (headingDepth > 0 && headingDepth >= section.level) {
77
+ headingDepth--;
78
+ headingPath.pop();
79
+ }
80
+ if (section.heading) {
81
+ headingDepth = section.level;
82
+ headingPath.push(section.heading);
83
+ }
84
+
85
+ const sectionText = section.heading ? `#${'#'.repeat(section.level - 1)} ${section.heading}\n\n${section.content}` : section.content;
86
+
87
+ if (estimateTokens(sectionText) <= opts.maxTokens) {
88
+ chunks.push({
89
+ content: sectionText,
90
+ headingPath: opts.includeHeadingPath ? [...headingPath] : [],
91
+ tokenCount: estimateTokens(sectionText),
92
+ index: chunks.length,
93
+ });
94
+ } else {
95
+ const paragraphs = splitByParagraphs(sectionText);
96
+ let currentChunk = '';
97
+
98
+ for (const paragraph of paragraphs) {
99
+ const testChunk = currentChunk ? `${currentChunk}\n\n${paragraph}` : paragraph;
100
+
101
+ if (estimateTokens(testChunk) <= opts.maxTokens) {
102
+ currentChunk = testChunk;
103
+ } else {
104
+ if (currentChunk) {
105
+ chunks.push({
106
+ content: currentChunk,
107
+ headingPath: opts.includeHeadingPath ? [...headingPath] : [],
108
+ tokenCount: estimateTokens(currentChunk),
109
+ index: chunks.length,
110
+ });
111
+ }
112
+
113
+ if (estimateTokens(paragraph) > opts.maxTokens) {
114
+ const sentences = splitBySentences(paragraph);
115
+ let sentenceChunk = '';
116
+ for (const sentence of sentences) {
117
+ const testSentence = sentenceChunk ? `${sentenceChunk} ${sentence}` : sentence;
118
+ if (estimateTokens(testSentence) <= opts.maxTokens) {
119
+ sentenceChunk = testSentence;
120
+ } else {
121
+ if (sentenceChunk) {
122
+ chunks.push({
123
+ content: sentenceChunk,
124
+ headingPath: opts.includeHeadingPath ? [...headingPath] : [],
125
+ tokenCount: estimateTokens(sentenceChunk),
126
+ index: chunks.length,
127
+ });
128
+ }
129
+ sentenceChunk = sentence;
130
+ }
131
+ }
132
+ currentChunk = sentenceChunk;
133
+ } else {
134
+ currentChunk = paragraph;
135
+ }
136
+ }
137
+ }
138
+
139
+ if (currentChunk) {
140
+ chunks.push({
141
+ content: currentChunk,
142
+ headingPath: opts.includeHeadingPath ? [...headingPath] : [],
143
+ tokenCount: estimateTokens(currentChunk),
144
+ index: chunks.length,
145
+ });
146
+ }
147
+ }
148
+ }
149
+
150
+ return chunks;
151
+ }
152
+
153
+ export function chunkText(text: string, options: ChunkOptions = {}): Chunk[] {
154
+ const opts = { ...DEFAULT_CHUNK_OPTIONS, ...options };
155
+ const chunks: Chunk[] = [];
156
+ const paragraphs = splitByParagraphs(text);
157
+ let currentChunk = '';
158
+
159
+ for (const paragraph of paragraphs) {
160
+ const testChunk = currentChunk ? `${currentChunk}\n\n${paragraph}` : paragraph;
161
+ if (estimateTokens(testChunk) <= opts.maxTokens) {
162
+ currentChunk = testChunk;
163
+ } else {
164
+ if (currentChunk) {
165
+ chunks.push({
166
+ content: currentChunk,
167
+ headingPath: [],
168
+ tokenCount: estimateTokens(currentChunk),
169
+ index: chunks.length,
170
+ });
171
+ }
172
+ currentChunk = paragraph;
173
+ }
174
+ }
175
+
176
+ if (currentChunk) {
177
+ chunks.push({
178
+ content: currentChunk,
179
+ headingPath: [],
180
+ tokenCount: estimateTokens(currentChunk),
181
+ index: chunks.length,
182
+ });
183
+ }
184
+
185
+ return chunks;
186
+ }
187
+
188
+ export function formatChunksForRAG(chunks: Chunk[]): string {
189
+ return chunks.map(chunk => {
190
+ const heading = chunk.headingPath.length > 0 ? `## ${chunk.headingPath.join(' > ')}\n\n` : '';
191
+ return `${heading}${chunk.content}`;
192
+ }).join('\n\n---\n\n');
193
+ }
package/scrape/cli.ts ADDED
@@ -0,0 +1,105 @@
1
+ #!/usr/bin/env node
2
+ import { Command } from 'commander';
3
+ import { readFileSync, existsSync } from 'node:fs';
4
+ import { resolve } from 'node:path';
5
+ import { scrapeUrl, scrapeHtml } from './scrape.js';
6
+ import { serializeResult } from './serialize.js';
7
+ import { resolveFormat, extensionForFormat, buildSerializeOptions } from './formats.js';
8
+ import { formatForAgent } from './agent.js';
9
+ import { writeOutput } from './utils.js';
10
+ import type { ScrapeOptions, ScrapeResult } from './types.js';
11
+
12
+ const program = new Command();
13
+
14
+ program
15
+ .name('webscrape')
16
+ .description('DevThink WebScrape — Universal AI-powered web scraping toolkit')
17
+ .version('2.0.0');
18
+
19
+ program
20
+ .argument('[url]', 'URL to scrape')
21
+ .option('-f, --format <type>', 'Output format: markdown, html, text, json, xml, redis', 'markdown')
22
+ .option('-o, --output <file>', 'Write output to file')
23
+ .option('--file <path>', 'Scrape HTML from local file')
24
+ .option('--mode <mode>', 'Scraping mode: auto, fetch, browser', 'auto')
25
+ .option('--no-headless', 'Show browser window')
26
+ .option('--timeout <ms>', 'Navigation timeout', '30000')
27
+ .option('--scroll', 'Scroll page to load dynamic content')
28
+ .option('--readable', 'Extract readable content only')
29
+ .option('--agent', 'Format output for AI agent consumption')
30
+ .option('--pretty', 'Pretty-print JSON output')
31
+ .option('--proxy <url>', 'Proxy server URL')
32
+ .option('--retries <n>', 'Number of retries', '3')
33
+ .option('--user-agent <ua>', 'Custom User-Agent string')
34
+ .action(async (url, options) => {
35
+ if (!url && !options.file) {
36
+ program.help();
37
+ }
38
+
39
+ const format = resolveFormat(options.format);
40
+ const scrapeOpts: ScrapeOptions = {
41
+ format,
42
+ timeout: parseInt(options.timeout),
43
+ scroll: options.scroll,
44
+ readable: options.readable,
45
+ mode: options.mode,
46
+ proxy: options.proxy,
47
+ retries: parseInt(options.retries),
48
+ userAgent: options.userAgent,
49
+ extractLinks: true,
50
+ extractImages: true,
51
+ extractTables: true,
52
+ };
53
+
54
+ if (options.readable) {
55
+ scrapeOpts.removeSelectors = [
56
+ 'script', 'style', 'nav', 'footer', 'header',
57
+ '.sidebar', '.advertisement', '.ads', '.menu',
58
+ '.comments', '.comment', '#comments',
59
+ ];
60
+ }
61
+
62
+ console.error(`Scraping: ${url || options.file}`);
63
+
64
+ let result: ScrapeResult;
65
+
66
+ if (options.file) {
67
+ const filePath = resolve(options.file);
68
+ if (!existsSync(filePath)) {
69
+ console.error(`File not found: ${filePath}`);
70
+ process.exit(1);
71
+ }
72
+ const html = readFileSync(filePath, 'utf-8');
73
+ result = await scrapeHtml(html, scrapeOpts);
74
+ } else {
75
+ result = await scrapeUrl(url, scrapeOpts);
76
+ }
77
+
78
+ console.error(`Done in ${result.duration}ms (${(result.size / 1024).toFixed(1)} KB)`);
79
+
80
+ if (options.agent) {
81
+ const agentOutput = formatForAgent(result);
82
+ const output = JSON.stringify(agentOutput, null, 2);
83
+
84
+ if (options.output) {
85
+ writeOutput(options.output, output);
86
+ console.error(`Saved to ${options.output}`);
87
+ } else {
88
+ console.log(output);
89
+ }
90
+ } else {
91
+ const serializeOpts = buildSerializeOptions(format, options.pretty);
92
+ const serialized = serializeResult(result, serializeOpts);
93
+
94
+ if (options.output) {
95
+ const ext = extensionForFormat(format);
96
+ const finalPath = options.output.endsWith(ext) ? options.output : options.output + ext;
97
+ writeOutput(finalPath, serialized.content);
98
+ console.error(`Saved to ${finalPath}`);
99
+ } else {
100
+ console.log(serialized.content);
101
+ }
102
+ }
103
+ });
104
+
105
+ program.parse();
@@ -0,0 +1,115 @@
1
+ import type { ScrapeOptions, ScrapeResult, CrawlStats } from './types.js';
2
+ import { scrapeUrl } from './scrape.js';
3
+
4
+ export interface CrawlOptions extends ScrapeOptions {
5
+ maxDepth?: number;
6
+ maxPages?: number;
7
+ maxConcurrent?: number;
8
+ sameDomain?: boolean;
9
+ delayMs?: number;
10
+ onDiscover?: (url: string, depth: number) => void;
11
+ onResult?: (result: ScrapeResult) => void;
12
+ onError?: (url: string, error: Error) => void;
13
+ }
14
+
15
+ interface CrawlEntry {
16
+ url: string;
17
+ depth: number;
18
+ }
19
+
20
+ function isSameDomain(url1: string, url2: string): boolean {
21
+ try {
22
+ return new URL(url1).hostname === new URL(url2).hostname;
23
+ } catch {
24
+ return false;
25
+ }
26
+ }
27
+
28
+ function normalizeForDedup(url: string): string {
29
+ try {
30
+ const u = new URL(url);
31
+ u.hash = '';
32
+ u.search = '';
33
+ return u.href.replace(/\/+$/, '') || u.href;
34
+ } catch {
35
+ return url;
36
+ }
37
+ }
38
+
39
+ export async function crawl(
40
+ startUrl: string,
41
+ options: CrawlOptions = {}
42
+ ): Promise<{ results: ScrapeResult[]; stats: CrawlStats }> {
43
+ const {
44
+ maxDepth = 2,
45
+ maxPages = 50,
46
+ maxConcurrent = 3,
47
+ sameDomain = true,
48
+ delayMs = 1000,
49
+ onDiscover,
50
+ onResult,
51
+ onError,
52
+ ...scrapeOpts
53
+ } = options;
54
+
55
+ const startTime = Date.now();
56
+ const visited = new Set<string>();
57
+ const queue: CrawlEntry[] = [{ url: startUrl, depth: 0 }];
58
+ const results: ScrapeResult[] = [];
59
+ let failed = 0;
60
+
61
+ visited.add(normalizeForDedup(startUrl));
62
+
63
+ while (queue.length > 0 && results.length < maxPages) {
64
+ const batch = queue.splice(0, maxConcurrent);
65
+ const promises = batch.map(async (entry) => {
66
+ if (entry.depth > maxDepth) return;
67
+
68
+ try {
69
+ const result = await scrapeUrl(entry.url, scrapeOpts);
70
+ results.push(result);
71
+ onResult?.(result);
72
+
73
+ // Extract links for further crawling
74
+ if (entry.depth < maxDepth) {
75
+ for (const link of result.links) {
76
+ const normalized = normalizeForDedup(link.href);
77
+ if (visited.has(normalized)) continue;
78
+ if (sameDomain && !isSameDomain(startUrl, link.href)) continue;
79
+ try {
80
+ new URL(link.href);
81
+ } catch {
82
+ continue;
83
+ }
84
+ visited.add(normalized);
85
+ queue.push({ url: link.href, depth: entry.depth + 1 });
86
+ onDiscover?.(link.href, entry.depth + 1);
87
+ }
88
+ }
89
+ } catch (error) {
90
+ failed++;
91
+ onError?.(entry.url, error as Error);
92
+ }
93
+ });
94
+
95
+ await Promise.all(promises);
96
+
97
+ if (delayMs > 0 && queue.length > 0) {
98
+ await new Promise(r => setTimeout(r, delayMs));
99
+ }
100
+ }
101
+
102
+ const duration = Date.now() - startTime;
103
+
104
+ return {
105
+ results,
106
+ stats: {
107
+ totalUrls: visited.size,
108
+ successful: results.length,
109
+ failed,
110
+ skipped: visited.size - results.length - failed,
111
+ duration,
112
+ avgResponseTime: results.length > 0 ? results.reduce((sum, r) => sum + r.duration, 0) / results.length : 0,
113
+ },
114
+ };
115
+ }
@@ -0,0 +1,94 @@
1
+ #!/usr/bin/env node
2
+ import { createServer } from 'node:http';
3
+
4
+ import { scrapeUrl } from './scrape.js';
5
+ import { serializeResult } from './serialize.js';
6
+ import { resolveFormat, buildSerializeOptions } from './formats.js';
7
+ import { formatForAgent } from './agent.js';
8
+ import { randomPort } from './utils/port.js';
9
+ import type { ScrapeOptions } from './types.js';
10
+
11
+ const PORT = randomPort();
12
+
13
+ const server = createServer(async (req, res) => {
14
+ res.setHeader('Access-Control-Allow-Origin', '*');
15
+ res.setHeader('Access-Control-Allow-Methods', 'POST, OPTIONS');
16
+ res.setHeader('Access-Control-Allow-Headers', 'Content-Type');
17
+
18
+ if (req.method === 'OPTIONS') {
19
+ res.writeHead(200);
20
+ res.end();
21
+ return;
22
+ }
23
+
24
+ if (req.method !== 'POST' || req.url !== '/api/scrape') {
25
+ res.writeHead(404);
26
+ res.end(JSON.stringify({ error: 'Not found' }));
27
+ return;
28
+ }
29
+
30
+ let body = '';
31
+ req.on('data', (chunk: Buffer) => { body += chunk.toString(); });
32
+ req.on('end', async () => {
33
+ try {
34
+ const { url, format: fmt, options = {} }: {
35
+ url: string;
36
+ format?: string;
37
+ options?: Record<string, unknown>;
38
+ } = JSON.parse(body);
39
+ if (!url) throw new Error('URL is required');
40
+
41
+ const format = resolveFormat(fmt || 'markdown');
42
+ const result = await scrapeUrl(url, {
43
+ timeout: 30000,
44
+ scroll: options.scroll as boolean,
45
+ extractLinks: (options.extractLinks as boolean) ?? true,
46
+ extractImages: (options.extractImages as boolean) ?? true,
47
+ extractTables: (options.extractTables as boolean) ?? true,
48
+ } as ScrapeOptions);
49
+
50
+ let response: Record<string, unknown>;
51
+
52
+ if (options.agent) {
53
+ const agentOutput = formatForAgent(result);
54
+ response = {
55
+ success: true,
56
+ title: result.title,
57
+ url: result.url,
58
+ duration: result.duration,
59
+ size: result.size,
60
+ agentOutput,
61
+ };
62
+ } else {
63
+ const serializeOpts = buildSerializeOptions(format);
64
+ const serialized = serializeResult(result, serializeOpts);
65
+ response = {
66
+ success: true,
67
+ title: result.title,
68
+ url: result.url,
69
+ format,
70
+ content: serialized.content,
71
+ duration: result.duration,
72
+ size: serialized.size,
73
+ metadata: result.metadata,
74
+ links: result.links,
75
+ images: result.images,
76
+ tables: result.tables,
77
+ result,
78
+ };
79
+ }
80
+
81
+ res.writeHead(200, { 'Content-Type': 'application/json' });
82
+ res.end(JSON.stringify(response));
83
+ } catch (err) {
84
+ const error = err as Error;
85
+ res.writeHead(500, { 'Content-Type': 'application/json' });
86
+ res.end(JSON.stringify({ error: error.message }));
87
+ }
88
+ });
89
+ });
90
+
91
+ server.listen(PORT, () => {
92
+ console.log(`API server running at http://localhost:${PORT}`);
93
+ console.log(`POST /api/scrape with { url, format?, options? }`);
94
+ });
@@ -0,0 +1,132 @@
1
+ export const ErrorCode = {
2
+ VALIDATION_FAILED: 'VALIDATION_FAILED',
3
+ INVALID_URL: 'INVALID_URL',
4
+ TIMEOUT: 'TIMEOUT',
5
+ BLOCKED: 'BLOCKED',
6
+ RATE_LIMITED: 'RATE_LIMITED',
7
+ PROXY_ERROR: 'PROXY_ERROR',
8
+ PARSE_ERROR: 'PARSE_ERROR',
9
+ AUTH_REQUIRED: 'AUTH_REQUIRED',
10
+ NETWORK_ERROR: 'NETWORK_ERROR',
11
+ BROWSER_NOT_AVAILABLE: 'BROWSER_NOT_AVAILABLE',
12
+ CRAWL_DEPTH_EXCEEDED: 'CRAWL_DEPTH_EXCEEDED',
13
+ MAX_RETRIES_EXCEEDED: 'MAX_RETRIES_EXCEEDED',
14
+ } as const;
15
+
16
+ export type ErrorCode = typeof ErrorCode[keyof typeof ErrorCode];
17
+
18
+ export class WebScrapeError extends Error {
19
+ public readonly code: ErrorCode;
20
+ public readonly statusCode: number;
21
+ public readonly isRetryable: boolean;
22
+ public readonly timestamp: string;
23
+ public readonly details?: Record<string, unknown>;
24
+ public readonly cause?: Error;
25
+
26
+ constructor(
27
+ message: string,
28
+ code: ErrorCode,
29
+ statusCode: number,
30
+ isRetryable: boolean,
31
+ details?: Record<string, unknown>,
32
+ cause?: Error
33
+ ) {
34
+ super(message);
35
+ this.name = 'WebScrapeError';
36
+ this.code = code;
37
+ this.statusCode = statusCode;
38
+ this.isRetryable = isRetryable;
39
+ this.timestamp = new Date().toISOString();
40
+ this.details = details;
41
+ this.cause = cause;
42
+ Object.setPrototypeOf(this, new.target.prototype);
43
+ Error.captureStackTrace?.(this, this.constructor);
44
+ }
45
+
46
+ toJSON() {
47
+ return {
48
+ error: {
49
+ name: this.name,
50
+ code: this.code,
51
+ message: this.message,
52
+ statusCode: this.statusCode,
53
+ isRetryable: this.isRetryable,
54
+ timestamp: this.timestamp,
55
+ details: this.details,
56
+ },
57
+ };
58
+ }
59
+ }
60
+
61
+ export class ValidationError extends WebScrapeError {
62
+ constructor(message: string, details?: Record<string, unknown>, cause?: Error) {
63
+ super(message, ErrorCode.VALIDATION_FAILED, 400, false, details, cause);
64
+ this.name = 'ValidationError';
65
+ }
66
+ }
67
+
68
+ export class TimeoutError extends WebScrapeError {
69
+ constructor(message: string, details?: Record<string, unknown>, cause?: Error) {
70
+ super(message, ErrorCode.TIMEOUT, 504, true, details, cause);
71
+ this.name = 'TimeoutError';
72
+ }
73
+ }
74
+
75
+ export class BlockedError extends WebScrapeError {
76
+ constructor(message: string, details?: Record<string, unknown>, cause?: Error) {
77
+ super(message, ErrorCode.BLOCKED, 403, false, details, cause);
78
+ this.name = 'BlockedError';
79
+ }
80
+ }
81
+
82
+ export class RateLimitError extends WebScrapeError {
83
+ public readonly retryAfterMs?: number;
84
+ constructor(
85
+ message: string,
86
+ retryAfterMs?: number,
87
+ details?: Record<string, unknown>,
88
+ cause?: Error
89
+ ) {
90
+ super(message, ErrorCode.RATE_LIMITED, 429, true, { ...details, retryAfterMs }, cause);
91
+ this.name = 'RateLimitError';
92
+ this.retryAfterMs = retryAfterMs;
93
+ }
94
+ }
95
+
96
+ export class ProxyError extends WebScrapeError {
97
+ constructor(message: string, details?: Record<string, unknown>, cause?: Error) {
98
+ super(message, ErrorCode.PROXY_ERROR, 502, true, details, cause);
99
+ this.name = 'ProxyError';
100
+ }
101
+ }
102
+
103
+ export class ParseError extends WebScrapeError {
104
+ constructor(message: string, details?: Record<string, unknown>, cause?: Error) {
105
+ super(message, ErrorCode.PARSE_ERROR, 422, false, details, cause);
106
+ this.name = 'ParseError';
107
+ }
108
+ }
109
+
110
+ export class AuthError extends WebScrapeError {
111
+ constructor(message: string, details?: Record<string, unknown>, cause?: Error) {
112
+ super(message, ErrorCode.AUTH_REQUIRED, 401, false, details, cause);
113
+ this.name = 'AuthError';
114
+ }
115
+ }
116
+
117
+ export class NetworkError extends WebScrapeError {
118
+ constructor(message: string, details?: Record<string, unknown>, cause?: Error) {
119
+ super(message, ErrorCode.NETWORK_ERROR, 503, true, details, cause);
120
+ this.name = 'NetworkError';
121
+ }
122
+ }
123
+
124
+ export class BrowserNotAvailableError extends WebScrapeError {
125
+ constructor(
126
+ message: string = 'Playwright is not installed. Install it with: npm install playwright',
127
+ details?: Record<string, unknown>
128
+ ) {
129
+ super(message, ErrorCode.BROWSER_NOT_AVAILABLE, 500, false, details);
130
+ this.name = 'BrowserNotAvailableError';
131
+ }
132
+ }
@@ -0,0 +1,26 @@
1
+ import Emittery from 'emittery';
2
+ import type { ScrapeOptions } from './types.js';
3
+
4
+ export interface WebScrapeEvents {
5
+ 'request:start': { url: string; options: ScrapeOptions; timestamp: number };
6
+ 'request:response': { url: string; status: number; attempt: number; duration: number };
7
+ 'request:error': { url: string; error: Error; attempt: number };
8
+ 'request:retry': { url: string; attempt: number; maxRetries: number; delayMs: number };
9
+ 'request:complete': { url: string; duration: number; success: boolean };
10
+ 'cache:hit': { key: string };
11
+ 'cache:miss': { key: string };
12
+ 'cache:set': { key: string; ttlMs?: number };
13
+ 'proxy:rotate': { proxy: string; reason: string };
14
+ 'proxy:error': { proxy: string; error: Error };
15
+ 'proxy:disabled': { proxy: string; reason: string };
16
+ 'crawl:discover': { url: string; depth: number; parentUrl?: string };
17
+ 'crawl:complete': { totalUrls: number; successful: number; failed: number; duration: number };
18
+ }
19
+
20
+ export type EventEmitter = Emittery<WebScrapeEvents>;
21
+
22
+ export function createEventEmitter(): EventEmitter {
23
+ return new Emittery<WebScrapeEvents>();
24
+ }
25
+
26
+ export const globalEmitter = createEventEmitter();