crawlberg 1.1.4 → 1.3.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/sig/types.rbs CHANGED
@@ -1,530 +1,554 @@
1
1
  # This file is auto-generated by alef — DO NOT EDIT.
2
- # alef:hash:899a9b0b3964b8cc10628ac686b74140e934aa910533d6b9333c2acf97f2773b
2
+ # alef:hash:78599b35c3e08365abebb762ea75ffe2c15e049fd992db89a8a26acf137f8c4f
3
3
  # To regenerate: alef generate
4
- # To verify freshness: alef verify --exit-code
4
+ # To verify freshness: alef verify
5
5
 
6
6
  module Crawlberg
7
7
 
8
- VERSION: String
8
+ VERSION: String
9
9
 
10
- type json_value = Hash[String, untyped] | Array[untyped] | String | Integer | Float | bool | nil
10
+ type json_value = Hash[String, untyped] | Array[untyped] | String | Integer | Float | bool | nil
11
11
 
12
- class ExtractionMeta
13
- attr_accessor cost: Float?
14
- attr_accessor prompt_tokens: Integer?
15
- attr_accessor completion_tokens: Integer?
16
- attr_accessor model: String?
17
- attr_accessor chunks_processed: Integer?
12
+ class ExtractionMeta
13
+ attr_reader cost: Float?
14
+ attr_reader prompt_tokens: Integer?
15
+ attr_reader completion_tokens: Integer?
16
+ attr_reader model: String?
17
+ attr_reader chunks_processed: Integer
18
18
 
19
19
  def initialize: (?cost: Float, ?prompt_tokens: Integer, ?completion_tokens: Integer, ?model: String, ?chunks_processed: Integer) -> void
20
- end
20
+ end
21
21
 
22
- class ProxyConfig
23
- attr_accessor url: String?
24
- attr_accessor username: String?
25
- attr_accessor password: String?
22
+ class ProxyConfig
23
+ attr_reader url: String
24
+ attr_reader username: String?
25
+ attr_reader password: String?
26
26
 
27
27
  def initialize: (?url: String, ?username: String, ?password: String) -> void
28
- end
29
-
30
- class ContentConfig
31
- attr_accessor output_format: String?
32
- attr_accessor preprocessing_preset: String?
33
- attr_accessor remove_navigation: bool?
34
- attr_accessor remove_forms: bool?
35
- attr_accessor strip_tags: Array[String]?
36
- attr_accessor preserve_tags: Array[String]?
37
- attr_accessor exclude_selectors: Array[String]?
38
- attr_accessor skip_images: bool?
39
- attr_accessor max_depth: Integer?
40
- attr_accessor wrap: bool?
41
- attr_accessor wrap_width: Integer?
42
- attr_accessor include_document_structure: bool?
43
-
44
- def initialize: (?output_format: String, ?preprocessing_preset: String, ?remove_navigation: bool, ?remove_forms: bool, ?strip_tags: Array[String], ?preserve_tags: Array[String], ?exclude_selectors: Array[String], ?skip_images: bool, ?max_depth: Integer, ?wrap: bool, ?wrap_width: Integer, ?include_document_structure: bool) -> void
45
- def self.default: () -> ContentConfig
46
- end
47
-
48
- class BrowserConfig
49
- attr_accessor mode: BrowserMode?
50
- attr_accessor backend: BrowserBackend?
51
- attr_accessor endpoint: String?
52
- attr_accessor timeout: Integer?
53
- attr_accessor wait: BrowserWait?
54
- attr_accessor wait_selector: String?
55
- attr_accessor extra_wait: Integer?
56
- attr_accessor proxy: ProxyConfig?
57
- attr_accessor block_url_patterns: Array[String]?
58
- attr_accessor eval_script: String?
59
- attr_accessor robots_user_agent: String?
60
- attr_accessor capture_network_events: bool?
61
- attr_accessor session_affinity: bool?
62
-
63
- def initialize: (?mode: BrowserMode, ?backend: BrowserBackend, ?endpoint: String, ?timeout: Integer, ?wait: BrowserWait, ?wait_selector: String, ?extra_wait: Integer, ?proxy: ProxyConfig, ?block_url_patterns: Array[String], ?eval_script: String, ?robots_user_agent: String, ?capture_network_events: bool, ?session_affinity: bool) -> void
64
- def self.default: () -> BrowserConfig
65
- end
66
-
67
- class CrawlConfig
68
- attr_accessor max_depth: Integer?
69
- attr_accessor max_pages: Integer?
70
- attr_accessor max_concurrent: Integer?
71
- attr_accessor respect_robots_txt: bool?
72
- attr_accessor soft_http_errors: bool?
73
- attr_accessor user_agent: String?
74
- attr_accessor stay_on_domain: bool?
75
- attr_accessor allow_subdomains: bool?
76
- attr_accessor include_paths: Array[String]?
77
- attr_accessor exclude_paths: Array[String]?
78
- attr_accessor custom_headers: Hash[String, String]?
79
- attr_accessor request_timeout: Integer?
80
- attr_accessor rate_limit_ms: Integer?
81
- attr_accessor max_redirects: Integer?
82
- attr_accessor retry_count: Integer?
83
- attr_accessor retry_codes: Array[Integer]?
84
- attr_accessor cookies_enabled: bool?
85
- attr_accessor auth: AuthConfig?
86
- attr_accessor max_body_size: Integer?
87
- attr_accessor remove_tags: Array[String]?
88
- attr_accessor content: ContentConfig?
89
- attr_accessor map_limit: Integer?
90
- attr_accessor map_search: String?
91
- attr_accessor download_assets: bool?
92
- attr_accessor asset_types: Array[AssetCategory]?
93
- attr_accessor max_asset_size: Integer?
94
- attr_accessor browser: BrowserConfig?
95
- attr_accessor proxy: ProxyConfig?
96
- attr_accessor user_agents: Array[String]?
97
- attr_accessor capture_screenshot: bool?
98
- attr_accessor follow_document_urls: bool?
99
- attr_accessor document_url_depth: Integer?
100
- attr_accessor download_documents: bool?
101
- attr_accessor document_max_size: Integer?
102
- attr_accessor document_mime_types: Array[String]?
103
- attr_accessor warc_output: String?
104
- attr_accessor browser_profile: String?
105
- attr_accessor save_browser_profile: bool?
106
- attr_accessor ssrf: SsrfPolicy?
107
-
108
- def initialize: (?max_depth: Integer, ?max_pages: Integer, ?max_concurrent: Integer, ?respect_robots_txt: bool, ?soft_http_errors: bool, ?user_agent: String, ?stay_on_domain: bool, ?allow_subdomains: bool, ?include_paths: Array[String], ?exclude_paths: Array[String], ?custom_headers: Hash[String, String], ?request_timeout: Integer, ?rate_limit_ms: Integer, ?max_redirects: Integer, ?retry_count: Integer, ?retry_codes: Array[Integer], ?cookies_enabled: bool, ?auth: AuthConfig, ?max_body_size: Integer, ?remove_tags: Array[String], ?content: ContentConfig, ?map_limit: Integer, ?map_search: String, ?download_assets: bool, ?asset_types: Array[AssetCategory], ?max_asset_size: Integer, ?browser: BrowserConfig, ?proxy: ProxyConfig, ?user_agents: Array[String], ?capture_screenshot: bool, ?follow_document_urls: bool, ?document_url_depth: Integer, ?download_documents: bool, ?document_max_size: Integer, ?document_mime_types: Array[String], ?warc_output: String, ?browser_profile: String, ?save_browser_profile: bool, ?ssrf: SsrfPolicy) -> void
28
+ end
29
+
30
+ class ContentConfig
31
+ attr_reader output_format: String
32
+ attr_reader preprocessing_preset: String
33
+ attr_reader remove_navigation: bool
34
+ attr_reader remove_forms: bool
35
+ attr_reader strip_tags: Array[String]
36
+ attr_reader preserve_tags: Array[String]
37
+ attr_reader exclude_selectors: Array[String]
38
+ attr_reader skip_images: bool
39
+ attr_reader max_depth: Integer?
40
+ attr_reader wrap: bool
41
+ attr_reader wrap_width: Integer
42
+ attr_reader include_document_structure: bool
43
+
44
+ def initialize: (?output_format: String, ?preprocessing_preset: String, ?remove_navigation: bool, ?remove_forms: bool, ?strip_tags: Array[String], ?preserve_tags: Array[String], ?exclude_selectors: Array[String], ?skip_images: bool, ?max_depth: Integer, ?wrap: bool, ?wrap_width: Integer, ?include_document_structure: bool) -> void
45
+ end
46
+
47
+ class BrowserConfig
48
+ attr_reader mode: BrowserMode
49
+ attr_reader backend: BrowserBackend
50
+ attr_reader endpoint: String?
51
+ attr_reader timeout: Integer
52
+ attr_reader wait: BrowserWait
53
+ attr_reader wait_selector: String?
54
+ attr_reader extra_wait: Integer?
55
+ attr_reader proxy: ProxyConfig?
56
+ attr_reader block_url_patterns: Array[String]
57
+ attr_reader eval_script: String?
58
+ attr_reader robots_user_agent: String?
59
+ attr_reader capture_network_events: bool
60
+ attr_reader session_affinity: bool
61
+
62
+ def initialize: (?mode: BrowserMode, ?backend: BrowserBackend, ?endpoint: String, ?timeout: Integer, ?wait: BrowserWait, ?wait_selector: String, ?extra_wait: Integer, ?proxy: ProxyConfig, ?block_url_patterns: Array[String], ?eval_script: String, ?robots_user_agent: String, ?capture_network_events: bool, ?session_affinity: bool) -> void
63
+ end
64
+
65
+ class CrawlConfig
66
+ attr_reader max_depth: Integer?
67
+ attr_reader max_pages: Integer?
68
+ attr_reader max_links_per_page: Integer?
69
+ attr_reader max_concurrent: Integer?
70
+ attr_reader crawl_strategy: CrawlStrategyKind
71
+ attr_reader content_filter: ContentFilterKind?
72
+ attr_reader bm25_query: String?
73
+ attr_reader bm25_threshold: Float?
74
+ attr_reader respect_robots_txt: bool
75
+ attr_reader soft_http_errors: bool
76
+ attr_reader user_agent: String?
77
+ attr_reader stay_on_domain: bool
78
+ attr_reader allow_subdomains: bool
79
+ attr_reader include_paths: Array[String]
80
+ attr_reader exclude_paths: Array[String]
81
+ attr_reader custom_headers: Hash[String, String]
82
+ attr_reader request_timeout: Integer
83
+ attr_reader rate_limit_ms: Integer?
84
+ attr_reader max_redirects: Integer
85
+ attr_reader retry_count: Integer
86
+ attr_reader retry_codes: Array[Integer]
87
+ attr_reader cookies_enabled: bool
88
+ attr_reader auth: AuthConfig?
89
+ attr_reader max_body_size: Integer?
90
+ attr_reader remove_tags: Array[String]
91
+ attr_reader content: ContentConfig
92
+ attr_reader map_limit: Integer?
93
+ attr_reader map_search: String?
94
+ attr_reader download_assets: bool
95
+ attr_reader asset_types: Array[AssetCategory]
96
+ attr_reader max_asset_size: Integer?
97
+ attr_reader browser: BrowserConfig
98
+ attr_reader proxy: ProxyConfig?
99
+ attr_reader user_agents: Array[String]
100
+ attr_reader capture_screenshot: bool
101
+ attr_reader follow_document_urls: bool
102
+ attr_reader document_url_depth: Integer?
103
+ attr_reader download_documents: bool
104
+ attr_reader document_max_size: Integer?
105
+ attr_reader document_mime_types: Array[String]
106
+ attr_reader document_output_dir: String?
107
+ attr_reader document_content_encoding: DocumentContentEncoding?
108
+ attr_reader warc_output: String?
109
+ attr_reader browser_profile: String?
110
+ attr_reader save_browser_profile: bool
111
+ attr_reader ssrf: SsrfPolicy
112
+ attr_reader ssrf_deny_private_explicit: bool?
113
+
114
+ def initialize: (?max_depth: Integer, ?max_pages: Integer, ?max_links_per_page: Integer, ?max_concurrent: Integer, ?crawl_strategy: CrawlStrategyKind, ?content_filter: ContentFilterKind, ?bm25_query: String, ?bm25_threshold: Float, ?respect_robots_txt: bool, ?soft_http_errors: bool, ?user_agent: String, ?stay_on_domain: bool, ?allow_subdomains: bool, ?include_paths: Array[String], ?exclude_paths: Array[String], ?custom_headers: Hash[String, String], ?request_timeout: Integer, ?rate_limit_ms: Integer, ?max_redirects: Integer, ?retry_count: Integer, ?retry_codes: Array[Integer], ?cookies_enabled: bool, ?auth: AuthConfig, ?max_body_size: Integer, ?remove_tags: Array[String], ?content: ContentConfig, ?map_limit: Integer, ?map_search: String, ?download_assets: bool, ?asset_types: Array[AssetCategory], ?max_asset_size: Integer, ?browser: BrowserConfig, ?proxy: ProxyConfig, ?user_agents: Array[String], ?capture_screenshot: bool, ?follow_document_urls: bool, ?document_url_depth: Integer, ?download_documents: bool, ?document_max_size: Integer, ?document_mime_types: Array[String], ?document_output_dir: String, ?document_content_encoding: DocumentContentEncoding, ?warc_output: String, ?browser_profile: String, ?save_browser_profile: bool, ?ssrf: SsrfPolicy, ?ssrf_deny_private_explicit: bool) -> void
109
115
  def validate: () -> void
110
- def self.default: () -> CrawlConfig
111
- end
112
-
113
- class BrowserExtras
114
- attr_accessor eval_result: json_value?
115
- attr_accessor network_events: Array[ResponseMeta]?
116
- attr_accessor cookies: Array[CookieInfo]?
117
-
118
- def initialize: (?eval_result: json_value, ?network_events: Array[ResponseMeta], ?cookies: Array[CookieInfo]) -> void
119
- end
120
-
121
- class DownloadedDocument
122
- attr_accessor url: String?
123
- attr_accessor mime_type: String?
124
- attr_accessor size: Integer?
125
- attr_accessor filename: String?
126
- attr_accessor content_hash: String?
127
- attr_accessor headers: Hash[String, String]?
128
-
129
- def initialize: (?url: String, ?mime_type: String, ?size: Integer, ?filename: String, ?content_hash: String, ?headers: Hash[String, String]) -> void
130
- end
131
-
132
- class InteractionResult
133
- attr_accessor action_results: Array[ActionResult]?
134
- attr_accessor final_html: String?
135
- attr_accessor final_url: String?
136
-
137
- def initialize: (?action_results: Array[ActionResult], ?final_html: String, ?final_url: String) -> void
138
- end
139
-
140
- class ActionResult
141
- attr_accessor action_index: Integer?
142
- attr_accessor action_type: String?
143
- attr_accessor success: bool?
144
- attr_accessor data: json_value?
145
- attr_accessor error: String?
146
-
147
- def initialize: (?action_index: Integer, ?action_type: String, ?success: bool, ?data: json_value, ?error: String) -> void
148
- end
149
-
150
- class ScrapeResult
151
- attr_accessor status_code: Integer?
152
- attr_accessor final_url: String?
153
- attr_accessor content_type: String?
154
- attr_accessor html: String?
155
- attr_accessor body_size: Integer?
156
- attr_accessor metadata: PageMetadata?
157
- attr_accessor links: Array[LinkInfo]?
158
- attr_accessor images: Array[ImageInfo]?
159
- attr_accessor feeds: Array[FeedInfo]?
160
- attr_accessor json_ld: Array[JsonLdEntry]?
161
- attr_accessor is_allowed: bool?
162
- attr_accessor crawl_delay: Integer?
163
- attr_accessor noindex_detected: bool?
164
- attr_accessor nofollow_detected: bool?
165
- attr_accessor x_robots_tag: String?
166
- attr_accessor is_pdf: bool?
167
- attr_accessor was_skipped: bool?
168
- attr_accessor detected_charset: String?
169
- attr_accessor auth_header_sent: bool?
170
- attr_accessor response_meta: ResponseMeta?
171
- attr_accessor assets: Array[DownloadedAsset]?
172
- attr_accessor js_render_hint: bool?
173
- attr_accessor browser_used: bool?
174
- attr_accessor markdown: MarkdownResult?
175
- attr_accessor extracted_data: json_value?
176
- attr_accessor extraction_meta: ExtractionMeta?
177
- attr_accessor downloaded_document: DownloadedDocument?
178
- attr_accessor browser: BrowserExtras?
179
-
180
- def initialize: (?status_code: Integer, ?final_url: String, ?content_type: String, ?html: String, ?body_size: Integer, ?metadata: PageMetadata, ?links: Array[LinkInfo], ?images: Array[ImageInfo], ?feeds: Array[FeedInfo], ?json_ld: Array[JsonLdEntry], ?is_allowed: bool, ?crawl_delay: Integer, ?noindex_detected: bool, ?nofollow_detected: bool, ?x_robots_tag: String, ?is_pdf: bool, ?was_skipped: bool, ?detected_charset: String, ?auth_header_sent: bool, ?response_meta: ResponseMeta, ?assets: Array[DownloadedAsset], ?js_render_hint: bool, ?browser_used: bool, ?markdown: MarkdownResult, ?extracted_data: json_value, ?extraction_meta: ExtractionMeta, ?downloaded_document: DownloadedDocument, ?browser: BrowserExtras) -> void
181
- end
182
-
183
- class CrawlPageResult
184
- attr_accessor url: String?
185
- attr_accessor normalized_url: String?
186
- attr_accessor status_code: Integer?
187
- attr_accessor content_type: String?
188
- attr_accessor html: String?
189
- attr_accessor body_size: Integer?
190
- attr_accessor metadata: PageMetadata?
191
- attr_accessor links: Array[LinkInfo]?
192
- attr_accessor images: Array[ImageInfo]?
193
- attr_accessor feeds: Array[FeedInfo]?
194
- attr_accessor json_ld: Array[JsonLdEntry]?
195
- attr_accessor depth: Integer?
196
- attr_accessor stayed_on_domain: bool?
197
- attr_accessor was_skipped: bool?
198
- attr_accessor is_pdf: bool?
199
- attr_accessor detected_charset: String?
200
- attr_accessor markdown: MarkdownResult?
201
- attr_accessor extracted_data: json_value?
202
- attr_accessor extraction_meta: ExtractionMeta?
203
- attr_accessor downloaded_document: DownloadedDocument?
204
- attr_accessor browser_used: bool?
205
-
206
- def initialize: (?url: String, ?normalized_url: String, ?status_code: Integer, ?content_type: String, ?html: String, ?body_size: Integer, ?metadata: PageMetadata, ?links: Array[LinkInfo], ?images: Array[ImageInfo], ?feeds: Array[FeedInfo], ?json_ld: Array[JsonLdEntry], ?depth: Integer, ?stayed_on_domain: bool, ?was_skipped: bool, ?is_pdf: bool, ?detected_charset: String, ?markdown: MarkdownResult, ?extracted_data: json_value, ?extraction_meta: ExtractionMeta, ?downloaded_document: DownloadedDocument, ?browser_used: bool) -> void
207
- end
208
-
209
- class CrawlResult
210
- attr_accessor pages: Array[CrawlPageResult]?
211
- attr_accessor final_url: String?
212
- attr_accessor redirect_count: Integer?
213
- attr_accessor was_skipped: bool?
214
- attr_accessor error: String?
215
- attr_accessor cookies: Array[CookieInfo]?
216
- attr_accessor stayed_on_domain: bool?
217
- attr_accessor browser_used: bool?
218
-
219
- def initialize: (?pages: Array[CrawlPageResult], ?final_url: String, ?redirect_count: Integer, ?was_skipped: bool, ?error: String, ?cookies: Array[CookieInfo], ?stayed_on_domain: bool, ?browser_used: bool) -> void
116
+ end
117
+
118
+ class BrowserExtras
119
+ attr_reader eval_result: String?
120
+ attr_reader network_events: Array[ResponseMeta]
121
+ attr_reader cookies: Array[CookieInfo]
122
+
123
+ def initialize: (?eval_result: String, ?network_events: Array[ResponseMeta], ?cookies: Array[CookieInfo]) -> void
124
+ end
125
+
126
+ class DownloadedDocument
127
+ attr_reader url: String
128
+ attr_reader mime_type: String
129
+ attr_reader size: Integer
130
+ attr_reader filename: String?
131
+ attr_reader content_hash: String
132
+ attr_reader headers: Hash[String, String]
133
+ attr_reader truncated: bool
134
+ attr_reader content_path: String?
135
+ attr_reader content_base64: String?
136
+
137
+ def initialize: (?url: String, ?mime_type: String, ?size: Integer, ?filename: String, ?content_hash: String, ?headers: Hash[String, String], ?truncated: bool, ?content_path: String, ?content_base64: String) -> void
138
+ end
139
+
140
+ class InteractionResult
141
+ attr_reader action_results: Array[ActionResult]
142
+ attr_reader final_html: String
143
+ attr_reader final_url: String
144
+ attr_reader screenshot_base64: String?
145
+
146
+ def initialize: (?action_results: Array[ActionResult], ?final_html: String, ?final_url: String, ?screenshot_base64: String) -> void
147
+ end
148
+
149
+ class ActionResult
150
+ attr_reader action_index: Integer
151
+ attr_reader action_type: String
152
+ attr_reader success: bool
153
+ attr_reader data: String?
154
+ attr_reader error: String?
155
+
156
+ def initialize: (?action_index: Integer, ?action_type: String, ?success: bool, ?data: String, ?error: String) -> void
157
+ end
158
+
159
+ class ScrapeResult
160
+ attr_reader status_code: Integer
161
+ attr_reader final_url: String
162
+ attr_reader content_type: String
163
+ attr_reader html: String
164
+ attr_reader body_size: Integer
165
+ attr_reader metadata: PageMetadata
166
+ attr_reader links: Array[LinkInfo]
167
+ attr_reader images: Array[ImageInfo]
168
+ attr_reader feeds: Array[FeedInfo]
169
+ attr_reader json_ld: Array[JsonLdEntry]
170
+ attr_reader is_allowed: bool
171
+ attr_reader crawl_delay: Integer?
172
+ attr_reader noindex_detected: bool
173
+ attr_reader nofollow_detected: bool
174
+ attr_reader x_robots_tag: String?
175
+ attr_reader is_pdf: bool
176
+ attr_reader was_skipped: bool
177
+ attr_reader detected_charset: String?
178
+ attr_reader auth_header_sent: bool
179
+ attr_reader response_meta: ResponseMeta?
180
+ attr_reader assets: Array[DownloadedAsset]
181
+ attr_reader js_render_hint: bool
182
+ attr_reader browser_used: bool
183
+ attr_reader markdown: MarkdownResult?
184
+ attr_reader extracted_data: String?
185
+ attr_reader extraction_meta: ExtractionMeta?
186
+ attr_reader screenshot_base64: String?
187
+ attr_reader downloaded_document: DownloadedDocument?
188
+ attr_reader browser: BrowserExtras?
189
+
190
+ def initialize: (?status_code: Integer, ?final_url: String, ?content_type: String, ?html: String, ?body_size: Integer, ?metadata: PageMetadata, ?links: Array[LinkInfo], ?images: Array[ImageInfo], ?feeds: Array[FeedInfo], ?json_ld: Array[JsonLdEntry], ?is_allowed: bool, ?crawl_delay: Integer, ?noindex_detected: bool, ?nofollow_detected: bool, ?x_robots_tag: String, ?is_pdf: bool, ?was_skipped: bool, ?detected_charset: String, ?auth_header_sent: bool, ?response_meta: ResponseMeta, ?assets: Array[DownloadedAsset], ?js_render_hint: bool, ?browser_used: bool, ?markdown: MarkdownResult, ?extracted_data: String, ?extraction_meta: ExtractionMeta, ?screenshot_base64: String, ?downloaded_document: DownloadedDocument, ?browser: BrowserExtras) -> void
191
+ end
192
+
193
+ class CrawlPageResult
194
+ attr_reader url: String
195
+ attr_reader normalized_url: String
196
+ attr_reader status_code: Integer
197
+ attr_reader content_type: String
198
+ attr_reader html: String
199
+ attr_reader body_size: Integer
200
+ attr_reader metadata: PageMetadata
201
+ attr_reader links: Array[LinkInfo]
202
+ attr_reader images: Array[ImageInfo]
203
+ attr_reader feeds: Array[FeedInfo]
204
+ attr_reader json_ld: Array[JsonLdEntry]
205
+ attr_reader depth: Integer
206
+ attr_reader stayed_on_domain: bool
207
+ attr_reader was_skipped: bool
208
+ attr_reader is_pdf: bool
209
+ attr_reader detected_charset: String?
210
+ attr_reader markdown: MarkdownResult?
211
+ attr_reader extracted_data: String?
212
+ attr_reader extraction_meta: ExtractionMeta?
213
+ attr_reader downloaded_document: DownloadedDocument?
214
+ attr_reader browser_used: bool
215
+
216
+ def initialize: (?url: String, ?normalized_url: String, ?status_code: Integer, ?content_type: String, ?html: String, ?body_size: Integer, ?metadata: PageMetadata, ?links: Array[LinkInfo], ?images: Array[ImageInfo], ?feeds: Array[FeedInfo], ?json_ld: Array[JsonLdEntry], ?depth: Integer, ?stayed_on_domain: bool, ?was_skipped: bool, ?is_pdf: bool, ?detected_charset: String, ?markdown: MarkdownResult, ?extracted_data: String, ?extraction_meta: ExtractionMeta, ?downloaded_document: DownloadedDocument, ?browser_used: bool) -> void
217
+ end
218
+
219
+ class CrawlResult
220
+ attr_reader pages: Array[CrawlPageResult]
221
+ attr_reader final_url: String
222
+ attr_reader redirect_count: Integer
223
+ attr_reader was_skipped: bool
224
+ attr_reader error: String?
225
+ attr_reader cookies: Array[CookieInfo]
226
+ attr_reader stayed_on_domain: bool
227
+ attr_reader browser_used: bool
228
+
229
+ def initialize: (?pages: Array[CrawlPageResult], ?final_url: String, ?redirect_count: Integer, ?was_skipped: bool, ?error: String, ?cookies: Array[CookieInfo], ?stayed_on_domain: bool, ?browser_used: bool) -> void
220
230
  def unique_normalized_urls: () -> Integer
221
- end
231
+ end
222
232
 
223
- class SitemapUrl
224
- attr_accessor url: String?
225
- attr_accessor lastmod: String?
226
- attr_accessor changefreq: String?
227
- attr_accessor priority: String?
233
+ class SitemapUrl
234
+ attr_reader url: String
235
+ attr_reader lastmod: String?
236
+ attr_reader changefreq: String?
237
+ attr_reader priority: String?
228
238
 
229
239
  def initialize: (?url: String, ?lastmod: String, ?changefreq: String, ?priority: String) -> void
230
- end
240
+ end
231
241
 
232
- class MapResult
233
- attr_accessor urls: Array[SitemapUrl]?
242
+ class MapResult
243
+ attr_reader urls: Array[SitemapUrl]
234
244
 
235
- def initialize: (?urls: Array[SitemapUrl]) -> void
236
- end
245
+ def initialize: (?urls: Array[SitemapUrl]) -> void
246
+ end
237
247
 
238
- class MarkdownResult
239
- attr_accessor content: String?
240
- attr_accessor document_structure: json_value?
241
- attr_accessor tables: Array[json_value]?
242
- attr_accessor warnings: Array[String]?
243
- attr_accessor citations: bool?
244
- attr_accessor fit_content: String?
248
+ class MarkdownResult
249
+ attr_reader content: String
250
+ attr_reader document_structure: String?
251
+ attr_reader tables: Array[String]
252
+ attr_reader warnings: Array[String]
253
+ attr_reader citations: bool
254
+ attr_reader fit_content: String?
245
255
 
246
- def initialize: (?content: String, ?document_structure: json_value, ?tables: Array[json_value], ?warnings: Array[String], ?citations: bool, ?fit_content: String) -> void
247
- end
256
+ def initialize: (?content: String, ?document_structure: String, ?tables: Array[String], ?warnings: Array[String], ?citations: bool, ?fit_content: String) -> void
257
+ end
248
258
 
249
- class LinkInfo
250
- attr_accessor url: String?
251
- attr_accessor text: String?
252
- attr_accessor link_type: LinkType?
253
- attr_accessor rel: String?
254
- attr_accessor nofollow: bool?
259
+ class LinkInfo
260
+ attr_reader url: String
261
+ attr_reader text: String
262
+ attr_reader link_type: LinkType
263
+ attr_reader rel: String?
264
+ attr_reader nofollow: bool
255
265
 
256
266
  def initialize: (?url: String, ?text: String, ?link_type: LinkType, ?rel: String, ?nofollow: bool) -> void
257
- end
267
+ end
258
268
 
259
- class ImageInfo
260
- attr_accessor url: String?
261
- attr_accessor alt: String?
262
- attr_accessor width: Integer?
263
- attr_accessor height: Integer?
264
- attr_accessor source: ImageSource?
269
+ class ImageInfo
270
+ attr_reader url: String
271
+ attr_reader alt: String?
272
+ attr_reader width: Integer?
273
+ attr_reader height: Integer?
274
+ attr_reader source: ImageSource
265
275
 
266
276
  def initialize: (?url: String, ?alt: String, ?width: Integer, ?height: Integer, ?source: ImageSource) -> void
267
- end
277
+ end
268
278
 
269
- class FeedInfo
270
- attr_accessor url: String?
271
- attr_accessor title: String?
272
- attr_accessor feed_type: FeedType?
279
+ class FeedInfo
280
+ attr_reader url: String
281
+ attr_reader title: String?
282
+ attr_reader feed_type: FeedType
273
283
 
274
284
  def initialize: (?url: String, ?title: String, ?feed_type: FeedType) -> void
275
- end
285
+ end
276
286
 
277
- class JsonLdEntry
278
- attr_accessor schema_type: String?
279
- attr_accessor name: String?
280
- attr_accessor raw: String?
287
+ class JsonLdEntry
288
+ attr_reader schema_type: String
289
+ attr_reader name: String?
290
+ attr_reader raw: String
281
291
 
282
292
  def initialize: (?schema_type: String, ?name: String, ?raw: String) -> void
283
- end
293
+ end
284
294
 
285
- class CookieInfo
286
- attr_accessor name: String?
287
- attr_accessor value: String?
288
- attr_accessor domain: String?
289
- attr_accessor path: String?
295
+ class CookieInfo
296
+ attr_reader name: String
297
+ attr_reader value: String
298
+ attr_reader domain: String?
299
+ attr_reader path: String?
290
300
 
291
301
  def initialize: (?name: String, ?value: String, ?domain: String, ?path: String) -> void
292
- end
302
+ end
293
303
 
294
- class DownloadedAsset
295
- attr_accessor url: String?
296
- attr_accessor content_hash: String?
297
- attr_accessor mime_type: String?
298
- attr_accessor size: Integer?
299
- attr_accessor asset_category: AssetCategory?
300
- attr_accessor html_tag: String?
304
+ class DownloadedAsset
305
+ attr_reader url: String
306
+ attr_reader content_hash: String
307
+ attr_reader mime_type: String?
308
+ attr_reader size: Integer
309
+ attr_reader asset_category: AssetCategory
310
+ attr_reader html_tag: String?
301
311
 
302
312
  def initialize: (?url: String, ?content_hash: String, ?mime_type: String, ?size: Integer, ?asset_category: AssetCategory, ?html_tag: String) -> void
303
- end
313
+ end
304
314
 
305
- class ArticleMetadata
306
- attr_accessor published_time: String?
307
- attr_accessor modified_time: String?
308
- attr_accessor author: String?
309
- attr_accessor section: String?
310
- attr_accessor tags: Array[String]?
315
+ class ArticleMetadata
316
+ attr_reader published_time: String?
317
+ attr_reader modified_time: String?
318
+ attr_reader author: String?
319
+ attr_reader section: String?
320
+ attr_reader tags: Array[String]
311
321
 
312
- def initialize: (?published_time: String, ?modified_time: String, ?author: String, ?section: String, ?tags: Array[String]) -> void
313
- end
322
+ def initialize: (?published_time: String, ?modified_time: String, ?author: String, ?section: String, ?tags: Array[String]) -> void
323
+ end
314
324
 
315
- class HreflangEntry
316
- attr_accessor lang: String?
317
- attr_accessor url: String?
325
+ class HreflangEntry
326
+ attr_reader lang: String
327
+ attr_reader url: String
318
328
 
319
329
  def initialize: (?lang: String, ?url: String) -> void
320
- end
330
+ end
321
331
 
322
- class FaviconInfo
323
- attr_accessor url: String?
324
- attr_accessor rel: String?
325
- attr_accessor sizes: String?
326
- attr_accessor mime_type: String?
332
+ class FaviconInfo
333
+ attr_reader url: String
334
+ attr_reader rel: String
335
+ attr_reader sizes: String?
336
+ attr_reader mime_type: String?
327
337
 
328
338
  def initialize: (?url: String, ?rel: String, ?sizes: String, ?mime_type: String) -> void
329
- end
339
+ end
330
340
 
331
- class HeadingInfo
332
- attr_accessor level: Integer?
333
- attr_accessor text: String?
341
+ class HeadingInfo
342
+ attr_reader level: Integer
343
+ attr_reader text: String
334
344
 
335
345
  def initialize: (?level: Integer, ?text: String) -> void
336
- end
346
+ end
337
347
 
338
- class ResponseMeta
339
- attr_accessor etag: String?
340
- attr_accessor last_modified: String?
341
- attr_accessor cache_control: String?
342
- attr_accessor server: String?
343
- attr_accessor x_powered_by: String?
344
- attr_accessor content_language: String?
345
- attr_accessor content_encoding: String?
348
+ class ResponseMeta
349
+ attr_reader etag: String?
350
+ attr_reader last_modified: String?
351
+ attr_reader cache_control: String?
352
+ attr_reader server: String?
353
+ attr_reader x_powered_by: String?
354
+ attr_reader content_language: String?
355
+ attr_reader content_encoding: String?
346
356
 
347
357
  def initialize: (?etag: String, ?last_modified: String, ?cache_control: String, ?server: String, ?x_powered_by: String, ?content_language: String, ?content_encoding: String) -> void
348
- end
349
-
350
- class PageMetadata
351
- attr_accessor title: String?
352
- attr_accessor description: String?
353
- attr_accessor canonical_url: String?
354
- attr_accessor keywords: String?
355
- attr_accessor author: String?
356
- attr_accessor viewport: String?
357
- attr_accessor theme_color: String?
358
- attr_accessor generator: String?
359
- attr_accessor robots: String?
360
- attr_accessor html_lang: String?
361
- attr_accessor html_dir: String?
362
- attr_accessor og_title: String?
363
- attr_accessor og_type: String?
364
- attr_accessor og_image: String?
365
- attr_accessor og_description: String?
366
- attr_accessor og_url: String?
367
- attr_accessor og_site_name: String?
368
- attr_accessor og_locale: String?
369
- attr_accessor og_video: String?
370
- attr_accessor og_audio: String?
371
- attr_accessor og_locale_alternates: Array[String]?
372
- attr_accessor twitter_card: String?
373
- attr_accessor twitter_title: String?
374
- attr_accessor twitter_description: String?
375
- attr_accessor twitter_image: String?
376
- attr_accessor twitter_site: String?
377
- attr_accessor twitter_creator: String?
378
- attr_accessor dc_title: String?
379
- attr_accessor dc_creator: String?
380
- attr_accessor dc_subject: String?
381
- attr_accessor dc_description: String?
382
- attr_accessor dc_publisher: String?
383
- attr_accessor dc_date: String?
384
- attr_accessor dc_type: String?
385
- attr_accessor dc_format: String?
386
- attr_accessor dc_identifier: String?
387
- attr_accessor dc_language: String?
388
- attr_accessor dc_rights: String?
389
- attr_accessor article: ArticleMetadata?
390
- attr_accessor hreflangs: Array[HreflangEntry]?
391
- attr_accessor favicons: Array[FaviconInfo]?
392
- attr_accessor headings: Array[HeadingInfo]?
393
- attr_accessor word_count: Integer?
394
-
395
- def initialize: (?title: String, ?description: String, ?canonical_url: String, ?keywords: String, ?author: String, ?viewport: String, ?theme_color: String, ?generator: String, ?robots: String, ?html_lang: String, ?html_dir: String, ?og_title: String, ?og_type: String, ?og_image: String, ?og_description: String, ?og_url: String, ?og_site_name: String, ?og_locale: String, ?og_video: String, ?og_audio: String, ?og_locale_alternates: Array[String], ?twitter_card: String, ?twitter_title: String, ?twitter_description: String, ?twitter_image: String, ?twitter_site: String, ?twitter_creator: String, ?dc_title: String, ?dc_creator: String, ?dc_subject: String, ?dc_description: String, ?dc_publisher: String, ?dc_date: String, ?dc_type: String, ?dc_format: String, ?dc_identifier: String, ?dc_language: String, ?dc_rights: String, ?article: ArticleMetadata, ?hreflangs: Array[HreflangEntry], ?favicons: Array[FaviconInfo], ?headings: Array[HeadingInfo], ?word_count: Integer) -> void
396
- end
397
-
398
- class CrawlStreamRequest
399
- attr_accessor url: String?
358
+ end
359
+
360
+ class PageMetadata
361
+ attr_reader title: String?
362
+ attr_reader description: String?
363
+ attr_reader canonical_url: String?
364
+ attr_reader keywords: String?
365
+ attr_reader author: String?
366
+ attr_reader viewport: String?
367
+ attr_reader theme_color: String?
368
+ attr_reader generator: String?
369
+ attr_reader robots: String?
370
+ attr_reader html_lang: String?
371
+ attr_reader html_dir: String?
372
+ attr_reader og_title: String?
373
+ attr_reader og_type: String?
374
+ attr_reader og_image: String?
375
+ attr_reader og_description: String?
376
+ attr_reader og_url: String?
377
+ attr_reader og_site_name: String?
378
+ attr_reader og_locale: String?
379
+ attr_reader og_video: String?
380
+ attr_reader og_audio: String?
381
+ attr_reader og_locale_alternates: Array[String]?
382
+ attr_reader twitter_card: String?
383
+ attr_reader twitter_title: String?
384
+ attr_reader twitter_description: String?
385
+ attr_reader twitter_image: String?
386
+ attr_reader twitter_site: String?
387
+ attr_reader twitter_creator: String?
388
+ attr_reader dc_title: String?
389
+ attr_reader dc_creator: String?
390
+ attr_reader dc_subject: String?
391
+ attr_reader dc_description: String?
392
+ attr_reader dc_publisher: String?
393
+ attr_reader dc_date: String?
394
+ attr_reader dc_type: String?
395
+ attr_reader dc_format: String?
396
+ attr_reader dc_identifier: String?
397
+ attr_reader dc_language: String?
398
+ attr_reader dc_rights: String?
399
+ attr_reader article: ArticleMetadata?
400
+ attr_reader hreflangs: Array[HreflangEntry]?
401
+ attr_reader favicons: Array[FaviconInfo]?
402
+ attr_reader headings: Array[HeadingInfo]?
403
+ attr_reader word_count: Integer?
404
+
405
+ def initialize: (?title: String, ?description: String, ?canonical_url: String, ?keywords: String, ?author: String, ?viewport: String, ?theme_color: String, ?generator: String, ?robots: String, ?html_lang: String, ?html_dir: String, ?og_title: String, ?og_type: String, ?og_image: String, ?og_description: String, ?og_url: String, ?og_site_name: String, ?og_locale: String, ?og_video: String, ?og_audio: String, ?og_locale_alternates: Array[String], ?twitter_card: String, ?twitter_title: String, ?twitter_description: String, ?twitter_image: String, ?twitter_site: String, ?twitter_creator: String, ?dc_title: String, ?dc_creator: String, ?dc_subject: String, ?dc_description: String, ?dc_publisher: String, ?dc_date: String, ?dc_type: String, ?dc_format: String, ?dc_identifier: String, ?dc_language: String, ?dc_rights: String, ?article: ArticleMetadata, ?hreflangs: Array[HreflangEntry], ?favicons: Array[FaviconInfo], ?headings: Array[HeadingInfo], ?word_count: Integer) -> void
406
+ end
407
+
408
+ class CrawlStreamRequest
409
+ attr_reader url: String
400
410
 
401
411
  def initialize: (?url: String) -> void
402
- end
412
+ end
403
413
 
404
- class BatchCrawlStreamRequest
405
- attr_accessor urls: Array[String]?
414
+ class BatchCrawlStreamRequest
415
+ attr_reader urls: Array[String]
406
416
 
407
- def initialize: (?urls: Array[String]) -> void
408
- end
417
+ def initialize: (?urls: Array[String]) -> void
418
+ end
409
419
 
410
- class CitationResult
411
- attr_accessor content: String?
412
- attr_accessor references: Array[CitationReference]?
420
+ class CitationResult
421
+ attr_reader content: String
422
+ attr_reader references: Array[CitationReference]
413
423
 
414
- def initialize: (?content: String, ?references: Array[CitationReference]) -> void
415
- end
424
+ def initialize: (?content: String, ?references: Array[CitationReference]) -> void
425
+ end
416
426
 
417
- class CitationReference
418
- attr_accessor index: Integer?
419
- attr_accessor url: String?
420
- attr_accessor text: String?
427
+ class CitationReference
428
+ attr_reader index: Integer
429
+ attr_reader url: String
430
+ attr_reader text: String
421
431
 
422
432
  def initialize: (?index: Integer, ?url: String, ?text: String) -> void
423
- end
433
+ end
424
434
 
425
- class CrawlEngineHandle
426
- def crawl_stream: (CrawlStreamRequest req) -> Enumerator[CrawlEvent]
427
- def batch_crawl_stream: (BatchCrawlStreamRequest req) -> Enumerator[CrawlEvent]
428
- end
435
+ class CrawlEngineHandle
436
+ def crawl_stream: (CrawlStreamRequest req) -> Enumerator[CrawlEvent]
437
+ def batch_crawl_stream: (BatchCrawlStreamRequest req) -> Enumerator[CrawlEvent]
438
+ end
429
439
 
430
- class BatchScrapeResult
431
- attr_accessor url: String?
432
- attr_accessor result: ScrapeResult?
433
- attr_accessor error: String?
440
+ class BatchScrapeResult
441
+ attr_reader url: String
442
+ attr_reader result: ScrapeResult?
443
+ attr_reader error: String?
434
444
 
435
445
  def initialize: (?url: String, ?result: ScrapeResult, ?error: String) -> void
436
- end
446
+ end
437
447
 
438
- class BatchCrawlResult
439
- attr_accessor url: String?
440
- attr_accessor result: CrawlResult?
441
- attr_accessor error: String?
448
+ class BatchCrawlResult
449
+ attr_reader url: String
450
+ attr_reader result: CrawlResult?
451
+ attr_reader error: String?
442
452
 
443
453
  def initialize: (?url: String, ?result: CrawlResult, ?error: String) -> void
444
- end
454
+ end
445
455
 
446
- class BatchScrapeResults
447
- attr_accessor results: Array[BatchScrapeResult]?
448
- attr_accessor total_count: Integer?
449
- attr_accessor completed_count: Integer?
450
- attr_accessor failed_count: Integer?
456
+ class BatchScrapeResults
457
+ attr_reader results: Array[BatchScrapeResult]
458
+ attr_reader total_count: Integer
459
+ attr_reader completed_count: Integer
460
+ attr_reader failed_count: Integer
451
461
 
452
- def initialize: (?results: Array[BatchScrapeResult], ?total_count: Integer, ?completed_count: Integer, ?failed_count: Integer) -> void
453
- end
462
+ def initialize: (?results: Array[BatchScrapeResult], ?total_count: Integer, ?completed_count: Integer, ?failed_count: Integer) -> void
463
+ end
454
464
 
455
- class BatchCrawlResults
456
- attr_accessor results: Array[BatchCrawlResult]?
457
- attr_accessor total_count: Integer?
458
- attr_accessor completed_count: Integer?
459
- attr_accessor failed_count: Integer?
465
+ class BatchCrawlResults
466
+ attr_reader results: Array[BatchCrawlResult]
467
+ attr_reader total_count: Integer
468
+ attr_reader completed_count: Integer
469
+ attr_reader failed_count: Integer
460
470
 
461
- def initialize: (?results: Array[BatchCrawlResult], ?total_count: Integer, ?completed_count: Integer, ?failed_count: Integer) -> void
462
- end
471
+ def initialize: (?results: Array[BatchCrawlResult], ?total_count: Integer, ?completed_count: Integer, ?failed_count: Integer) -> void
472
+ end
463
473
 
464
- class SsrfPolicy
465
- attr_accessor deny_private: bool?
466
- attr_accessor max_redirects: Integer?
474
+ class SsrfPolicy
475
+ attr_reader deny_private: bool
476
+ attr_reader allowlist: Array[HostMatcher]
477
+ attr_reader max_redirects: Integer
467
478
 
468
- def initialize: (?deny_private: bool, ?max_redirects: Integer) -> void
469
- def self.default: () -> SsrfPolicy
470
- def self.from_env: () -> SsrfPolicy
471
- end
479
+ def initialize: (?deny_private: bool, ?allowlist: Array[HostMatcher], ?max_redirects: Integer) -> void
480
+ end
472
481
 
473
- class BrowserMode
474
- type value = :auto | :always | :never | :stealth
475
- end
482
+ class BrowserMode
483
+ type value = :auto | :always | :never | :stealth
484
+ end
476
485
 
477
- class BrowserWait
478
- type value = :network_idle | :selector | :fixed
479
- end
486
+ class BrowserWait
487
+ type value = :network_idle | :selector | :fixed
488
+ end
480
489
 
481
- class BrowserBackend
482
- type value = :chromiumoxide | :native
483
- end
490
+ class BrowserBackend
491
+ type value = :chromiumoxide | :native
492
+ end
484
493
 
485
- class AuthConfig
486
- end
494
+ class DocumentContentEncoding
495
+ type value = :base64
496
+ end
487
497
 
488
- class LinkType
489
- type value = :internal | :external | :anchor | :document
490
- end
498
+ class CrawlStrategyKind
499
+ type value = :bfs | :dfs | :best_first | :adaptive
500
+ end
491
501
 
492
- class ImageSource
493
- type value = :img | :picture_source | :og_image | :twitter_image
494
- end
502
+ class ContentFilterKind
503
+ type value = :bm25
504
+ end
495
505
 
496
- class FeedType
497
- type value = :rss | :atom | :json_feed
498
- end
506
+ class AuthConfig
507
+ end
499
508
 
500
- class AssetCategory
501
- type value = :document | :image | :audio | :video | :font | :stylesheet | :script | :archive | :data | :other
502
- end
509
+ class LinkType
510
+ type value = :internal | :external | :anchor | :document
511
+ end
503
512
 
504
- class CrawlEvent
505
- end
513
+ class ImageSource
514
+ type value = :img | :picture_source | :og:image | :twitter:image
515
+ end
506
516
 
507
- class PageAction
508
- end
517
+ class FeedType
518
+ type value = :rss | :atom | :json_feed
519
+ end
509
520
 
510
- class ScrollDirection
511
- type value = :up | :down
512
- end
521
+ class AssetCategory
522
+ type value = :document | :image | :audio | :video | :font | :stylesheet | :script | :archive | :data | :other
523
+ end
513
524
 
514
- def self.generate_citations: (String markdown) -> CitationResult
525
+ class CrawlEvent
526
+ end
515
527
 
516
- def self.create_engine: (?CrawlConfig config) -> CrawlEngineHandle
528
+ class PageAction
529
+ end
517
530
 
518
- def self.scrape: (CrawlEngineHandle engine, String url) -> ScrapeResult
531
+ class ScrollDirection
532
+ type value = :up | :down
533
+ end
519
534
 
520
- def self.crawl: (CrawlEngineHandle engine, String url) -> CrawlResult
535
+ class HostMatcher
536
+ end
521
537
 
522
- def self.map_urls: (CrawlEngineHandle engine, String url) -> MapResult
538
+ def self.generate_citations: (String markdown) -> CitationResult
523
539
 
524
- def self.interact: (CrawlEngineHandle engine, String url, Array[PageAction] actions) -> InteractionResult
540
+ def self.create_engine: (?CrawlConfig config) -> CrawlEngineHandle
525
541
 
526
- def self.batch_scrape: (CrawlEngineHandle engine, Array[String] urls) -> BatchScrapeResults
542
+ def self.scrape: (CrawlEngineHandle engine, String url) -> ScrapeResult
527
543
 
528
- def self.batch_crawl: (CrawlEngineHandle engine, Array[String] urls) -> BatchCrawlResults
544
+ def self.crawl: (CrawlEngineHandle engine, String url) -> CrawlResult
545
+
546
+ def self.map_urls: (CrawlEngineHandle engine, String url) -> MapResult
547
+
548
+ def self.interact: (CrawlEngineHandle engine, String url, Array[PageAction] actions) -> InteractionResult
549
+
550
+ def self.batch_scrape: (CrawlEngineHandle engine, Array[String] urls) -> BatchScrapeResults
551
+
552
+ def self.batch_crawl: (CrawlEngineHandle engine, Array[String] urls) -> BatchCrawlResults
529
553
 
530
554
  end