crawlberg 1.3.3 → 1.4.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +1 -1
- data/ext/crawlberg_rb/native/Cargo.lock +20 -20
- data/ext/crawlberg_rb/native/Cargo.toml +3 -3
- data/ext/crawlberg_rb/native/extconf.rb +1 -1
- data/ext/crawlberg_rb/src/lib.rs +3630 -3627
- data/lib/crawlberg/native.rb +191 -142
- data/lib/crawlberg/version.rb +2 -2
- data/lib/crawlberg.rb +1 -1
- data/sig/types.rbs +458 -457
- metadata +2 -2
data/sig/types.rbs
CHANGED
|
@@ -1,554 +1,555 @@
|
|
|
1
1
|
# This file is auto-generated by alef — DO NOT EDIT.
|
|
2
|
-
# alef:hash:
|
|
2
|
+
# alef:hash:1a2c1bf0fb6b53091ffe21fcecbee4a7e5f8214d544addbfe5c39f34e5aa7590
|
|
3
3
|
# To regenerate: alef generate
|
|
4
4
|
# To verify freshness: alef verify
|
|
5
5
|
|
|
6
6
|
module Crawlberg
|
|
7
7
|
|
|
8
|
-
|
|
8
|
+
VERSION: String
|
|
9
9
|
|
|
10
|
-
|
|
10
|
+
type json_value = Hash[String, untyped] | Array[untyped] | String | Integer | Float | bool | nil
|
|
11
11
|
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
12
|
+
class ActionResult
|
|
13
|
+
attr_reader action_index: Integer
|
|
14
|
+
attr_reader action_type: String
|
|
15
|
+
attr_reader success: bool
|
|
16
|
+
attr_reader data: String?
|
|
17
|
+
attr_reader error: String?
|
|
18
18
|
|
|
19
|
-
def initialize: (?
|
|
20
|
-
|
|
19
|
+
def initialize: (?action_index: Integer, ?action_type: String, ?success: bool, ?data: String, ?error: String) -> void
|
|
20
|
+
end
|
|
21
21
|
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
22
|
+
class ArticleMetadata
|
|
23
|
+
attr_reader published_time: String?
|
|
24
|
+
attr_reader modified_time: String?
|
|
25
|
+
attr_reader author: String?
|
|
26
|
+
attr_reader section: String?
|
|
27
|
+
attr_reader tags: Array[String]
|
|
26
28
|
|
|
27
|
-
|
|
28
|
-
|
|
29
|
+
def initialize: (?published_time: String, ?modified_time: String, ?author: String, ?section: String, ?tags: Array[String]) -> void
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
class BatchCrawlResult
|
|
33
|
+
attr_reader url: String
|
|
34
|
+
attr_reader result: CrawlResult?
|
|
35
|
+
attr_reader error: String?
|
|
36
|
+
|
|
37
|
+
def initialize: (?url: String, ?result: CrawlResult, ?error: String) -> void
|
|
38
|
+
end
|
|
39
|
+
|
|
40
|
+
class BatchCrawlResults
|
|
41
|
+
attr_reader results: Array[BatchCrawlResult]
|
|
42
|
+
attr_reader total_count: Integer
|
|
43
|
+
attr_reader completed_count: Integer
|
|
44
|
+
attr_reader failed_count: Integer
|
|
45
|
+
|
|
46
|
+
def initialize: (?results: Array[BatchCrawlResult], ?total_count: Integer, ?completed_count: Integer, ?failed_count: Integer) -> void
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
class BatchCrawlStreamRequest
|
|
50
|
+
attr_reader urls: Array[String]
|
|
51
|
+
|
|
52
|
+
def initialize: (?urls: Array[String]) -> void
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
class BatchScrapeResult
|
|
56
|
+
attr_reader url: String
|
|
57
|
+
attr_reader result: ScrapeResult?
|
|
58
|
+
attr_reader error: String?
|
|
59
|
+
|
|
60
|
+
def initialize: (?url: String, ?result: ScrapeResult, ?error: String) -> void
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
class BatchScrapeResults
|
|
64
|
+
attr_reader results: Array[BatchScrapeResult]
|
|
65
|
+
attr_reader total_count: Integer
|
|
66
|
+
attr_reader completed_count: Integer
|
|
67
|
+
attr_reader failed_count: Integer
|
|
68
|
+
|
|
69
|
+
def initialize: (?results: Array[BatchScrapeResult], ?total_count: Integer, ?completed_count: Integer, ?failed_count: Integer) -> void
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
class BrowserConfig
|
|
73
|
+
attr_reader mode: BrowserMode
|
|
74
|
+
attr_reader backend: BrowserBackend
|
|
75
|
+
attr_reader endpoint: String?
|
|
76
|
+
attr_reader timeout: Integer
|
|
77
|
+
attr_reader wait: BrowserWait
|
|
78
|
+
attr_reader wait_selector: String?
|
|
79
|
+
attr_reader extra_wait: Integer?
|
|
80
|
+
attr_reader proxy: ProxyConfig?
|
|
81
|
+
attr_reader block_url_patterns: Array[String]
|
|
82
|
+
attr_reader eval_script: String?
|
|
83
|
+
attr_reader robots_user_agent: String?
|
|
84
|
+
attr_reader capture_network_events: bool
|
|
85
|
+
attr_reader session_affinity: bool
|
|
86
|
+
|
|
87
|
+
def initialize: (?mode: BrowserMode, ?backend: BrowserBackend, ?endpoint: String, ?timeout: Integer, ?wait: BrowserWait, ?wait_selector: String, ?extra_wait: Integer, ?proxy: ProxyConfig, ?block_url_patterns: Array[String], ?eval_script: String, ?robots_user_agent: String, ?capture_network_events: bool, ?session_affinity: bool) -> void
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
class BrowserExtras
|
|
91
|
+
attr_reader eval_result: String?
|
|
92
|
+
attr_reader network_events: Array[ResponseMeta]
|
|
93
|
+
attr_reader cookies: Array[CookieInfo]
|
|
94
|
+
|
|
95
|
+
def initialize: (?eval_result: String, ?network_events: Array[ResponseMeta], ?cookies: Array[CookieInfo]) -> void
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
class CitationReference
|
|
99
|
+
attr_reader index: Integer
|
|
100
|
+
attr_reader url: String
|
|
101
|
+
attr_reader text: String
|
|
102
|
+
|
|
103
|
+
def initialize: (?index: Integer, ?url: String, ?text: String) -> void
|
|
104
|
+
end
|
|
105
|
+
|
|
106
|
+
class CitationResult
|
|
107
|
+
attr_reader content: String
|
|
108
|
+
attr_reader references: Array[CitationReference]
|
|
109
|
+
|
|
110
|
+
def initialize: (?content: String, ?references: Array[CitationReference]) -> void
|
|
111
|
+
end
|
|
29
112
|
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
113
|
+
class ContentConfig
|
|
114
|
+
attr_reader output_format: String
|
|
115
|
+
attr_reader preprocessing_preset: String
|
|
116
|
+
attr_reader remove_navigation: bool
|
|
117
|
+
attr_reader remove_forms: bool
|
|
35
118
|
attr_reader strip_tags: Array[String]
|
|
36
119
|
attr_reader preserve_tags: Array[String]
|
|
37
120
|
attr_reader exclude_selectors: Array[String]
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
53
|
-
|
|
54
|
-
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
attr_reader crawl_strategy: CrawlStrategyKind
|
|
71
|
-
attr_reader content_filter: ContentFilterKind?
|
|
72
|
-
attr_reader bm25_query: String?
|
|
73
|
-
attr_reader bm25_threshold: Float?
|
|
74
|
-
attr_reader respect_robots_txt: bool
|
|
75
|
-
attr_reader soft_http_errors: bool
|
|
76
|
-
attr_reader user_agent: String?
|
|
77
|
-
attr_reader stay_on_domain: bool
|
|
78
|
-
attr_reader allow_subdomains: bool
|
|
121
|
+
attr_reader skip_images: bool
|
|
122
|
+
attr_reader max_depth: Integer?
|
|
123
|
+
attr_reader wrap: bool
|
|
124
|
+
attr_reader wrap_width: Integer
|
|
125
|
+
attr_reader include_document_structure: bool
|
|
126
|
+
|
|
127
|
+
def initialize: (?output_format: String, ?preprocessing_preset: String, ?remove_navigation: bool, ?remove_forms: bool, ?strip_tags: Array[String], ?preserve_tags: Array[String], ?exclude_selectors: Array[String], ?skip_images: bool, ?max_depth: Integer, ?wrap: bool, ?wrap_width: Integer, ?include_document_structure: bool) -> void
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
class CookieInfo
|
|
131
|
+
attr_reader name: String
|
|
132
|
+
attr_reader value: String
|
|
133
|
+
attr_reader domain: String?
|
|
134
|
+
attr_reader path: String?
|
|
135
|
+
|
|
136
|
+
def initialize: (?name: String, ?value: String, ?domain: String, ?path: String) -> void
|
|
137
|
+
end
|
|
138
|
+
|
|
139
|
+
class CrawlConfig
|
|
140
|
+
attr_reader max_depth: Integer?
|
|
141
|
+
attr_reader max_pages: Integer?
|
|
142
|
+
attr_reader max_links_per_page: Integer?
|
|
143
|
+
attr_reader max_concurrent: Integer?
|
|
144
|
+
attr_reader crawl_strategy: CrawlStrategyKind
|
|
145
|
+
attr_reader content_filter: ContentFilterKind?
|
|
146
|
+
attr_reader bm25_query: String?
|
|
147
|
+
attr_reader bm25_threshold: Float?
|
|
148
|
+
attr_reader respect_robots_txt: bool
|
|
149
|
+
attr_reader soft_http_errors: bool
|
|
150
|
+
attr_reader user_agent: String?
|
|
151
|
+
attr_reader stay_on_domain: bool
|
|
152
|
+
attr_reader allow_subdomains: bool
|
|
79
153
|
attr_reader include_paths: Array[String]
|
|
80
154
|
attr_reader exclude_paths: Array[String]
|
|
81
155
|
attr_reader custom_headers: Hash[String, String]
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
156
|
+
attr_reader request_timeout: Integer
|
|
157
|
+
attr_reader rate_limit_ms: Integer?
|
|
158
|
+
attr_reader max_redirects: Integer
|
|
159
|
+
attr_reader retry_count: Integer
|
|
86
160
|
attr_reader retry_codes: Array[Integer]
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
161
|
+
attr_reader cookies_enabled: bool
|
|
162
|
+
attr_reader auth: AuthConfig?
|
|
163
|
+
attr_reader max_body_size: Integer?
|
|
90
164
|
attr_reader remove_tags: Array[String]
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
165
|
+
attr_reader content: ContentConfig
|
|
166
|
+
attr_reader map_limit: Integer?
|
|
167
|
+
attr_reader map_search: String?
|
|
168
|
+
attr_reader download_assets: bool
|
|
95
169
|
attr_reader asset_types: Array[AssetCategory]
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
170
|
+
attr_reader max_asset_size: Integer?
|
|
171
|
+
attr_reader browser: BrowserConfig
|
|
172
|
+
attr_reader proxy: ProxyConfig?
|
|
99
173
|
attr_reader user_agents: Array[String]
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
174
|
+
attr_reader capture_screenshot: bool
|
|
175
|
+
attr_reader follow_document_urls: bool
|
|
176
|
+
attr_reader document_url_depth: Integer?
|
|
177
|
+
attr_reader download_documents: bool
|
|
178
|
+
attr_reader document_max_size: Integer?
|
|
105
179
|
attr_reader document_mime_types: Array[String]
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
180
|
+
attr_reader document_output_dir: String?
|
|
181
|
+
attr_reader document_content_encoding: DocumentContentEncoding?
|
|
182
|
+
attr_reader warc_output: String?
|
|
183
|
+
attr_reader browser_profile: String?
|
|
184
|
+
attr_reader save_browser_profile: bool
|
|
185
|
+
attr_reader ssrf: SsrfPolicy
|
|
186
|
+
attr_reader ssrf_deny_private_explicit: bool?
|
|
187
|
+
|
|
188
|
+
def initialize: (?max_depth: Integer, ?max_pages: Integer, ?max_links_per_page: Integer, ?max_concurrent: Integer, ?crawl_strategy: CrawlStrategyKind, ?content_filter: ContentFilterKind, ?bm25_query: String, ?bm25_threshold: Float, ?respect_robots_txt: bool, ?soft_http_errors: bool, ?user_agent: String, ?stay_on_domain: bool, ?allow_subdomains: bool, ?include_paths: Array[String], ?exclude_paths: Array[String], ?custom_headers: Hash[String, String], ?request_timeout: Integer, ?rate_limit_ms: Integer, ?max_redirects: Integer, ?retry_count: Integer, ?retry_codes: Array[Integer], ?cookies_enabled: bool, ?auth: AuthConfig, ?max_body_size: Integer, ?remove_tags: Array[String], ?content: ContentConfig, ?map_limit: Integer, ?map_search: String, ?download_assets: bool, ?asset_types: Array[AssetCategory], ?max_asset_size: Integer, ?browser: BrowserConfig, ?proxy: ProxyConfig, ?user_agents: Array[String], ?capture_screenshot: bool, ?follow_document_urls: bool, ?document_url_depth: Integer, ?download_documents: bool, ?document_max_size: Integer, ?document_mime_types: Array[String], ?document_output_dir: String, ?document_content_encoding: DocumentContentEncoding, ?warc_output: String, ?browser_profile: String, ?save_browser_profile: bool, ?ssrf: SsrfPolicy, ?ssrf_deny_private_explicit: bool) -> void
|
|
115
189
|
def validate: () -> void
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
attr_reader content_hash: String
|
|
132
|
-
attr_reader headers: Hash[String, String]
|
|
133
|
-
attr_reader truncated: bool
|
|
134
|
-
attr_reader content_path: String?
|
|
135
|
-
attr_reader content_base64: String?
|
|
136
|
-
|
|
137
|
-
def initialize: (?url: String, ?mime_type: String, ?size: Integer, ?filename: String, ?content_hash: String, ?headers: Hash[String, String], ?truncated: bool, ?content_path: String, ?content_base64: String) -> void
|
|
138
|
-
end
|
|
139
|
-
|
|
140
|
-
class InteractionResult
|
|
141
|
-
attr_reader action_results: Array[ActionResult]
|
|
142
|
-
attr_reader final_html: String
|
|
143
|
-
attr_reader final_url: String
|
|
144
|
-
attr_reader screenshot_base64: String?
|
|
145
|
-
|
|
146
|
-
def initialize: (?action_results: Array[ActionResult], ?final_html: String, ?final_url: String, ?screenshot_base64: String) -> void
|
|
147
|
-
end
|
|
148
|
-
|
|
149
|
-
class ActionResult
|
|
150
|
-
attr_reader action_index: Integer
|
|
151
|
-
attr_reader action_type: String
|
|
152
|
-
attr_reader success: bool
|
|
153
|
-
attr_reader data: String?
|
|
154
|
-
attr_reader error: String?
|
|
155
|
-
|
|
156
|
-
def initialize: (?action_index: Integer, ?action_type: String, ?success: bool, ?data: String, ?error: String) -> void
|
|
157
|
-
end
|
|
158
|
-
|
|
159
|
-
class ScrapeResult
|
|
160
|
-
attr_reader status_code: Integer
|
|
161
|
-
attr_reader final_url: String
|
|
162
|
-
attr_reader content_type: String
|
|
163
|
-
attr_reader html: String
|
|
164
|
-
attr_reader body_size: Integer
|
|
165
|
-
attr_reader metadata: PageMetadata
|
|
166
|
-
attr_reader links: Array[LinkInfo]
|
|
167
|
-
attr_reader images: Array[ImageInfo]
|
|
168
|
-
attr_reader feeds: Array[FeedInfo]
|
|
169
|
-
attr_reader json_ld: Array[JsonLdEntry]
|
|
170
|
-
attr_reader is_allowed: bool
|
|
171
|
-
attr_reader crawl_delay: Integer?
|
|
172
|
-
attr_reader noindex_detected: bool
|
|
173
|
-
attr_reader nofollow_detected: bool
|
|
174
|
-
attr_reader x_robots_tag: String?
|
|
175
|
-
attr_reader is_pdf: bool
|
|
176
|
-
attr_reader was_skipped: bool
|
|
177
|
-
attr_reader detected_charset: String?
|
|
178
|
-
attr_reader auth_header_sent: bool
|
|
179
|
-
attr_reader response_meta: ResponseMeta?
|
|
180
|
-
attr_reader assets: Array[DownloadedAsset]
|
|
181
|
-
attr_reader js_render_hint: bool
|
|
182
|
-
attr_reader browser_used: bool
|
|
183
|
-
attr_reader markdown: MarkdownResult?
|
|
184
|
-
attr_reader extracted_data: String?
|
|
185
|
-
attr_reader extraction_meta: ExtractionMeta?
|
|
186
|
-
attr_reader screenshot_base64: String?
|
|
187
|
-
attr_reader downloaded_document: DownloadedDocument?
|
|
188
|
-
attr_reader browser: BrowserExtras?
|
|
189
|
-
|
|
190
|
-
def initialize: (?status_code: Integer, ?final_url: String, ?content_type: String, ?html: String, ?body_size: Integer, ?metadata: PageMetadata, ?links: Array[LinkInfo], ?images: Array[ImageInfo], ?feeds: Array[FeedInfo], ?json_ld: Array[JsonLdEntry], ?is_allowed: bool, ?crawl_delay: Integer, ?noindex_detected: bool, ?nofollow_detected: bool, ?x_robots_tag: String, ?is_pdf: bool, ?was_skipped: bool, ?detected_charset: String, ?auth_header_sent: bool, ?response_meta: ResponseMeta, ?assets: Array[DownloadedAsset], ?js_render_hint: bool, ?browser_used: bool, ?markdown: MarkdownResult, ?extracted_data: String, ?extraction_meta: ExtractionMeta, ?screenshot_base64: String, ?downloaded_document: DownloadedDocument, ?browser: BrowserExtras) -> void
|
|
191
|
-
end
|
|
192
|
-
|
|
193
|
-
class CrawlPageResult
|
|
194
|
-
attr_reader url: String
|
|
195
|
-
attr_reader normalized_url: String
|
|
196
|
-
attr_reader status_code: Integer
|
|
197
|
-
attr_reader content_type: String
|
|
198
|
-
attr_reader html: String
|
|
199
|
-
attr_reader body_size: Integer
|
|
200
|
-
attr_reader metadata: PageMetadata
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
class CrawlEngineHandle
|
|
193
|
+
def crawl_stream: (CrawlStreamRequest req) -> Enumerator[CrawlEvent]
|
|
194
|
+
def batch_crawl_stream: (BatchCrawlStreamRequest req) -> Enumerator[CrawlEvent]
|
|
195
|
+
end
|
|
196
|
+
|
|
197
|
+
class CrawlPageResult
|
|
198
|
+
attr_reader url: String
|
|
199
|
+
attr_reader normalized_url: String
|
|
200
|
+
attr_reader status_code: Integer
|
|
201
|
+
attr_reader content_type: String
|
|
202
|
+
attr_reader html: String
|
|
203
|
+
attr_reader body_size: Integer
|
|
204
|
+
attr_reader metadata: PageMetadata
|
|
201
205
|
attr_reader links: Array[LinkInfo]
|
|
202
206
|
attr_reader images: Array[ImageInfo]
|
|
203
207
|
attr_reader feeds: Array[FeedInfo]
|
|
204
208
|
attr_reader json_ld: Array[JsonLdEntry]
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
209
|
+
attr_reader depth: Integer
|
|
210
|
+
attr_reader stayed_on_domain: bool
|
|
211
|
+
attr_reader was_skipped: bool
|
|
212
|
+
attr_reader is_pdf: bool
|
|
213
|
+
attr_reader detected_charset: String?
|
|
214
|
+
attr_reader markdown: MarkdownResult?
|
|
215
|
+
attr_reader extracted_data: String?
|
|
216
|
+
attr_reader extraction_meta: ExtractionMeta?
|
|
217
|
+
attr_reader downloaded_document: DownloadedDocument?
|
|
218
|
+
attr_reader browser_used: bool
|
|
219
|
+
|
|
220
|
+
def initialize: (?url: String, ?normalized_url: String, ?status_code: Integer, ?content_type: String, ?html: String, ?body_size: Integer, ?metadata: PageMetadata, ?links: Array[LinkInfo], ?images: Array[ImageInfo], ?feeds: Array[FeedInfo], ?json_ld: Array[JsonLdEntry], ?depth: Integer, ?stayed_on_domain: bool, ?was_skipped: bool, ?is_pdf: bool, ?detected_charset: String, ?markdown: MarkdownResult, ?extracted_data: String, ?extraction_meta: ExtractionMeta, ?downloaded_document: DownloadedDocument, ?browser_used: bool) -> void
|
|
221
|
+
end
|
|
222
|
+
|
|
223
|
+
class CrawlResult
|
|
220
224
|
attr_reader pages: Array[CrawlPageResult]
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
+
attr_reader final_url: String
|
|
226
|
+
attr_reader redirect_count: Integer
|
|
227
|
+
attr_reader was_skipped: bool
|
|
228
|
+
attr_reader error: String?
|
|
225
229
|
attr_reader cookies: Array[CookieInfo]
|
|
226
|
-
|
|
227
|
-
|
|
230
|
+
attr_reader stayed_on_domain: bool
|
|
231
|
+
attr_reader browser_used: bool
|
|
228
232
|
|
|
229
|
-
|
|
233
|
+
def initialize: (?pages: Array[CrawlPageResult], ?final_url: String, ?redirect_count: Integer, ?was_skipped: bool, ?error: String, ?cookies: Array[CookieInfo], ?stayed_on_domain: bool, ?browser_used: bool) -> void
|
|
230
234
|
def unique_normalized_urls: () -> Integer
|
|
231
|
-
|
|
235
|
+
end
|
|
232
236
|
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
attr_reader lastmod: String?
|
|
236
|
-
attr_reader changefreq: String?
|
|
237
|
-
attr_reader priority: String?
|
|
237
|
+
class CrawlStreamRequest
|
|
238
|
+
attr_reader url: String
|
|
238
239
|
|
|
239
|
-
def initialize: (?url: String
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
class MapResult
|
|
243
|
-
attr_reader urls: Array[SitemapUrl]
|
|
240
|
+
def initialize: (?url: String) -> void
|
|
241
|
+
end
|
|
244
242
|
|
|
245
|
-
|
|
246
|
-
|
|
243
|
+
class DownloadedAsset
|
|
244
|
+
attr_reader url: String
|
|
245
|
+
attr_reader content_hash: String
|
|
246
|
+
attr_reader mime_type: String?
|
|
247
|
+
attr_reader size: Integer
|
|
248
|
+
attr_reader asset_category: AssetCategory
|
|
249
|
+
attr_reader html_tag: String?
|
|
247
250
|
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
251
|
+
def initialize: (?url: String, ?content_hash: String, ?mime_type: String, ?size: Integer, ?asset_category: AssetCategory, ?html_tag: String) -> void
|
|
252
|
+
end
|
|
253
|
+
|
|
254
|
+
class DownloadedDocument
|
|
255
|
+
attr_reader url: String
|
|
256
|
+
attr_reader mime_type: String
|
|
257
|
+
attr_reader size: Integer
|
|
258
|
+
attr_reader filename: String?
|
|
259
|
+
attr_reader content_hash: String
|
|
260
|
+
attr_reader headers: Hash[String, String]
|
|
261
|
+
attr_reader truncated: bool
|
|
262
|
+
attr_reader content_path: String?
|
|
263
|
+
attr_reader content_base64: String?
|
|
255
264
|
|
|
256
|
-
|
|
257
|
-
|
|
265
|
+
def initialize: (?url: String, ?mime_type: String, ?size: Integer, ?filename: String, ?content_hash: String, ?headers: Hash[String, String], ?truncated: bool, ?content_path: String, ?content_base64: String) -> void
|
|
266
|
+
end
|
|
258
267
|
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
268
|
+
class ExtractionMeta
|
|
269
|
+
attr_reader cost: Float?
|
|
270
|
+
attr_reader prompt_tokens: Integer?
|
|
271
|
+
attr_reader completion_tokens: Integer?
|
|
272
|
+
attr_reader model: String?
|
|
273
|
+
attr_reader chunks_processed: Integer
|
|
265
274
|
|
|
266
|
-
def initialize: (?
|
|
267
|
-
|
|
275
|
+
def initialize: (?cost: Float, ?prompt_tokens: Integer, ?completion_tokens: Integer, ?model: String, ?chunks_processed: Integer) -> void
|
|
276
|
+
end
|
|
268
277
|
|
|
269
|
-
|
|
270
|
-
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
attr_reader source: ImageSource
|
|
278
|
+
class FaviconInfo
|
|
279
|
+
attr_reader url: String
|
|
280
|
+
attr_reader rel: String
|
|
281
|
+
attr_reader sizes: String?
|
|
282
|
+
attr_reader mime_type: String?
|
|
275
283
|
|
|
276
|
-
def initialize: (?url: String, ?
|
|
277
|
-
|
|
284
|
+
def initialize: (?url: String, ?rel: String, ?sizes: String, ?mime_type: String) -> void
|
|
285
|
+
end
|
|
278
286
|
|
|
279
|
-
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
|
|
287
|
+
class FeedInfo
|
|
288
|
+
attr_reader url: String
|
|
289
|
+
attr_reader title: String?
|
|
290
|
+
attr_reader feed_type: FeedType
|
|
283
291
|
|
|
284
292
|
def initialize: (?url: String, ?title: String, ?feed_type: FeedType) -> void
|
|
285
|
-
|
|
293
|
+
end
|
|
286
294
|
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
attr_reader raw: String
|
|
291
|
-
|
|
292
|
-
def initialize: (?schema_type: String, ?name: String, ?raw: String) -> void
|
|
293
|
-
end
|
|
295
|
+
class HeadingInfo
|
|
296
|
+
attr_reader level: Integer
|
|
297
|
+
attr_reader text: String
|
|
294
298
|
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
attr_reader value: String
|
|
298
|
-
attr_reader domain: String?
|
|
299
|
-
attr_reader path: String?
|
|
299
|
+
def initialize: (?level: Integer, ?text: String) -> void
|
|
300
|
+
end
|
|
300
301
|
|
|
301
|
-
|
|
302
|
-
|
|
302
|
+
class HreflangEntry
|
|
303
|
+
attr_reader lang: String
|
|
304
|
+
attr_reader url: String
|
|
303
305
|
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
attr_reader content_hash: String
|
|
307
|
-
attr_reader mime_type: String?
|
|
308
|
-
attr_reader size: Integer
|
|
309
|
-
attr_reader asset_category: AssetCategory
|
|
310
|
-
attr_reader html_tag: String?
|
|
306
|
+
def initialize: (?lang: String, ?url: String) -> void
|
|
307
|
+
end
|
|
311
308
|
|
|
312
|
-
|
|
313
|
-
|
|
309
|
+
class ImageInfo
|
|
310
|
+
attr_reader url: String
|
|
311
|
+
attr_reader alt: String?
|
|
312
|
+
attr_reader width: Integer?
|
|
313
|
+
attr_reader height: Integer?
|
|
314
|
+
attr_reader source: ImageSource
|
|
314
315
|
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
attr_reader modified_time: String?
|
|
318
|
-
attr_reader author: String?
|
|
319
|
-
attr_reader section: String?
|
|
320
|
-
attr_reader tags: Array[String]
|
|
316
|
+
def initialize: (?url: String, ?alt: String, ?width: Integer, ?height: Integer, ?source: ImageSource) -> void
|
|
317
|
+
end
|
|
321
318
|
|
|
322
|
-
|
|
323
|
-
|
|
319
|
+
class InteractionResult
|
|
320
|
+
attr_reader action_results: Array[ActionResult]
|
|
321
|
+
attr_reader final_html: String
|
|
322
|
+
attr_reader final_url: String
|
|
323
|
+
attr_reader screenshot_base64: String?
|
|
324
324
|
|
|
325
|
-
|
|
326
|
-
|
|
327
|
-
attr_reader url: String
|
|
325
|
+
def initialize: (?action_results: Array[ActionResult], ?final_html: String, ?final_url: String, ?screenshot_base64: String) -> void
|
|
326
|
+
end
|
|
328
327
|
|
|
329
|
-
|
|
330
|
-
|
|
328
|
+
class JsonLdEntry
|
|
329
|
+
attr_reader schema_type: String
|
|
330
|
+
attr_reader name: String?
|
|
331
|
+
attr_reader raw: String
|
|
331
332
|
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
attr_reader rel: String
|
|
335
|
-
attr_reader sizes: String?
|
|
336
|
-
attr_reader mime_type: String?
|
|
333
|
+
def initialize: (?schema_type: String, ?name: String, ?raw: String) -> void
|
|
334
|
+
end
|
|
337
335
|
|
|
338
|
-
|
|
339
|
-
|
|
336
|
+
class LinkInfo
|
|
337
|
+
attr_reader url: String
|
|
338
|
+
attr_reader text: String
|
|
339
|
+
attr_reader link_type: LinkType
|
|
340
|
+
attr_reader rel: String?
|
|
341
|
+
attr_reader nofollow: bool
|
|
340
342
|
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
attr_reader text: String
|
|
343
|
+
def initialize: (?url: String, ?text: String, ?link_type: LinkType, ?rel: String, ?nofollow: bool) -> void
|
|
344
|
+
end
|
|
344
345
|
|
|
345
|
-
|
|
346
|
-
|
|
346
|
+
class MapResult
|
|
347
|
+
attr_reader urls: Array[SitemapUrl]
|
|
347
348
|
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
attr_reader last_modified: String?
|
|
351
|
-
attr_reader cache_control: String?
|
|
352
|
-
attr_reader server: String?
|
|
353
|
-
attr_reader x_powered_by: String?
|
|
354
|
-
attr_reader content_language: String?
|
|
355
|
-
attr_reader content_encoding: String?
|
|
349
|
+
def initialize: (?urls: Array[SitemapUrl]) -> void
|
|
350
|
+
end
|
|
356
351
|
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
attr_reader
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
|
|
379
|
-
|
|
380
|
-
|
|
352
|
+
class MarkdownResult
|
|
353
|
+
attr_reader content: String
|
|
354
|
+
attr_reader document_structure: String?
|
|
355
|
+
attr_reader tables: Array[String]
|
|
356
|
+
attr_reader warnings: Array[String]
|
|
357
|
+
attr_reader citations: bool
|
|
358
|
+
attr_reader fit_content: String?
|
|
359
|
+
|
|
360
|
+
def initialize: (?content: String, ?document_structure: String, ?tables: Array[String], ?warnings: Array[String], ?citations: bool, ?fit_content: String) -> void
|
|
361
|
+
end
|
|
362
|
+
|
|
363
|
+
class PageMetadata
|
|
364
|
+
attr_reader title: String?
|
|
365
|
+
attr_reader description: String?
|
|
366
|
+
attr_reader canonical_url: String?
|
|
367
|
+
attr_reader keywords: String?
|
|
368
|
+
attr_reader author: String?
|
|
369
|
+
attr_reader viewport: String?
|
|
370
|
+
attr_reader theme_color: String?
|
|
371
|
+
attr_reader generator: String?
|
|
372
|
+
attr_reader robots: String?
|
|
373
|
+
attr_reader html_lang: String?
|
|
374
|
+
attr_reader html_dir: String?
|
|
375
|
+
attr_reader og_title: String?
|
|
376
|
+
attr_reader og_type: String?
|
|
377
|
+
attr_reader og_image: String?
|
|
378
|
+
attr_reader og_description: String?
|
|
379
|
+
attr_reader og_url: String?
|
|
380
|
+
attr_reader og_site_name: String?
|
|
381
|
+
attr_reader og_locale: String?
|
|
382
|
+
attr_reader og_video: String?
|
|
383
|
+
attr_reader og_audio: String?
|
|
381
384
|
attr_reader og_locale_alternates: Array[String]?
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
390
|
-
|
|
391
|
-
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
385
|
+
attr_reader twitter_card: String?
|
|
386
|
+
attr_reader twitter_title: String?
|
|
387
|
+
attr_reader twitter_description: String?
|
|
388
|
+
attr_reader twitter_image: String?
|
|
389
|
+
attr_reader twitter_site: String?
|
|
390
|
+
attr_reader twitter_creator: String?
|
|
391
|
+
attr_reader dc_title: String?
|
|
392
|
+
attr_reader dc_creator: String?
|
|
393
|
+
attr_reader dc_subject: String?
|
|
394
|
+
attr_reader dc_description: String?
|
|
395
|
+
attr_reader dc_publisher: String?
|
|
396
|
+
attr_reader dc_date: String?
|
|
397
|
+
attr_reader dc_type: String?
|
|
398
|
+
attr_reader dc_format: String?
|
|
399
|
+
attr_reader dc_identifier: String?
|
|
400
|
+
attr_reader dc_language: String?
|
|
401
|
+
attr_reader dc_rights: String?
|
|
402
|
+
attr_reader article: ArticleMetadata?
|
|
400
403
|
attr_reader hreflangs: Array[HreflangEntry]?
|
|
401
404
|
attr_reader favicons: Array[FaviconInfo]?
|
|
402
405
|
attr_reader headings: Array[HeadingInfo]?
|
|
403
|
-
|
|
406
|
+
attr_reader word_count: Integer?
|
|
404
407
|
|
|
405
|
-
|
|
406
|
-
|
|
408
|
+
def initialize: (?title: String, ?description: String, ?canonical_url: String, ?keywords: String, ?author: String, ?viewport: String, ?theme_color: String, ?generator: String, ?robots: String, ?html_lang: String, ?html_dir: String, ?og_title: String, ?og_type: String, ?og_image: String, ?og_description: String, ?og_url: String, ?og_site_name: String, ?og_locale: String, ?og_video: String, ?og_audio: String, ?og_locale_alternates: Array[String], ?twitter_card: String, ?twitter_title: String, ?twitter_description: String, ?twitter_image: String, ?twitter_site: String, ?twitter_creator: String, ?dc_title: String, ?dc_creator: String, ?dc_subject: String, ?dc_description: String, ?dc_publisher: String, ?dc_date: String, ?dc_type: String, ?dc_format: String, ?dc_identifier: String, ?dc_language: String, ?dc_rights: String, ?article: ArticleMetadata, ?hreflangs: Array[HreflangEntry], ?favicons: Array[FaviconInfo], ?headings: Array[HeadingInfo], ?word_count: Integer) -> void
|
|
409
|
+
end
|
|
407
410
|
|
|
408
|
-
|
|
409
|
-
|
|
411
|
+
class ProxyConfig
|
|
412
|
+
attr_reader url: String
|
|
413
|
+
attr_reader username: String?
|
|
414
|
+
attr_reader password: String?
|
|
410
415
|
|
|
411
|
-
def initialize: (?url: String) -> void
|
|
412
|
-
|
|
413
|
-
|
|
414
|
-
class BatchCrawlStreamRequest
|
|
415
|
-
attr_reader urls: Array[String]
|
|
416
|
-
|
|
417
|
-
def initialize: (?urls: Array[String]) -> void
|
|
418
|
-
end
|
|
419
|
-
|
|
420
|
-
class CitationResult
|
|
421
|
-
attr_reader content: String
|
|
422
|
-
attr_reader references: Array[CitationReference]
|
|
423
|
-
|
|
424
|
-
def initialize: (?content: String, ?references: Array[CitationReference]) -> void
|
|
425
|
-
end
|
|
426
|
-
|
|
427
|
-
class CitationReference
|
|
428
|
-
attr_reader index: Integer
|
|
429
|
-
attr_reader url: String
|
|
430
|
-
attr_reader text: String
|
|
431
|
-
|
|
432
|
-
def initialize: (?index: Integer, ?url: String, ?text: String) -> void
|
|
433
|
-
end
|
|
434
|
-
|
|
435
|
-
class CrawlEngineHandle
|
|
436
|
-
def crawl_stream: (CrawlStreamRequest req) -> Enumerator[CrawlEvent]
|
|
437
|
-
def batch_crawl_stream: (BatchCrawlStreamRequest req) -> Enumerator[CrawlEvent]
|
|
438
|
-
end
|
|
439
|
-
|
|
440
|
-
class BatchScrapeResult
|
|
441
|
-
attr_reader url: String
|
|
442
|
-
attr_reader result: ScrapeResult?
|
|
443
|
-
attr_reader error: String?
|
|
444
|
-
|
|
445
|
-
def initialize: (?url: String, ?result: ScrapeResult, ?error: String) -> void
|
|
446
|
-
end
|
|
447
|
-
|
|
448
|
-
class BatchCrawlResult
|
|
449
|
-
attr_reader url: String
|
|
450
|
-
attr_reader result: CrawlResult?
|
|
451
|
-
attr_reader error: String?
|
|
452
|
-
|
|
453
|
-
def initialize: (?url: String, ?result: CrawlResult, ?error: String) -> void
|
|
454
|
-
end
|
|
455
|
-
|
|
456
|
-
class BatchScrapeResults
|
|
457
|
-
attr_reader results: Array[BatchScrapeResult]
|
|
458
|
-
attr_reader total_count: Integer
|
|
459
|
-
attr_reader completed_count: Integer
|
|
460
|
-
attr_reader failed_count: Integer
|
|
416
|
+
def initialize: (?url: String, ?username: String, ?password: String) -> void
|
|
417
|
+
end
|
|
461
418
|
|
|
462
|
-
|
|
463
|
-
|
|
419
|
+
class ResponseMeta
|
|
420
|
+
attr_reader etag: String?
|
|
421
|
+
attr_reader last_modified: String?
|
|
422
|
+
attr_reader cache_control: String?
|
|
423
|
+
attr_reader server: String?
|
|
424
|
+
attr_reader x_powered_by: String?
|
|
425
|
+
attr_reader content_language: String?
|
|
426
|
+
attr_reader content_encoding: String?
|
|
464
427
|
|
|
465
|
-
|
|
466
|
-
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
428
|
+
def initialize: (?etag: String, ?last_modified: String, ?cache_control: String, ?server: String, ?x_powered_by: String, ?content_language: String, ?content_encoding: String) -> void
|
|
429
|
+
end
|
|
430
|
+
|
|
431
|
+
class ScrapeResult
|
|
432
|
+
attr_reader status_code: Integer
|
|
433
|
+
attr_reader final_url: String
|
|
434
|
+
attr_reader content_type: String
|
|
435
|
+
attr_reader html: String
|
|
436
|
+
attr_reader body_size: Integer
|
|
437
|
+
attr_reader metadata: PageMetadata
|
|
438
|
+
attr_reader links: Array[LinkInfo]
|
|
439
|
+
attr_reader images: Array[ImageInfo]
|
|
440
|
+
attr_reader feeds: Array[FeedInfo]
|
|
441
|
+
attr_reader json_ld: Array[JsonLdEntry]
|
|
442
|
+
attr_reader is_allowed: bool
|
|
443
|
+
attr_reader crawl_delay: Integer?
|
|
444
|
+
attr_reader noindex_detected: bool
|
|
445
|
+
attr_reader nofollow_detected: bool
|
|
446
|
+
attr_reader x_robots_tag: String?
|
|
447
|
+
attr_reader is_pdf: bool
|
|
448
|
+
attr_reader was_skipped: bool
|
|
449
|
+
attr_reader detected_charset: String?
|
|
450
|
+
attr_reader auth_header_sent: bool
|
|
451
|
+
attr_reader response_meta: ResponseMeta?
|
|
452
|
+
attr_reader assets: Array[DownloadedAsset]
|
|
453
|
+
attr_reader js_render_hint: bool
|
|
454
|
+
attr_reader browser_used: bool
|
|
455
|
+
attr_reader markdown: MarkdownResult?
|
|
456
|
+
attr_reader extracted_data: String?
|
|
457
|
+
attr_reader extraction_meta: ExtractionMeta?
|
|
458
|
+
attr_reader screenshot_base64: String?
|
|
459
|
+
attr_reader downloaded_document: DownloadedDocument?
|
|
460
|
+
attr_reader browser: BrowserExtras?
|
|
461
|
+
|
|
462
|
+
def initialize: (?status_code: Integer, ?final_url: String, ?content_type: String, ?html: String, ?body_size: Integer, ?metadata: PageMetadata, ?links: Array[LinkInfo], ?images: Array[ImageInfo], ?feeds: Array[FeedInfo], ?json_ld: Array[JsonLdEntry], ?is_allowed: bool, ?crawl_delay: Integer, ?noindex_detected: bool, ?nofollow_detected: bool, ?x_robots_tag: String, ?is_pdf: bool, ?was_skipped: bool, ?detected_charset: String, ?auth_header_sent: bool, ?response_meta: ResponseMeta, ?assets: Array[DownloadedAsset], ?js_render_hint: bool, ?browser_used: bool, ?markdown: MarkdownResult, ?extracted_data: String, ?extraction_meta: ExtractionMeta, ?screenshot_base64: String, ?downloaded_document: DownloadedDocument, ?browser: BrowserExtras) -> void
|
|
463
|
+
end
|
|
464
|
+
|
|
465
|
+
class SitemapUrl
|
|
466
|
+
attr_reader url: String
|
|
467
|
+
attr_reader lastmod: String?
|
|
468
|
+
attr_reader changefreq: String?
|
|
469
|
+
attr_reader priority: String?
|
|
470
470
|
|
|
471
|
-
def initialize: (?
|
|
472
|
-
|
|
471
|
+
def initialize: (?url: String, ?lastmod: String, ?changefreq: String, ?priority: String) -> void
|
|
472
|
+
end
|
|
473
473
|
|
|
474
|
-
|
|
475
|
-
|
|
474
|
+
class SsrfPolicy
|
|
475
|
+
attr_reader deny_private: bool
|
|
476
476
|
attr_reader allowlist: Array[HostMatcher]
|
|
477
|
-
|
|
477
|
+
attr_reader max_redirects: Integer
|
|
478
|
+
attr_reader scheme_allowlist: Array[String]
|
|
478
479
|
|
|
479
|
-
|
|
480
|
-
|
|
480
|
+
def initialize: (?deny_private: bool, ?allowlist: Array[HostMatcher], ?max_redirects: Integer, ?scheme_allowlist: Array[String]) -> void
|
|
481
|
+
end
|
|
481
482
|
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
483
|
+
class AssetCategory
|
|
484
|
+
type value = :document | :image | :audio | :video | :font | :stylesheet | :script | :archive | :data | :other
|
|
485
|
+
end
|
|
485
486
|
|
|
486
|
-
|
|
487
|
-
|
|
488
|
-
end
|
|
487
|
+
class AuthConfig
|
|
488
|
+
end
|
|
489
489
|
|
|
490
|
-
|
|
491
|
-
|
|
492
|
-
|
|
490
|
+
class BrowserBackend
|
|
491
|
+
type value = :chromiumoxide | :native
|
|
492
|
+
end
|
|
493
493
|
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
494
|
+
class BrowserMode
|
|
495
|
+
type value = :auto | :always | :never | :stealth
|
|
496
|
+
end
|
|
497
497
|
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
498
|
+
class BrowserWait
|
|
499
|
+
type value = :network_idle | :selector | :fixed
|
|
500
|
+
end
|
|
501
501
|
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
502
|
+
class ContentFilterKind
|
|
503
|
+
type value = :bm25
|
|
504
|
+
end
|
|
505
505
|
|
|
506
|
-
|
|
507
|
-
|
|
506
|
+
class CrawlEvent
|
|
507
|
+
end
|
|
508
508
|
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
|
|
509
|
+
class CrawlStrategyKind
|
|
510
|
+
type value = :bfs | :dfs | :best_first | :adaptive
|
|
511
|
+
end
|
|
512
512
|
|
|
513
|
-
|
|
514
|
-
|
|
515
|
-
|
|
513
|
+
class DocumentContentEncoding
|
|
514
|
+
type value = :base64
|
|
515
|
+
end
|
|
516
516
|
|
|
517
|
-
|
|
518
|
-
|
|
519
|
-
|
|
517
|
+
class FeedType
|
|
518
|
+
type value = :rss | :atom | :json_feed
|
|
519
|
+
end
|
|
520
520
|
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
end
|
|
521
|
+
class HostMatcher
|
|
522
|
+
end
|
|
524
523
|
|
|
525
|
-
|
|
526
|
-
|
|
524
|
+
class ImageSource
|
|
525
|
+
type value = :img | :picture_source | :"og:image" | :"twitter:image"
|
|
526
|
+
end
|
|
527
527
|
|
|
528
|
-
|
|
529
|
-
|
|
528
|
+
class LinkType
|
|
529
|
+
type value = :internal | :external | :anchor | :document
|
|
530
|
+
end
|
|
530
531
|
|
|
531
|
-
|
|
532
|
-
|
|
533
|
-
end
|
|
532
|
+
class PageAction
|
|
533
|
+
end
|
|
534
534
|
|
|
535
|
-
|
|
536
|
-
|
|
535
|
+
class ScrollDirection
|
|
536
|
+
type value = :up | :down
|
|
537
|
+
end
|
|
537
538
|
|
|
538
|
-
|
|
539
|
+
def self.batch_crawl: (CrawlEngineHandle engine, Array[String] urls) -> BatchCrawlResults
|
|
539
540
|
|
|
540
|
-
|
|
541
|
+
def self.batch_scrape: (CrawlEngineHandle engine, Array[String] urls) -> BatchScrapeResults
|
|
541
542
|
|
|
542
|
-
|
|
543
|
+
def self.crawl: (CrawlEngineHandle engine, String url) -> CrawlResult
|
|
543
544
|
|
|
544
|
-
|
|
545
|
+
def self.create_engine: (?CrawlConfig config) -> CrawlEngineHandle
|
|
545
546
|
|
|
546
|
-
|
|
547
|
+
def self.generate_citations: (String markdown) -> CitationResult
|
|
547
548
|
|
|
548
|
-
|
|
549
|
+
def self.interact: (CrawlEngineHandle engine, String url, Array[PageAction] actions) -> InteractionResult
|
|
549
550
|
|
|
550
|
-
|
|
551
|
+
def self.map_urls: (CrawlEngineHandle engine, String url) -> MapResult
|
|
551
552
|
|
|
552
|
-
|
|
553
|
+
def self.scrape: (CrawlEngineHandle engine, String url) -> ScrapeResult
|
|
553
554
|
|
|
554
555
|
end
|