crawlfox 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +23 -2
- data/lib/crawlfox.rb +73 -1
- metadata +4 -3
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 3f00152d3e38ba69cf8c03b41ffafd218d383f00da3229716440bd4c931c4c38
|
|
4
|
+
data.tar.gz: 3094b245a4b034b50f05a38bd93b6593e9796c5ca7184f01beaafa887ac3bd6f
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 55be4fd634749a72da737def709b7c772e76317c19b1684acbfcb7b8d571ceb64ac1489c908b6dc5f92ff9effab624c96ee798a060c8d18895bc12ef243fe524
|
|
7
|
+
data.tar.gz: 984f31a8c6149f885fb6b62a9f976c59745e562c3e43401c039e5bcd5d563c9d6a6c0b589780045181b7db1ce0c789005148a8fe671d65a2d6c65277188a1e7c
|
data/README.md
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
# crawlfox (Ruby)
|
|
2
2
|
|
|
3
|
-
Official Ruby SDK for the [CrawlFox](https://crawlfox.io) API: scrape, search (Google, Bing, or DuckDuckGo), batch scrape, and logs.
|
|
3
|
+
Official Ruby SDK for the [CrawlFox](https://crawlfox.io) API: scrape, search (Google, Bing, or DuckDuckGo), batch scrape, map site links, crawl multi-page jobs, and logs.
|
|
4
4
|
|
|
5
5
|
Get a key from the [dashboard](https://crawlfox.io). Set `CRAWLFOX_API_KEY`.
|
|
6
6
|
|
|
@@ -30,7 +30,28 @@ batch = app.batch(
|
|
|
30
30
|
)
|
|
31
31
|
```
|
|
32
32
|
|
|
33
|
-
Formats: `markdown`, `html`, `rawHtml`, `json`, `links`, `images`, `emails`. Use `json_options` / CSS selectors for `json`.
|
|
33
|
+
Formats: `markdown`, `html`, `rawHtml`, `json`, `links`, `images`, `emails`. Use `json_options` / CSS selectors for `json`.
|
|
34
|
+
|
|
35
|
+
## Map
|
|
36
|
+
|
|
37
|
+
`POST /v1/map`. One credit per call.
|
|
38
|
+
|
|
39
|
+
```ruby
|
|
40
|
+
mapped = app.map("https://example.com/", limit: 10)
|
|
41
|
+
mapped["links"].each { |link| puts link["url"] }
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
## Crawl
|
|
45
|
+
|
|
46
|
+
```ruby
|
|
47
|
+
job = app.crawl("https://example.com/", limit: 3, scrapeOptions: { formats: ["markdown"] })
|
|
48
|
+
status = app.get_crawl(job["id"])
|
|
49
|
+
done = app.wait_for_crawl(job["id"])
|
|
50
|
+
app.cancel_crawl(job["id"])
|
|
51
|
+
errors = app.get_crawl_errors(job["id"])
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Credits: scrape 1 per page, search 1 per 10 requested results, map 1 per call, crawl 1 per page per format.
|
|
34
55
|
|
|
35
56
|
```bash
|
|
36
57
|
ruby -Ilib:test test/client_test.rb
|
data/lib/crawlfox.rb
CHANGED
|
@@ -5,7 +5,7 @@ require "net/http"
|
|
|
5
5
|
require "uri"
|
|
6
6
|
|
|
7
7
|
module Crawlfox
|
|
8
|
-
VERSION = "0.
|
|
8
|
+
VERSION = "0.2.0"
|
|
9
9
|
DEFAULT_API_URL = "https://api.crawlfox.io"
|
|
10
10
|
|
|
11
11
|
class Error < StandardError
|
|
@@ -74,6 +74,78 @@ module Crawlfox
|
|
|
74
74
|
request("GET", "/v1/logs/#{URI.encode_www_form_component(id)}/result")
|
|
75
75
|
end
|
|
76
76
|
|
|
77
|
+
def map(url, **options)
|
|
78
|
+
env = request("POST", "/v1/map", { url: url }.merge(options))
|
|
79
|
+
{
|
|
80
|
+
"success" => env.fetch("success", true),
|
|
81
|
+
"id" => env["id"],
|
|
82
|
+
"links" => env["links"] || []
|
|
83
|
+
}
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
def crawl(url, **options)
|
|
87
|
+
env = request("POST", "/v1/crawl", { url: url }.merge(options))
|
|
88
|
+
raise Error.new("crawl response missing id", status: 500, retryable: false) if env["id"].to_s.empty?
|
|
89
|
+
env
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
def get_crawl(id, skip: nil)
|
|
93
|
+
path = "/v1/crawl/#{URI.encode_www_form_component(id)}"
|
|
94
|
+
path += "?skip=#{skip}" if skip && skip.to_i > 0
|
|
95
|
+
request("GET", path)
|
|
96
|
+
end
|
|
97
|
+
|
|
98
|
+
def cancel_crawl(id)
|
|
99
|
+
request("DELETE", "/v1/crawl/#{URI.encode_www_form_component(id)}")
|
|
100
|
+
nil
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def get_crawl_errors(id)
|
|
104
|
+
request("GET", "/v1/crawl/#{URI.encode_www_form_component(id)}/errors")
|
|
105
|
+
end
|
|
106
|
+
|
|
107
|
+
def wait_for_crawl(id, poll_interval_ms: 1500, timeout_ms: 180_000)
|
|
108
|
+
started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
109
|
+
skip = 0
|
|
110
|
+
merged = nil
|
|
111
|
+
|
|
112
|
+
loop do
|
|
113
|
+
page = get_crawl(id, skip: skip > 0 ? skip : nil)
|
|
114
|
+
if merged.nil?
|
|
115
|
+
merged = page.dup
|
|
116
|
+
merged["data"] = Array(page["data"]).dup
|
|
117
|
+
else
|
|
118
|
+
merged["status"] = page["status"]
|
|
119
|
+
merged["total"] = page["total"]
|
|
120
|
+
merged["completed"] = page["completed"]
|
|
121
|
+
merged["creditsUsed"] = page["creditsUsed"]
|
|
122
|
+
merged["next"] = page["next"]
|
|
123
|
+
merged["data"].concat(Array(page["data"]))
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
status = page["status"]
|
|
127
|
+
nxt = page["next"]
|
|
128
|
+
if %w[completed failed cancelled].include?(status) && nxt.to_s.empty?
|
|
129
|
+
return merged
|
|
130
|
+
end
|
|
131
|
+
if nxt && !nxt.empty?
|
|
132
|
+
begin
|
|
133
|
+
u = URI.parse(nxt)
|
|
134
|
+
s = u.query ? URI.decode_www_form(u.query).to_h["skip"] : nil
|
|
135
|
+
skip = s ? s.to_i : skip + Array(page["data"]).length
|
|
136
|
+
rescue StandardError
|
|
137
|
+
skip += Array(page["data"]).length
|
|
138
|
+
end
|
|
139
|
+
elsif %w[completed failed cancelled].include?(status)
|
|
140
|
+
return merged
|
|
141
|
+
end
|
|
142
|
+
if (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) * 1000 > timeout_ms
|
|
143
|
+
raise Error.new("crawl poll timed out", status: 408, retryable: true)
|
|
144
|
+
end
|
|
145
|
+
sleep(poll_interval_ms / 1000.0)
|
|
146
|
+
end
|
|
147
|
+
end
|
|
148
|
+
|
|
77
149
|
private
|
|
78
150
|
|
|
79
151
|
def document(envelope)
|
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: crawlfox
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.2.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Automote LLC
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-09-
|
|
11
|
+
date: 2026-09-18 00:00:00.000000000 Z
|
|
12
12
|
dependencies: []
|
|
13
13
|
description:
|
|
14
14
|
email:
|
|
@@ -45,5 +45,6 @@ requirements: []
|
|
|
45
45
|
rubygems_version: 3.4.20
|
|
46
46
|
signing_key:
|
|
47
47
|
specification_version: 4
|
|
48
|
-
summary: Official Ruby SDK for
|
|
48
|
+
summary: 'Official Ruby SDK for CrawlFox: scrape, batch, search, map site URLs, and
|
|
49
|
+
crawl sites with polling helpers.'
|
|
49
50
|
test_files: []
|