crawlfox 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (4) hide show
  1. checksums.yaml +4 -4
  2. data/README.md +23 -2
  3. data/lib/crawlfox.rb +73 -1
  4. metadata +4 -3
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: bee7b5227e2463cfd1bd2f1bc30eebe985d9dc40672af60446174027a80c2e61
4
- data.tar.gz: 64cbd128cf1ed931049d053f9ffd2e6baa93a40fa15f2621dda7fa99edcb441d
3
+ metadata.gz: 3f00152d3e38ba69cf8c03b41ffafd218d383f00da3229716440bd4c931c4c38
4
+ data.tar.gz: 3094b245a4b034b50f05a38bd93b6593e9796c5ca7184f01beaafa887ac3bd6f
5
5
  SHA512:
6
- metadata.gz: 957829da07e3df0c495235ad1df257be32a699db15893681034a7e466278c20bba86b24047737e22a238bd9796a42ef533f314eb19bffc67adb20a8e6cd91f46
7
- data.tar.gz: 422d16cc631d9bd98c83e0f1606d8b43b5f6dbdac321f93bc00800fe1d4799058530e1f31960e700bedfe3fdf14231c587965faae5ca81e7370073df277fca1a
6
+ metadata.gz: 55be4fd634749a72da737def709b7c772e76317c19b1684acbfcb7b8d571ceb64ac1489c908b6dc5f92ff9effab624c96ee798a060c8d18895bc12ef243fe524
7
+ data.tar.gz: 984f31a8c6149f885fb6b62a9f976c59745e562c3e43401c039e5bcd5d563c9d6a6c0b589780045181b7db1ce0c789005148a8fe671d65a2d6c65277188a1e7c
data/README.md CHANGED
@@ -1,6 +1,6 @@
1
1
  # crawlfox (Ruby)
2
2
 
3
- Official Ruby SDK for the [CrawlFox](https://crawlfox.io) API: scrape, search (Google, Bing, or DuckDuckGo), batch scrape, and logs.
3
+ Official Ruby SDK for the [CrawlFox](https://crawlfox.io) API: scrape, search (Google, Bing, or DuckDuckGo), batch scrape, map site links, crawl multi-page jobs, and logs.
4
4
 
5
5
  Get a key from the [dashboard](https://crawlfox.io). Set `CRAWLFOX_API_KEY`.
6
6
 
@@ -30,7 +30,28 @@ batch = app.batch(
30
30
  )
31
31
  ```
32
32
 
33
- Formats: `markdown`, `html`, `rawHtml`, `json`, `links`, `images`, `emails`. Use `json_options` / CSS selectors for `json`. Credits: 1 per page, 1 per 10 requested search results.
33
+ Formats: `markdown`, `html`, `rawHtml`, `json`, `links`, `images`, `emails`. Use `json_options` / CSS selectors for `json`.
34
+
35
+ ## Map
36
+
37
+ `POST /v1/map`. One credit per call.
38
+
39
+ ```ruby
40
+ mapped = app.map("https://example.com/", limit: 10)
41
+ mapped["links"].each { |link| puts link["url"] }
42
+ ```
43
+
44
+ ## Crawl
45
+
46
+ ```ruby
47
+ job = app.crawl("https://example.com/", limit: 3, scrapeOptions: { formats: ["markdown"] })
48
+ status = app.get_crawl(job["id"])
49
+ done = app.wait_for_crawl(job["id"])
50
+ app.cancel_crawl(job["id"])
51
+ errors = app.get_crawl_errors(job["id"])
52
+ ```
53
+
54
+ Credits: scrape 1 per page, search 1 per 10 requested results, map 1 per call, crawl 1 per page per format.
34
55
 
35
56
  ```bash
36
57
  ruby -Ilib:test test/client_test.rb
data/lib/crawlfox.rb CHANGED
@@ -5,7 +5,7 @@ require "net/http"
5
5
  require "uri"
6
6
 
7
7
  module Crawlfox
8
- VERSION = "0.1.0"
8
+ VERSION = "0.2.0"
9
9
  DEFAULT_API_URL = "https://api.crawlfox.io"
10
10
 
11
11
  class Error < StandardError
@@ -74,6 +74,78 @@ module Crawlfox
74
74
  request("GET", "/v1/logs/#{URI.encode_www_form_component(id)}/result")
75
75
  end
76
76
 
77
+ def map(url, **options)
78
+ env = request("POST", "/v1/map", { url: url }.merge(options))
79
+ {
80
+ "success" => env.fetch("success", true),
81
+ "id" => env["id"],
82
+ "links" => env["links"] || []
83
+ }
84
+ end
85
+
86
+ def crawl(url, **options)
87
+ env = request("POST", "/v1/crawl", { url: url }.merge(options))
88
+ raise Error.new("crawl response missing id", status: 500, retryable: false) if env["id"].to_s.empty?
89
+ env
90
+ end
91
+
92
+ def get_crawl(id, skip: nil)
93
+ path = "/v1/crawl/#{URI.encode_www_form_component(id)}"
94
+ path += "?skip=#{skip}" if skip && skip.to_i > 0
95
+ request("GET", path)
96
+ end
97
+
98
+ def cancel_crawl(id)
99
+ request("DELETE", "/v1/crawl/#{URI.encode_www_form_component(id)}")
100
+ nil
101
+ end
102
+
103
+ def get_crawl_errors(id)
104
+ request("GET", "/v1/crawl/#{URI.encode_www_form_component(id)}/errors")
105
+ end
106
+
107
+ def wait_for_crawl(id, poll_interval_ms: 1500, timeout_ms: 180_000)
108
+ started = Process.clock_gettime(Process::CLOCK_MONOTONIC)
109
+ skip = 0
110
+ merged = nil
111
+
112
+ loop do
113
+ page = get_crawl(id, skip: skip > 0 ? skip : nil)
114
+ if merged.nil?
115
+ merged = page.dup
116
+ merged["data"] = Array(page["data"]).dup
117
+ else
118
+ merged["status"] = page["status"]
119
+ merged["total"] = page["total"]
120
+ merged["completed"] = page["completed"]
121
+ merged["creditsUsed"] = page["creditsUsed"]
122
+ merged["next"] = page["next"]
123
+ merged["data"].concat(Array(page["data"]))
124
+ end
125
+
126
+ status = page["status"]
127
+ nxt = page["next"]
128
+ if %w[completed failed cancelled].include?(status) && nxt.to_s.empty?
129
+ return merged
130
+ end
131
+ if nxt && !nxt.empty?
132
+ begin
133
+ u = URI.parse(nxt)
134
+ s = u.query ? URI.decode_www_form(u.query).to_h["skip"] : nil
135
+ skip = s ? s.to_i : skip + Array(page["data"]).length
136
+ rescue StandardError
137
+ skip += Array(page["data"]).length
138
+ end
139
+ elsif %w[completed failed cancelled].include?(status)
140
+ return merged
141
+ end
142
+ if (Process.clock_gettime(Process::CLOCK_MONOTONIC) - started) * 1000 > timeout_ms
143
+ raise Error.new("crawl poll timed out", status: 408, retryable: true)
144
+ end
145
+ sleep(poll_interval_ms / 1000.0)
146
+ end
147
+ end
148
+
77
149
  private
78
150
 
79
151
  def document(envelope)
metadata CHANGED
@@ -1,14 +1,14 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: crawlfox
3
3
  version: !ruby/object:Gem::Version
4
- version: 0.1.0
4
+ version: 0.2.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Automote LLC
8
8
  autorequire:
9
9
  bindir: bin
10
10
  cert_chain: []
11
- date: 2026-09-10 00:00:00.000000000 Z
11
+ date: 2026-09-18 00:00:00.000000000 Z
12
12
  dependencies: []
13
13
  description:
14
14
  email:
@@ -45,5 +45,6 @@ requirements: []
45
45
  rubygems_version: 3.4.20
46
46
  signing_key:
47
47
  specification_version: 4
48
- summary: Official Ruby SDK for the CrawlFox scrape and search API
48
+ summary: 'Official Ruby SDK for CrawlFox: scrape, batch, search, map site URLs, and
49
+ crawl sites with polling helpers.'
49
50
  test_files: []