context.dev 1.17.0 → 1.19.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +19 -0
- data/README.md +3 -3
- data/lib/context_dev/models/web_screenshot_params.rb +24 -1
- data/lib/context_dev/models/web_web_crawl_md_params.rb +53 -8
- data/lib/context_dev/models/web_web_crawl_md_response.rb +3 -2
- data/lib/context_dev/models/web_web_scrape_html_params.rb +42 -8
- data/lib/context_dev/models/web_web_scrape_md_params.rb +42 -8
- data/lib/context_dev/resources/web.rb +12 -9
- data/lib/context_dev/version.rb +1 -1
- data/rbi/context_dev/models/web_screenshot_params.rbi +62 -0
- data/rbi/context_dev/models/web_web_crawl_md_params.rbi +90 -13
- data/rbi/context_dev/models/web_web_crawl_md_response.rbi +4 -2
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +73 -13
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +73 -13
- data/rbi/context_dev/resources/web.rbi +24 -15
- data/sig/context_dev/models/web_screenshot_params.rbs +20 -0
- data/sig/context_dev/models/web_web_crawl_md_params.rbs +38 -5
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +31 -5
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +31 -5
- data/sig/context_dev/resources/web.rbs +5 -3
- metadata +2 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 1f8d1ca708475c531193510f12a13c95d03e0a28c10b5411d6f1de6fc922dcab
|
|
4
|
+
data.tar.gz: 55b4b7999f774bd387bd3cc03a0c4095b4fb33295a5bb27d3a1ed69968557b14
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: f6b4313f8b005eb2680db0468f0854b1d64bae299a89997ada0e7fb7d82559deca0e59431bd6d5b7c8342362233e2289cfd365d1afbaf2da1961e3481366ebfb
|
|
7
|
+
data.tar.gz: 20d55c97248661498d7ff43a627d63c97457900625bc53f7fb1a6c35ae28229ad2ece229c26474468330c6a23e4e50f0921307f7c78f3cf3ade343f893bea481
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,24 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.19.0 (2026-05-11)
|
|
4
|
+
|
|
5
|
+
Full Changelog: [v1.18.0...v1.19.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.18.0...v1.19.0)
|
|
6
|
+
|
|
7
|
+
### Features
|
|
8
|
+
|
|
9
|
+
* **api:** api update ([ab9c393](https://github.com/context-dot-dev/context-ruby-sdk/commit/ab9c3935b669fbe2c7a8ffdd5bff7ea05fc2ae56))
|
|
10
|
+
* **api:** api update ([7f8b7b7](https://github.com/context-dot-dev/context-ruby-sdk/commit/7f8b7b7a720f3fdf5364ffaf7c7e55f171b72c0e))
|
|
11
|
+
|
|
12
|
+
## 1.18.0 (2026-05-10)
|
|
13
|
+
|
|
14
|
+
Full Changelog: [v1.17.0...v1.18.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.17.0...v1.18.0)
|
|
15
|
+
|
|
16
|
+
### Features
|
|
17
|
+
|
|
18
|
+
* **api:** api update ([b582c05](https://github.com/context-dot-dev/context-ruby-sdk/commit/b582c05376102bb0cb6f8d4d8c9a2cefdef8c1ec))
|
|
19
|
+
* **api:** api update ([4a4e4bb](https://github.com/context-dot-dev/context-ruby-sdk/commit/4a4e4bbc547662de263a307b213dd7eecd03a61d))
|
|
20
|
+
* **api:** manual updates ([ec963bb](https://github.com/context-dot-dev/context-ruby-sdk/commit/ec963bb99ac36d162552c76fb067e87144f21089))
|
|
21
|
+
|
|
3
22
|
## 1.17.0 (2026-05-09)
|
|
4
23
|
|
|
5
24
|
Full Changelog: [v1.16.0...v1.17.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.16.0...v1.17.0)
|
data/README.md
CHANGED
|
@@ -8,8 +8,8 @@ It is generated with [Stainless](https://www.stainless.com/).
|
|
|
8
8
|
|
|
9
9
|
Use the Context Dev MCP Server to enable AI assistants to interact with this API, allowing them to explore endpoints, make test requests, and use documentation to help integrate this SDK into your application.
|
|
10
10
|
|
|
11
|
-
[](https://cursor.com/en-US/install-mcp?name=context
|
|
12
|
-
[](https://vscode.stainless.com/mcp/%7B%22name%22%3A%22context
|
|
11
|
+
[](https://cursor.com/en-US/install-mcp?name=context-dev-mcp&config=eyJuYW1lIjoiY29udGV4dC1kZXYtbWNwIiwidHJhbnNwb3J0IjoiaHR0cCIsInVybCI6Imh0dHBzOi8vY29udGV4dC1kZXYuc3RsbWNwLmNvbSIsImhlYWRlcnMiOnsieC1jb250ZXh0LWRldi1hcGkta2V5IjoiTXkgQVBJIEtleSJ9fQ)
|
|
12
|
+
[](https://vscode.stainless.com/mcp/%7B%22name%22%3A%22context-dev-mcp%22%2C%22type%22%3A%22http%22%2C%22url%22%3A%22https%3A%2F%2Fcontext-dev.stlmcp.com%22%2C%22headers%22%3A%7B%22x-context-dev-api-key%22%3A%22My%20API%20Key%22%7D%7D)
|
|
13
13
|
|
|
14
14
|
> Note: You may need to set environment variables in your MCP client.
|
|
15
15
|
|
|
@@ -26,7 +26,7 @@ To use this gem, install via Bundler by adding the following to your application
|
|
|
26
26
|
<!-- x-release-please-start-version -->
|
|
27
27
|
|
|
28
28
|
```ruby
|
|
29
|
-
gem "context.dev", "~> 1.
|
|
29
|
+
gem "context.dev", "~> 1.19.0"
|
|
30
30
|
```
|
|
31
31
|
|
|
32
32
|
<!-- x-release-please-end -->
|
|
@@ -31,6 +31,14 @@ module ContextDev
|
|
|
31
31
|
# @return [Symbol, ContextDev::Models::WebScreenshotParams::FullScreenshot, nil]
|
|
32
32
|
optional :full_screenshot, enum: -> { ContextDev::WebScreenshotParams::FullScreenshot }
|
|
33
33
|
|
|
34
|
+
# @!attribute handle_cookie_popup
|
|
35
|
+
# Optional parameter to control cookie/consent popup handling. If 'true', we
|
|
36
|
+
# dismiss cookie banner before capture. If 'false' or not provided, captures the
|
|
37
|
+
# page without that step.
|
|
38
|
+
#
|
|
39
|
+
# @return [Symbol, ContextDev::Models::WebScreenshotParams::HandleCookiePopup, nil]
|
|
40
|
+
optional :handle_cookie_popup, enum: -> { ContextDev::WebScreenshotParams::HandleCookiePopup }
|
|
41
|
+
|
|
34
42
|
# @!attribute max_age_ms
|
|
35
43
|
# Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
36
44
|
# and is younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
@@ -71,7 +79,7 @@ module ContextDev
|
|
|
71
79
|
# @return [Integer, nil]
|
|
72
80
|
optional :wait_for_ms, Integer
|
|
73
81
|
|
|
74
|
-
# @!method initialize(direct_url: nil, domain: nil, full_screenshot: nil, max_age_ms: nil, page: nil, timeout_ms: nil, viewport: nil, wait_for_ms: nil, request_options: {})
|
|
82
|
+
# @!method initialize(direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, max_age_ms: nil, page: nil, timeout_ms: nil, viewport: nil, wait_for_ms: nil, request_options: {})
|
|
75
83
|
# Some parameter documentations has been truncated, see
|
|
76
84
|
# {ContextDev::Models::WebScreenshotParams} for more details.
|
|
77
85
|
#
|
|
@@ -81,6 +89,8 @@ module ContextDev
|
|
|
81
89
|
#
|
|
82
90
|
# @param full_screenshot [Symbol, ContextDev::Models::WebScreenshotParams::FullScreenshot] Optional parameter to determine screenshot type. If 'true', takes a full page sc
|
|
83
91
|
#
|
|
92
|
+
# @param handle_cookie_popup [Symbol, ContextDev::Models::WebScreenshotParams::HandleCookiePopup] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
|
|
93
|
+
#
|
|
84
94
|
# @param max_age_ms [Integer] Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
85
95
|
#
|
|
86
96
|
# @param page [Symbol, ContextDev::Models::WebScreenshotParams::Page] Optional parameter to specify which page type to screenshot. If provided, the sy
|
|
@@ -106,6 +116,19 @@ module ContextDev
|
|
|
106
116
|
# @return [Array<Symbol>]
|
|
107
117
|
end
|
|
108
118
|
|
|
119
|
+
# Optional parameter to control cookie/consent popup handling. If 'true', we
|
|
120
|
+
# dismiss cookie banner before capture. If 'false' or not provided, captures the
|
|
121
|
+
# page without that step.
|
|
122
|
+
module HandleCookiePopup
|
|
123
|
+
extend ContextDev::Internal::Type::Enum
|
|
124
|
+
|
|
125
|
+
TRUE = :true
|
|
126
|
+
FALSE = :false
|
|
127
|
+
|
|
128
|
+
# @!method self.values
|
|
129
|
+
# @return [Array<Symbol>]
|
|
130
|
+
end
|
|
131
|
+
|
|
109
132
|
# Optional parameter to specify which page type to screenshot. If provided, the
|
|
110
133
|
# system will scrape the domain's links and use heuristics to find the most
|
|
111
134
|
# appropriate URL for the specified page type (30 supported languages). If not
|
|
@@ -60,13 +60,12 @@ module ContextDev
|
|
|
60
60
|
# @return [Integer, nil]
|
|
61
61
|
optional :max_pages, Integer, api_name: :maxPages
|
|
62
62
|
|
|
63
|
-
# @!attribute
|
|
64
|
-
#
|
|
65
|
-
#
|
|
66
|
-
# entirely (not included in results and not counted as failures).
|
|
63
|
+
# @!attribute pdf
|
|
64
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
65
|
+
# inclusive 1-based page range.
|
|
67
66
|
#
|
|
68
|
-
# @return [
|
|
69
|
-
optional :
|
|
67
|
+
# @return [ContextDev::Models::WebWebCrawlMdParams::Pdf, nil]
|
|
68
|
+
optional :pdf, -> { ContextDev::WebWebCrawlMdParams::Pdf }
|
|
70
69
|
|
|
71
70
|
# @!attribute shorten_base64_images
|
|
72
71
|
# Truncate base64-encoded image data in the Markdown output
|
|
@@ -74,6 +73,15 @@ module ContextDev
|
|
|
74
73
|
# @return [Boolean, nil]
|
|
75
74
|
optional :shorten_base64_images, ContextDev::Internal::Type::Boolean, api_name: :shortenBase64Images
|
|
76
75
|
|
|
76
|
+
# @!attribute stop_after_ms
|
|
77
|
+
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
78
|
+
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
79
|
+
# instead of continuing. Min: 10000 (10s). Max: 240000 (4 min). Default: 120000 (2
|
|
80
|
+
# min).
|
|
81
|
+
#
|
|
82
|
+
# @return [Integer, nil]
|
|
83
|
+
optional :stop_after_ms, Integer, api_name: :stopAfterMs
|
|
84
|
+
|
|
77
85
|
# @!attribute timeout_ms
|
|
78
86
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
79
87
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
@@ -102,7 +110,7 @@ module ContextDev
|
|
|
102
110
|
# @return [Integer, nil]
|
|
103
111
|
optional :wait_for_ms, Integer, api_name: :waitForMs
|
|
104
112
|
|
|
105
|
-
# @!method initialize(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil,
|
|
113
|
+
# @!method initialize(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
106
114
|
# Some parameter documentations has been truncated, see
|
|
107
115
|
# {ContextDev::Models::WebWebCrawlMdParams} for more details.
|
|
108
116
|
#
|
|
@@ -122,10 +130,12 @@ module ContextDev
|
|
|
122
130
|
#
|
|
123
131
|
# @param max_pages [Integer] Maximum number of pages to crawl. Hard cap: 500.
|
|
124
132
|
#
|
|
125
|
-
# @param
|
|
133
|
+
# @param pdf [ContextDev::Models::WebWebCrawlMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
126
134
|
#
|
|
127
135
|
# @param shorten_base64_images [Boolean] Truncate base64-encoded image data in the Markdown output
|
|
128
136
|
#
|
|
137
|
+
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds. After each scrape, the crawler c
|
|
138
|
+
#
|
|
129
139
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
130
140
|
#
|
|
131
141
|
# @param url_regex [String] Regex pattern. Only URLs matching this pattern will be followed and scraped.
|
|
@@ -135,6 +145,41 @@ module ContextDev
|
|
|
135
145
|
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
|
|
136
146
|
#
|
|
137
147
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
|
|
148
|
+
|
|
149
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
150
|
+
# @!attribute end_
|
|
151
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
152
|
+
# Must be greater than or equal to start when both are provided.
|
|
153
|
+
#
|
|
154
|
+
# @return [Integer, nil]
|
|
155
|
+
optional :end_, Integer, api_name: :end
|
|
156
|
+
|
|
157
|
+
# @!attribute should_parse
|
|
158
|
+
# When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
|
|
159
|
+
# entirely (not included in results and not counted as failures).
|
|
160
|
+
#
|
|
161
|
+
# @return [Boolean, nil]
|
|
162
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
163
|
+
|
|
164
|
+
# @!attribute start
|
|
165
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
166
|
+
#
|
|
167
|
+
# @return [Integer, nil]
|
|
168
|
+
optional :start, Integer
|
|
169
|
+
|
|
170
|
+
# @!method initialize(end_: nil, should_parse: nil, start: nil)
|
|
171
|
+
# Some parameter documentations has been truncated, see
|
|
172
|
+
# {ContextDev::Models::WebWebCrawlMdParams::Pdf} for more details.
|
|
173
|
+
#
|
|
174
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
175
|
+
# inclusive 1-based page range.
|
|
176
|
+
#
|
|
177
|
+
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
178
|
+
#
|
|
179
|
+
# @param should_parse [Boolean] When true, PDF pages are fetched and parsed. When false, PDF pages are skipped e
|
|
180
|
+
#
|
|
181
|
+
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
182
|
+
end
|
|
138
183
|
end
|
|
139
184
|
end
|
|
140
185
|
end
|
|
@@ -34,7 +34,8 @@ module ContextDev
|
|
|
34
34
|
required :num_failed, Integer, api_name: :numFailed
|
|
35
35
|
|
|
36
36
|
# @!attribute num_skipped
|
|
37
|
-
# Number of URLs skipped (PDFs when
|
|
37
|
+
# Number of URLs skipped (PDFs when pdf.shouldParse=false, or URLs not matching
|
|
38
|
+
# urlRegex)
|
|
38
39
|
#
|
|
39
40
|
# @return [Integer]
|
|
40
41
|
required :num_skipped, Integer, api_name: :numSkipped
|
|
@@ -59,7 +60,7 @@ module ContextDev
|
|
|
59
60
|
#
|
|
60
61
|
# @param num_failed [Integer] Number of pages that failed to crawl
|
|
61
62
|
#
|
|
62
|
-
# @param num_skipped [Integer] Number of URLs skipped (PDFs when
|
|
63
|
+
# @param num_skipped [Integer] Number of URLs skipped (PDFs when pdf.shouldParse=false, or URLs not matching ur
|
|
63
64
|
#
|
|
64
65
|
# @param num_succeeded [Integer] Number of pages successfully crawled
|
|
65
66
|
#
|
|
@@ -27,13 +27,12 @@ module ContextDev
|
|
|
27
27
|
# @return [Integer, nil]
|
|
28
28
|
optional :max_age_ms, Integer
|
|
29
29
|
|
|
30
|
-
# @!attribute
|
|
31
|
-
#
|
|
32
|
-
#
|
|
33
|
-
# and a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
30
|
+
# @!attribute pdf
|
|
31
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
32
|
+
# inclusive 1-based page range.
|
|
34
33
|
#
|
|
35
|
-
# @return [
|
|
36
|
-
optional :
|
|
34
|
+
# @return [ContextDev::Models::WebWebScrapeHTMLParams::Pdf, nil]
|
|
35
|
+
optional :pdf, -> { ContextDev::WebWebScrapeHTMLParams::Pdf }
|
|
37
36
|
|
|
38
37
|
# @!attribute timeout_ms
|
|
39
38
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
@@ -50,7 +49,7 @@ module ContextDev
|
|
|
50
49
|
# @return [Integer, nil]
|
|
51
50
|
optional :wait_for_ms, Integer
|
|
52
51
|
|
|
53
|
-
# @!method initialize(url:, include_frames: nil, max_age_ms: nil,
|
|
52
|
+
# @!method initialize(url:, include_frames: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
54
53
|
# Some parameter documentations has been truncated, see
|
|
55
54
|
# {ContextDev::Models::WebWebScrapeHTMLParams} for more details.
|
|
56
55
|
#
|
|
@@ -60,13 +59,48 @@ module ContextDev
|
|
|
60
59
|
#
|
|
61
60
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
62
61
|
#
|
|
63
|
-
# @param
|
|
62
|
+
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
64
63
|
#
|
|
65
64
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
66
65
|
#
|
|
67
66
|
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
68
67
|
#
|
|
69
68
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
|
|
69
|
+
|
|
70
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
71
|
+
# @!attribute end_
|
|
72
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
73
|
+
# Must be greater than or equal to start when both are provided.
|
|
74
|
+
#
|
|
75
|
+
# @return [Integer, nil]
|
|
76
|
+
optional :end_, Integer, api_name: :end
|
|
77
|
+
|
|
78
|
+
# @!attribute should_parse
|
|
79
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
80
|
+
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
81
|
+
#
|
|
82
|
+
# @return [Boolean, nil]
|
|
83
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
84
|
+
|
|
85
|
+
# @!attribute start
|
|
86
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
87
|
+
#
|
|
88
|
+
# @return [Integer, nil]
|
|
89
|
+
optional :start, Integer
|
|
90
|
+
|
|
91
|
+
# @!method initialize(end_: nil, should_parse: nil, start: nil)
|
|
92
|
+
# Some parameter documentations has been truncated, see
|
|
93
|
+
# {ContextDev::Models::WebWebScrapeHTMLParams::Pdf} for more details.
|
|
94
|
+
#
|
|
95
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
96
|
+
# inclusive 1-based page range.
|
|
97
|
+
#
|
|
98
|
+
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
99
|
+
#
|
|
100
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
101
|
+
#
|
|
102
|
+
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
103
|
+
end
|
|
70
104
|
end
|
|
71
105
|
end
|
|
72
106
|
end
|
|
@@ -40,13 +40,12 @@ module ContextDev
|
|
|
40
40
|
# @return [Integer, nil]
|
|
41
41
|
optional :max_age_ms, Integer
|
|
42
42
|
|
|
43
|
-
# @!attribute
|
|
44
|
-
#
|
|
45
|
-
#
|
|
46
|
-
# WEBSITE_ACCESS_ERROR is returned.
|
|
43
|
+
# @!attribute pdf
|
|
44
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
45
|
+
# inclusive 1-based page range.
|
|
47
46
|
#
|
|
48
|
-
# @return [
|
|
49
|
-
optional :
|
|
47
|
+
# @return [ContextDev::Models::WebWebScrapeMdParams::Pdf, nil]
|
|
48
|
+
optional :pdf, -> { ContextDev::WebWebScrapeMdParams::Pdf }
|
|
50
49
|
|
|
51
50
|
# @!attribute shorten_base64_images
|
|
52
51
|
# Shorten base64-encoded image data in the Markdown output
|
|
@@ -76,7 +75,7 @@ module ContextDev
|
|
|
76
75
|
# @return [Integer, nil]
|
|
77
76
|
optional :wait_for_ms, Integer
|
|
78
77
|
|
|
79
|
-
# @!method initialize(url:, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil,
|
|
78
|
+
# @!method initialize(url:, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
80
79
|
# Some parameter documentations has been truncated, see
|
|
81
80
|
# {ContextDev::Models::WebWebScrapeMdParams} for more details.
|
|
82
81
|
#
|
|
@@ -90,7 +89,7 @@ module ContextDev
|
|
|
90
89
|
#
|
|
91
90
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
92
91
|
#
|
|
93
|
-
# @param
|
|
92
|
+
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
94
93
|
#
|
|
95
94
|
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
96
95
|
#
|
|
@@ -101,6 +100,41 @@ module ContextDev
|
|
|
101
100
|
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load before conver
|
|
102
101
|
#
|
|
103
102
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
|
|
103
|
+
|
|
104
|
+
class Pdf < ContextDev::Internal::Type::BaseModel
|
|
105
|
+
# @!attribute end_
|
|
106
|
+
# Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
|
|
107
|
+
# Must be greater than or equal to start when both are provided.
|
|
108
|
+
#
|
|
109
|
+
# @return [Integer, nil]
|
|
110
|
+
optional :end_, Integer, api_name: :end
|
|
111
|
+
|
|
112
|
+
# @!attribute should_parse
|
|
113
|
+
# When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
114
|
+
# a 400 WEBSITE_ACCESS_ERROR is returned.
|
|
115
|
+
#
|
|
116
|
+
# @return [Boolean, nil]
|
|
117
|
+
optional :should_parse, ContextDev::Internal::Type::Boolean, api_name: :shouldParse
|
|
118
|
+
|
|
119
|
+
# @!attribute start
|
|
120
|
+
# First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
121
|
+
#
|
|
122
|
+
# @return [Integer, nil]
|
|
123
|
+
optional :start, Integer
|
|
124
|
+
|
|
125
|
+
# @!method initialize(end_: nil, should_parse: nil, start: nil)
|
|
126
|
+
# Some parameter documentations has been truncated, see
|
|
127
|
+
# {ContextDev::Models::WebWebScrapeMdParams::Pdf} for more details.
|
|
128
|
+
#
|
|
129
|
+
# PDF parsing controls. Use start/end to limit text extraction and OCR to an
|
|
130
|
+
# inclusive 1-based page range.
|
|
131
|
+
#
|
|
132
|
+
# @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
|
|
133
|
+
#
|
|
134
|
+
# @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
|
|
135
|
+
#
|
|
136
|
+
# @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
|
|
137
|
+
end
|
|
104
138
|
end
|
|
105
139
|
end
|
|
106
140
|
end
|
|
@@ -70,7 +70,7 @@ module ContextDev
|
|
|
70
70
|
#
|
|
71
71
|
# Capture a screenshot of a website.
|
|
72
72
|
#
|
|
73
|
-
# @overload screenshot(direct_url: nil, domain: nil, full_screenshot: nil, max_age_ms: nil, page: nil, timeout_ms: nil, viewport: nil, wait_for_ms: nil, request_options: {})
|
|
73
|
+
# @overload screenshot(direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, max_age_ms: nil, page: nil, timeout_ms: nil, viewport: nil, wait_for_ms: nil, request_options: {})
|
|
74
74
|
#
|
|
75
75
|
# @param direct_url [String] A specific URL to screenshot directly, bypassing domain resolution (e.g., 'https
|
|
76
76
|
#
|
|
@@ -78,6 +78,8 @@ module ContextDev
|
|
|
78
78
|
#
|
|
79
79
|
# @param full_screenshot [Symbol, ContextDev::Models::WebScreenshotParams::FullScreenshot] Optional parameter to determine screenshot type. If 'true', takes a full page sc
|
|
80
80
|
#
|
|
81
|
+
# @param handle_cookie_popup [Symbol, ContextDev::Models::WebScreenshotParams::HandleCookiePopup] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
|
|
82
|
+
#
|
|
81
83
|
# @param max_age_ms [Integer] Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
82
84
|
#
|
|
83
85
|
# @param page [Symbol, ContextDev::Models::WebScreenshotParams::Page] Optional parameter to specify which page type to screenshot. If provided, the sy
|
|
@@ -102,6 +104,7 @@ module ContextDev
|
|
|
102
104
|
query: query.transform_keys(
|
|
103
105
|
direct_url: "directUrl",
|
|
104
106
|
full_screenshot: "fullScreenshot",
|
|
107
|
+
handle_cookie_popup: "handleCookiePopup",
|
|
105
108
|
max_age_ms: "maxAgeMs",
|
|
106
109
|
timeout_ms: "timeoutMS",
|
|
107
110
|
wait_for_ms: "waitForMs"
|
|
@@ -117,7 +120,7 @@ module ContextDev
|
|
|
117
120
|
# Performs a crawl starting from a given URL, extracts page content as Markdown,
|
|
118
121
|
# and returns results for all crawled pages.
|
|
119
122
|
#
|
|
120
|
-
# @overload web_crawl_md(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil,
|
|
123
|
+
# @overload web_crawl_md(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
121
124
|
#
|
|
122
125
|
# @param url [String] The starting URL for the crawl (must include http:// or https:// protocol)
|
|
123
126
|
#
|
|
@@ -135,10 +138,12 @@ module ContextDev
|
|
|
135
138
|
#
|
|
136
139
|
# @param max_pages [Integer] Maximum number of pages to crawl. Hard cap: 500.
|
|
137
140
|
#
|
|
138
|
-
# @param
|
|
141
|
+
# @param pdf [ContextDev::Models::WebWebCrawlMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
139
142
|
#
|
|
140
143
|
# @param shorten_base64_images [Boolean] Truncate base64-encoded image data in the Markdown output
|
|
141
144
|
#
|
|
145
|
+
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds. After each scrape, the crawler c
|
|
146
|
+
#
|
|
142
147
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
143
148
|
#
|
|
144
149
|
# @param url_regex [String] Regex pattern. Only URLs matching this pattern will be followed and scraped.
|
|
@@ -168,7 +173,7 @@ module ContextDev
|
|
|
168
173
|
#
|
|
169
174
|
# Scrapes the given URL and returns the raw HTML content of the page.
|
|
170
175
|
#
|
|
171
|
-
# @overload web_scrape_html(url:, include_frames: nil, max_age_ms: nil,
|
|
176
|
+
# @overload web_scrape_html(url:, include_frames: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
172
177
|
#
|
|
173
178
|
# @param url [String] Full URL to scrape (must include http:// or https:// protocol)
|
|
174
179
|
#
|
|
@@ -176,7 +181,7 @@ module ContextDev
|
|
|
176
181
|
#
|
|
177
182
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
178
183
|
#
|
|
179
|
-
# @param
|
|
184
|
+
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
180
185
|
#
|
|
181
186
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
182
187
|
#
|
|
@@ -196,7 +201,6 @@ module ContextDev
|
|
|
196
201
|
query: query.transform_keys(
|
|
197
202
|
include_frames: "includeFrames",
|
|
198
203
|
max_age_ms: "maxAgeMs",
|
|
199
|
-
parse_pdf: "parsePDF",
|
|
200
204
|
timeout_ms: "timeoutMS",
|
|
201
205
|
wait_for_ms: "waitForMs"
|
|
202
206
|
),
|
|
@@ -251,7 +255,7 @@ module ContextDev
|
|
|
251
255
|
#
|
|
252
256
|
# Scrapes the given URL into LLM usable Markdown.
|
|
253
257
|
#
|
|
254
|
-
# @overload web_scrape_md(url:, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil,
|
|
258
|
+
# @overload web_scrape_md(url:, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
255
259
|
#
|
|
256
260
|
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
257
261
|
#
|
|
@@ -263,7 +267,7 @@ module ContextDev
|
|
|
263
267
|
#
|
|
264
268
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
265
269
|
#
|
|
266
|
-
# @param
|
|
270
|
+
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
267
271
|
#
|
|
268
272
|
# @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
|
|
269
273
|
#
|
|
@@ -289,7 +293,6 @@ module ContextDev
|
|
|
289
293
|
include_images: "includeImages",
|
|
290
294
|
include_links: "includeLinks",
|
|
291
295
|
max_age_ms: "maxAgeMs",
|
|
292
|
-
parse_pdf: "parsePDF",
|
|
293
296
|
shorten_base64_images: "shortenBase64Images",
|
|
294
297
|
timeout_ms: "timeoutMS",
|
|
295
298
|
use_main_content_only: "useMainContentOnly",
|
data/lib/context_dev/version.rb
CHANGED
|
@@ -47,6 +47,26 @@ module ContextDev
|
|
|
47
47
|
end
|
|
48
48
|
attr_writer :full_screenshot
|
|
49
49
|
|
|
50
|
+
# Optional parameter to control cookie/consent popup handling. If 'true', we
|
|
51
|
+
# dismiss cookie banner before capture. If 'false' or not provided, captures the
|
|
52
|
+
# page without that step.
|
|
53
|
+
sig do
|
|
54
|
+
returns(
|
|
55
|
+
T.nilable(
|
|
56
|
+
ContextDev::WebScreenshotParams::HandleCookiePopup::OrSymbol
|
|
57
|
+
)
|
|
58
|
+
)
|
|
59
|
+
end
|
|
60
|
+
attr_reader :handle_cookie_popup
|
|
61
|
+
|
|
62
|
+
sig do
|
|
63
|
+
params(
|
|
64
|
+
handle_cookie_popup:
|
|
65
|
+
ContextDev::WebScreenshotParams::HandleCookiePopup::OrSymbol
|
|
66
|
+
).void
|
|
67
|
+
end
|
|
68
|
+
attr_writer :handle_cookie_popup
|
|
69
|
+
|
|
50
70
|
# Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
51
71
|
# and is younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
52
72
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always capture fresh.
|
|
@@ -102,6 +122,8 @@ module ContextDev
|
|
|
102
122
|
domain: String,
|
|
103
123
|
full_screenshot:
|
|
104
124
|
ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
|
|
125
|
+
handle_cookie_popup:
|
|
126
|
+
ContextDev::WebScreenshotParams::HandleCookiePopup::OrSymbol,
|
|
105
127
|
max_age_ms: Integer,
|
|
106
128
|
page: ContextDev::WebScreenshotParams::Page::OrSymbol,
|
|
107
129
|
timeout_ms: Integer,
|
|
@@ -123,6 +145,10 @@ module ContextDev
|
|
|
123
145
|
# screenshot capturing all content. If 'false' or not provided, takes a viewport
|
|
124
146
|
# screenshot (standard browser view).
|
|
125
147
|
full_screenshot: nil,
|
|
148
|
+
# Optional parameter to control cookie/consent popup handling. If 'true', we
|
|
149
|
+
# dismiss cookie banner before capture. If 'false' or not provided, captures the
|
|
150
|
+
# page without that step.
|
|
151
|
+
handle_cookie_popup: nil,
|
|
126
152
|
# Return a cached screenshot if a prior screenshot for the same parameters exists
|
|
127
153
|
# and is younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
128
154
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always capture fresh.
|
|
@@ -154,6 +180,8 @@ module ContextDev
|
|
|
154
180
|
domain: String,
|
|
155
181
|
full_screenshot:
|
|
156
182
|
ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
|
|
183
|
+
handle_cookie_popup:
|
|
184
|
+
ContextDev::WebScreenshotParams::HandleCookiePopup::OrSymbol,
|
|
157
185
|
max_age_ms: Integer,
|
|
158
186
|
page: ContextDev::WebScreenshotParams::Page::OrSymbol,
|
|
159
187
|
timeout_ms: Integer,
|
|
@@ -200,6 +228,40 @@ module ContextDev
|
|
|
200
228
|
end
|
|
201
229
|
end
|
|
202
230
|
|
|
231
|
+
# Optional parameter to control cookie/consent popup handling. If 'true', we
|
|
232
|
+
# dismiss cookie banner before capture. If 'false' or not provided, captures the
|
|
233
|
+
# page without that step.
|
|
234
|
+
module HandleCookiePopup
|
|
235
|
+
extend ContextDev::Internal::Type::Enum
|
|
236
|
+
|
|
237
|
+
TaggedSymbol =
|
|
238
|
+
T.type_alias do
|
|
239
|
+
T.all(Symbol, ContextDev::WebScreenshotParams::HandleCookiePopup)
|
|
240
|
+
end
|
|
241
|
+
OrSymbol = T.type_alias { T.any(Symbol, String) }
|
|
242
|
+
|
|
243
|
+
TRUE =
|
|
244
|
+
T.let(
|
|
245
|
+
:true,
|
|
246
|
+
ContextDev::WebScreenshotParams::HandleCookiePopup::TaggedSymbol
|
|
247
|
+
)
|
|
248
|
+
FALSE =
|
|
249
|
+
T.let(
|
|
250
|
+
:false,
|
|
251
|
+
ContextDev::WebScreenshotParams::HandleCookiePopup::TaggedSymbol
|
|
252
|
+
)
|
|
253
|
+
|
|
254
|
+
sig do
|
|
255
|
+
override.returns(
|
|
256
|
+
T::Array[
|
|
257
|
+
ContextDev::WebScreenshotParams::HandleCookiePopup::TaggedSymbol
|
|
258
|
+
]
|
|
259
|
+
)
|
|
260
|
+
end
|
|
261
|
+
def self.values
|
|
262
|
+
end
|
|
263
|
+
end
|
|
264
|
+
|
|
203
265
|
# Optional parameter to specify which page type to screenshot. If provided, the
|
|
204
266
|
# system will scrape the domain's links and use heuristics to find the most
|
|
205
267
|
# appropriate URL for the specified page type (30 supported languages). If not
|