context.dev 1.29.0 → 1.31.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +17 -0
- data/README.md +1 -1
- data/lib/context_dev/models/web_extract_params.rb +3 -2
- data/lib/context_dev/models/web_web_crawl_md_params.rb +24 -3
- data/lib/context_dev/models/web_web_scrape_html_params.rb +30 -1
- data/lib/context_dev/models/web_web_scrape_md_params.rb +21 -1
- data/lib/context_dev/resources/web.rb +23 -4
- data/lib/context_dev/version.rb +1 -1
- data/rbi/context_dev/models/web_extract_params.rbi +4 -2
- data/rbi/context_dev/models/web_web_crawl_md_params.rbi +36 -4
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +43 -0
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +30 -0
- data/rbi/context_dev/resources/web.rbi +39 -3
- data/sig/context_dev/models/web_web_crawl_md_params.rbs +14 -0
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +21 -0
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +14 -0
- data/sig/context_dev/resources/web.rbs +7 -0
- metadata +2 -2
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 113f1babdb07592f8be5f9e71e90c91dbb71e247c23d4c8607ab3e41454d0134
|
|
4
|
+
data.tar.gz: 3d360daae866082593f3bd1c96aca5e8344a47ea8ba6b1d0d3a25f5caa8ce65c
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 5482ddcb84ae21fa52f1b15af283d0a9106674c9a59403183d6496d3c1d125fb070e926246ee9348b1848eff0c9859f8e26496e61e24c9198a0e0afcc6f0f204
|
|
7
|
+
data.tar.gz: 047b20f851d1d4af45eb23986823e664a2bdcf08bffd470f519367806328ecb8a27635d3fd1e9a5a724129b05bad8242b9d6f51d0ee80391fc5adea0ff567b1e
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,22 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.31.0 (2026-06-08)
|
|
4
|
+
|
|
5
|
+
Full Changelog: [v1.30.0...v1.31.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.30.0...v1.31.0)
|
|
6
|
+
|
|
7
|
+
### Features
|
|
8
|
+
|
|
9
|
+
* **api:** api update ([2170040](https://github.com/context-dot-dev/context-ruby-sdk/commit/2170040142dab272a55f903f0d062b0d36d751aa))
|
|
10
|
+
* **api:** api update ([c57bae4](https://github.com/context-dot-dev/context-ruby-sdk/commit/c57bae40d99770e18c92d402eb71dc0aa651f522))
|
|
11
|
+
|
|
12
|
+
## 1.30.0 (2026-06-07)
|
|
13
|
+
|
|
14
|
+
Full Changelog: [v1.29.0...v1.30.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.29.0...v1.30.0)
|
|
15
|
+
|
|
16
|
+
### Features
|
|
17
|
+
|
|
18
|
+
* **api:** api update ([c897bbf](https://github.com/context-dot-dev/context-ruby-sdk/commit/c897bbf03235960c8036bd808c5239b935d94bf3))
|
|
19
|
+
|
|
3
20
|
## 1.29.0 (2026-06-07)
|
|
4
21
|
|
|
5
22
|
Full Changelog: [v1.28.0...v1.29.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.28.0...v1.29.0)
|
data/README.md
CHANGED
|
@@ -65,7 +65,8 @@ module ContextDev
|
|
|
65
65
|
optional :pdf, -> { ContextDev::WebExtractParams::Pdf }
|
|
66
66
|
|
|
67
67
|
# @!attribute stop_after_ms
|
|
68
|
-
# Soft time budget for the crawl in milliseconds.
|
|
68
|
+
# Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000
|
|
69
|
+
# (110s). Default: 80000 (80s).
|
|
69
70
|
#
|
|
70
71
|
# @return [Integer, nil]
|
|
71
72
|
optional :stop_after_ms, Integer, api_name: :stopAfterMs
|
|
@@ -105,7 +106,7 @@ module ContextDev
|
|
|
105
106
|
#
|
|
106
107
|
# @param pdf [ContextDev::Models::WebExtractParams::Pdf]
|
|
107
108
|
#
|
|
108
|
-
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds.
|
|
109
|
+
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000 (1
|
|
109
110
|
#
|
|
110
111
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
111
112
|
#
|
|
@@ -13,6 +13,14 @@ module ContextDev
|
|
|
13
13
|
# @return [String]
|
|
14
14
|
required :url, String
|
|
15
15
|
|
|
16
|
+
# @!attribute exclude_selectors
|
|
17
|
+
# CSS selectors to remove before each crawled page is converted to Markdown.
|
|
18
|
+
# Applied after includeSelectors. Exclusion takes precedence: an element matching
|
|
19
|
+
# both is removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
20
|
+
#
|
|
21
|
+
# @return [Array<String>, nil]
|
|
22
|
+
optional :exclude_selectors, ContextDev::Internal::Type::ArrayOf[String], api_name: :excludeSelectors
|
|
23
|
+
|
|
16
24
|
# @!attribute follow_subdomains
|
|
17
25
|
# When true, follow links on subdomains of the starting URL's domain (e.g.
|
|
18
26
|
# docs.example.com when starting from example.com). www and apex are always
|
|
@@ -40,6 +48,15 @@ module ContextDev
|
|
|
40
48
|
# @return [Boolean, nil]
|
|
41
49
|
optional :include_links, ContextDev::Internal::Type::Boolean, api_name: :includeLinks
|
|
42
50
|
|
|
51
|
+
# @!attribute include_selectors
|
|
52
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
53
|
+
# descendants) are kept before each crawled page is converted to Markdown. When
|
|
54
|
+
# omitted, the entire document is kept. Examples: "article.main", "#content",
|
|
55
|
+
# "[role=main]".
|
|
56
|
+
#
|
|
57
|
+
# @return [Array<String>, nil]
|
|
58
|
+
optional :include_selectors, ContextDev::Internal::Type::ArrayOf[String], api_name: :includeSelectors
|
|
59
|
+
|
|
43
60
|
# @!attribute max_age_ms
|
|
44
61
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
45
62
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
@@ -76,8 +93,8 @@ module ContextDev
|
|
|
76
93
|
# @!attribute stop_after_ms
|
|
77
94
|
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
78
95
|
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
79
|
-
# instead of continuing. Min: 10000 (10s). Max:
|
|
80
|
-
#
|
|
96
|
+
# instead of continuing. Min: 10000 (10s). Max: 110000 (110s). Default: 80000
|
|
97
|
+
# (80s).
|
|
81
98
|
#
|
|
82
99
|
# @return [Integer, nil]
|
|
83
100
|
optional :stop_after_ms, Integer, api_name: :stopAfterMs
|
|
@@ -110,12 +127,14 @@ module ContextDev
|
|
|
110
127
|
# @return [Integer, nil]
|
|
111
128
|
optional :wait_for_ms, Integer, api_name: :waitForMs
|
|
112
129
|
|
|
113
|
-
# @!method initialize(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
130
|
+
# @!method initialize(url:, exclude_selectors: nil, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
114
131
|
# Some parameter documentations has been truncated, see
|
|
115
132
|
# {ContextDev::Models::WebWebCrawlMdParams} for more details.
|
|
116
133
|
#
|
|
117
134
|
# @param url [String] The starting URL for the crawl (must include http:// or https:// protocol)
|
|
118
135
|
#
|
|
136
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove before each crawled page is converted to Markdown. Appli
|
|
137
|
+
#
|
|
119
138
|
# @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain (e.g. docs.ex
|
|
120
139
|
#
|
|
121
140
|
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown for each crawled pag
|
|
@@ -124,6 +143,8 @@ module ContextDev
|
|
|
124
143
|
#
|
|
125
144
|
# @param include_links [Boolean] Preserve hyperlinks in the Markdown output
|
|
126
145
|
#
|
|
146
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
147
|
+
#
|
|
127
148
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
128
149
|
#
|
|
129
150
|
# @param max_depth [Integer] Maximum link depth from the starting URL (0 = only the starting page)
|
|
@@ -13,6 +13,14 @@ module ContextDev
|
|
|
13
13
|
# @return [String]
|
|
14
14
|
required :url, String
|
|
15
15
|
|
|
16
|
+
# @!attribute exclude_selectors
|
|
17
|
+
# CSS selectors to remove from the result. Applied after includeSelectors.
|
|
18
|
+
# Exclusion takes precedence: an element matching both is removed. Examples:
|
|
19
|
+
# "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
20
|
+
#
|
|
21
|
+
# @return [Array<String>, nil]
|
|
22
|
+
optional :exclude_selectors, ContextDev::Internal::Type::ArrayOf[String]
|
|
23
|
+
|
|
16
24
|
# @!attribute headers
|
|
17
25
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
18
26
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
@@ -27,6 +35,14 @@ module ContextDev
|
|
|
27
35
|
# @return [Boolean, nil]
|
|
28
36
|
optional :include_frames, ContextDev::Internal::Type::Boolean
|
|
29
37
|
|
|
38
|
+
# @!attribute include_selectors
|
|
39
|
+
# CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
40
|
+
# kept and everything else is dropped. When omitted, the entire document is kept.
|
|
41
|
+
# Examples: "article.main", "#content", "[role=main]".
|
|
42
|
+
#
|
|
43
|
+
# @return [Array<String>, nil]
|
|
44
|
+
optional :include_selectors, ContextDev::Internal::Type::ArrayOf[String]
|
|
45
|
+
|
|
30
46
|
# @!attribute max_age_ms
|
|
31
47
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
32
48
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
@@ -50,6 +66,13 @@ module ContextDev
|
|
|
50
66
|
# @return [Integer, nil]
|
|
51
67
|
optional :timeout_ms, Integer
|
|
52
68
|
|
|
69
|
+
# @!attribute use_main_content_only
|
|
70
|
+
# When true, return only the page's main content in the HTML response, excluding
|
|
71
|
+
# headers, footers, sidebars, and navigation when detectable.
|
|
72
|
+
#
|
|
73
|
+
# @return [Boolean, nil]
|
|
74
|
+
optional :use_main_content_only, ContextDev::Internal::Type::Boolean
|
|
75
|
+
|
|
53
76
|
# @!attribute wait_for_ms
|
|
54
77
|
# Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
55
78
|
# 30000 (30 seconds).
|
|
@@ -57,22 +80,28 @@ module ContextDev
|
|
|
57
80
|
# @return [Integer, nil]
|
|
58
81
|
optional :wait_for_ms, Integer
|
|
59
82
|
|
|
60
|
-
# @!method initialize(url:, headers: nil, include_frames: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
83
|
+
# @!method initialize(url:, exclude_selectors: nil, headers: nil, include_frames: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
61
84
|
# Some parameter documentations has been truncated, see
|
|
62
85
|
# {ContextDev::Models::WebWebScrapeHTMLParams} for more details.
|
|
63
86
|
#
|
|
64
87
|
# @param url [String] Full URL to scrape (must include http:// or https:// protocol)
|
|
65
88
|
#
|
|
89
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove from the result. Applied after includeSelectors. Exclusi
|
|
90
|
+
#
|
|
66
91
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
67
92
|
#
|
|
68
93
|
# @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
|
|
69
94
|
#
|
|
95
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
96
|
+
#
|
|
70
97
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
71
98
|
#
|
|
72
99
|
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
73
100
|
#
|
|
74
101
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
75
102
|
#
|
|
103
|
+
# @param use_main_content_only [Boolean] When true, return only the page's main content in the HTML response, excluding h
|
|
104
|
+
#
|
|
76
105
|
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
77
106
|
#
|
|
78
107
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
|
|
@@ -14,6 +14,14 @@ module ContextDev
|
|
|
14
14
|
# @return [String]
|
|
15
15
|
required :url, String
|
|
16
16
|
|
|
17
|
+
# @!attribute exclude_selectors
|
|
18
|
+
# CSS selectors to remove before conversion to Markdown. Applied after
|
|
19
|
+
# includeSelectors. Exclusion takes precedence: an element matching both is
|
|
20
|
+
# removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
21
|
+
#
|
|
22
|
+
# @return [Array<String>, nil]
|
|
23
|
+
optional :exclude_selectors, ContextDev::Internal::Type::ArrayOf[String]
|
|
24
|
+
|
|
17
25
|
# @!attribute headers
|
|
18
26
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
19
27
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
@@ -40,6 +48,14 @@ module ContextDev
|
|
|
40
48
|
# @return [Boolean, nil]
|
|
41
49
|
optional :include_links, ContextDev::Internal::Type::Boolean
|
|
42
50
|
|
|
51
|
+
# @!attribute include_selectors
|
|
52
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
53
|
+
# descendants) are kept before conversion to Markdown. When omitted, the entire
|
|
54
|
+
# document is kept. Examples: "article.main", "#content", "[role=main]".
|
|
55
|
+
#
|
|
56
|
+
# @return [Array<String>, nil]
|
|
57
|
+
optional :include_selectors, ContextDev::Internal::Type::ArrayOf[String]
|
|
58
|
+
|
|
43
59
|
# @!attribute max_age_ms
|
|
44
60
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
45
61
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
@@ -83,12 +99,14 @@ module ContextDev
|
|
|
83
99
|
# @return [Integer, nil]
|
|
84
100
|
optional :wait_for_ms, Integer
|
|
85
101
|
|
|
86
|
-
# @!method initialize(url:, headers: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
102
|
+
# @!method initialize(url:, exclude_selectors: nil, headers: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
87
103
|
# Some parameter documentations has been truncated, see
|
|
88
104
|
# {ContextDev::Models::WebWebScrapeMdParams} for more details.
|
|
89
105
|
#
|
|
90
106
|
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
91
107
|
#
|
|
108
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove before conversion to Markdown. Applied after includeSele
|
|
109
|
+
#
|
|
92
110
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
93
111
|
#
|
|
94
112
|
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
|
|
@@ -97,6 +115,8 @@ module ContextDev
|
|
|
97
115
|
#
|
|
98
116
|
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
99
117
|
#
|
|
118
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
119
|
+
#
|
|
100
120
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
101
121
|
#
|
|
102
122
|
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
@@ -27,7 +27,7 @@ module ContextDev
|
|
|
27
27
|
#
|
|
28
28
|
# @param pdf [ContextDev::Models::WebExtractParams::Pdf]
|
|
29
29
|
#
|
|
30
|
-
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds.
|
|
30
|
+
# @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000 (1
|
|
31
31
|
#
|
|
32
32
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
33
33
|
#
|
|
@@ -246,10 +246,12 @@ module ContextDev
|
|
|
246
246
|
# Performs a crawl starting from a given URL, extracts page content as Markdown,
|
|
247
247
|
# and returns results for all crawled pages.
|
|
248
248
|
#
|
|
249
|
-
# @overload web_crawl_md(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
249
|
+
# @overload web_crawl_md(url:, exclude_selectors: nil, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
250
250
|
#
|
|
251
251
|
# @param url [String] The starting URL for the crawl (must include http:// or https:// protocol)
|
|
252
252
|
#
|
|
253
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove before each crawled page is converted to Markdown. Appli
|
|
254
|
+
#
|
|
253
255
|
# @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain (e.g. docs.ex
|
|
254
256
|
#
|
|
255
257
|
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown for each crawled pag
|
|
@@ -258,6 +260,8 @@ module ContextDev
|
|
|
258
260
|
#
|
|
259
261
|
# @param include_links [Boolean] Preserve hyperlinks in the Markdown output
|
|
260
262
|
#
|
|
263
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
264
|
+
#
|
|
261
265
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
262
266
|
#
|
|
263
267
|
# @param max_depth [Integer] Maximum link depth from the starting URL (0 = only the starting page)
|
|
@@ -299,20 +303,26 @@ module ContextDev
|
|
|
299
303
|
#
|
|
300
304
|
# Scrapes the given URL and returns the raw HTML content of the page.
|
|
301
305
|
#
|
|
302
|
-
# @overload web_scrape_html(url:, headers: nil, include_frames: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
306
|
+
# @overload web_scrape_html(url:, exclude_selectors: nil, headers: nil, include_frames: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
303
307
|
#
|
|
304
308
|
# @param url [String] Full URL to scrape (must include http:// or https:// protocol)
|
|
305
309
|
#
|
|
310
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove from the result. Applied after includeSelectors. Exclusi
|
|
311
|
+
#
|
|
306
312
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
307
313
|
#
|
|
308
314
|
# @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
|
|
309
315
|
#
|
|
316
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
317
|
+
#
|
|
310
318
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
311
319
|
#
|
|
312
320
|
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
313
321
|
#
|
|
314
322
|
# @param timeout_ms [Integer] Optional timeout in milliseconds for the request. If the request takes longer th
|
|
315
323
|
#
|
|
324
|
+
# @param use_main_content_only [Boolean] When true, return only the page's main content in the HTML response, excluding h
|
|
325
|
+
#
|
|
316
326
|
# @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
317
327
|
#
|
|
318
328
|
# @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
|
|
@@ -327,9 +337,12 @@ module ContextDev
|
|
|
327
337
|
method: :get,
|
|
328
338
|
path: "web/scrape/html",
|
|
329
339
|
query: query.transform_keys(
|
|
340
|
+
exclude_selectors: "excludeSelectors",
|
|
330
341
|
include_frames: "includeFrames",
|
|
342
|
+
include_selectors: "includeSelectors",
|
|
331
343
|
max_age_ms: "maxAgeMs",
|
|
332
344
|
timeout_ms: "timeoutMS",
|
|
345
|
+
use_main_content_only: "useMainContentOnly",
|
|
333
346
|
wait_for_ms: "waitForMs"
|
|
334
347
|
),
|
|
335
348
|
model: ContextDev::Models::WebWebScrapeHTMLResponse,
|
|
@@ -385,10 +398,12 @@ module ContextDev
|
|
|
385
398
|
#
|
|
386
399
|
# Scrapes the given URL into LLM usable Markdown.
|
|
387
400
|
#
|
|
388
|
-
# @overload web_scrape_md(url:, headers: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
401
|
+
# @overload web_scrape_md(url:, exclude_selectors: nil, headers: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
389
402
|
#
|
|
390
403
|
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
391
404
|
#
|
|
405
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove before conversion to Markdown. Applied after includeSele
|
|
406
|
+
#
|
|
392
407
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
393
408
|
#
|
|
394
409
|
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
|
|
@@ -397,6 +412,8 @@ module ContextDev
|
|
|
397
412
|
#
|
|
398
413
|
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
399
414
|
#
|
|
415
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
416
|
+
#
|
|
400
417
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
401
418
|
#
|
|
402
419
|
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
@@ -421,9 +438,11 @@ module ContextDev
|
|
|
421
438
|
method: :get,
|
|
422
439
|
path: "web/scrape/markdown",
|
|
423
440
|
query: query.transform_keys(
|
|
441
|
+
exclude_selectors: "excludeSelectors",
|
|
424
442
|
include_frames: "includeFrames",
|
|
425
443
|
include_images: "includeImages",
|
|
426
444
|
include_links: "includeLinks",
|
|
445
|
+
include_selectors: "includeSelectors",
|
|
427
446
|
max_age_ms: "maxAgeMs",
|
|
428
447
|
shorten_base64_images: "shortenBase64Images",
|
|
429
448
|
timeout_ms: "timeoutMS",
|
data/lib/context_dev/version.rb
CHANGED
|
@@ -70,7 +70,8 @@ module ContextDev
|
|
|
70
70
|
sig { params(pdf: ContextDev::WebExtractParams::Pdf::OrHash).void }
|
|
71
71
|
attr_writer :pdf
|
|
72
72
|
|
|
73
|
-
# Soft time budget for the crawl in milliseconds.
|
|
73
|
+
# Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000
|
|
74
|
+
# (110s). Default: 80000 (80s).
|
|
74
75
|
sig { returns(T.nilable(Integer)) }
|
|
75
76
|
attr_reader :stop_after_ms
|
|
76
77
|
|
|
@@ -136,7 +137,8 @@ module ContextDev
|
|
|
136
137
|
# younger than this many milliseconds. Defaults to 7 days (604800000 ms).
|
|
137
138
|
max_age_ms: nil,
|
|
138
139
|
pdf: nil,
|
|
139
|
-
# Soft time budget for the crawl in milliseconds.
|
|
140
|
+
# Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000
|
|
141
|
+
# (110s). Default: 80000 (80s).
|
|
140
142
|
stop_after_ms: nil,
|
|
141
143
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
142
144
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
@@ -15,6 +15,15 @@ module ContextDev
|
|
|
15
15
|
sig { returns(String) }
|
|
16
16
|
attr_accessor :url
|
|
17
17
|
|
|
18
|
+
# CSS selectors to remove before each crawled page is converted to Markdown.
|
|
19
|
+
# Applied after includeSelectors. Exclusion takes precedence: an element matching
|
|
20
|
+
# both is removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
21
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
22
|
+
attr_reader :exclude_selectors
|
|
23
|
+
|
|
24
|
+
sig { params(exclude_selectors: T::Array[String]).void }
|
|
25
|
+
attr_writer :exclude_selectors
|
|
26
|
+
|
|
18
27
|
# When true, follow links on subdomains of the starting URL's domain (e.g.
|
|
19
28
|
# docs.example.com when starting from example.com). www and apex are always
|
|
20
29
|
# treated as equivalent.
|
|
@@ -46,6 +55,16 @@ module ContextDev
|
|
|
46
55
|
sig { params(include_links: T::Boolean).void }
|
|
47
56
|
attr_writer :include_links
|
|
48
57
|
|
|
58
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
59
|
+
# descendants) are kept before each crawled page is converted to Markdown. When
|
|
60
|
+
# omitted, the entire document is kept. Examples: "article.main", "#content",
|
|
61
|
+
# "[role=main]".
|
|
62
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
63
|
+
attr_reader :include_selectors
|
|
64
|
+
|
|
65
|
+
sig { params(include_selectors: T::Array[String]).void }
|
|
66
|
+
attr_writer :include_selectors
|
|
67
|
+
|
|
49
68
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
50
69
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
51
70
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -86,8 +105,8 @@ module ContextDev
|
|
|
86
105
|
|
|
87
106
|
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
88
107
|
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
89
|
-
# instead of continuing. Min: 10000 (10s). Max:
|
|
90
|
-
#
|
|
108
|
+
# instead of continuing. Min: 10000 (10s). Max: 110000 (110s). Default: 80000
|
|
109
|
+
# (80s).
|
|
91
110
|
sig { returns(T.nilable(Integer)) }
|
|
92
111
|
attr_reader :stop_after_ms
|
|
93
112
|
|
|
@@ -129,10 +148,12 @@ module ContextDev
|
|
|
129
148
|
sig do
|
|
130
149
|
params(
|
|
131
150
|
url: String,
|
|
151
|
+
exclude_selectors: T::Array[String],
|
|
132
152
|
follow_subdomains: T::Boolean,
|
|
133
153
|
include_frames: T::Boolean,
|
|
134
154
|
include_images: T::Boolean,
|
|
135
155
|
include_links: T::Boolean,
|
|
156
|
+
include_selectors: T::Array[String],
|
|
136
157
|
max_age_ms: Integer,
|
|
137
158
|
max_depth: Integer,
|
|
138
159
|
max_pages: Integer,
|
|
@@ -149,6 +170,10 @@ module ContextDev
|
|
|
149
170
|
def self.new(
|
|
150
171
|
# The starting URL for the crawl (must include http:// or https:// protocol)
|
|
151
172
|
url:,
|
|
173
|
+
# CSS selectors to remove before each crawled page is converted to Markdown.
|
|
174
|
+
# Applied after includeSelectors. Exclusion takes precedence: an element matching
|
|
175
|
+
# both is removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
176
|
+
exclude_selectors: nil,
|
|
152
177
|
# When true, follow links on subdomains of the starting URL's domain (e.g.
|
|
153
178
|
# docs.example.com when starting from example.com). www and apex are always
|
|
154
179
|
# treated as equivalent.
|
|
@@ -160,6 +185,11 @@ module ContextDev
|
|
|
160
185
|
include_images: nil,
|
|
161
186
|
# Preserve hyperlinks in the Markdown output
|
|
162
187
|
include_links: nil,
|
|
188
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
189
|
+
# descendants) are kept before each crawled page is converted to Markdown. When
|
|
190
|
+
# omitted, the entire document is kept. Examples: "article.main", "#content",
|
|
191
|
+
# "[role=main]".
|
|
192
|
+
include_selectors: nil,
|
|
163
193
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
164
194
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
165
195
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -175,8 +205,8 @@ module ContextDev
|
|
|
175
205
|
shorten_base64_images: nil,
|
|
176
206
|
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
177
207
|
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
178
|
-
# instead of continuing. Min: 10000 (10s). Max:
|
|
179
|
-
#
|
|
208
|
+
# instead of continuing. Min: 10000 (10s). Max: 110000 (110s). Default: 80000
|
|
209
|
+
# (80s).
|
|
180
210
|
stop_after_ms: nil,
|
|
181
211
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
182
212
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
@@ -198,10 +228,12 @@ module ContextDev
|
|
|
198
228
|
override.returns(
|
|
199
229
|
{
|
|
200
230
|
url: String,
|
|
231
|
+
exclude_selectors: T::Array[String],
|
|
201
232
|
follow_subdomains: T::Boolean,
|
|
202
233
|
include_frames: T::Boolean,
|
|
203
234
|
include_images: T::Boolean,
|
|
204
235
|
include_links: T::Boolean,
|
|
236
|
+
include_selectors: T::Array[String],
|
|
205
237
|
max_age_ms: Integer,
|
|
206
238
|
max_depth: Integer,
|
|
207
239
|
max_pages: Integer,
|
|
@@ -18,6 +18,15 @@ module ContextDev
|
|
|
18
18
|
sig { returns(String) }
|
|
19
19
|
attr_accessor :url
|
|
20
20
|
|
|
21
|
+
# CSS selectors to remove from the result. Applied after includeSelectors.
|
|
22
|
+
# Exclusion takes precedence: an element matching both is removed. Examples:
|
|
23
|
+
# "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
24
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
25
|
+
attr_reader :exclude_selectors
|
|
26
|
+
|
|
27
|
+
sig { params(exclude_selectors: T::Array[String]).void }
|
|
28
|
+
attr_writer :exclude_selectors
|
|
29
|
+
|
|
21
30
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
22
31
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
23
32
|
# is bypassed: the result is neither read from nor written to cache.
|
|
@@ -34,6 +43,15 @@ module ContextDev
|
|
|
34
43
|
sig { params(include_frames: T::Boolean).void }
|
|
35
44
|
attr_writer :include_frames
|
|
36
45
|
|
|
46
|
+
# CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
47
|
+
# kept and everything else is dropped. When omitted, the entire document is kept.
|
|
48
|
+
# Examples: "article.main", "#content", "[role=main]".
|
|
49
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
50
|
+
attr_reader :include_selectors
|
|
51
|
+
|
|
52
|
+
sig { params(include_selectors: T::Array[String]).void }
|
|
53
|
+
attr_writer :include_selectors
|
|
54
|
+
|
|
37
55
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
38
56
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
39
57
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -60,6 +78,14 @@ module ContextDev
|
|
|
60
78
|
sig { params(timeout_ms: Integer).void }
|
|
61
79
|
attr_writer :timeout_ms
|
|
62
80
|
|
|
81
|
+
# When true, return only the page's main content in the HTML response, excluding
|
|
82
|
+
# headers, footers, sidebars, and navigation when detectable.
|
|
83
|
+
sig { returns(T.nilable(T::Boolean)) }
|
|
84
|
+
attr_reader :use_main_content_only
|
|
85
|
+
|
|
86
|
+
sig { params(use_main_content_only: T::Boolean).void }
|
|
87
|
+
attr_writer :use_main_content_only
|
|
88
|
+
|
|
63
89
|
# Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
64
90
|
# 30000 (30 seconds).
|
|
65
91
|
sig { returns(T.nilable(Integer)) }
|
|
@@ -71,11 +97,14 @@ module ContextDev
|
|
|
71
97
|
sig do
|
|
72
98
|
params(
|
|
73
99
|
url: String,
|
|
100
|
+
exclude_selectors: T::Array[String],
|
|
74
101
|
headers: T::Hash[Symbol, String],
|
|
75
102
|
include_frames: T::Boolean,
|
|
103
|
+
include_selectors: T::Array[String],
|
|
76
104
|
max_age_ms: Integer,
|
|
77
105
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
|
|
78
106
|
timeout_ms: Integer,
|
|
107
|
+
use_main_content_only: T::Boolean,
|
|
79
108
|
wait_for_ms: Integer,
|
|
80
109
|
request_options: ContextDev::RequestOptions::OrHash
|
|
81
110
|
).returns(T.attached_class)
|
|
@@ -83,12 +112,20 @@ module ContextDev
|
|
|
83
112
|
def self.new(
|
|
84
113
|
# Full URL to scrape (must include http:// or https:// protocol)
|
|
85
114
|
url:,
|
|
115
|
+
# CSS selectors to remove from the result. Applied after includeSelectors.
|
|
116
|
+
# Exclusion takes precedence: an element matching both is removed. Examples:
|
|
117
|
+
# "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
118
|
+
exclude_selectors: nil,
|
|
86
119
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
87
120
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
88
121
|
# is bypassed: the result is neither read from nor written to cache.
|
|
89
122
|
headers: nil,
|
|
90
123
|
# When true, iframes are rendered inline into the returned HTML.
|
|
91
124
|
include_frames: nil,
|
|
125
|
+
# CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
126
|
+
# kept and everything else is dropped. When omitted, the entire document is kept.
|
|
127
|
+
# Examples: "article.main", "#content", "[role=main]".
|
|
128
|
+
include_selectors: nil,
|
|
92
129
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
93
130
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
94
131
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -100,6 +137,9 @@ module ContextDev
|
|
|
100
137
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
101
138
|
# value is 300000ms (5 minutes).
|
|
102
139
|
timeout_ms: nil,
|
|
140
|
+
# When true, return only the page's main content in the HTML response, excluding
|
|
141
|
+
# headers, footers, sidebars, and navigation when detectable.
|
|
142
|
+
use_main_content_only: nil,
|
|
103
143
|
# Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
104
144
|
# 30000 (30 seconds).
|
|
105
145
|
wait_for_ms: nil,
|
|
@@ -111,11 +151,14 @@ module ContextDev
|
|
|
111
151
|
override.returns(
|
|
112
152
|
{
|
|
113
153
|
url: String,
|
|
154
|
+
exclude_selectors: T::Array[String],
|
|
114
155
|
headers: T::Hash[Symbol, String],
|
|
115
156
|
include_frames: T::Boolean,
|
|
157
|
+
include_selectors: T::Array[String],
|
|
116
158
|
max_age_ms: Integer,
|
|
117
159
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
118
160
|
timeout_ms: Integer,
|
|
161
|
+
use_main_content_only: T::Boolean,
|
|
119
162
|
wait_for_ms: Integer,
|
|
120
163
|
request_options: ContextDev::RequestOptions
|
|
121
164
|
}
|
|
@@ -16,6 +16,15 @@ module ContextDev
|
|
|
16
16
|
sig { returns(String) }
|
|
17
17
|
attr_accessor :url
|
|
18
18
|
|
|
19
|
+
# CSS selectors to remove before conversion to Markdown. Applied after
|
|
20
|
+
# includeSelectors. Exclusion takes precedence: an element matching both is
|
|
21
|
+
# removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
22
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
23
|
+
attr_reader :exclude_selectors
|
|
24
|
+
|
|
25
|
+
sig { params(exclude_selectors: T::Array[String]).void }
|
|
26
|
+
attr_writer :exclude_selectors
|
|
27
|
+
|
|
19
28
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
20
29
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
21
30
|
# is bypassed: the result is neither read from nor written to cache.
|
|
@@ -46,6 +55,15 @@ module ContextDev
|
|
|
46
55
|
sig { params(include_links: T::Boolean).void }
|
|
47
56
|
attr_writer :include_links
|
|
48
57
|
|
|
58
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
59
|
+
# descendants) are kept before conversion to Markdown. When omitted, the entire
|
|
60
|
+
# document is kept. Examples: "article.main", "#content", "[role=main]".
|
|
61
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
62
|
+
attr_reader :include_selectors
|
|
63
|
+
|
|
64
|
+
sig { params(include_selectors: T::Array[String]).void }
|
|
65
|
+
attr_writer :include_selectors
|
|
66
|
+
|
|
49
67
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
50
68
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
51
69
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -98,10 +116,12 @@ module ContextDev
|
|
|
98
116
|
sig do
|
|
99
117
|
params(
|
|
100
118
|
url: String,
|
|
119
|
+
exclude_selectors: T::Array[String],
|
|
101
120
|
headers: T::Hash[Symbol, String],
|
|
102
121
|
include_frames: T::Boolean,
|
|
103
122
|
include_images: T::Boolean,
|
|
104
123
|
include_links: T::Boolean,
|
|
124
|
+
include_selectors: T::Array[String],
|
|
105
125
|
max_age_ms: Integer,
|
|
106
126
|
pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
|
|
107
127
|
shorten_base64_images: T::Boolean,
|
|
@@ -115,6 +135,10 @@ module ContextDev
|
|
|
115
135
|
# Full URL to scrape into LLM usable Markdown (must include http:// or https://
|
|
116
136
|
# protocol)
|
|
117
137
|
url:,
|
|
138
|
+
# CSS selectors to remove before conversion to Markdown. Applied after
|
|
139
|
+
# includeSelectors. Exclusion takes precedence: an element matching both is
|
|
140
|
+
# removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
141
|
+
exclude_selectors: nil,
|
|
118
142
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
119
143
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
120
144
|
# is bypassed: the result is neither read from nor written to cache.
|
|
@@ -125,6 +149,10 @@ module ContextDev
|
|
|
125
149
|
include_images: nil,
|
|
126
150
|
# Preserve hyperlinks in Markdown output
|
|
127
151
|
include_links: nil,
|
|
152
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
153
|
+
# descendants) are kept before conversion to Markdown. When omitted, the entire
|
|
154
|
+
# document is kept. Examples: "article.main", "#content", "[role=main]".
|
|
155
|
+
include_selectors: nil,
|
|
128
156
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
129
157
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
130
158
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -152,10 +180,12 @@ module ContextDev
|
|
|
152
180
|
override.returns(
|
|
153
181
|
{
|
|
154
182
|
url: String,
|
|
183
|
+
exclude_selectors: T::Array[String],
|
|
155
184
|
headers: T::Hash[Symbol, String],
|
|
156
185
|
include_frames: T::Boolean,
|
|
157
186
|
include_images: T::Boolean,
|
|
158
187
|
include_links: T::Boolean,
|
|
188
|
+
include_selectors: T::Array[String],
|
|
159
189
|
max_age_ms: Integer,
|
|
160
190
|
pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
161
191
|
shorten_base64_images: T::Boolean,
|
|
@@ -47,7 +47,8 @@ module ContextDev
|
|
|
47
47
|
# younger than this many milliseconds. Defaults to 7 days (604800000 ms).
|
|
48
48
|
max_age_ms: nil,
|
|
49
49
|
pdf: nil,
|
|
50
|
-
# Soft time budget for the crawl in milliseconds.
|
|
50
|
+
# Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000
|
|
51
|
+
# (110s). Default: 80000 (80s).
|
|
51
52
|
stop_after_ms: nil,
|
|
52
53
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
53
54
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
@@ -251,10 +252,12 @@ module ContextDev
|
|
|
251
252
|
sig do
|
|
252
253
|
params(
|
|
253
254
|
url: String,
|
|
255
|
+
exclude_selectors: T::Array[String],
|
|
254
256
|
follow_subdomains: T::Boolean,
|
|
255
257
|
include_frames: T::Boolean,
|
|
256
258
|
include_images: T::Boolean,
|
|
257
259
|
include_links: T::Boolean,
|
|
260
|
+
include_selectors: T::Array[String],
|
|
258
261
|
max_age_ms: Integer,
|
|
259
262
|
max_depth: Integer,
|
|
260
263
|
max_pages: Integer,
|
|
@@ -271,6 +274,10 @@ module ContextDev
|
|
|
271
274
|
def web_crawl_md(
|
|
272
275
|
# The starting URL for the crawl (must include http:// or https:// protocol)
|
|
273
276
|
url:,
|
|
277
|
+
# CSS selectors to remove before each crawled page is converted to Markdown.
|
|
278
|
+
# Applied after includeSelectors. Exclusion takes precedence: an element matching
|
|
279
|
+
# both is removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
280
|
+
exclude_selectors: nil,
|
|
274
281
|
# When true, follow links on subdomains of the starting URL's domain (e.g.
|
|
275
282
|
# docs.example.com when starting from example.com). www and apex are always
|
|
276
283
|
# treated as equivalent.
|
|
@@ -282,6 +289,11 @@ module ContextDev
|
|
|
282
289
|
include_images: nil,
|
|
283
290
|
# Preserve hyperlinks in the Markdown output
|
|
284
291
|
include_links: nil,
|
|
292
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
293
|
+
# descendants) are kept before each crawled page is converted to Markdown. When
|
|
294
|
+
# omitted, the entire document is kept. Examples: "article.main", "#content",
|
|
295
|
+
# "[role=main]".
|
|
296
|
+
include_selectors: nil,
|
|
285
297
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
286
298
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
287
299
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -297,8 +309,8 @@ module ContextDev
|
|
|
297
309
|
shorten_base64_images: nil,
|
|
298
310
|
# Soft time budget for the crawl in milliseconds. After each scrape, the crawler
|
|
299
311
|
# checks the elapsed time and, if exceeded, returns the pages collected so far
|
|
300
|
-
# instead of continuing. Min: 10000 (10s). Max:
|
|
301
|
-
#
|
|
312
|
+
# instead of continuing. Min: 10000 (10s). Max: 110000 (110s). Default: 80000
|
|
313
|
+
# (80s).
|
|
302
314
|
stop_after_ms: nil,
|
|
303
315
|
# Optional timeout in milliseconds for the request. If the request takes longer
|
|
304
316
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
@@ -320,11 +332,14 @@ module ContextDev
|
|
|
320
332
|
sig do
|
|
321
333
|
params(
|
|
322
334
|
url: String,
|
|
335
|
+
exclude_selectors: T::Array[String],
|
|
323
336
|
headers: T::Hash[Symbol, String],
|
|
324
337
|
include_frames: T::Boolean,
|
|
338
|
+
include_selectors: T::Array[String],
|
|
325
339
|
max_age_ms: Integer,
|
|
326
340
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
|
|
327
341
|
timeout_ms: Integer,
|
|
342
|
+
use_main_content_only: T::Boolean,
|
|
328
343
|
wait_for_ms: Integer,
|
|
329
344
|
request_options: ContextDev::RequestOptions::OrHash
|
|
330
345
|
).returns(ContextDev::Models::WebWebScrapeHTMLResponse)
|
|
@@ -332,12 +347,20 @@ module ContextDev
|
|
|
332
347
|
def web_scrape_html(
|
|
333
348
|
# Full URL to scrape (must include http:// or https:// protocol)
|
|
334
349
|
url:,
|
|
350
|
+
# CSS selectors to remove from the result. Applied after includeSelectors.
|
|
351
|
+
# Exclusion takes precedence: an element matching both is removed. Examples:
|
|
352
|
+
# "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
353
|
+
exclude_selectors: nil,
|
|
335
354
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
336
355
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
337
356
|
# is bypassed: the result is neither read from nor written to cache.
|
|
338
357
|
headers: nil,
|
|
339
358
|
# When true, iframes are rendered inline into the returned HTML.
|
|
340
359
|
include_frames: nil,
|
|
360
|
+
# CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
361
|
+
# kept and everything else is dropped. When omitted, the entire document is kept.
|
|
362
|
+
# Examples: "article.main", "#content", "[role=main]".
|
|
363
|
+
include_selectors: nil,
|
|
341
364
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
342
365
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
343
366
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -349,6 +372,9 @@ module ContextDev
|
|
|
349
372
|
# than this value, it will be aborted with a 408 status code. Maximum allowed
|
|
350
373
|
# value is 300000ms (5 minutes).
|
|
351
374
|
timeout_ms: nil,
|
|
375
|
+
# When true, return only the page's main content in the HTML response, excluding
|
|
376
|
+
# headers, footers, sidebars, and navigation when detectable.
|
|
377
|
+
use_main_content_only: nil,
|
|
352
378
|
# Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
|
|
353
379
|
# 30000 (30 seconds).
|
|
354
380
|
wait_for_ms: nil,
|
|
@@ -399,10 +425,12 @@ module ContextDev
|
|
|
399
425
|
sig do
|
|
400
426
|
params(
|
|
401
427
|
url: String,
|
|
428
|
+
exclude_selectors: T::Array[String],
|
|
402
429
|
headers: T::Hash[Symbol, String],
|
|
403
430
|
include_frames: T::Boolean,
|
|
404
431
|
include_images: T::Boolean,
|
|
405
432
|
include_links: T::Boolean,
|
|
433
|
+
include_selectors: T::Array[String],
|
|
406
434
|
max_age_ms: Integer,
|
|
407
435
|
pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
|
|
408
436
|
shorten_base64_images: T::Boolean,
|
|
@@ -416,6 +444,10 @@ module ContextDev
|
|
|
416
444
|
# Full URL to scrape into LLM usable Markdown (must include http:// or https://
|
|
417
445
|
# protocol)
|
|
418
446
|
url:,
|
|
447
|
+
# CSS selectors to remove before conversion to Markdown. Applied after
|
|
448
|
+
# includeSelectors. Exclusion takes precedence: an element matching both is
|
|
449
|
+
# removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
450
|
+
exclude_selectors: nil,
|
|
419
451
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
420
452
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
421
453
|
# is bypassed: the result is neither read from nor written to cache.
|
|
@@ -426,6 +458,10 @@ module ContextDev
|
|
|
426
458
|
include_images: nil,
|
|
427
459
|
# Preserve hyperlinks in Markdown output
|
|
428
460
|
include_links: nil,
|
|
461
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
462
|
+
# descendants) are kept before conversion to Markdown. When omitted, the entire
|
|
463
|
+
# document is kept. Examples: "article.main", "#content", "[role=main]".
|
|
464
|
+
include_selectors: nil,
|
|
429
465
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
430
466
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
431
467
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -3,10 +3,12 @@ module ContextDev
|
|
|
3
3
|
type web_web_crawl_md_params =
|
|
4
4
|
{
|
|
5
5
|
url: String,
|
|
6
|
+
exclude_selectors: ::Array[String],
|
|
6
7
|
follow_subdomains: bool,
|
|
7
8
|
include_frames: bool,
|
|
8
9
|
include_images: bool,
|
|
9
10
|
include_links: bool,
|
|
11
|
+
include_selectors: ::Array[String],
|
|
10
12
|
max_age_ms: Integer,
|
|
11
13
|
max_depth: Integer,
|
|
12
14
|
max_pages: Integer,
|
|
@@ -26,6 +28,10 @@ module ContextDev
|
|
|
26
28
|
|
|
27
29
|
attr_accessor url: String
|
|
28
30
|
|
|
31
|
+
attr_reader exclude_selectors: ::Array[String]?
|
|
32
|
+
|
|
33
|
+
def exclude_selectors=: (::Array[String]) -> ::Array[String]
|
|
34
|
+
|
|
29
35
|
attr_reader follow_subdomains: bool?
|
|
30
36
|
|
|
31
37
|
def follow_subdomains=: (bool) -> bool
|
|
@@ -42,6 +48,10 @@ module ContextDev
|
|
|
42
48
|
|
|
43
49
|
def include_links=: (bool) -> bool
|
|
44
50
|
|
|
51
|
+
attr_reader include_selectors: ::Array[String]?
|
|
52
|
+
|
|
53
|
+
def include_selectors=: (::Array[String]) -> ::Array[String]
|
|
54
|
+
|
|
45
55
|
attr_reader max_age_ms: Integer?
|
|
46
56
|
|
|
47
57
|
def max_age_ms=: (Integer) -> Integer
|
|
@@ -86,10 +96,12 @@ module ContextDev
|
|
|
86
96
|
|
|
87
97
|
def initialize: (
|
|
88
98
|
url: String,
|
|
99
|
+
?exclude_selectors: ::Array[String],
|
|
89
100
|
?follow_subdomains: bool,
|
|
90
101
|
?include_frames: bool,
|
|
91
102
|
?include_images: bool,
|
|
92
103
|
?include_links: bool,
|
|
104
|
+
?include_selectors: ::Array[String],
|
|
93
105
|
?max_age_ms: Integer,
|
|
94
106
|
?max_depth: Integer,
|
|
95
107
|
?max_pages: Integer,
|
|
@@ -105,10 +117,12 @@ module ContextDev
|
|
|
105
117
|
|
|
106
118
|
def to_hash: -> {
|
|
107
119
|
url: String,
|
|
120
|
+
exclude_selectors: ::Array[String],
|
|
108
121
|
follow_subdomains: bool,
|
|
109
122
|
include_frames: bool,
|
|
110
123
|
include_images: bool,
|
|
111
124
|
include_links: bool,
|
|
125
|
+
include_selectors: ::Array[String],
|
|
112
126
|
max_age_ms: Integer,
|
|
113
127
|
max_depth: Integer,
|
|
114
128
|
max_pages: Integer,
|
|
@@ -3,11 +3,14 @@ module ContextDev
|
|
|
3
3
|
type web_web_scrape_html_params =
|
|
4
4
|
{
|
|
5
5
|
url: String,
|
|
6
|
+
exclude_selectors: ::Array[String],
|
|
6
7
|
headers: ::Hash[Symbol, String],
|
|
7
8
|
include_frames: bool,
|
|
9
|
+
include_selectors: ::Array[String],
|
|
8
10
|
max_age_ms: Integer,
|
|
9
11
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
10
12
|
timeout_ms: Integer,
|
|
13
|
+
use_main_content_only: bool,
|
|
11
14
|
wait_for_ms: Integer
|
|
12
15
|
}
|
|
13
16
|
& ContextDev::Internal::Type::request_parameters
|
|
@@ -18,6 +21,10 @@ module ContextDev
|
|
|
18
21
|
|
|
19
22
|
attr_accessor url: String
|
|
20
23
|
|
|
24
|
+
attr_reader exclude_selectors: ::Array[String]?
|
|
25
|
+
|
|
26
|
+
def exclude_selectors=: (::Array[String]) -> ::Array[String]
|
|
27
|
+
|
|
21
28
|
attr_reader headers: ::Hash[Symbol, String]?
|
|
22
29
|
|
|
23
30
|
def headers=: (::Hash[Symbol, String]) -> ::Hash[Symbol, String]
|
|
@@ -26,6 +33,10 @@ module ContextDev
|
|
|
26
33
|
|
|
27
34
|
def include_frames=: (bool) -> bool
|
|
28
35
|
|
|
36
|
+
attr_reader include_selectors: ::Array[String]?
|
|
37
|
+
|
|
38
|
+
def include_selectors=: (::Array[String]) -> ::Array[String]
|
|
39
|
+
|
|
29
40
|
attr_reader max_age_ms: Integer?
|
|
30
41
|
|
|
31
42
|
def max_age_ms=: (Integer) -> Integer
|
|
@@ -40,28 +51,38 @@ module ContextDev
|
|
|
40
51
|
|
|
41
52
|
def timeout_ms=: (Integer) -> Integer
|
|
42
53
|
|
|
54
|
+
attr_reader use_main_content_only: bool?
|
|
55
|
+
|
|
56
|
+
def use_main_content_only=: (bool) -> bool
|
|
57
|
+
|
|
43
58
|
attr_reader wait_for_ms: Integer?
|
|
44
59
|
|
|
45
60
|
def wait_for_ms=: (Integer) -> Integer
|
|
46
61
|
|
|
47
62
|
def initialize: (
|
|
48
63
|
url: String,
|
|
64
|
+
?exclude_selectors: ::Array[String],
|
|
49
65
|
?headers: ::Hash[Symbol, String],
|
|
50
66
|
?include_frames: bool,
|
|
67
|
+
?include_selectors: ::Array[String],
|
|
51
68
|
?max_age_ms: Integer,
|
|
52
69
|
?pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
53
70
|
?timeout_ms: Integer,
|
|
71
|
+
?use_main_content_only: bool,
|
|
54
72
|
?wait_for_ms: Integer,
|
|
55
73
|
?request_options: ContextDev::request_opts
|
|
56
74
|
) -> void
|
|
57
75
|
|
|
58
76
|
def to_hash: -> {
|
|
59
77
|
url: String,
|
|
78
|
+
exclude_selectors: ::Array[String],
|
|
60
79
|
headers: ::Hash[Symbol, String],
|
|
61
80
|
include_frames: bool,
|
|
81
|
+
include_selectors: ::Array[String],
|
|
62
82
|
max_age_ms: Integer,
|
|
63
83
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
64
84
|
timeout_ms: Integer,
|
|
85
|
+
use_main_content_only: bool,
|
|
65
86
|
wait_for_ms: Integer,
|
|
66
87
|
request_options: ContextDev::RequestOptions
|
|
67
88
|
}
|
|
@@ -3,10 +3,12 @@ module ContextDev
|
|
|
3
3
|
type web_web_scrape_md_params =
|
|
4
4
|
{
|
|
5
5
|
url: String,
|
|
6
|
+
exclude_selectors: ::Array[String],
|
|
6
7
|
headers: ::Hash[Symbol, String],
|
|
7
8
|
include_frames: bool,
|
|
8
9
|
include_images: bool,
|
|
9
10
|
include_links: bool,
|
|
11
|
+
include_selectors: ::Array[String],
|
|
10
12
|
max_age_ms: Integer,
|
|
11
13
|
pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
12
14
|
:shorten_base64_images => bool,
|
|
@@ -22,6 +24,10 @@ module ContextDev
|
|
|
22
24
|
|
|
23
25
|
attr_accessor url: String
|
|
24
26
|
|
|
27
|
+
attr_reader exclude_selectors: ::Array[String]?
|
|
28
|
+
|
|
29
|
+
def exclude_selectors=: (::Array[String]) -> ::Array[String]
|
|
30
|
+
|
|
25
31
|
attr_reader headers: ::Hash[Symbol, String]?
|
|
26
32
|
|
|
27
33
|
def headers=: (::Hash[Symbol, String]) -> ::Hash[Symbol, String]
|
|
@@ -38,6 +44,10 @@ module ContextDev
|
|
|
38
44
|
|
|
39
45
|
def include_links=: (bool) -> bool
|
|
40
46
|
|
|
47
|
+
attr_reader include_selectors: ::Array[String]?
|
|
48
|
+
|
|
49
|
+
def include_selectors=: (::Array[String]) -> ::Array[String]
|
|
50
|
+
|
|
41
51
|
attr_reader max_age_ms: Integer?
|
|
42
52
|
|
|
43
53
|
def max_age_ms=: (Integer) -> Integer
|
|
@@ -66,10 +76,12 @@ module ContextDev
|
|
|
66
76
|
|
|
67
77
|
def initialize: (
|
|
68
78
|
url: String,
|
|
79
|
+
?exclude_selectors: ::Array[String],
|
|
69
80
|
?headers: ::Hash[Symbol, String],
|
|
70
81
|
?include_frames: bool,
|
|
71
82
|
?include_images: bool,
|
|
72
83
|
?include_links: bool,
|
|
84
|
+
?include_selectors: ::Array[String],
|
|
73
85
|
?max_age_ms: Integer,
|
|
74
86
|
?pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
75
87
|
?shorten_base64_images: bool,
|
|
@@ -81,10 +93,12 @@ module ContextDev
|
|
|
81
93
|
|
|
82
94
|
def to_hash: -> {
|
|
83
95
|
url: String,
|
|
96
|
+
exclude_selectors: ::Array[String],
|
|
84
97
|
headers: ::Hash[Symbol, String],
|
|
85
98
|
include_frames: bool,
|
|
86
99
|
include_images: bool,
|
|
87
100
|
include_links: bool,
|
|
101
|
+
include_selectors: ::Array[String],
|
|
88
102
|
max_age_ms: Integer,
|
|
89
103
|
pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
90
104
|
:shorten_base64_images => bool,
|
|
@@ -65,10 +65,12 @@ module ContextDev
|
|
|
65
65
|
|
|
66
66
|
def web_crawl_md: (
|
|
67
67
|
url: String,
|
|
68
|
+
?exclude_selectors: ::Array[String],
|
|
68
69
|
?follow_subdomains: bool,
|
|
69
70
|
?include_frames: bool,
|
|
70
71
|
?include_images: bool,
|
|
71
72
|
?include_links: bool,
|
|
73
|
+
?include_selectors: ::Array[String],
|
|
72
74
|
?max_age_ms: Integer,
|
|
73
75
|
?max_depth: Integer,
|
|
74
76
|
?max_pages: Integer,
|
|
@@ -84,11 +86,14 @@ module ContextDev
|
|
|
84
86
|
|
|
85
87
|
def web_scrape_html: (
|
|
86
88
|
url: String,
|
|
89
|
+
?exclude_selectors: ::Array[String],
|
|
87
90
|
?headers: ::Hash[Symbol, String],
|
|
88
91
|
?include_frames: bool,
|
|
92
|
+
?include_selectors: ::Array[String],
|
|
89
93
|
?max_age_ms: Integer,
|
|
90
94
|
?pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
91
95
|
?timeout_ms: Integer,
|
|
96
|
+
?use_main_content_only: bool,
|
|
92
97
|
?wait_for_ms: Integer,
|
|
93
98
|
?request_options: ContextDev::request_opts
|
|
94
99
|
) -> ContextDev::Models::WebWebScrapeHTMLResponse
|
|
@@ -105,10 +110,12 @@ module ContextDev
|
|
|
105
110
|
|
|
106
111
|
def web_scrape_md: (
|
|
107
112
|
url: String,
|
|
113
|
+
?exclude_selectors: ::Array[String],
|
|
108
114
|
?headers: ::Hash[Symbol, String],
|
|
109
115
|
?include_frames: bool,
|
|
110
116
|
?include_images: bool,
|
|
111
117
|
?include_links: bool,
|
|
118
|
+
?include_selectors: ::Array[String],
|
|
112
119
|
?max_age_ms: Integer,
|
|
113
120
|
?pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
114
121
|
?shorten_base64_images: bool,
|
metadata
CHANGED
|
@@ -1,14 +1,14 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: context.dev
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 1.
|
|
4
|
+
version: 1.31.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Context Dev
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-06-
|
|
11
|
+
date: 2026-06-08 00:00:00.000000000 Z
|
|
12
12
|
dependencies:
|
|
13
13
|
- !ruby/object:Gem::Dependency
|
|
14
14
|
name: cgi
|