context.dev 1.29.0 → 1.30.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +8 -0
- data/README.md +1 -1
- data/lib/context_dev/models/web_web_crawl_md_params.rb +22 -1
- data/lib/context_dev/models/web_web_scrape_html_params.rb +21 -1
- data/lib/context_dev/models/web_web_scrape_md_params.rb +21 -1
- data/lib/context_dev/resources/web.rb +19 -3
- data/lib/context_dev/version.rb +1 -1
- data/rbi/context_dev/models/web_web_crawl_md_params.rbi +32 -0
- data/rbi/context_dev/models/web_web_scrape_html_params.rbi +30 -0
- data/rbi/context_dev/models/web_web_scrape_md_params.rbi +30 -0
- data/rbi/context_dev/resources/web.rbi +31 -0
- data/sig/context_dev/models/web_web_crawl_md_params.rbs +14 -0
- data/sig/context_dev/models/web_web_scrape_html_params.rbs +14 -0
- data/sig/context_dev/models/web_web_scrape_md_params.rbs +14 -0
- data/sig/context_dev/resources/web.rbs +6 -0
- metadata +1 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 074be977167979846299e7e3c6eb549b3ae6ac29e1c928aec4d692111250c305
|
|
4
|
+
data.tar.gz: c6837a38aea2ae93f3faf15d835e845d5333332fc188bf92ccfa999134d33cfc
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 867c490615c21f264743296a01d0038280546f75eac1c7bf8355c6c32bf6fb2cf24312b66336d02298bb284db9015dd98cb9e3d1f11a588c4fd72989a8123549
|
|
7
|
+
data.tar.gz: 228294367be4138af290bfe6d0775f9266c019b4c739121f6464ad19a8998ffcb34b3b6dc377edcac4acb7a17368f211066c5499f49bd6b416c9399f15963906
|
data/CHANGELOG.md
CHANGED
|
@@ -1,5 +1,13 @@
|
|
|
1
1
|
# Changelog
|
|
2
2
|
|
|
3
|
+
## 1.30.0 (2026-06-07)
|
|
4
|
+
|
|
5
|
+
Full Changelog: [v1.29.0...v1.30.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.29.0...v1.30.0)
|
|
6
|
+
|
|
7
|
+
### Features
|
|
8
|
+
|
|
9
|
+
* **api:** api update ([c897bbf](https://github.com/context-dot-dev/context-ruby-sdk/commit/c897bbf03235960c8036bd808c5239b935d94bf3))
|
|
10
|
+
|
|
3
11
|
## 1.29.0 (2026-06-07)
|
|
4
12
|
|
|
5
13
|
Full Changelog: [v1.28.0...v1.29.0](https://github.com/context-dot-dev/context-ruby-sdk/compare/v1.28.0...v1.29.0)
|
data/README.md
CHANGED
|
@@ -13,6 +13,14 @@ module ContextDev
|
|
|
13
13
|
# @return [String]
|
|
14
14
|
required :url, String
|
|
15
15
|
|
|
16
|
+
# @!attribute exclude_selectors
|
|
17
|
+
# CSS selectors to remove before each crawled page is converted to Markdown.
|
|
18
|
+
# Applied after includeSelectors. Exclusion takes precedence: an element matching
|
|
19
|
+
# both is removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
20
|
+
#
|
|
21
|
+
# @return [Array<String>, nil]
|
|
22
|
+
optional :exclude_selectors, ContextDev::Internal::Type::ArrayOf[String], api_name: :excludeSelectors
|
|
23
|
+
|
|
16
24
|
# @!attribute follow_subdomains
|
|
17
25
|
# When true, follow links on subdomains of the starting URL's domain (e.g.
|
|
18
26
|
# docs.example.com when starting from example.com). www and apex are always
|
|
@@ -40,6 +48,15 @@ module ContextDev
|
|
|
40
48
|
# @return [Boolean, nil]
|
|
41
49
|
optional :include_links, ContextDev::Internal::Type::Boolean, api_name: :includeLinks
|
|
42
50
|
|
|
51
|
+
# @!attribute include_selectors
|
|
52
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
53
|
+
# descendants) are kept before each crawled page is converted to Markdown. When
|
|
54
|
+
# omitted, the entire document is kept. Examples: "article.main", "#content",
|
|
55
|
+
# "[role=main]".
|
|
56
|
+
#
|
|
57
|
+
# @return [Array<String>, nil]
|
|
58
|
+
optional :include_selectors, ContextDev::Internal::Type::ArrayOf[String], api_name: :includeSelectors
|
|
59
|
+
|
|
43
60
|
# @!attribute max_age_ms
|
|
44
61
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
45
62
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
@@ -110,12 +127,14 @@ module ContextDev
|
|
|
110
127
|
# @return [Integer, nil]
|
|
111
128
|
optional :wait_for_ms, Integer, api_name: :waitForMs
|
|
112
129
|
|
|
113
|
-
# @!method initialize(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
130
|
+
# @!method initialize(url:, exclude_selectors: nil, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
114
131
|
# Some parameter documentations has been truncated, see
|
|
115
132
|
# {ContextDev::Models::WebWebCrawlMdParams} for more details.
|
|
116
133
|
#
|
|
117
134
|
# @param url [String] The starting URL for the crawl (must include http:// or https:// protocol)
|
|
118
135
|
#
|
|
136
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove before each crawled page is converted to Markdown. Appli
|
|
137
|
+
#
|
|
119
138
|
# @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain (e.g. docs.ex
|
|
120
139
|
#
|
|
121
140
|
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown for each crawled pag
|
|
@@ -124,6 +143,8 @@ module ContextDev
|
|
|
124
143
|
#
|
|
125
144
|
# @param include_links [Boolean] Preserve hyperlinks in the Markdown output
|
|
126
145
|
#
|
|
146
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
147
|
+
#
|
|
127
148
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
128
149
|
#
|
|
129
150
|
# @param max_depth [Integer] Maximum link depth from the starting URL (0 = only the starting page)
|
|
@@ -13,6 +13,14 @@ module ContextDev
|
|
|
13
13
|
# @return [String]
|
|
14
14
|
required :url, String
|
|
15
15
|
|
|
16
|
+
# @!attribute exclude_selectors
|
|
17
|
+
# CSS selectors to remove from the result. Applied after includeSelectors.
|
|
18
|
+
# Exclusion takes precedence: an element matching both is removed. Examples:
|
|
19
|
+
# "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
20
|
+
#
|
|
21
|
+
# @return [Array<String>, nil]
|
|
22
|
+
optional :exclude_selectors, ContextDev::Internal::Type::ArrayOf[String]
|
|
23
|
+
|
|
16
24
|
# @!attribute headers
|
|
17
25
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
18
26
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
@@ -27,6 +35,14 @@ module ContextDev
|
|
|
27
35
|
# @return [Boolean, nil]
|
|
28
36
|
optional :include_frames, ContextDev::Internal::Type::Boolean
|
|
29
37
|
|
|
38
|
+
# @!attribute include_selectors
|
|
39
|
+
# CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
40
|
+
# kept and everything else is dropped. When omitted, the entire document is kept.
|
|
41
|
+
# Examples: "article.main", "#content", "[role=main]".
|
|
42
|
+
#
|
|
43
|
+
# @return [Array<String>, nil]
|
|
44
|
+
optional :include_selectors, ContextDev::Internal::Type::ArrayOf[String]
|
|
45
|
+
|
|
30
46
|
# @!attribute max_age_ms
|
|
31
47
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
32
48
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
@@ -57,16 +73,20 @@ module ContextDev
|
|
|
57
73
|
# @return [Integer, nil]
|
|
58
74
|
optional :wait_for_ms, Integer
|
|
59
75
|
|
|
60
|
-
# @!method initialize(url:, headers: nil, include_frames: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
76
|
+
# @!method initialize(url:, exclude_selectors: nil, headers: nil, include_frames: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
61
77
|
# Some parameter documentations has been truncated, see
|
|
62
78
|
# {ContextDev::Models::WebWebScrapeHTMLParams} for more details.
|
|
63
79
|
#
|
|
64
80
|
# @param url [String] Full URL to scrape (must include http:// or https:// protocol)
|
|
65
81
|
#
|
|
82
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove from the result. Applied after includeSelectors. Exclusi
|
|
83
|
+
#
|
|
66
84
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
67
85
|
#
|
|
68
86
|
# @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
|
|
69
87
|
#
|
|
88
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
89
|
+
#
|
|
70
90
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
71
91
|
#
|
|
72
92
|
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
@@ -14,6 +14,14 @@ module ContextDev
|
|
|
14
14
|
# @return [String]
|
|
15
15
|
required :url, String
|
|
16
16
|
|
|
17
|
+
# @!attribute exclude_selectors
|
|
18
|
+
# CSS selectors to remove before conversion to Markdown. Applied after
|
|
19
|
+
# includeSelectors. Exclusion takes precedence: an element matching both is
|
|
20
|
+
# removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
21
|
+
#
|
|
22
|
+
# @return [Array<String>, nil]
|
|
23
|
+
optional :exclude_selectors, ContextDev::Internal::Type::ArrayOf[String]
|
|
24
|
+
|
|
17
25
|
# @!attribute headers
|
|
18
26
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
19
27
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
@@ -40,6 +48,14 @@ module ContextDev
|
|
|
40
48
|
# @return [Boolean, nil]
|
|
41
49
|
optional :include_links, ContextDev::Internal::Type::Boolean
|
|
42
50
|
|
|
51
|
+
# @!attribute include_selectors
|
|
52
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
53
|
+
# descendants) are kept before conversion to Markdown. When omitted, the entire
|
|
54
|
+
# document is kept. Examples: "article.main", "#content", "[role=main]".
|
|
55
|
+
#
|
|
56
|
+
# @return [Array<String>, nil]
|
|
57
|
+
optional :include_selectors, ContextDev::Internal::Type::ArrayOf[String]
|
|
58
|
+
|
|
43
59
|
# @!attribute max_age_ms
|
|
44
60
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
45
61
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
@@ -83,12 +99,14 @@ module ContextDev
|
|
|
83
99
|
# @return [Integer, nil]
|
|
84
100
|
optional :wait_for_ms, Integer
|
|
85
101
|
|
|
86
|
-
# @!method initialize(url:, headers: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
102
|
+
# @!method initialize(url:, exclude_selectors: nil, headers: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
87
103
|
# Some parameter documentations has been truncated, see
|
|
88
104
|
# {ContextDev::Models::WebWebScrapeMdParams} for more details.
|
|
89
105
|
#
|
|
90
106
|
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
91
107
|
#
|
|
108
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove before conversion to Markdown. Applied after includeSele
|
|
109
|
+
#
|
|
92
110
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
93
111
|
#
|
|
94
112
|
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
|
|
@@ -97,6 +115,8 @@ module ContextDev
|
|
|
97
115
|
#
|
|
98
116
|
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
99
117
|
#
|
|
118
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
119
|
+
#
|
|
100
120
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
101
121
|
#
|
|
102
122
|
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
@@ -246,10 +246,12 @@ module ContextDev
|
|
|
246
246
|
# Performs a crawl starting from a given URL, extracts page content as Markdown,
|
|
247
247
|
# and returns results for all crawled pages.
|
|
248
248
|
#
|
|
249
|
-
# @overload web_crawl_md(url:, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
249
|
+
# @overload web_crawl_md(url:, exclude_selectors: nil, follow_subdomains: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, shorten_base64_images: nil, stop_after_ms: nil, timeout_ms: nil, url_regex: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
250
250
|
#
|
|
251
251
|
# @param url [String] The starting URL for the crawl (must include http:// or https:// protocol)
|
|
252
252
|
#
|
|
253
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove before each crawled page is converted to Markdown. Appli
|
|
254
|
+
#
|
|
253
255
|
# @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain (e.g. docs.ex
|
|
254
256
|
#
|
|
255
257
|
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown for each crawled pag
|
|
@@ -258,6 +260,8 @@ module ContextDev
|
|
|
258
260
|
#
|
|
259
261
|
# @param include_links [Boolean] Preserve hyperlinks in the Markdown output
|
|
260
262
|
#
|
|
263
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
264
|
+
#
|
|
261
265
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
262
266
|
#
|
|
263
267
|
# @param max_depth [Integer] Maximum link depth from the starting URL (0 = only the starting page)
|
|
@@ -299,14 +303,18 @@ module ContextDev
|
|
|
299
303
|
#
|
|
300
304
|
# Scrapes the given URL and returns the raw HTML content of the page.
|
|
301
305
|
#
|
|
302
|
-
# @overload web_scrape_html(url:, headers: nil, include_frames: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
306
|
+
# @overload web_scrape_html(url:, exclude_selectors: nil, headers: nil, include_frames: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, timeout_ms: nil, wait_for_ms: nil, request_options: {})
|
|
303
307
|
#
|
|
304
308
|
# @param url [String] Full URL to scrape (must include http:// or https:// protocol)
|
|
305
309
|
#
|
|
310
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove from the result. Applied after includeSelectors. Exclusi
|
|
311
|
+
#
|
|
306
312
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
307
313
|
#
|
|
308
314
|
# @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
|
|
309
315
|
#
|
|
316
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
317
|
+
#
|
|
310
318
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
311
319
|
#
|
|
312
320
|
# @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
@@ -327,7 +335,9 @@ module ContextDev
|
|
|
327
335
|
method: :get,
|
|
328
336
|
path: "web/scrape/html",
|
|
329
337
|
query: query.transform_keys(
|
|
338
|
+
exclude_selectors: "excludeSelectors",
|
|
330
339
|
include_frames: "includeFrames",
|
|
340
|
+
include_selectors: "includeSelectors",
|
|
331
341
|
max_age_ms: "maxAgeMs",
|
|
332
342
|
timeout_ms: "timeoutMS",
|
|
333
343
|
wait_for_ms: "waitForMs"
|
|
@@ -385,10 +395,12 @@ module ContextDev
|
|
|
385
395
|
#
|
|
386
396
|
# Scrapes the given URL into LLM usable Markdown.
|
|
387
397
|
#
|
|
388
|
-
# @overload web_scrape_md(url:, headers: nil, include_frames: nil, include_images: nil, include_links: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
398
|
+
# @overload web_scrape_md(url:, exclude_selectors: nil, headers: nil, include_frames: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, shorten_base64_images: nil, timeout_ms: nil, use_main_content_only: nil, wait_for_ms: nil, request_options: {})
|
|
389
399
|
#
|
|
390
400
|
# @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
|
|
391
401
|
#
|
|
402
|
+
# @param exclude_selectors [Array<String>] CSS selectors to remove before conversion to Markdown. Applied after includeSele
|
|
403
|
+
#
|
|
392
404
|
# @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
|
|
393
405
|
#
|
|
394
406
|
# @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
|
|
@@ -397,6 +409,8 @@ module ContextDev
|
|
|
397
409
|
#
|
|
398
410
|
# @param include_links [Boolean] Preserve hyperlinks in Markdown output
|
|
399
411
|
#
|
|
412
|
+
# @param include_selectors [Array<String>] CSS selectors. When provided, only matching HTML subtrees (and their descendants
|
|
413
|
+
#
|
|
400
414
|
# @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
|
|
401
415
|
#
|
|
402
416
|
# @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
|
|
@@ -421,9 +435,11 @@ module ContextDev
|
|
|
421
435
|
method: :get,
|
|
422
436
|
path: "web/scrape/markdown",
|
|
423
437
|
query: query.transform_keys(
|
|
438
|
+
exclude_selectors: "excludeSelectors",
|
|
424
439
|
include_frames: "includeFrames",
|
|
425
440
|
include_images: "includeImages",
|
|
426
441
|
include_links: "includeLinks",
|
|
442
|
+
include_selectors: "includeSelectors",
|
|
427
443
|
max_age_ms: "maxAgeMs",
|
|
428
444
|
shorten_base64_images: "shortenBase64Images",
|
|
429
445
|
timeout_ms: "timeoutMS",
|
data/lib/context_dev/version.rb
CHANGED
|
@@ -15,6 +15,15 @@ module ContextDev
|
|
|
15
15
|
sig { returns(String) }
|
|
16
16
|
attr_accessor :url
|
|
17
17
|
|
|
18
|
+
# CSS selectors to remove before each crawled page is converted to Markdown.
|
|
19
|
+
# Applied after includeSelectors. Exclusion takes precedence: an element matching
|
|
20
|
+
# both is removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
21
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
22
|
+
attr_reader :exclude_selectors
|
|
23
|
+
|
|
24
|
+
sig { params(exclude_selectors: T::Array[String]).void }
|
|
25
|
+
attr_writer :exclude_selectors
|
|
26
|
+
|
|
18
27
|
# When true, follow links on subdomains of the starting URL's domain (e.g.
|
|
19
28
|
# docs.example.com when starting from example.com). www and apex are always
|
|
20
29
|
# treated as equivalent.
|
|
@@ -46,6 +55,16 @@ module ContextDev
|
|
|
46
55
|
sig { params(include_links: T::Boolean).void }
|
|
47
56
|
attr_writer :include_links
|
|
48
57
|
|
|
58
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
59
|
+
# descendants) are kept before each crawled page is converted to Markdown. When
|
|
60
|
+
# omitted, the entire document is kept. Examples: "article.main", "#content",
|
|
61
|
+
# "[role=main]".
|
|
62
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
63
|
+
attr_reader :include_selectors
|
|
64
|
+
|
|
65
|
+
sig { params(include_selectors: T::Array[String]).void }
|
|
66
|
+
attr_writer :include_selectors
|
|
67
|
+
|
|
49
68
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
50
69
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
51
70
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -129,10 +148,12 @@ module ContextDev
|
|
|
129
148
|
sig do
|
|
130
149
|
params(
|
|
131
150
|
url: String,
|
|
151
|
+
exclude_selectors: T::Array[String],
|
|
132
152
|
follow_subdomains: T::Boolean,
|
|
133
153
|
include_frames: T::Boolean,
|
|
134
154
|
include_images: T::Boolean,
|
|
135
155
|
include_links: T::Boolean,
|
|
156
|
+
include_selectors: T::Array[String],
|
|
136
157
|
max_age_ms: Integer,
|
|
137
158
|
max_depth: Integer,
|
|
138
159
|
max_pages: Integer,
|
|
@@ -149,6 +170,10 @@ module ContextDev
|
|
|
149
170
|
def self.new(
|
|
150
171
|
# The starting URL for the crawl (must include http:// or https:// protocol)
|
|
151
172
|
url:,
|
|
173
|
+
# CSS selectors to remove before each crawled page is converted to Markdown.
|
|
174
|
+
# Applied after includeSelectors. Exclusion takes precedence: an element matching
|
|
175
|
+
# both is removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
176
|
+
exclude_selectors: nil,
|
|
152
177
|
# When true, follow links on subdomains of the starting URL's domain (e.g.
|
|
153
178
|
# docs.example.com when starting from example.com). www and apex are always
|
|
154
179
|
# treated as equivalent.
|
|
@@ -160,6 +185,11 @@ module ContextDev
|
|
|
160
185
|
include_images: nil,
|
|
161
186
|
# Preserve hyperlinks in the Markdown output
|
|
162
187
|
include_links: nil,
|
|
188
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
189
|
+
# descendants) are kept before each crawled page is converted to Markdown. When
|
|
190
|
+
# omitted, the entire document is kept. Examples: "article.main", "#content",
|
|
191
|
+
# "[role=main]".
|
|
192
|
+
include_selectors: nil,
|
|
163
193
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
164
194
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
165
195
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -198,10 +228,12 @@ module ContextDev
|
|
|
198
228
|
override.returns(
|
|
199
229
|
{
|
|
200
230
|
url: String,
|
|
231
|
+
exclude_selectors: T::Array[String],
|
|
201
232
|
follow_subdomains: T::Boolean,
|
|
202
233
|
include_frames: T::Boolean,
|
|
203
234
|
include_images: T::Boolean,
|
|
204
235
|
include_links: T::Boolean,
|
|
236
|
+
include_selectors: T::Array[String],
|
|
205
237
|
max_age_ms: Integer,
|
|
206
238
|
max_depth: Integer,
|
|
207
239
|
max_pages: Integer,
|
|
@@ -18,6 +18,15 @@ module ContextDev
|
|
|
18
18
|
sig { returns(String) }
|
|
19
19
|
attr_accessor :url
|
|
20
20
|
|
|
21
|
+
# CSS selectors to remove from the result. Applied after includeSelectors.
|
|
22
|
+
# Exclusion takes precedence: an element matching both is removed. Examples:
|
|
23
|
+
# "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
24
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
25
|
+
attr_reader :exclude_selectors
|
|
26
|
+
|
|
27
|
+
sig { params(exclude_selectors: T::Array[String]).void }
|
|
28
|
+
attr_writer :exclude_selectors
|
|
29
|
+
|
|
21
30
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
22
31
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
23
32
|
# is bypassed: the result is neither read from nor written to cache.
|
|
@@ -34,6 +43,15 @@ module ContextDev
|
|
|
34
43
|
sig { params(include_frames: T::Boolean).void }
|
|
35
44
|
attr_writer :include_frames
|
|
36
45
|
|
|
46
|
+
# CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
47
|
+
# kept and everything else is dropped. When omitted, the entire document is kept.
|
|
48
|
+
# Examples: "article.main", "#content", "[role=main]".
|
|
49
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
50
|
+
attr_reader :include_selectors
|
|
51
|
+
|
|
52
|
+
sig { params(include_selectors: T::Array[String]).void }
|
|
53
|
+
attr_writer :include_selectors
|
|
54
|
+
|
|
37
55
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
38
56
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
39
57
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -71,8 +89,10 @@ module ContextDev
|
|
|
71
89
|
sig do
|
|
72
90
|
params(
|
|
73
91
|
url: String,
|
|
92
|
+
exclude_selectors: T::Array[String],
|
|
74
93
|
headers: T::Hash[Symbol, String],
|
|
75
94
|
include_frames: T::Boolean,
|
|
95
|
+
include_selectors: T::Array[String],
|
|
76
96
|
max_age_ms: Integer,
|
|
77
97
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
|
|
78
98
|
timeout_ms: Integer,
|
|
@@ -83,12 +103,20 @@ module ContextDev
|
|
|
83
103
|
def self.new(
|
|
84
104
|
# Full URL to scrape (must include http:// or https:// protocol)
|
|
85
105
|
url:,
|
|
106
|
+
# CSS selectors to remove from the result. Applied after includeSelectors.
|
|
107
|
+
# Exclusion takes precedence: an element matching both is removed. Examples:
|
|
108
|
+
# "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
109
|
+
exclude_selectors: nil,
|
|
86
110
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
87
111
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
88
112
|
# is bypassed: the result is neither read from nor written to cache.
|
|
89
113
|
headers: nil,
|
|
90
114
|
# When true, iframes are rendered inline into the returned HTML.
|
|
91
115
|
include_frames: nil,
|
|
116
|
+
# CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
117
|
+
# kept and everything else is dropped. When omitted, the entire document is kept.
|
|
118
|
+
# Examples: "article.main", "#content", "[role=main]".
|
|
119
|
+
include_selectors: nil,
|
|
92
120
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
93
121
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
94
122
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -111,8 +139,10 @@ module ContextDev
|
|
|
111
139
|
override.returns(
|
|
112
140
|
{
|
|
113
141
|
url: String,
|
|
142
|
+
exclude_selectors: T::Array[String],
|
|
114
143
|
headers: T::Hash[Symbol, String],
|
|
115
144
|
include_frames: T::Boolean,
|
|
145
|
+
include_selectors: T::Array[String],
|
|
116
146
|
max_age_ms: Integer,
|
|
117
147
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
118
148
|
timeout_ms: Integer,
|
|
@@ -16,6 +16,15 @@ module ContextDev
|
|
|
16
16
|
sig { returns(String) }
|
|
17
17
|
attr_accessor :url
|
|
18
18
|
|
|
19
|
+
# CSS selectors to remove before conversion to Markdown. Applied after
|
|
20
|
+
# includeSelectors. Exclusion takes precedence: an element matching both is
|
|
21
|
+
# removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
22
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
23
|
+
attr_reader :exclude_selectors
|
|
24
|
+
|
|
25
|
+
sig { params(exclude_selectors: T::Array[String]).void }
|
|
26
|
+
attr_writer :exclude_selectors
|
|
27
|
+
|
|
19
28
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
20
29
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
21
30
|
# is bypassed: the result is neither read from nor written to cache.
|
|
@@ -46,6 +55,15 @@ module ContextDev
|
|
|
46
55
|
sig { params(include_links: T::Boolean).void }
|
|
47
56
|
attr_writer :include_links
|
|
48
57
|
|
|
58
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
59
|
+
# descendants) are kept before conversion to Markdown. When omitted, the entire
|
|
60
|
+
# document is kept. Examples: "article.main", "#content", "[role=main]".
|
|
61
|
+
sig { returns(T.nilable(T::Array[String])) }
|
|
62
|
+
attr_reader :include_selectors
|
|
63
|
+
|
|
64
|
+
sig { params(include_selectors: T::Array[String]).void }
|
|
65
|
+
attr_writer :include_selectors
|
|
66
|
+
|
|
49
67
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
50
68
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
51
69
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -98,10 +116,12 @@ module ContextDev
|
|
|
98
116
|
sig do
|
|
99
117
|
params(
|
|
100
118
|
url: String,
|
|
119
|
+
exclude_selectors: T::Array[String],
|
|
101
120
|
headers: T::Hash[Symbol, String],
|
|
102
121
|
include_frames: T::Boolean,
|
|
103
122
|
include_images: T::Boolean,
|
|
104
123
|
include_links: T::Boolean,
|
|
124
|
+
include_selectors: T::Array[String],
|
|
105
125
|
max_age_ms: Integer,
|
|
106
126
|
pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
|
|
107
127
|
shorten_base64_images: T::Boolean,
|
|
@@ -115,6 +135,10 @@ module ContextDev
|
|
|
115
135
|
# Full URL to scrape into LLM usable Markdown (must include http:// or https://
|
|
116
136
|
# protocol)
|
|
117
137
|
url:,
|
|
138
|
+
# CSS selectors to remove before conversion to Markdown. Applied after
|
|
139
|
+
# includeSelectors. Exclusion takes precedence: an element matching both is
|
|
140
|
+
# removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
141
|
+
exclude_selectors: nil,
|
|
118
142
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
119
143
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
120
144
|
# is bypassed: the result is neither read from nor written to cache.
|
|
@@ -125,6 +149,10 @@ module ContextDev
|
|
|
125
149
|
include_images: nil,
|
|
126
150
|
# Preserve hyperlinks in Markdown output
|
|
127
151
|
include_links: nil,
|
|
152
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
153
|
+
# descendants) are kept before conversion to Markdown. When omitted, the entire
|
|
154
|
+
# document is kept. Examples: "article.main", "#content", "[role=main]".
|
|
155
|
+
include_selectors: nil,
|
|
128
156
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
129
157
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
130
158
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -152,10 +180,12 @@ module ContextDev
|
|
|
152
180
|
override.returns(
|
|
153
181
|
{
|
|
154
182
|
url: String,
|
|
183
|
+
exclude_selectors: T::Array[String],
|
|
155
184
|
headers: T::Hash[Symbol, String],
|
|
156
185
|
include_frames: T::Boolean,
|
|
157
186
|
include_images: T::Boolean,
|
|
158
187
|
include_links: T::Boolean,
|
|
188
|
+
include_selectors: T::Array[String],
|
|
159
189
|
max_age_ms: Integer,
|
|
160
190
|
pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
161
191
|
shorten_base64_images: T::Boolean,
|
|
@@ -251,10 +251,12 @@ module ContextDev
|
|
|
251
251
|
sig do
|
|
252
252
|
params(
|
|
253
253
|
url: String,
|
|
254
|
+
exclude_selectors: T::Array[String],
|
|
254
255
|
follow_subdomains: T::Boolean,
|
|
255
256
|
include_frames: T::Boolean,
|
|
256
257
|
include_images: T::Boolean,
|
|
257
258
|
include_links: T::Boolean,
|
|
259
|
+
include_selectors: T::Array[String],
|
|
258
260
|
max_age_ms: Integer,
|
|
259
261
|
max_depth: Integer,
|
|
260
262
|
max_pages: Integer,
|
|
@@ -271,6 +273,10 @@ module ContextDev
|
|
|
271
273
|
def web_crawl_md(
|
|
272
274
|
# The starting URL for the crawl (must include http:// or https:// protocol)
|
|
273
275
|
url:,
|
|
276
|
+
# CSS selectors to remove before each crawled page is converted to Markdown.
|
|
277
|
+
# Applied after includeSelectors. Exclusion takes precedence: an element matching
|
|
278
|
+
# both is removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
279
|
+
exclude_selectors: nil,
|
|
274
280
|
# When true, follow links on subdomains of the starting URL's domain (e.g.
|
|
275
281
|
# docs.example.com when starting from example.com). www and apex are always
|
|
276
282
|
# treated as equivalent.
|
|
@@ -282,6 +288,11 @@ module ContextDev
|
|
|
282
288
|
include_images: nil,
|
|
283
289
|
# Preserve hyperlinks in the Markdown output
|
|
284
290
|
include_links: nil,
|
|
291
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
292
|
+
# descendants) are kept before each crawled page is converted to Markdown. When
|
|
293
|
+
# omitted, the entire document is kept. Examples: "article.main", "#content",
|
|
294
|
+
# "[role=main]".
|
|
295
|
+
include_selectors: nil,
|
|
285
296
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
286
297
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
287
298
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -320,8 +331,10 @@ module ContextDev
|
|
|
320
331
|
sig do
|
|
321
332
|
params(
|
|
322
333
|
url: String,
|
|
334
|
+
exclude_selectors: T::Array[String],
|
|
323
335
|
headers: T::Hash[Symbol, String],
|
|
324
336
|
include_frames: T::Boolean,
|
|
337
|
+
include_selectors: T::Array[String],
|
|
325
338
|
max_age_ms: Integer,
|
|
326
339
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
|
|
327
340
|
timeout_ms: Integer,
|
|
@@ -332,12 +345,20 @@ module ContextDev
|
|
|
332
345
|
def web_scrape_html(
|
|
333
346
|
# Full URL to scrape (must include http:// or https:// protocol)
|
|
334
347
|
url:,
|
|
348
|
+
# CSS selectors to remove from the result. Applied after includeSelectors.
|
|
349
|
+
# Exclusion takes precedence: an element matching both is removed. Examples:
|
|
350
|
+
# "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
351
|
+
exclude_selectors: nil,
|
|
335
352
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
336
353
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
337
354
|
# is bypassed: the result is neither read from nor written to cache.
|
|
338
355
|
headers: nil,
|
|
339
356
|
# When true, iframes are rendered inline into the returned HTML.
|
|
340
357
|
include_frames: nil,
|
|
358
|
+
# CSS selectors. When provided, only matching subtrees (and their descendants) are
|
|
359
|
+
# kept and everything else is dropped. When omitted, the entire document is kept.
|
|
360
|
+
# Examples: "article.main", "#content", "[role=main]".
|
|
361
|
+
include_selectors: nil,
|
|
341
362
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
342
363
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
343
364
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -399,10 +420,12 @@ module ContextDev
|
|
|
399
420
|
sig do
|
|
400
421
|
params(
|
|
401
422
|
url: String,
|
|
423
|
+
exclude_selectors: T::Array[String],
|
|
402
424
|
headers: T::Hash[Symbol, String],
|
|
403
425
|
include_frames: T::Boolean,
|
|
404
426
|
include_images: T::Boolean,
|
|
405
427
|
include_links: T::Boolean,
|
|
428
|
+
include_selectors: T::Array[String],
|
|
406
429
|
max_age_ms: Integer,
|
|
407
430
|
pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
|
|
408
431
|
shorten_base64_images: T::Boolean,
|
|
@@ -416,6 +439,10 @@ module ContextDev
|
|
|
416
439
|
# Full URL to scrape into LLM usable Markdown (must include http:// or https://
|
|
417
440
|
# protocol)
|
|
418
441
|
url:,
|
|
442
|
+
# CSS selectors to remove before conversion to Markdown. Applied after
|
|
443
|
+
# includeSelectors. Exclusion takes precedence: an element matching both is
|
|
444
|
+
# removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
|
|
445
|
+
exclude_selectors: nil,
|
|
419
446
|
# Optional outbound HTTP headers forwarded only to the target URL, sent as
|
|
420
447
|
# deep-object query params such as headers[X-Custom]=value. When provided, caching
|
|
421
448
|
# is bypassed: the result is neither read from nor written to cache.
|
|
@@ -426,6 +453,10 @@ module ContextDev
|
|
|
426
453
|
include_images: nil,
|
|
427
454
|
# Preserve hyperlinks in Markdown output
|
|
428
455
|
include_links: nil,
|
|
456
|
+
# CSS selectors. When provided, only matching HTML subtrees (and their
|
|
457
|
+
# descendants) are kept before conversion to Markdown. When omitted, the entire
|
|
458
|
+
# document is kept. Examples: "article.main", "#content", "[role=main]".
|
|
459
|
+
include_selectors: nil,
|
|
429
460
|
# Return a cached result if a prior scrape for the same parameters exists and is
|
|
430
461
|
# younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
|
|
431
462
|
# omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
|
|
@@ -3,10 +3,12 @@ module ContextDev
|
|
|
3
3
|
type web_web_crawl_md_params =
|
|
4
4
|
{
|
|
5
5
|
url: String,
|
|
6
|
+
exclude_selectors: ::Array[String],
|
|
6
7
|
follow_subdomains: bool,
|
|
7
8
|
include_frames: bool,
|
|
8
9
|
include_images: bool,
|
|
9
10
|
include_links: bool,
|
|
11
|
+
include_selectors: ::Array[String],
|
|
10
12
|
max_age_ms: Integer,
|
|
11
13
|
max_depth: Integer,
|
|
12
14
|
max_pages: Integer,
|
|
@@ -26,6 +28,10 @@ module ContextDev
|
|
|
26
28
|
|
|
27
29
|
attr_accessor url: String
|
|
28
30
|
|
|
31
|
+
attr_reader exclude_selectors: ::Array[String]?
|
|
32
|
+
|
|
33
|
+
def exclude_selectors=: (::Array[String]) -> ::Array[String]
|
|
34
|
+
|
|
29
35
|
attr_reader follow_subdomains: bool?
|
|
30
36
|
|
|
31
37
|
def follow_subdomains=: (bool) -> bool
|
|
@@ -42,6 +48,10 @@ module ContextDev
|
|
|
42
48
|
|
|
43
49
|
def include_links=: (bool) -> bool
|
|
44
50
|
|
|
51
|
+
attr_reader include_selectors: ::Array[String]?
|
|
52
|
+
|
|
53
|
+
def include_selectors=: (::Array[String]) -> ::Array[String]
|
|
54
|
+
|
|
45
55
|
attr_reader max_age_ms: Integer?
|
|
46
56
|
|
|
47
57
|
def max_age_ms=: (Integer) -> Integer
|
|
@@ -86,10 +96,12 @@ module ContextDev
|
|
|
86
96
|
|
|
87
97
|
def initialize: (
|
|
88
98
|
url: String,
|
|
99
|
+
?exclude_selectors: ::Array[String],
|
|
89
100
|
?follow_subdomains: bool,
|
|
90
101
|
?include_frames: bool,
|
|
91
102
|
?include_images: bool,
|
|
92
103
|
?include_links: bool,
|
|
104
|
+
?include_selectors: ::Array[String],
|
|
93
105
|
?max_age_ms: Integer,
|
|
94
106
|
?max_depth: Integer,
|
|
95
107
|
?max_pages: Integer,
|
|
@@ -105,10 +117,12 @@ module ContextDev
|
|
|
105
117
|
|
|
106
118
|
def to_hash: -> {
|
|
107
119
|
url: String,
|
|
120
|
+
exclude_selectors: ::Array[String],
|
|
108
121
|
follow_subdomains: bool,
|
|
109
122
|
include_frames: bool,
|
|
110
123
|
include_images: bool,
|
|
111
124
|
include_links: bool,
|
|
125
|
+
include_selectors: ::Array[String],
|
|
112
126
|
max_age_ms: Integer,
|
|
113
127
|
max_depth: Integer,
|
|
114
128
|
max_pages: Integer,
|
|
@@ -3,8 +3,10 @@ module ContextDev
|
|
|
3
3
|
type web_web_scrape_html_params =
|
|
4
4
|
{
|
|
5
5
|
url: String,
|
|
6
|
+
exclude_selectors: ::Array[String],
|
|
6
7
|
headers: ::Hash[Symbol, String],
|
|
7
8
|
include_frames: bool,
|
|
9
|
+
include_selectors: ::Array[String],
|
|
8
10
|
max_age_ms: Integer,
|
|
9
11
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
10
12
|
timeout_ms: Integer,
|
|
@@ -18,6 +20,10 @@ module ContextDev
|
|
|
18
20
|
|
|
19
21
|
attr_accessor url: String
|
|
20
22
|
|
|
23
|
+
attr_reader exclude_selectors: ::Array[String]?
|
|
24
|
+
|
|
25
|
+
def exclude_selectors=: (::Array[String]) -> ::Array[String]
|
|
26
|
+
|
|
21
27
|
attr_reader headers: ::Hash[Symbol, String]?
|
|
22
28
|
|
|
23
29
|
def headers=: (::Hash[Symbol, String]) -> ::Hash[Symbol, String]
|
|
@@ -26,6 +32,10 @@ module ContextDev
|
|
|
26
32
|
|
|
27
33
|
def include_frames=: (bool) -> bool
|
|
28
34
|
|
|
35
|
+
attr_reader include_selectors: ::Array[String]?
|
|
36
|
+
|
|
37
|
+
def include_selectors=: (::Array[String]) -> ::Array[String]
|
|
38
|
+
|
|
29
39
|
attr_reader max_age_ms: Integer?
|
|
30
40
|
|
|
31
41
|
def max_age_ms=: (Integer) -> Integer
|
|
@@ -46,8 +56,10 @@ module ContextDev
|
|
|
46
56
|
|
|
47
57
|
def initialize: (
|
|
48
58
|
url: String,
|
|
59
|
+
?exclude_selectors: ::Array[String],
|
|
49
60
|
?headers: ::Hash[Symbol, String],
|
|
50
61
|
?include_frames: bool,
|
|
62
|
+
?include_selectors: ::Array[String],
|
|
51
63
|
?max_age_ms: Integer,
|
|
52
64
|
?pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
53
65
|
?timeout_ms: Integer,
|
|
@@ -57,8 +69,10 @@ module ContextDev
|
|
|
57
69
|
|
|
58
70
|
def to_hash: -> {
|
|
59
71
|
url: String,
|
|
72
|
+
exclude_selectors: ::Array[String],
|
|
60
73
|
headers: ::Hash[Symbol, String],
|
|
61
74
|
include_frames: bool,
|
|
75
|
+
include_selectors: ::Array[String],
|
|
62
76
|
max_age_ms: Integer,
|
|
63
77
|
pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
64
78
|
timeout_ms: Integer,
|
|
@@ -3,10 +3,12 @@ module ContextDev
|
|
|
3
3
|
type web_web_scrape_md_params =
|
|
4
4
|
{
|
|
5
5
|
url: String,
|
|
6
|
+
exclude_selectors: ::Array[String],
|
|
6
7
|
headers: ::Hash[Symbol, String],
|
|
7
8
|
include_frames: bool,
|
|
8
9
|
include_images: bool,
|
|
9
10
|
include_links: bool,
|
|
11
|
+
include_selectors: ::Array[String],
|
|
10
12
|
max_age_ms: Integer,
|
|
11
13
|
pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
12
14
|
:shorten_base64_images => bool,
|
|
@@ -22,6 +24,10 @@ module ContextDev
|
|
|
22
24
|
|
|
23
25
|
attr_accessor url: String
|
|
24
26
|
|
|
27
|
+
attr_reader exclude_selectors: ::Array[String]?
|
|
28
|
+
|
|
29
|
+
def exclude_selectors=: (::Array[String]) -> ::Array[String]
|
|
30
|
+
|
|
25
31
|
attr_reader headers: ::Hash[Symbol, String]?
|
|
26
32
|
|
|
27
33
|
def headers=: (::Hash[Symbol, String]) -> ::Hash[Symbol, String]
|
|
@@ -38,6 +44,10 @@ module ContextDev
|
|
|
38
44
|
|
|
39
45
|
def include_links=: (bool) -> bool
|
|
40
46
|
|
|
47
|
+
attr_reader include_selectors: ::Array[String]?
|
|
48
|
+
|
|
49
|
+
def include_selectors=: (::Array[String]) -> ::Array[String]
|
|
50
|
+
|
|
41
51
|
attr_reader max_age_ms: Integer?
|
|
42
52
|
|
|
43
53
|
def max_age_ms=: (Integer) -> Integer
|
|
@@ -66,10 +76,12 @@ module ContextDev
|
|
|
66
76
|
|
|
67
77
|
def initialize: (
|
|
68
78
|
url: String,
|
|
79
|
+
?exclude_selectors: ::Array[String],
|
|
69
80
|
?headers: ::Hash[Symbol, String],
|
|
70
81
|
?include_frames: bool,
|
|
71
82
|
?include_images: bool,
|
|
72
83
|
?include_links: bool,
|
|
84
|
+
?include_selectors: ::Array[String],
|
|
73
85
|
?max_age_ms: Integer,
|
|
74
86
|
?pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
75
87
|
?shorten_base64_images: bool,
|
|
@@ -81,10 +93,12 @@ module ContextDev
|
|
|
81
93
|
|
|
82
94
|
def to_hash: -> {
|
|
83
95
|
url: String,
|
|
96
|
+
exclude_selectors: ::Array[String],
|
|
84
97
|
headers: ::Hash[Symbol, String],
|
|
85
98
|
include_frames: bool,
|
|
86
99
|
include_images: bool,
|
|
87
100
|
include_links: bool,
|
|
101
|
+
include_selectors: ::Array[String],
|
|
88
102
|
max_age_ms: Integer,
|
|
89
103
|
pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
90
104
|
:shorten_base64_images => bool,
|
|
@@ -65,10 +65,12 @@ module ContextDev
|
|
|
65
65
|
|
|
66
66
|
def web_crawl_md: (
|
|
67
67
|
url: String,
|
|
68
|
+
?exclude_selectors: ::Array[String],
|
|
68
69
|
?follow_subdomains: bool,
|
|
69
70
|
?include_frames: bool,
|
|
70
71
|
?include_images: bool,
|
|
71
72
|
?include_links: bool,
|
|
73
|
+
?include_selectors: ::Array[String],
|
|
72
74
|
?max_age_ms: Integer,
|
|
73
75
|
?max_depth: Integer,
|
|
74
76
|
?max_pages: Integer,
|
|
@@ -84,8 +86,10 @@ module ContextDev
|
|
|
84
86
|
|
|
85
87
|
def web_scrape_html: (
|
|
86
88
|
url: String,
|
|
89
|
+
?exclude_selectors: ::Array[String],
|
|
87
90
|
?headers: ::Hash[Symbol, String],
|
|
88
91
|
?include_frames: bool,
|
|
92
|
+
?include_selectors: ::Array[String],
|
|
89
93
|
?max_age_ms: Integer,
|
|
90
94
|
?pdf: ContextDev::WebWebScrapeHTMLParams::Pdf,
|
|
91
95
|
?timeout_ms: Integer,
|
|
@@ -105,10 +109,12 @@ module ContextDev
|
|
|
105
109
|
|
|
106
110
|
def web_scrape_md: (
|
|
107
111
|
url: String,
|
|
112
|
+
?exclude_selectors: ::Array[String],
|
|
108
113
|
?headers: ::Hash[Symbol, String],
|
|
109
114
|
?include_frames: bool,
|
|
110
115
|
?include_images: bool,
|
|
111
116
|
?include_links: bool,
|
|
117
|
+
?include_selectors: ::Array[String],
|
|
112
118
|
?max_age_ms: Integer,
|
|
113
119
|
?pdf: ContextDev::WebWebScrapeMdParams::Pdf,
|
|
114
120
|
?shorten_base64_images: bool,
|