context.dev 2.17.1 → 2.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +33 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +0 -4
  5. data/lib/context_dev/models/industry_retrieve_naics_params.rb +28 -1
  6. data/lib/context_dev/models/industry_retrieve_sic_params.rb +28 -1
  7. data/lib/context_dev/models/monitor_retrieve_run_params.rb +26 -0
  8. data/lib/context_dev/models/monitor_retrieve_run_response.rb +232 -0
  9. data/lib/context_dev/models/monitor_rotate_webhook_secret_params.rb +20 -0
  10. data/lib/context_dev/models/monitor_rotate_webhook_secret_response.rb +765 -0
  11. data/lib/context_dev/models/parse_handle_params.rb +8 -6
  12. data/lib/context_dev/models/person_enrich_params.rb +28 -1
  13. data/lib/context_dev/models/web_answers_params.rb +28 -1
  14. data/lib/context_dev/models/web_extract_competitors_params.rb +28 -1
  15. data/lib/context_dev/models/web_extract_styleguide_params.rb +30 -3
  16. data/lib/context_dev/models/{web_web_scrape_sitemap_params.rb → web_map_urls_params.rb} +22 -20
  17. data/lib/context_dev/models/web_map_urls_response.rb +123 -0
  18. data/lib/context_dev/models/web_scrape_params.rb +906 -0
  19. data/lib/context_dev/models/web_scrape_response.rb +659 -0
  20. data/lib/context_dev/models/web_screenshot_params.rb +25 -9
  21. data/lib/context_dev/models/web_screenshot_response.rb +3 -3
  22. data/lib/context_dev/models/web_search_params.rb +30 -3
  23. data/lib/context_dev/models.rb +8 -20
  24. data/lib/context_dev/resources/brand.rb +0 -36
  25. data/lib/context_dev/resources/industry.rb +6 -2
  26. data/lib/context_dev/resources/monitors.rb +47 -0
  27. data/lib/context_dev/resources/people.rb +3 -1
  28. data/lib/context_dev/resources/web.rb +116 -413
  29. data/lib/context_dev/version.rb +1 -1
  30. data/lib/context_dev.rb +8 -21
  31. data/rbi/context_dev/client.rbi +0 -3
  32. data/rbi/context_dev/models/industry_retrieve_naics_params.rbi +59 -0
  33. data/rbi/context_dev/models/industry_retrieve_sic_params.rbi +57 -0
  34. data/rbi/context_dev/models/monitor_retrieve_run_params.rbi +46 -0
  35. data/rbi/context_dev/models/monitor_retrieve_run_response.rbi +452 -0
  36. data/rbi/context_dev/models/monitor_rotate_webhook_secret_params.rbi +38 -0
  37. data/rbi/context_dev/models/monitor_rotate_webhook_secret_response.rbi +1367 -0
  38. data/rbi/context_dev/models/parse_handle_params.rbi +12 -9
  39. data/rbi/context_dev/models/person_enrich_params.rbi +45 -0
  40. data/rbi/context_dev/models/web_answers_params.rbi +45 -0
  41. data/rbi/context_dev/models/web_extract_competitors_params.rbi +59 -0
  42. data/rbi/context_dev/models/web_extract_styleguide_params.rbi +62 -3
  43. data/rbi/context_dev/models/{web_web_scrape_sitemap_params.rbi → web_map_urls_params.rbi} +35 -54
  44. data/rbi/context_dev/models/web_map_urls_response.rbi +226 -0
  45. data/rbi/context_dev/models/web_scrape_params.rbi +2000 -0
  46. data/rbi/context_dev/models/{web_web_scrape_html_response.rbi → web_scrape_response.rbi} +550 -418
  47. data/rbi/context_dev/models/web_screenshot_params.rbi +38 -12
  48. data/rbi/context_dev/models/web_screenshot_response.rbi +4 -4
  49. data/rbi/context_dev/models/web_search_params.rbi +48 -3
  50. data/rbi/context_dev/models.rbi +9 -21
  51. data/rbi/context_dev/resources/brand.rbi +0 -35
  52. data/rbi/context_dev/resources/industry.rbi +14 -0
  53. data/rbi/context_dev/resources/monitors.rbi +24 -0
  54. data/rbi/context_dev/resources/parse.rbi +4 -4
  55. data/rbi/context_dev/resources/people.rbi +7 -0
  56. data/rbi/context_dev/resources/web.rbi +166 -533
  57. data/sig/context_dev/client.rbs +0 -2
  58. data/sig/context_dev/models/industry_retrieve_naics_params.rbs +21 -1
  59. data/sig/context_dev/models/industry_retrieve_sic_params.rbs +21 -1
  60. data/sig/context_dev/models/monitor_retrieve_run_params.rbs +28 -0
  61. data/sig/context_dev/models/monitor_retrieve_run_response.rbs +182 -0
  62. data/sig/context_dev/models/monitor_rotate_webhook_secret_params.rbs +23 -0
  63. data/sig/context_dev/models/monitor_rotate_webhook_secret_response.rbs +545 -0
  64. data/sig/context_dev/models/person_enrich_params.rbs +21 -1
  65. data/sig/context_dev/models/web_answers_params.rbs +21 -1
  66. data/sig/context_dev/models/web_extract_competitors_params.rbs +21 -1
  67. data/sig/context_dev/models/web_extract_styleguide_params.rbs +21 -1
  68. data/sig/context_dev/models/{web_web_scrape_sitemap_params.rbs → web_map_urls_params.rbs} +22 -22
  69. data/sig/context_dev/models/web_map_urls_response.rbs +125 -0
  70. data/sig/context_dev/models/web_scrape_params.rbs +834 -0
  71. data/sig/context_dev/models/web_scrape_response.rbs +552 -0
  72. data/sig/context_dev/models/web_screenshot_params.rbs +7 -0
  73. data/sig/context_dev/models/web_search_params.rbs +21 -1
  74. data/sig/context_dev/models.rbs +8 -20
  75. data/sig/context_dev/resources/brand.rbs +0 -9
  76. data/sig/context_dev/resources/industry.rbs +2 -0
  77. data/sig/context_dev/resources/monitors.rbs +11 -0
  78. data/sig/context_dev/resources/people.rbs +1 -0
  79. data/sig/context_dev/resources/web.rbs +34 -108
  80. metadata +26 -65
  81. data/lib/context_dev/models/ai_extract_product_params.rb +0 -98
  82. data/lib/context_dev/models/ai_extract_product_response.rb +0 -352
  83. data/lib/context_dev/models/ai_extract_products_params.rb +0 -233
  84. data/lib/context_dev/models/ai_extract_products_response.rb +0 -320
  85. data/lib/context_dev/models/brand_retrieve_simplified_params.rb +0 -120
  86. data/lib/context_dev/models/brand_retrieve_simplified_response.rb +0 -441
  87. data/lib/context_dev/models/web_extract_fonts_params.rb +0 -114
  88. data/lib/context_dev/models/web_extract_fonts_response.rb +0 -291
  89. data/lib/context_dev/models/web_extract_params.rb +0 -392
  90. data/lib/context_dev/models/web_extract_response.rb +0 -287
  91. data/lib/context_dev/models/web_web_scrape_bytes_params.rb +0 -344
  92. data/lib/context_dev/models/web_web_scrape_bytes_response.rb +0 -135
  93. data/lib/context_dev/models/web_web_scrape_html_params.rb +0 -633
  94. data/lib/context_dev/models/web_web_scrape_html_response.rb +0 -588
  95. data/lib/context_dev/models/web_web_scrape_images_params.rb +0 -346
  96. data/lib/context_dev/models/web_web_scrape_images_response.rb +0 -409
  97. data/lib/context_dev/models/web_web_scrape_md_params.rb +0 -668
  98. data/lib/context_dev/models/web_web_scrape_md_response.rb +0 -567
  99. data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +0 -144
  100. data/lib/context_dev/resources/ai.rb +0 -69
  101. data/rbi/context_dev/models/ai_extract_product_params.rbi +0 -199
  102. data/rbi/context_dev/models/ai_extract_product_response.rbi +0 -696
  103. data/rbi/context_dev/models/ai_extract_products_params.rbi +0 -490
  104. data/rbi/context_dev/models/ai_extract_products_response.rbi +0 -623
  105. data/rbi/context_dev/models/brand_retrieve_simplified_params.rbi +0 -256
  106. data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +0 -1040
  107. data/rbi/context_dev/models/web_extract_fonts_params.rbi +0 -223
  108. data/rbi/context_dev/models/web_extract_fonts_response.rbi +0 -563
  109. data/rbi/context_dev/models/web_extract_params.rbi +0 -736
  110. data/rbi/context_dev/models/web_extract_response.rbi +0 -512
  111. data/rbi/context_dev/models/web_web_scrape_bytes_params.rbi +0 -699
  112. data/rbi/context_dev/models/web_web_scrape_bytes_response.rbi +0 -238
  113. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +0 -1217
  114. data/rbi/context_dev/models/web_web_scrape_images_params.rbi +0 -671
  115. data/rbi/context_dev/models/web_web_scrape_images_response.rbi +0 -910
  116. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +0 -1049
  117. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +0 -1088
  118. data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +0 -258
  119. data/rbi/context_dev/resources/ai.rbi +0 -56
  120. data/sig/context_dev/models/ai_extract_product_params.rbs +0 -86
  121. data/sig/context_dev/models/ai_extract_product_response.rbs +0 -273
  122. data/sig/context_dev/models/ai_extract_products_params.rbs +0 -202
  123. data/sig/context_dev/models/ai_extract_products_response.rbs +0 -252
  124. data/sig/context_dev/models/brand_retrieve_simplified_params.rbs +0 -104
  125. data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +0 -413
  126. data/sig/context_dev/models/web_extract_fonts_params.rbs +0 -93
  127. data/sig/context_dev/models/web_extract_fonts_response.rbs +0 -230
  128. data/sig/context_dev/models/web_extract_params.rbs +0 -304
  129. data/sig/context_dev/models/web_extract_response.rbs +0 -239
  130. data/sig/context_dev/models/web_web_scrape_bytes_params.rbs +0 -531
  131. data/sig/context_dev/models/web_web_scrape_bytes_response.rbs +0 -108
  132. data/sig/context_dev/models/web_web_scrape_html_params.rbs +0 -730
  133. data/sig/context_dev/models/web_web_scrape_html_response.rbs +0 -495
  134. data/sig/context_dev/models/web_web_scrape_images_params.rbs +0 -261
  135. data/sig/context_dev/models/web_web_scrape_images_response.rbs +0 -371
  136. data/sig/context_dev/models/web_web_scrape_md_params.rbs +0 -758
  137. data/sig/context_dev/models/web_web_scrape_md_response.rbs +0 -465
  138. data/sig/context_dev/models/web_web_scrape_sitemap_response.rbs +0 -117
  139. data/sig/context_dev/resources/ai.rbs +0 -20
@@ -12,7 +12,7 @@ module ContextDev
12
12
  # 30 seconds and ultra to 50 seconds; timeoutOpts.milliseconds can shorten either
13
13
  # deadline.
14
14
  #
15
- # @overload answers(task:, json_format: nil, mode: nil, tags: nil, timeout_opts: nil, request_options: {})
15
+ # @overload answers(task:, json_format: nil, mode: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
16
16
  #
17
17
  # @param task [String] What to research and answer, in plain language. Naming a domain in the task (for
18
18
  #
@@ -24,6 +24,8 @@ module ContextDev
24
24
  #
25
25
  # @param timeout_opts [ContextDev::Models::WebAnswersParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
26
26
  #
27
+ # @param zdr [Symbol, ContextDev::Models::WebAnswersParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
28
+ #
27
29
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
28
30
  #
29
31
  # @return [ContextDev::Models::WebAnswersResponse]
@@ -40,69 +42,13 @@ module ContextDev
40
42
  )
41
43
  end
42
44
 
43
- # Some parameter documentations has been truncated, see
44
- # {ContextDev::Models::WebExtractParams} for more details.
45
- #
46
- # Crawl a website, use the provided JSON Schema and instructions to prioritize
47
- # relevant internal links, and extract structured data from the selected pages.
48
- #
49
- # @overload extract(schema:, url:, actions: nil, fact_check: nil, follow_subdomains: nil, include_frames: nil, instructions: nil, max_age_ms: nil, max_depth: nil, max_pages: nil, pdf: nil, settle_animations: nil, stop_after_ms: nil, tags: nil, timeout_opts: nil, wait_for_ms: nil, request_options: {})
50
- #
51
- # @param schema [Hash{Symbol=>Object}] JSON Schema for the returned data object. Image fields such as `image_urls` or `
52
- #
53
- # @param url [String] The starting website URL to crawl and extract from. Must include http:// or http
54
- #
55
- # @param actions [Array<ContextDev::Models::WebExtractParams::Action::Wait, ContextDev::Models::WebExtractParams::Action::Perform, ContextDev::Models::WebExtractParams::Action::Scroll>] Optional browser actions executed in order on the requested page after it loads,
56
- #
57
- # @param fact_check [Boolean] When true, every returned value must be grounded in facts stated on the page; fi
58
- #
59
- # @param follow_subdomains [Boolean] When true, follow links on subdomains of the starting URL's domain.
60
- #
61
- # @param include_frames [Boolean] When true, iframe contents are included in Markdown before extraction.
62
- #
63
- # @param instructions [String] Optional extraction guidance, such as which facts to prioritize or how to interp
64
- #
65
- # @param max_age_ms [Integer] Return cached scrape results if a prior scrape for the same parameters is younge
66
- #
67
- # @param max_depth [Integer] Optional maximum link depth from the starting URL (0 = only the starting page).
68
- #
69
- # @param max_pages [Integer] Maximum number of pages to analyze for extraction. Hard cap: 50. Defaults to 5.
70
- #
71
- # @param pdf [ContextDev::Models::WebExtractParams::Pdf]
72
- #
73
- # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
74
- #
75
- # @param stop_after_ms [Integer] Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000 (1
76
- #
77
- # @param tags [Array<String>] Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
78
- #
79
- # @param timeout_opts [ContextDev::Models::WebExtractParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
80
- #
81
- # @param wait_for_ms [Integer] Optional browser wait time in milliseconds after initial page load for each craw
82
- #
83
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
84
- #
85
- # @return [ContextDev::Models::WebExtractResponse]
86
- #
87
- # @see ContextDev::Models::WebExtractParams
88
- def extract(params)
89
- parsed, options = ContextDev::WebExtractParams.dump_request(params)
90
- @client.request(
91
- method: :post,
92
- path: "web/extract",
93
- body: parsed,
94
- model: ContextDev::Models::WebExtractResponse,
95
- options: options
96
- )
97
- end
98
-
99
45
  # Some parameter documentations has been truncated, see
100
46
  # {ContextDev::Models::WebExtractCompetitorsParams} for more details.
101
47
  #
102
48
  # Analyze a company's landing page and web search evidence to return direct
103
49
  # competitors for the same product or market.
104
50
  #
105
- # @overload extract_competitors(domain:, num_competitors: nil, tags: nil, timeout_opts: nil, request_options: {})
51
+ # @overload extract_competitors(domain:, num_competitors: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
106
52
  #
107
53
  # @param domain [String] Company domain to analyze, such as `stripe.com`. Full http(s) URLs are accepted
108
54
  #
@@ -112,6 +58,8 @@ module ContextDev
112
58
  #
113
59
  # @param timeout_opts [ContextDev::Models::WebExtractCompetitorsParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
114
60
  #
61
+ # @param zdr [Symbol, ContextDev::Models::WebExtractCompetitorsParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
62
+ #
115
63
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
116
64
  #
117
65
  # @return [ContextDev::Models::WebExtractCompetitorsResponse]
@@ -130,82 +78,154 @@ module ContextDev
130
78
  end
131
79
 
132
80
  # Some parameter documentations has been truncated, see
133
- # {ContextDev::Models::WebExtractFontsParams} for more details.
81
+ # {ContextDev::Models::WebExtractStyleguideParams} for more details.
134
82
  #
135
- # Scrape font information from a website including font families, usage
136
- # statistics, fallbacks, and element/word counts.
83
+ # Extract a comprehensive design system from a website including colors,
84
+ # typography, spacing, shadows, and UI components.
137
85
  #
138
- # @overload extract_fonts(direct_url: nil, domain: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, request_options: {})
86
+ # @overload extract_styleguide(color_scheme: nil, direct_url: nil, domain: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
139
87
  #
140
- # @param direct_url [String] A specific URL to fetch fonts from directly, bypassing domain resolution (e.g.,
88
+ # @param color_scheme [Symbol, ContextDev::Models::WebExtractStyleguideParams::ColorScheme] Optional browser color scheme to emulate for websites that respond to prefers-co
141
89
  #
142
- # @param domain [String] Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The domai
90
+ # @param direct_url [String] A specific URL to fetch the styleguide from directly, bypassing domain resolutio
91
+ #
92
+ # @param domain [String] Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
143
93
  #
144
94
  # @param max_age_ms [Integer, nil] Maximum age in milliseconds for cached brand data before the API performs a hard
145
95
  #
146
96
  # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
147
97
  #
148
- # @param timeout_opts [ContextDev::Models::WebExtractFontsParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
98
+ # @param timeout_opts [ContextDev::Models::WebExtractStyleguideParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
99
+ #
100
+ # @param zdr [Symbol, ContextDev::Models::WebExtractStyleguideParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
149
101
  #
150
102
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
151
103
  #
152
- # @return [ContextDev::Models::WebExtractFontsResponse]
104
+ # @return [ContextDev::Models::WebExtractStyleguideResponse]
153
105
  #
154
- # @see ContextDev::Models::WebExtractFontsParams
155
- def extract_fonts(params = {})
156
- parsed, options = ContextDev::WebExtractFontsParams.dump_request(params)
106
+ # @see ContextDev::Models::WebExtractStyleguideParams
107
+ def extract_styleguide(params = {})
108
+ parsed, options = ContextDev::WebExtractStyleguideParams.dump_request(params)
157
109
  query = ContextDev::Internal::Util.encode_query_params(parsed)
158
110
  @client.request(
159
111
  method: :get,
160
- path: "web/fonts",
112
+ path: "web/styleguide",
161
113
  query: query.transform_keys(
114
+ color_scheme: "colorScheme",
162
115
  direct_url: "directUrl",
163
116
  max_age_ms: "maxAgeMs",
164
117
  timeout_opts: "timeoutOpts"
165
118
  ),
166
- model: ContextDev::Models::WebExtractFontsResponse,
119
+ model: ContextDev::Models::WebExtractStyleguideResponse,
167
120
  options: options
168
121
  )
169
122
  end
170
123
 
171
124
  # Some parameter documentations has been truncated, see
172
- # {ContextDev::Models::WebExtractStyleguideParams} for more details.
125
+ # {ContextDev::Models::WebMapURLsParams} for more details.
173
126
  #
174
- # Extract a comprehensive design system from a website including colors,
175
- # typography, spacing, shadows, and UI components.
127
+ # Discovers URLs using the same sitemap crawl, filters, and limits as
128
+ # /web/scrape/sitemap. Each URL includes its available title, description,
129
+ # keywords, and language. URLs without stored enrichment are returned immediately
130
+ # with only the URL and queued for background HTML scraping, so later requests can
131
+ # include their metadata. Responses are never cached as a whole; every request
132
+ # reads the current per-URL enrichment. Zero data retention and credential-bearing
133
+ # discovery requests return URLs without reading or storing shared enrichment or
134
+ # queuing background scrapes. Costs 1 credit, or 2 credits with search.
176
135
  #
177
- # @overload extract_styleguide(color_scheme: nil, direct_url: nil, domain: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, request_options: {})
136
+ # @overload map_urls(domain:, headers: nil, include_subdomains: nil, max_links: nil, search: nil, sitemap_url: nil, tags: nil, timeout_opts: nil, url_regex: nil, zdr: nil, request_options: {})
178
137
  #
179
- # @param color_scheme [Symbol, ContextDev::Models::WebExtractStyleguideParams::ColorScheme] Optional browser color scheme to emulate for websites that respond to prefers-co
138
+ # @param domain [String] Domain to build a sitemap for
180
139
  #
181
- # @param direct_url [String] A specific URL to fetch the styleguide from directly, bypassing domain resolutio
140
+ # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
182
141
  #
183
- # @param domain [String] Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
142
+ # @param include_subdomains [Boolean] When true, discover and include public pages and sitemaps on subdomains of the r
184
143
  #
185
- # @param max_age_ms [Integer, nil] Maximum age in milliseconds for cached brand data before the API performs a hard
144
+ # @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
145
+ #
146
+ # @param search [String] Optional search phrase. When provided, the crawled sitemap is filtered to the pa
147
+ #
148
+ # @param sitemap_url [String] Optional explicit sitemap URL. When provided, exactly this sitemap is crawled in
186
149
  #
187
150
  # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
188
151
  #
189
- # @param timeout_opts [ContextDev::Models::WebExtractStyleguideParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
152
+ # @param timeout_opts [ContextDev::Models::WebMapURLsParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
153
+ #
154
+ # @param url_regex [String] Optional RE2-compatible regex pattern. Only URLs matching this pattern are retur
155
+ #
156
+ # @param zdr [Symbol, ContextDev::Models::WebMapURLsParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
190
157
  #
191
158
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
192
159
  #
193
- # @return [ContextDev::Models::WebExtractStyleguideResponse]
160
+ # @return [ContextDev::Models::WebMapURLsResponse]
194
161
  #
195
- # @see ContextDev::Models::WebExtractStyleguideParams
196
- def extract_styleguide(params = {})
197
- parsed, options = ContextDev::WebExtractStyleguideParams.dump_request(params)
162
+ # @see ContextDev::Models::WebMapURLsParams
163
+ def map_urls(params)
164
+ parsed, options = ContextDev::WebMapURLsParams.dump_request(params)
198
165
  query = ContextDev::Internal::Util.encode_query_params(parsed)
199
166
  @client.request(
200
167
  method: :get,
201
- path: "web/styleguide",
168
+ path: "web/urls",
202
169
  query: query.transform_keys(
203
- color_scheme: "colorScheme",
204
- direct_url: "directUrl",
205
- max_age_ms: "maxAgeMs",
206
- timeout_opts: "timeoutOpts"
170
+ include_subdomains: "includeSubdomains",
171
+ max_links: "maxLinks",
172
+ sitemap_url: "sitemapUrl",
173
+ timeout_opts: "timeoutOpts",
174
+ url_regex: "urlRegex"
207
175
  ),
208
- model: ContextDev::Models::WebExtractStyleguideResponse,
176
+ model: ContextDev::Models::WebMapURLsResponse,
177
+ options: options
178
+ )
179
+ end
180
+
181
+ # Some parameter documentations has been truncated, see
182
+ # {ContextDev::Models::WebScrapeParams} for more details.
183
+ #
184
+ # Reuse cached outputs independently and capture missing formats in one page
185
+ # visit. Each cache key includes only the settings that affect that output. HTML
186
+ # is shared with Markdown and parsed fields. Cached outputs can come from
187
+ # different visits within maxAgeMs; use 0 for a fresh capture. HTML-only requests
188
+ # use the existing fast acquisition path. One credit per request, including cache
189
+ # hits, or two with browser actions; PDF OCR adds one credit per recovered page on
190
+ # fresh extraction. Original response bytes and screenshots are limited to 20 MiB
191
+ # each, screenshots to 40 megapixels, and the combined browser capture to 60 MiB.
192
+ #
193
+ # @overload scrape(formats:, url:, image_params: nil, markdown_params: nil, max_age_ms: nil, parse_params: nil, screenshot_params: nil, shared_params: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
194
+ #
195
+ # @param formats [ContextDev::Models::WebScrapeParams::Formats] Outputs to return. Enable at least one; omitted formats are false.
196
+ #
197
+ # @param url [String] The URL to scrape.
198
+ #
199
+ # @param image_params [ContextDev::Models::WebScrapeParams::ImageParams] Image options. Requires formats.images: true.
200
+ #
201
+ # @param markdown_params [ContextDev::Models::WebScrapeParams::MarkdownParams] Markdown options. Requires formats.markdown: true.
202
+ #
203
+ # @param max_age_ms [Integer] Maximum age of each cached output. Defaults to 1 day; 0 fetches fresh and update
204
+ #
205
+ # @param parse_params [ContextDev::Models::WebScrapeParams::ParseParams] Required when formats.parse is true.
206
+ #
207
+ # @param screenshot_params [ContextDev::Models::WebScrapeParams::ScreenshotParams] Screenshot options. Requires formats.screenshot: true.
208
+ #
209
+ # @param shared_params [ContextDev::Models::WebScrapeParams::SharedParams] Shared browser and content settings. Content filters leave screenshots and origi
210
+ #
211
+ # @param tags [Array<String>] Labels for tracking request usage. Not retained when zdr is enabled.
212
+ #
213
+ # @param timeout_opts [ContextDev::Models::WebScrapeParams::TimeoutOpts] Total deadline, including navigation, actions, waiting, and all outputs. Default
214
+ #
215
+ # @param zdr [Symbol, ContextDev::Models::WebScrapeParams::Zdr] Zero data retention. Bypasses caches and uploads; excludes request/response cont
216
+ #
217
+ # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
218
+ #
219
+ # @return [ContextDev::Models::WebScrapeResponse]
220
+ #
221
+ # @see ContextDev::Models::WebScrapeParams
222
+ def scrape(params)
223
+ parsed, options = ContextDev::WebScrapeParams.dump_request(params)
224
+ @client.request(
225
+ method: :post,
226
+ path: "web/scrape",
227
+ body: parsed,
228
+ model: ContextDev::Models::WebScrapeResponse,
209
229
  options: options
210
230
  )
211
231
  end
@@ -215,7 +235,7 @@ module ContextDev
215
235
  #
216
236
  # Capture a screenshot of a website.
217
237
  #
218
- # @overload screenshot(clear_popups: nil, color_scheme: nil, country: nil, direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, max_age_ms: nil, page: nil, scroll_offset: nil, tags: nil, timeout_opts: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
238
+ # @overload screenshot(clear_popups: nil, color_scheme: nil, country: nil, direct_url: nil, domain: nil, full_screenshot: nil, handle_cookie_popup: nil, headers: nil, max_age_ms: nil, page: nil, scroll_offset: nil, tags: nil, timeout_opts: nil, viewport: nil, wait_for_ms: nil, zdr: nil, request_options: {})
219
239
  #
220
240
  # @param clear_popups [Boolean] Optional parameter for comprehensive popup cleanup. If 'true', the browser dismi
221
241
  #
@@ -231,6 +251,8 @@ module ContextDev
231
251
  #
232
252
  # @param handle_cookie_popup [Boolean] Optional parameter to control cookie/consent popup handling. If 'true', we dismi
233
253
  #
254
+ # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers, using the same JSON object or deep-object query
255
+ #
234
256
  # @param max_age_ms [Integer, nil] Return a cached screenshot if a prior screenshot for the same parameters exists
235
257
  #
236
258
  # @param page [Symbol, ContextDev::Models::WebScreenshotParams::Page] Optional parameter to specify which page type to screenshot. If provided, the sy
@@ -279,7 +301,7 @@ module ContextDev
279
301
  #
280
302
  # Search the web and optionally scrape each result to Markdown in one round-trip.
281
303
  #
282
- # @overload search(query:, country: nil, exclude_domains: nil, freshness: nil, include_domains: nil, markdown_options: nil, num_results: nil, query_fanout: nil, tags: nil, timeout_opts: nil, request_options: {})
304
+ # @overload search(query:, country: nil, exclude_domains: nil, freshness: nil, include_domains: nil, markdown_options: nil, num_results: nil, query_fanout: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
283
305
  #
284
306
  # @param query [String] Search query. Accepts natural language as well as Google-style search operators
285
307
  #
@@ -301,6 +323,8 @@ module ContextDev
301
323
  #
302
324
  # @param timeout_opts [ContextDev::Models::WebSearchParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
303
325
  #
326
+ # @param zdr [Symbol, ContextDev::Models::WebSearchParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
327
+ #
304
328
  # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
305
329
  #
306
330
  # @return [ContextDev::Models::WebSearchResponse]
@@ -383,327 +407,6 @@ module ContextDev
383
407
  )
384
408
  end
385
409
 
386
- # Some parameter documentations has been truncated, see
387
- # {ContextDev::Models::WebWebScrapeBytesParams} for more details.
388
- #
389
- # Downloads a resource and returns its bytes as base64. Supports images, PDFs,
390
- # HTML pages, and any other content type without image conversion, text
391
- # extraction, or character-encoding changes. HTTP compression is decoded before
392
- # base64 encoding. HTML is the original HTTP response; JavaScript is not rendered.
393
- # Follows public redirects and retries failed downloads through ISP and
394
- # residential proxies, with a direct fallback. When country is specified, only a
395
- # residential proxy in that country is used. Supply headers such as Referer for
396
- # images that require a referring page. Downloads are not cached. Maximum decoded
397
- # resource size: 20 MiB (20971520 bytes), before base64 encoding. Successful
398
- # requests cost 1 credit; errors are not billed.
399
- #
400
- # @overload web_scrape_bytes(url:, country: nil, headers: nil, tags: nil, timeout_opts: nil, zdr: nil, request_options: {})
401
- #
402
- # @param url [String] Full HTTP(S) URL of the resource to download, such as an image, PDF, or page.
403
- #
404
- # @param country [Symbol, ContextDev::Models::WebWebScrapeBytesParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
405
- #
406
- # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers, such as Referer, Cookie, or Authorization. Send
407
- #
408
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
409
- #
410
- # @param timeout_opts [ContextDev::Models::WebWebScrapeBytesParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
411
- #
412
- # @param zdr [Symbol, ContextDev::Models::WebWebScrapeBytesParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
413
- #
414
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
415
- #
416
- # @return [ContextDev::Models::WebWebScrapeBytesResponse]
417
- #
418
- # @see ContextDev::Models::WebWebScrapeBytesParams
419
- def web_scrape_bytes(params)
420
- parsed, options = ContextDev::WebWebScrapeBytesParams.dump_request(params)
421
- query = ContextDev::Internal::Util.encode_query_params(parsed)
422
- @client.request(
423
- method: :get,
424
- path: "web/scrape/bytes",
425
- query: query.transform_keys(timeout_opts: "timeoutOpts"),
426
- model: ContextDev::Models::WebWebScrapeBytesResponse,
427
- options: options
428
- )
429
- end
430
-
431
- # Some parameter documentations has been truncated, see
432
- # {ContextDev::Models::WebWebScrapeHTMLParams} for more details.
433
- #
434
- # Scrapes the given URL and returns the raw HTML content of the page. The base
435
- # request costs 1 credit; requests with browser actions cost 2 credits. A request
436
- # that hits its timeoutOpts.milliseconds deadline fails with 408 and is not
437
- # billed, unless timeoutOpts.behavior=return-partial is set — then the page as
438
- # rendered so far is returned with `finalDOMState: "still-loading"` and billed at
439
- # the base cost of 1 credit.
440
- #
441
- # @overload web_scrape_html(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, tags: nil, timeout_opts: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
442
- #
443
- # @param url [String] Full URL to scrape (must include http:// or https:// protocol)
444
- #
445
- # @param actions [Array<ContextDev::Models::WebWebScrapeHTMLParams::Action::Wait, ContextDev::Models::WebWebScrapeHTMLParams::Action::Perform, ContextDev::Models::WebWebScrapeHTMLParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
446
- #
447
- # @param country [Symbol, ContextDev::Models::WebWebScrapeHTMLParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
448
- #
449
- # @param exclude_selectors [Array<String>, nil] CSS selectors to remove from the result. Applied after includeSelectors. Exclusi
450
- #
451
- # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
452
- #
453
- # @param include_frames [Boolean] When true, iframes are rendered inline into the returned HTML.
454
- #
455
- # @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching subtrees (and their descendants) are
456
- #
457
- # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
458
- #
459
- # @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
460
- #
461
- # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
462
- #
463
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
464
- #
465
- # @param timeout_opts [ContextDev::Models::WebWebScrapeHTMLParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
466
- #
467
- # @param use_main_content_only [Boolean] When true, return only the page's main content in the HTML response, excluding h
468
- #
469
- # @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
470
- #
471
- # @param zdr [Symbol, ContextDev::Models::WebWebScrapeHTMLParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
472
- #
473
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
474
- #
475
- # @return [ContextDev::Models::WebWebScrapeHTMLResponse]
476
- #
477
- # @see ContextDev::Models::WebWebScrapeHTMLParams
478
- def web_scrape_html(params)
479
- parsed, options = ContextDev::WebWebScrapeHTMLParams.dump_request(params)
480
- query = ContextDev::Internal::Util.encode_query_params(parsed)
481
- @client.request(
482
- method: :get,
483
- path: "web/scrape/html",
484
- query: query.transform_keys(
485
- exclude_selectors: "excludeSelectors",
486
- include_frames: "includeFrames",
487
- include_selectors: "includeSelectors",
488
- max_age_ms: "maxAgeMs",
489
- settle_animations: "settleAnimations",
490
- timeout_opts: "timeoutOpts",
491
- use_main_content_only: "useMainContentOnly",
492
- wait_for_ms: "waitForMs"
493
- ),
494
- model: ContextDev::Models::WebWebScrapeHTMLResponse,
495
- options: options
496
- )
497
- end
498
-
499
- # Some parameter documentations has been truncated, see
500
- # {ContextDev::Models::WebWebScrapeImagesParams} for more details.
501
- #
502
- # Extract image assets from a web page, including standard URLs, inline SVGs, data
503
- # URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
504
- # embeds. The base request costs 1 credit, or 2 credits with browser actions. When
505
- # enrichment is enabled, the entire call costs 5 credits, including requests that
506
- # also use actions.
507
- #
508
- # @overload web_scrape_images(url:, actions: nil, dedupe: nil, enrichment: nil, headers: nil, max_age_ms: nil, tags: nil, timeout_opts: nil, wait_for_ms: nil, request_options: {})
509
- #
510
- # @param url [String] Page URL to inspect. Must include http:// or https://.
511
- #
512
- # @param actions [Array<ContextDev::Models::WebWebScrapeImagesParams::Action::Wait, ContextDev::Models::WebWebScrapeImagesParams::Action::Perform, ContextDev::Models::WebWebScrapeImagesParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
513
- #
514
- # @param dedupe [Boolean] When true, visually duplicate images are removed: every image is loaded and perc
515
- #
516
- # @param enrichment [ContextDev::Models::WebWebScrapeImagesParams::Enrichment, nil] Optional per-image processing, sent as deep-object query params such as enrichme
517
- #
518
- # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
519
- #
520
- # @param max_age_ms [Integer, nil] Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
521
- #
522
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
523
- #
524
- # @param timeout_opts [ContextDev::Models::WebWebScrapeImagesParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
525
- #
526
- # @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before collec
527
- #
528
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
529
- #
530
- # @return [ContextDev::Models::WebWebScrapeImagesResponse]
531
- #
532
- # @see ContextDev::Models::WebWebScrapeImagesParams
533
- def web_scrape_images(params)
534
- parsed, options = ContextDev::WebWebScrapeImagesParams.dump_request(params)
535
- query = ContextDev::Internal::Util.encode_query_params(parsed)
536
- @client.request(
537
- method: :get,
538
- path: "web/scrape/images",
539
- query: query.transform_keys(
540
- max_age_ms: "maxAgeMs",
541
- timeout_opts: "timeoutOpts",
542
- wait_for_ms: "waitForMs"
543
- ),
544
- model: ContextDev::Models::WebWebScrapeImagesResponse,
545
- options: options
546
- )
547
- end
548
-
549
- # Some parameter documentations has been truncated, see
550
- # {ContextDev::Models::WebWebScrapeMdParams} for more details.
551
- #
552
- # Scrapes the given URL into LLM usable Markdown. Inspect key_metadata on JSON
553
- # responses from a recognized API key; use error_code to distinguish stable
554
- # failure categories.
555
- #
556
- # ### YouTube
557
- #
558
- # YouTube URLs return the video or channel itself rather than the surrounding
559
- # player and navigation chrome. A URL addressing a single video (`/watch`,
560
- # `youtu.be`, `/shorts`, `/embed`, `/live`) returns its title, channel, duration,
561
- # view count, keywords, full description, and the transcript when the video has
562
- # captions that can be retrieved; videos without captions return everything except
563
- # the transcript. A channel URL (`/channel/UC…`, `/@handle`, `/c/…`, `/user/…`)
564
- # returns its name, handle, subscriber count, video count, and full description.
565
- # When `includeImages=true`, video responses also include the thumbnail and
566
- # channel responses include the avatar. Costs the same as any other scrape.
567
- #
568
- # ### Billing & errors
569
- #
570
- # | HTTP status | Billed? | Meaning |
571
- # | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
572
- # | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing. A partial result (`finalDOMState: "still-loading"`, only with timeoutOpts.behavior=return-partial) is billed at the base 1 credit with no OCR or actions surcharge |
573
- # | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
574
- # | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
575
- # | 404 | No | Target page returned or fingerprinted as not found |
576
- # | 408 | No | Request timed out. With timeoutOpts.behavior=return-partial this only happens when nothing usable had rendered by the deadline |
577
- # | 413 | No | Target content exceeds the maximum supported size (20 MB) |
578
- # | 415 | No | Unsupported content type |
579
- # | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
580
- # | 500 | No | Internal error |
581
- #
582
- # @overload web_scrape_md(url:, actions: nil, country: nil, exclude_selectors: nil, headers: nil, include_frames: nil, include_html: nil, include_images: nil, include_links: nil, include_selectors: nil, max_age_ms: nil, pdf: nil, settle_animations: nil, shorten_base64_images: nil, tags: nil, timeout_opts: nil, use_main_content_only: nil, wait_for_ms: nil, zdr: nil, request_options: {})
583
- #
584
- # @param url [String] Full URL to scrape into LLM usable Markdown (must include http:// or https:// pr
585
- #
586
- # @param actions [Array<ContextDev::Models::WebWebScrapeMdParams::Action::Wait, ContextDev::Models::WebWebScrapeMdParams::Action::Perform, ContextDev::Models::WebWebScrapeMdParams::Action::Scroll>, nil] Optional browser actions executed in array order after the page loads and before
587
- #
588
- # @param country [Symbol, ContextDev::Models::WebWebScrapeMdParams::Country] Fetch the target page through a residential proxy in this country (ISO 3166-1 al
589
- #
590
- # @param exclude_selectors [Array<String>, nil] CSS selectors to remove before conversion to Markdown. Applied after includeSele
591
- #
592
- # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
593
- #
594
- # @param include_frames [Boolean] When true, the contents of iframes are rendered to Markdown.
595
- #
596
- # @param include_html [Boolean] When true, the response also includes an `html` field with the page HTML the Mar
597
- #
598
- # @param include_images [Boolean] Include image references in Markdown output
599
- #
600
- # @param include_links [Boolean] Preserve hyperlinks in Markdown output
601
- #
602
- # @param include_selectors [Array<String>, nil] CSS selectors. When provided, only matching HTML subtrees (and their descendants
603
- #
604
- # @param max_age_ms [Integer, nil] Return a cached result if a prior scrape for the same parameters exists and is y
605
- #
606
- # @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
607
- #
608
- # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
609
- #
610
- # @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
611
- #
612
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
613
- #
614
- # @param timeout_opts [ContextDev::Models::WebWebScrapeMdParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
615
- #
616
- # @param use_main_content_only [Boolean] Extract only the main content of the page, excluding headers, footers, sidebars,
617
- #
618
- # @param wait_for_ms [Integer, nil] Optional browser wait time in milliseconds after initial page load before conver
619
- #
620
- # @param zdr [Symbol, ContextDev::Models::WebWebScrapeMdParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
621
- #
622
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
623
- #
624
- # @return [ContextDev::Models::WebWebScrapeMdResponse]
625
- #
626
- # @see ContextDev::Models::WebWebScrapeMdParams
627
- def web_scrape_md(params)
628
- parsed, options = ContextDev::WebWebScrapeMdParams.dump_request(params)
629
- query = ContextDev::Internal::Util.encode_query_params(parsed)
630
- @client.request(
631
- method: :get,
632
- path: "web/scrape/markdown",
633
- query: query.transform_keys(
634
- exclude_selectors: "excludeSelectors",
635
- include_frames: "includeFrames",
636
- include_html: "includeHTML",
637
- include_images: "includeImages",
638
- include_links: "includeLinks",
639
- include_selectors: "includeSelectors",
640
- max_age_ms: "maxAgeMs",
641
- settle_animations: "settleAnimations",
642
- shorten_base64_images: "shortenBase64Images",
643
- timeout_opts: "timeoutOpts",
644
- use_main_content_only: "useMainContentOnly",
645
- wait_for_ms: "waitForMs"
646
- ),
647
- model: ContextDev::Models::WebWebScrapeMdResponse,
648
- options: options
649
- )
650
- end
651
-
652
- # Some parameter documentations has been truncated, see
653
- # {ContextDev::Models::WebWebScrapeSitemapParams} for more details.
654
- #
655
- # Crawl an entire website's sitemap and return all discovered page URLs. Set
656
- # `includeSubdomains=true` to also discover public pages and sitemaps on child
657
- # hosts such as `docs.example.com` or `brand.example.com`. Pass `search` to have
658
- # the discovered URLs filtered down to the pages about a phrase (for example
659
- # `pricing and plans` or `api authentication docs`), most relevant first — a
660
- # searched crawl scans the whole sitemap and costs 2 credits instead of 1.
661
- #
662
- # @overload web_scrape_sitemap(domain:, headers: nil, include_subdomains: nil, max_links: nil, search: nil, sitemap_url: nil, tags: nil, timeout_opts: nil, url_regex: nil, zdr: nil, request_options: {})
663
- #
664
- # @param domain [String] Domain to build a sitemap for
665
- #
666
- # @param headers [Hash{Symbol=>String}] Optional outbound HTTP headers forwarded only to the target URL, sent as deep-ob
667
- #
668
- # @param include_subdomains [Boolean] When true, discover and include public pages and sitemaps on subdomains of the r
669
- #
670
- # @param max_links [Integer] Maximum number of links to return from the sitemap crawl. Defaults to 10,000. Mi
671
- #
672
- # @param search [String] Optional search phrase. When provided, the crawled sitemap is filtered to the pa
673
- #
674
- # @param sitemap_url [String] Optional explicit sitemap URL. When provided, exactly this sitemap is crawled in
675
- #
676
- # @param tags [Array<String>] Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50 charac
677
- #
678
- # @param timeout_opts [ContextDev::Models::WebWebScrapeSitemapParams::TimeoutOpts] Optional request deadline and behavior on timeout. For GET requests, use timeout
679
- #
680
- # @param url_regex [String] Optional RE2-compatible regex pattern. Only URLs matching this pattern are retur
681
- #
682
- # @param zdr [Symbol, ContextDev::Models::WebWebScrapeSitemapParams::Zdr] Set to enabled to bypass shared caches and omit request and response content fro
683
- #
684
- # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}, nil]
685
- #
686
- # @return [ContextDev::Models::WebWebScrapeSitemapResponse]
687
- #
688
- # @see ContextDev::Models::WebWebScrapeSitemapParams
689
- def web_scrape_sitemap(params)
690
- parsed, options = ContextDev::WebWebScrapeSitemapParams.dump_request(params)
691
- query = ContextDev::Internal::Util.encode_query_params(parsed)
692
- @client.request(
693
- method: :get,
694
- path: "web/scrape/sitemap",
695
- query: query.transform_keys(
696
- include_subdomains: "includeSubdomains",
697
- max_links: "maxLinks",
698
- sitemap_url: "sitemapUrl",
699
- timeout_opts: "timeoutOpts",
700
- url_regex: "urlRegex"
701
- ),
702
- model: ContextDev::Models::WebWebScrapeSitemapResponse,
703
- options: options
704
- )
705
- end
706
-
707
410
  # @api private
708
411
  #
709
412
  # @param client [ContextDev::Client]