context.dev 2.17.1 → 2.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (139) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +33 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +0 -4
  5. data/lib/context_dev/models/industry_retrieve_naics_params.rb +28 -1
  6. data/lib/context_dev/models/industry_retrieve_sic_params.rb +28 -1
  7. data/lib/context_dev/models/monitor_retrieve_run_params.rb +26 -0
  8. data/lib/context_dev/models/monitor_retrieve_run_response.rb +232 -0
  9. data/lib/context_dev/models/monitor_rotate_webhook_secret_params.rb +20 -0
  10. data/lib/context_dev/models/monitor_rotate_webhook_secret_response.rb +765 -0
  11. data/lib/context_dev/models/parse_handle_params.rb +8 -6
  12. data/lib/context_dev/models/person_enrich_params.rb +28 -1
  13. data/lib/context_dev/models/web_answers_params.rb +28 -1
  14. data/lib/context_dev/models/web_extract_competitors_params.rb +28 -1
  15. data/lib/context_dev/models/web_extract_styleguide_params.rb +30 -3
  16. data/lib/context_dev/models/{web_web_scrape_sitemap_params.rb → web_map_urls_params.rb} +22 -20
  17. data/lib/context_dev/models/web_map_urls_response.rb +123 -0
  18. data/lib/context_dev/models/web_scrape_params.rb +906 -0
  19. data/lib/context_dev/models/web_scrape_response.rb +659 -0
  20. data/lib/context_dev/models/web_screenshot_params.rb +25 -9
  21. data/lib/context_dev/models/web_screenshot_response.rb +3 -3
  22. data/lib/context_dev/models/web_search_params.rb +30 -3
  23. data/lib/context_dev/models.rb +8 -20
  24. data/lib/context_dev/resources/brand.rb +0 -36
  25. data/lib/context_dev/resources/industry.rb +6 -2
  26. data/lib/context_dev/resources/monitors.rb +47 -0
  27. data/lib/context_dev/resources/people.rb +3 -1
  28. data/lib/context_dev/resources/web.rb +116 -413
  29. data/lib/context_dev/version.rb +1 -1
  30. data/lib/context_dev.rb +8 -21
  31. data/rbi/context_dev/client.rbi +0 -3
  32. data/rbi/context_dev/models/industry_retrieve_naics_params.rbi +59 -0
  33. data/rbi/context_dev/models/industry_retrieve_sic_params.rbi +57 -0
  34. data/rbi/context_dev/models/monitor_retrieve_run_params.rbi +46 -0
  35. data/rbi/context_dev/models/monitor_retrieve_run_response.rbi +452 -0
  36. data/rbi/context_dev/models/monitor_rotate_webhook_secret_params.rbi +38 -0
  37. data/rbi/context_dev/models/monitor_rotate_webhook_secret_response.rbi +1367 -0
  38. data/rbi/context_dev/models/parse_handle_params.rbi +12 -9
  39. data/rbi/context_dev/models/person_enrich_params.rbi +45 -0
  40. data/rbi/context_dev/models/web_answers_params.rbi +45 -0
  41. data/rbi/context_dev/models/web_extract_competitors_params.rbi +59 -0
  42. data/rbi/context_dev/models/web_extract_styleguide_params.rbi +62 -3
  43. data/rbi/context_dev/models/{web_web_scrape_sitemap_params.rbi → web_map_urls_params.rbi} +35 -54
  44. data/rbi/context_dev/models/web_map_urls_response.rbi +226 -0
  45. data/rbi/context_dev/models/web_scrape_params.rbi +2000 -0
  46. data/rbi/context_dev/models/{web_web_scrape_html_response.rbi → web_scrape_response.rbi} +550 -418
  47. data/rbi/context_dev/models/web_screenshot_params.rbi +38 -12
  48. data/rbi/context_dev/models/web_screenshot_response.rbi +4 -4
  49. data/rbi/context_dev/models/web_search_params.rbi +48 -3
  50. data/rbi/context_dev/models.rbi +9 -21
  51. data/rbi/context_dev/resources/brand.rbi +0 -35
  52. data/rbi/context_dev/resources/industry.rbi +14 -0
  53. data/rbi/context_dev/resources/monitors.rbi +24 -0
  54. data/rbi/context_dev/resources/parse.rbi +4 -4
  55. data/rbi/context_dev/resources/people.rbi +7 -0
  56. data/rbi/context_dev/resources/web.rbi +166 -533
  57. data/sig/context_dev/client.rbs +0 -2
  58. data/sig/context_dev/models/industry_retrieve_naics_params.rbs +21 -1
  59. data/sig/context_dev/models/industry_retrieve_sic_params.rbs +21 -1
  60. data/sig/context_dev/models/monitor_retrieve_run_params.rbs +28 -0
  61. data/sig/context_dev/models/monitor_retrieve_run_response.rbs +182 -0
  62. data/sig/context_dev/models/monitor_rotate_webhook_secret_params.rbs +23 -0
  63. data/sig/context_dev/models/monitor_rotate_webhook_secret_response.rbs +545 -0
  64. data/sig/context_dev/models/person_enrich_params.rbs +21 -1
  65. data/sig/context_dev/models/web_answers_params.rbs +21 -1
  66. data/sig/context_dev/models/web_extract_competitors_params.rbs +21 -1
  67. data/sig/context_dev/models/web_extract_styleguide_params.rbs +21 -1
  68. data/sig/context_dev/models/{web_web_scrape_sitemap_params.rbs → web_map_urls_params.rbs} +22 -22
  69. data/sig/context_dev/models/web_map_urls_response.rbs +125 -0
  70. data/sig/context_dev/models/web_scrape_params.rbs +834 -0
  71. data/sig/context_dev/models/web_scrape_response.rbs +552 -0
  72. data/sig/context_dev/models/web_screenshot_params.rbs +7 -0
  73. data/sig/context_dev/models/web_search_params.rbs +21 -1
  74. data/sig/context_dev/models.rbs +8 -20
  75. data/sig/context_dev/resources/brand.rbs +0 -9
  76. data/sig/context_dev/resources/industry.rbs +2 -0
  77. data/sig/context_dev/resources/monitors.rbs +11 -0
  78. data/sig/context_dev/resources/people.rbs +1 -0
  79. data/sig/context_dev/resources/web.rbs +34 -108
  80. metadata +26 -65
  81. data/lib/context_dev/models/ai_extract_product_params.rb +0 -98
  82. data/lib/context_dev/models/ai_extract_product_response.rb +0 -352
  83. data/lib/context_dev/models/ai_extract_products_params.rb +0 -233
  84. data/lib/context_dev/models/ai_extract_products_response.rb +0 -320
  85. data/lib/context_dev/models/brand_retrieve_simplified_params.rb +0 -120
  86. data/lib/context_dev/models/brand_retrieve_simplified_response.rb +0 -441
  87. data/lib/context_dev/models/web_extract_fonts_params.rb +0 -114
  88. data/lib/context_dev/models/web_extract_fonts_response.rb +0 -291
  89. data/lib/context_dev/models/web_extract_params.rb +0 -392
  90. data/lib/context_dev/models/web_extract_response.rb +0 -287
  91. data/lib/context_dev/models/web_web_scrape_bytes_params.rb +0 -344
  92. data/lib/context_dev/models/web_web_scrape_bytes_response.rb +0 -135
  93. data/lib/context_dev/models/web_web_scrape_html_params.rb +0 -633
  94. data/lib/context_dev/models/web_web_scrape_html_response.rb +0 -588
  95. data/lib/context_dev/models/web_web_scrape_images_params.rb +0 -346
  96. data/lib/context_dev/models/web_web_scrape_images_response.rb +0 -409
  97. data/lib/context_dev/models/web_web_scrape_md_params.rb +0 -668
  98. data/lib/context_dev/models/web_web_scrape_md_response.rb +0 -567
  99. data/lib/context_dev/models/web_web_scrape_sitemap_response.rb +0 -144
  100. data/lib/context_dev/resources/ai.rb +0 -69
  101. data/rbi/context_dev/models/ai_extract_product_params.rbi +0 -199
  102. data/rbi/context_dev/models/ai_extract_product_response.rbi +0 -696
  103. data/rbi/context_dev/models/ai_extract_products_params.rbi +0 -490
  104. data/rbi/context_dev/models/ai_extract_products_response.rbi +0 -623
  105. data/rbi/context_dev/models/brand_retrieve_simplified_params.rbi +0 -256
  106. data/rbi/context_dev/models/brand_retrieve_simplified_response.rbi +0 -1040
  107. data/rbi/context_dev/models/web_extract_fonts_params.rbi +0 -223
  108. data/rbi/context_dev/models/web_extract_fonts_response.rbi +0 -563
  109. data/rbi/context_dev/models/web_extract_params.rbi +0 -736
  110. data/rbi/context_dev/models/web_extract_response.rbi +0 -512
  111. data/rbi/context_dev/models/web_web_scrape_bytes_params.rbi +0 -699
  112. data/rbi/context_dev/models/web_web_scrape_bytes_response.rbi +0 -238
  113. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +0 -1217
  114. data/rbi/context_dev/models/web_web_scrape_images_params.rbi +0 -671
  115. data/rbi/context_dev/models/web_web_scrape_images_response.rbi +0 -910
  116. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +0 -1049
  117. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +0 -1088
  118. data/rbi/context_dev/models/web_web_scrape_sitemap_response.rbi +0 -258
  119. data/rbi/context_dev/resources/ai.rbi +0 -56
  120. data/sig/context_dev/models/ai_extract_product_params.rbs +0 -86
  121. data/sig/context_dev/models/ai_extract_product_response.rbs +0 -273
  122. data/sig/context_dev/models/ai_extract_products_params.rbs +0 -202
  123. data/sig/context_dev/models/ai_extract_products_response.rbs +0 -252
  124. data/sig/context_dev/models/brand_retrieve_simplified_params.rbs +0 -104
  125. data/sig/context_dev/models/brand_retrieve_simplified_response.rbs +0 -413
  126. data/sig/context_dev/models/web_extract_fonts_params.rbs +0 -93
  127. data/sig/context_dev/models/web_extract_fonts_response.rbs +0 -230
  128. data/sig/context_dev/models/web_extract_params.rbs +0 -304
  129. data/sig/context_dev/models/web_extract_response.rbs +0 -239
  130. data/sig/context_dev/models/web_web_scrape_bytes_params.rbs +0 -531
  131. data/sig/context_dev/models/web_web_scrape_bytes_response.rbs +0 -108
  132. data/sig/context_dev/models/web_web_scrape_html_params.rbs +0 -730
  133. data/sig/context_dev/models/web_web_scrape_html_response.rbs +0 -495
  134. data/sig/context_dev/models/web_web_scrape_images_params.rbs +0 -261
  135. data/sig/context_dev/models/web_web_scrape_images_response.rbs +0 -371
  136. data/sig/context_dev/models/web_web_scrape_md_params.rbs +0 -758
  137. data/sig/context_dev/models/web_web_scrape_md_response.rbs +0 -465
  138. data/sig/context_dev/models/web_web_scrape_sitemap_response.rbs +0 -117
  139. data/sig/context_dev/resources/ai.rbs +0 -20
@@ -15,6 +15,7 @@ module ContextDev
15
15
  mode: ContextDev::WebAnswersParams::Mode::OrSymbol,
16
16
  tags: T::Array[String],
17
17
  timeout_opts: ContextDev::WebAnswersParams::TimeoutOpts::OrHash,
18
+ zdr: ContextDev::WebAnswersParams::Zdr::OrSymbol,
18
19
  request_options: ContextDev::RequestOptions::OrHash
19
20
  ).returns(ContextDev::Models::WebAnswersResponse)
20
21
  end
@@ -38,95 +39,12 @@ module ContextDev
38
39
  # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
39
40
  # timeoutOpts object.
40
41
  timeout_opts: nil,
41
- request_options: {}
42
- )
43
- end
44
-
45
- # Crawl a website, use the provided JSON Schema and instructions to prioritize
46
- # relevant internal links, and extract structured data from the selected pages.
47
- sig do
48
- params(
49
- schema: T::Hash[Symbol, T.anything],
50
- url: String,
51
- actions:
52
- T::Array[
53
- T.any(
54
- ContextDev::WebExtractParams::Action::Wait::OrHash,
55
- ContextDev::WebExtractParams::Action::Perform::OrHash,
56
- ContextDev::WebExtractParams::Action::Scroll::OrHash
57
- )
58
- ],
59
- fact_check: T::Boolean,
60
- follow_subdomains: T::Boolean,
61
- include_frames: T::Boolean,
62
- instructions: String,
63
- max_age_ms: Integer,
64
- max_depth: Integer,
65
- max_pages: Integer,
66
- pdf: ContextDev::WebExtractParams::Pdf::OrHash,
67
- settle_animations: T::Boolean,
68
- stop_after_ms: Integer,
69
- tags: T::Array[String],
70
- timeout_opts: ContextDev::WebExtractParams::TimeoutOpts::OrHash,
71
- wait_for_ms: Integer,
72
- request_options: ContextDev::RequestOptions::OrHash
73
- ).returns(ContextDev::Models::WebExtractResponse)
74
- end
75
- def extract(
76
- # JSON Schema for the returned data object. Image fields such as `image_urls` or
77
- # `product_photos` automatically make page image references available to
78
- # extraction, so product data and photos can be returned in one call. TypeScript
79
- # Zod users can pass a JSON Schema generated from a Zod object; Python users can
80
- # pass the equivalent JSON Schema object.
81
- schema:,
82
- # The starting website URL to crawl and extract from. Must include http:// or
83
- # https://.
84
- url:,
85
- # Optional browser actions executed in order on the requested page after it loads,
86
- # before links are discovered or additional pages are crawled. Requires a paid
87
- # plan. When actions are provided and stopAfterMs is omitted, the crawl budget
88
- # defaults to 110000 ms.
89
- actions: nil,
90
- # When true, every returned value must be grounded in facts stated on the page;
91
- # fields that cannot be supported by the page are returned as null/empty. When
92
- # false (default), the model may make reasonable inferences and derivations from
93
- # the page content (e.g. ideal customer, competitor analysis, recommendations)
94
- # while keeping verifiable specifics (names, quotes, URLs, dates, metrics)
95
- # faithful to the source.
96
- fact_check: nil,
97
- # When true, follow links on subdomains of the starting URL's domain.
98
- follow_subdomains: nil,
99
- # When true, iframe contents are included in Markdown before extraction.
100
- include_frames: nil,
101
- # Optional extraction guidance, such as which facts to prioritize or how to
102
- # interpret fields in the schema.
103
- instructions: nil,
104
- # Return cached scrape results if a prior scrape for the same parameters is
105
- # younger than this many milliseconds. Defaults to 7 days (604800000 ms).
106
- max_age_ms: nil,
107
- # Optional maximum link depth from the starting URL (0 = only the starting page).
108
- # If omitted, there is no crawl depth limit.
109
- max_depth: nil,
110
- # Maximum number of pages to analyze for extraction. Hard cap: 50. Defaults to 5.
111
- max_pages: nil,
112
- pdf: nil,
113
- # When true, waits briefly for CSS and transition animations to settle before
114
- # extracting each crawled page. Defaults to false. This adds a bit of latency in
115
- # exchange for more stable output on animated pages.
116
- settle_animations: nil,
117
- # Soft time budget for the crawl in milliseconds. Min: 10000 (10s). Max: 110000
118
- # (110s). Defaults to 80000 (80s), or 110000 (110s) when browser actions are
119
- # provided.
120
- stop_after_ms: nil,
121
- # Optional tags for tracking usage. Up to 20 tags, each 1 to 50 characters.
122
- tags: nil,
123
- # Optional request deadline and behavior on timeout. For GET requests, use
124
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
125
- # timeoutOpts object.
126
- timeout_opts: nil,
127
- # Optional browser wait time in milliseconds after initial page load for each
128
- # crawled page.
129
- wait_for_ms: nil,
42
+ # Set to enabled to bypass shared caches and omit request and response content
43
+ # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
44
+ # omitted. Requires zero data retention to be enabled for your organization
45
+ # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
46
+ # Successful ZDR responses include X-Context-ZDR: true.
47
+ zdr: nil,
130
48
  request_options: {}
131
49
  )
132
50
  end
@@ -140,6 +58,7 @@ module ContextDev
140
58
  tags: T::Array[String],
141
59
  timeout_opts:
142
60
  ContextDev::WebExtractCompetitorsParams::TimeoutOpts::OrHash,
61
+ zdr: ContextDev::WebExtractCompetitorsParams::Zdr::OrSymbol,
143
62
  request_options: ContextDev::RequestOptions::OrHash
144
63
  ).returns(ContextDev::Models::WebExtractCompetitorsResponse)
145
64
  end
@@ -156,28 +75,42 @@ module ContextDev
156
75
  # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
157
76
  # timeoutOpts object.
158
77
  timeout_opts: nil,
78
+ # Set to enabled to bypass shared caches and omit request and response content
79
+ # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
80
+ # omitted. Requires zero data retention to be enabled for your organization
81
+ # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
82
+ # Successful ZDR responses include X-Context-ZDR: true.
83
+ zdr: nil,
159
84
  request_options: {}
160
85
  )
161
86
  end
162
87
 
163
- # Scrape font information from a website including font families, usage
164
- # statistics, fallbacks, and element/word counts.
88
+ # Extract a comprehensive design system from a website including colors,
89
+ # typography, spacing, shadows, and UI components.
165
90
  sig do
166
91
  params(
92
+ color_scheme:
93
+ ContextDev::WebExtractStyleguideParams::ColorScheme::OrSymbol,
167
94
  direct_url: String,
168
95
  domain: String,
169
96
  max_age_ms: T.nilable(Integer),
170
97
  tags: T::Array[String],
171
- timeout_opts: ContextDev::WebExtractFontsParams::TimeoutOpts::OrHash,
98
+ timeout_opts:
99
+ ContextDev::WebExtractStyleguideParams::TimeoutOpts::OrHash,
100
+ zdr: ContextDev::WebExtractStyleguideParams::Zdr::OrSymbol,
172
101
  request_options: ContextDev::RequestOptions::OrHash
173
- ).returns(ContextDev::Models::WebExtractFontsResponse)
102
+ ).returns(ContextDev::Models::WebExtractStyleguideResponse)
174
103
  end
175
- def extract_fonts(
176
- # A specific URL to fetch fonts from directly, bypassing domain resolution (e.g.,
177
- # 'https://example.com/design-system'). When provided, fonts are extracted from
178
- # this exact URL. You must provide either 'domain' or 'directUrl', but not both.
104
+ def extract_styleguide(
105
+ # Optional browser color scheme to emulate for websites that respond to
106
+ # prefers-color-scheme. This value is part of the styleguide cache key.
107
+ color_scheme: nil,
108
+ # A specific URL to fetch the styleguide from directly, bypassing domain
109
+ # resolution (e.g., 'https://example.com/design-system'). When provided, the
110
+ # styleguide is extracted from this exact URL. You must provide either 'domain' or
111
+ # 'directUrl', but not both.
179
112
  direct_url: nil,
180
- # Domain name to extract fonts from (e.g., 'example.com', 'google.com'). The
113
+ # Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
181
114
  # domain will be automatically normalized and validated. You must provide either
182
115
  # 'domain' or 'directUrl', but not both.
183
116
  domain: nil,
@@ -193,43 +126,59 @@ module ContextDev
193
126
  # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
194
127
  # timeoutOpts object.
195
128
  timeout_opts: nil,
129
+ # Set to enabled to bypass shared caches and omit request and response content
130
+ # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
131
+ # omitted. Requires zero data retention to be enabled for your organization
132
+ # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
133
+ # Successful ZDR responses include X-Context-ZDR: true.
134
+ zdr: nil,
196
135
  request_options: {}
197
136
  )
198
137
  end
199
138
 
200
- # Extract a comprehensive design system from a website including colors,
201
- # typography, spacing, shadows, and UI components.
139
+ # Discovers URLs using the same sitemap crawl, filters, and limits as
140
+ # /web/scrape/sitemap. Each URL includes its available title, description,
141
+ # keywords, and language. URLs without stored enrichment are returned immediately
142
+ # with only the URL and queued for background HTML scraping, so later requests can
143
+ # include their metadata. Responses are never cached as a whole; every request
144
+ # reads the current per-URL enrichment. Zero data retention and credential-bearing
145
+ # discovery requests return URLs without reading or storing shared enrichment or
146
+ # queuing background scrapes. Costs 1 credit, or 2 credits with search.
202
147
  sig do
203
148
  params(
204
- color_scheme:
205
- ContextDev::WebExtractStyleguideParams::ColorScheme::OrSymbol,
206
- direct_url: String,
207
149
  domain: String,
208
- max_age_ms: T.nilable(Integer),
150
+ headers: T::Hash[Symbol, String],
151
+ include_subdomains: T::Boolean,
152
+ max_links: Integer,
153
+ search: String,
154
+ sitemap_url: String,
209
155
  tags: T::Array[String],
210
- timeout_opts:
211
- ContextDev::WebExtractStyleguideParams::TimeoutOpts::OrHash,
156
+ timeout_opts: ContextDev::WebMapURLsParams::TimeoutOpts::OrHash,
157
+ url_regex: String,
158
+ zdr: ContextDev::WebMapURLsParams::Zdr::OrSymbol,
212
159
  request_options: ContextDev::RequestOptions::OrHash
213
- ).returns(ContextDev::Models::WebExtractStyleguideResponse)
160
+ ).returns(ContextDev::Models::WebMapURLsResponse)
214
161
  end
215
- def extract_styleguide(
216
- # Optional browser color scheme to emulate for websites that respond to
217
- # prefers-color-scheme. This value is part of the styleguide cache key.
218
- color_scheme: nil,
219
- # A specific URL to fetch the styleguide from directly, bypassing domain
220
- # resolution (e.g., 'https://example.com/design-system'). When provided, the
221
- # styleguide is extracted from this exact URL. You must provide either 'domain' or
222
- # 'directUrl', but not both.
223
- direct_url: nil,
224
- # Domain name to extract styleguide from (e.g., 'example.com', 'google.com'). The
225
- # domain will be automatically normalized and validated. You must provide either
226
- # 'domain' or 'directUrl', but not both.
227
- domain: nil,
228
- # Maximum age in milliseconds for cached brand data before the API performs a hard
229
- # refresh. Defaults to 3 months (7776000000 ms). Set to 0 to always perform a hard
230
- # refresh. Negative values are clamped to 0; values above 1 year (31536000000 ms)
231
- # are clamped to 1 year.
232
- max_age_ms: nil,
162
+ def map_urls(
163
+ # Domain to build a sitemap for
164
+ domain:,
165
+ # Optional outbound HTTP headers forwarded only to the target URL, sent as
166
+ # deep-object query params such as headers[X-Custom]=value. When provided, caching
167
+ # is bypassed: the result is neither read from nor written to cache.
168
+ headers: nil,
169
+ # When true, discover and include public pages and sitemaps on subdomains of the
170
+ # requested domain. Defaults to false.
171
+ include_subdomains: nil,
172
+ # Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
173
+ # Minimum is 1, maximum is 100,000.
174
+ max_links: nil,
175
+ # Optional search phrase. When provided, the crawled sitemap is filtered to the
176
+ # pages whose URLs are about that phrase, most relevant first, and the request
177
+ # costs 2 credits instead of 1.
178
+ search: nil,
179
+ # Optional explicit sitemap URL. When provided, exactly this sitemap is crawled
180
+ # instead of discovering the domain's sitemaps.
181
+ sitemap_url: nil,
233
182
  # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
234
183
  # characters.
235
184
  tags: nil,
@@ -237,6 +186,78 @@ module ContextDev
237
186
  # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
238
187
  # timeoutOpts object.
239
188
  timeout_opts: nil,
189
+ # Optional RE2-compatible regex pattern. Only URLs matching this pattern are
190
+ # returned and counted against maxLinks.
191
+ url_regex: nil,
192
+ # Set to enabled to bypass shared caches and omit request and response content
193
+ # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
194
+ # omitted. Requires zero data retention to be enabled for your organization
195
+ # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
196
+ # Successful ZDR responses include X-Context-ZDR: true.
197
+ zdr: nil,
198
+ request_options: {}
199
+ )
200
+ end
201
+
202
+ # Reuse cached outputs independently and capture missing formats in one page
203
+ # visit. Each cache key includes only the settings that affect that output. HTML
204
+ # is shared with Markdown and parsed fields. Cached outputs can come from
205
+ # different visits within maxAgeMs; use 0 for a fresh capture. HTML-only requests
206
+ # use the existing fast acquisition path. One credit per request, including cache
207
+ # hits, or two with browser actions; PDF OCR adds one credit per recovered page on
208
+ # fresh extraction. Original response bytes and screenshots are limited to 20 MiB
209
+ # each, screenshots to 40 megapixels, and the combined browser capture to 60 MiB.
210
+ sig do
211
+ params(
212
+ formats: ContextDev::WebScrapeParams::Formats::OrHash,
213
+ url: String,
214
+ image_params: ContextDev::WebScrapeParams::ImageParams::OrHash,
215
+ markdown_params: ContextDev::WebScrapeParams::MarkdownParams::OrHash,
216
+ max_age_ms: Integer,
217
+ parse_params: ContextDev::WebScrapeParams::ParseParams::OrHash,
218
+ screenshot_params:
219
+ ContextDev::WebScrapeParams::ScreenshotParams::OrHash,
220
+ shared_params: ContextDev::WebScrapeParams::SharedParams::OrHash,
221
+ tags: T::Array[String],
222
+ timeout_opts: ContextDev::WebScrapeParams::TimeoutOpts::OrHash,
223
+ zdr: ContextDev::WebScrapeParams::Zdr::OrSymbol,
224
+ request_options: ContextDev::RequestOptions::OrHash
225
+ ).returns(ContextDev::Models::WebScrapeResponse)
226
+ end
227
+ def scrape(
228
+ # Outputs to return. Enable at least one; omitted formats are false.
229
+ formats:,
230
+ # The URL to scrape.
231
+ url:,
232
+ # Image options. Requires formats.images: true.
233
+ image_params: nil,
234
+ # Markdown options. Requires formats.markdown: true.
235
+ markdown_params: nil,
236
+ # Maximum age of each cached output. Defaults to 1 day; 0 fetches fresh and
237
+ # updates the requested outputs. Compatible outputs are shared with the individual
238
+ # scrape endpoints. Image results with hosted files refresh after 23 hours; other
239
+ # outputs retain their own freshness.
240
+ max_age_ms: nil,
241
+ # Required when formats.parse is true.
242
+ parse_params: nil,
243
+ # Screenshot options. Requires formats.screenshot: true.
244
+ screenshot_params: nil,
245
+ # Shared browser and content settings. Content filters leave screenshots and
246
+ # original bytes unchanged.
247
+ shared_params: nil,
248
+ # Labels for tracking request usage. Not retained when zdr is enabled.
249
+ tags: nil,
250
+ # Total deadline, including navigation, actions, waiting, and all outputs.
251
+ # Defaults to 60000 milliseconds with behavior fail. Use return-partial to capture
252
+ # the current page state and return captured images if image processing cannot
253
+ # finish before the deadline; these responses set isPartial and are not cached.
254
+ # Every requested format must still be available. Fixed waits must fit before a
255
+ # response reserve of up to 5000 milliseconds (at most one quarter of the timeout)
256
+ # when using return-partial.
257
+ timeout_opts: nil,
258
+ # Zero data retention. Bypasses caches and uploads; excludes request/response
259
+ # content and tags from logs. Must be enabled for your organization.
260
+ zdr: nil,
240
261
  request_options: {}
241
262
  )
242
263
  end
@@ -252,6 +273,7 @@ module ContextDev
252
273
  full_screenshot:
253
274
  ContextDev::WebScreenshotParams::FullScreenshot::OrSymbol,
254
275
  handle_cookie_popup: T::Boolean,
276
+ headers: T::Hash[Symbol, String],
255
277
  max_age_ms: T.nilable(Integer),
256
278
  page: ContextDev::WebScreenshotParams::Page::OrSymbol,
257
279
  scroll_offset: T.nilable(Integer),
@@ -292,6 +314,14 @@ module ContextDev
292
314
  # dismiss cookie banner before capture. If 'false' or not provided, captures the
293
315
  # page without that step.
294
316
  handle_cookie_popup: nil,
317
+ # Optional outbound HTTP headers, using the same JSON object or deep-object query
318
+ # format as other scrape endpoints (for example headers[Authorization]=Bearer
319
+ # token). Headers are scoped to the target origin during capture. For domain/page
320
+ # requests, discovery receives no custom headers and only pages on the resolved
321
+ # origin are eligible. Non-empty headers bypass screenshot caching and return an
322
+ # in-memory data URL; no screenshot is uploaded. Empty objects behave like omitted
323
+ # headers.
324
+ headers: nil,
295
325
  # Return a cached screenshot if a prior screenshot for the same parameters exists
296
326
  # and is younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
297
327
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always capture fresh.
@@ -325,9 +355,10 @@ module ContextDev
325
355
  # TIMEOUT_TOO_SHORT_FOR_WAIT.
326
356
  wait_for_ms: nil,
327
357
  # Set to enabled to bypass shared caches and omit request and response content
328
- # from retained usage logs. Requires zero data retention to be enabled for your
329
- # organization (contact support@context.dev), otherwise the request fails with
330
- # ZDR_NOT_ENABLED. Successful ZDR responses include X-Context-ZDR: true.
358
+ # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
359
+ # omitted. Requires zero data retention to be enabled for your organization
360
+ # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
361
+ # Successful ZDR responses include X-Context-ZDR: true.
331
362
  zdr: nil,
332
363
  request_options: {}
333
364
  )
@@ -347,6 +378,7 @@ module ContextDev
347
378
  query_fanout: T::Boolean,
348
379
  tags: T::Array[String],
349
380
  timeout_opts: ContextDev::WebSearchParams::TimeoutOpts::OrHash,
381
+ zdr: ContextDev::WebSearchParams::Zdr::OrSymbol,
350
382
  request_options: ContextDev::RequestOptions::OrHash
351
383
  ).returns(ContextDev::Models::WebSearchResponse)
352
384
  end
@@ -377,6 +409,12 @@ module ContextDev
377
409
  # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
378
410
  # timeoutOpts object.
379
411
  timeout_opts: nil,
412
+ # Set to enabled to bypass shared caches and omit request and response content
413
+ # from retained usage logs. Asset uploads are skipped, so hosted image URLs are
414
+ # omitted. Requires zero data retention to be enabled for your organization
415
+ # (contact support@context.dev), otherwise the request fails with ZDR_NOT_ENABLED.
416
+ # Successful ZDR responses include X-Context-ZDR: true.
417
+ zdr: nil,
380
418
  request_options: {}
381
419
  )
382
420
  end
@@ -482,411 +520,6 @@ module ContextDev
482
520
  )
483
521
  end
484
522
 
485
- # Downloads a resource and returns its bytes as base64. Supports images, PDFs,
486
- # HTML pages, and any other content type without image conversion, text
487
- # extraction, or character-encoding changes. HTTP compression is decoded before
488
- # base64 encoding. HTML is the original HTTP response; JavaScript is not rendered.
489
- # Follows public redirects and retries failed downloads through ISP and
490
- # residential proxies, with a direct fallback. When country is specified, only a
491
- # residential proxy in that country is used. Supply headers such as Referer for
492
- # images that require a referring page. Downloads are not cached. Maximum decoded
493
- # resource size: 20 MiB (20971520 bytes), before base64 encoding. Successful
494
- # requests cost 1 credit; errors are not billed.
495
- sig do
496
- params(
497
- url: String,
498
- country: ContextDev::WebWebScrapeBytesParams::Country::OrSymbol,
499
- headers: T::Hash[Symbol, String],
500
- tags: T::Array[String],
501
- timeout_opts:
502
- ContextDev::WebWebScrapeBytesParams::TimeoutOpts::OrHash,
503
- zdr: ContextDev::WebWebScrapeBytesParams::Zdr::OrSymbol,
504
- request_options: ContextDev::RequestOptions::OrHash
505
- ).returns(ContextDev::Models::WebWebScrapeBytesResponse)
506
- end
507
- def web_scrape_bytes(
508
- # Full HTTP(S) URL of the resource to download, such as an image, PDF, or page.
509
- url:,
510
- # Fetch the target page through a residential proxy in this country (ISO 3166-1
511
- # alpha-2).
512
- country: nil,
513
- # Optional outbound HTTP headers, such as Referer, Cookie, or Authorization. Send
514
- # as a JSON object or deep-object query params such as
515
- # headers[Referer]=https://example.com/. Host, Content-Length, and hop-by-hop
516
- # transport headers are rejected. Authorization and cookies are removed when a
517
- # redirect changes origin.
518
- headers: nil,
519
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
520
- # characters.
521
- tags: nil,
522
- # Optional request deadline and behavior on timeout. For GET requests, use
523
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
524
- # timeoutOpts object.
525
- timeout_opts: nil,
526
- # Set to enabled to bypass shared caches and omit request and response content
527
- # from retained usage logs. Requires zero data retention to be enabled for your
528
- # organization (contact support@context.dev), otherwise the request fails with
529
- # ZDR_NOT_ENABLED. Successful ZDR responses include X-Context-ZDR: true.
530
- zdr: nil,
531
- request_options: {}
532
- )
533
- end
534
-
535
- # Scrapes the given URL and returns the raw HTML content of the page. The base
536
- # request costs 1 credit; requests with browser actions cost 2 credits. A request
537
- # that hits its timeoutOpts.milliseconds deadline fails with 408 and is not
538
- # billed, unless timeoutOpts.behavior=return-partial is set — then the page as
539
- # rendered so far is returned with `finalDOMState: "still-loading"` and billed at
540
- # the base cost of 1 credit.
541
- sig do
542
- params(
543
- url: String,
544
- actions:
545
- T.nilable(
546
- T::Array[
547
- T.any(
548
- ContextDev::WebWebScrapeHTMLParams::Action::Wait::OrHash,
549
- ContextDev::WebWebScrapeHTMLParams::Action::Perform::OrHash,
550
- ContextDev::WebWebScrapeHTMLParams::Action::Scroll::OrHash
551
- )
552
- ]
553
- ),
554
- country: ContextDev::WebWebScrapeHTMLParams::Country::OrSymbol,
555
- exclude_selectors: T.nilable(T::Array[String]),
556
- headers: T::Hash[Symbol, String],
557
- include_frames: T::Boolean,
558
- include_selectors: T.nilable(T::Array[String]),
559
- max_age_ms: T.nilable(Integer),
560
- pdf: ContextDev::WebWebScrapeHTMLParams::Pdf::OrHash,
561
- settle_animations: T::Boolean,
562
- tags: T::Array[String],
563
- timeout_opts: ContextDev::WebWebScrapeHTMLParams::TimeoutOpts::OrHash,
564
- use_main_content_only: T::Boolean,
565
- wait_for_ms: T.nilable(Integer),
566
- zdr: ContextDev::WebWebScrapeHTMLParams::Zdr::OrSymbol,
567
- request_options: ContextDev::RequestOptions::OrHash
568
- ).returns(ContextDev::Models::WebWebScrapeHTMLResponse)
569
- end
570
- def web_scrape_html(
571
- # Full URL to scrape (must include http:// or https:// protocol)
572
- url:,
573
- # Optional browser actions executed in array order after the page loads and before
574
- # content is captured. Requires a paid plan. Send a JSON array in the query
575
- # parameter. Maximum: 5 actions.
576
- actions: nil,
577
- # Fetch the target page through a residential proxy in this country (ISO 3166-1
578
- # alpha-2).
579
- country: nil,
580
- # CSS selectors to remove from the result. Applied after includeSelectors.
581
- # Exclusion takes precedence: an element matching both is removed. Examples:
582
- # "nav", "footer", ".ad-banner", "[aria-hidden=true]".
583
- exclude_selectors: nil,
584
- # Optional outbound HTTP headers forwarded only to the target URL, sent as
585
- # deep-object query params such as headers[X-Custom]=value. When provided, caching
586
- # is bypassed: the result is neither read from nor written to cache.
587
- headers: nil,
588
- # When true, iframes are rendered inline into the returned HTML.
589
- include_frames: nil,
590
- # CSS selectors. When provided, only matching subtrees (and their descendants) are
591
- # kept and everything else is dropped. When omitted, the entire document is kept.
592
- # Examples: "article.main", "#content", "[role=main]".
593
- include_selectors: nil,
594
- # Return a cached result if a prior scrape for the same parameters exists and is
595
- # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
596
- # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
597
- max_age_ms: nil,
598
- # PDF parsing controls. Use start/end to limit text extraction and embedded-image
599
- # detection/OCR to an inclusive 1-based page range.
600
- pdf: nil,
601
- # When true, waits briefly for CSS and transition animations to settle before
602
- # extracting HTML. Defaults to false. This adds a bit of latency in exchange for
603
- # more stable output on animated pages.
604
- settle_animations: nil,
605
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
606
- # characters.
607
- tags: nil,
608
- # Optional request deadline and behavior on timeout. For GET requests, use
609
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
610
- # timeoutOpts object.
611
- timeout_opts: nil,
612
- # When true, return only the page's main content in the HTML response, excluding
613
- # headers, footers, sidebars, and navigation when detectable.
614
- use_main_content_only: nil,
615
- # Optional browser wait time in milliseconds after initial page load. Min: 0. Max:
616
- # 30000 (30 seconds). When combined with timeoutOpts, timeoutOpts.milliseconds
617
- # must be at least waitForMs + 10000 ms; a shorter deadline is rejected with 400
618
- # TIMEOUT_TOO_SHORT_FOR_WAIT.
619
- wait_for_ms: nil,
620
- # Set to enabled to bypass shared caches and omit request and response content
621
- # from retained usage logs. Requires zero data retention to be enabled for your
622
- # organization (contact support@context.dev), otherwise the request fails with
623
- # ZDR_NOT_ENABLED. Successful ZDR responses include X-Context-ZDR: true.
624
- zdr: nil,
625
- request_options: {}
626
- )
627
- end
628
-
629
- # Extract image assets from a web page, including standard URLs, inline SVGs, data
630
- # URIs, responsive image sources, metadata, CSS backgrounds, video posters, and
631
- # embeds. The base request costs 1 credit, or 2 credits with browser actions. When
632
- # enrichment is enabled, the entire call costs 5 credits, including requests that
633
- # also use actions.
634
- sig do
635
- params(
636
- url: String,
637
- actions:
638
- T.nilable(
639
- T::Array[
640
- T.any(
641
- ContextDev::WebWebScrapeImagesParams::Action::Wait::OrHash,
642
- ContextDev::WebWebScrapeImagesParams::Action::Perform::OrHash,
643
- ContextDev::WebWebScrapeImagesParams::Action::Scroll::OrHash
644
- )
645
- ]
646
- ),
647
- dedupe: T::Boolean,
648
- enrichment:
649
- T.nilable(ContextDev::WebWebScrapeImagesParams::Enrichment::OrHash),
650
- headers: T::Hash[Symbol, String],
651
- max_age_ms: T.nilable(Integer),
652
- tags: T::Array[String],
653
- timeout_opts:
654
- ContextDev::WebWebScrapeImagesParams::TimeoutOpts::OrHash,
655
- wait_for_ms: T.nilable(Integer),
656
- request_options: ContextDev::RequestOptions::OrHash
657
- ).returns(ContextDev::Models::WebWebScrapeImagesResponse)
658
- end
659
- def web_scrape_images(
660
- # Page URL to inspect. Must include http:// or https://.
661
- url:,
662
- # Optional browser actions executed in array order after the page loads and before
663
- # content is captured. Requires a paid plan. Send a JSON array in the query
664
- # parameter. Maximum: 5 actions.
665
- actions: nil,
666
- # When true, visually duplicate images are removed: every image is loaded and
667
- # perceptually hashed, and only the highest-resolution copy of each duplicate
668
- # group is kept. Images that cannot be downloaded or hashed are kept. Default:
669
- # false.
670
- dedupe: nil,
671
- # Optional per-image processing, sent as deep-object query params such as
672
- # enrichment[resolution]=true.
673
- enrichment: nil,
674
- # Optional outbound HTTP headers forwarded only to the target URL, sent as
675
- # deep-object query params such as headers[X-Custom]=value. When provided, caching
676
- # is bypassed: the result is neither read from nor written to cache.
677
- headers: nil,
678
- # Reuse a cached result this many milliseconds old or newer. Default: 86400000 (1
679
- # day). Set to 0 to bypass cache. Maximum: 2592000000 (30 days).
680
- max_age_ms: nil,
681
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
682
- # characters.
683
- tags: nil,
684
- # Optional request deadline and behavior on timeout. For GET requests, use
685
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
686
- # timeoutOpts object.
687
- timeout_opts: nil,
688
- # Optional browser wait time in milliseconds after initial page load before
689
- # collecting images. Min: 0. Max: 30000 (30 seconds). When combined with
690
- # timeoutOpts, timeoutOpts.milliseconds must be at least waitForMs + 10000 ms; a
691
- # shorter deadline is rejected with 400 TIMEOUT_TOO_SHORT_FOR_WAIT.
692
- wait_for_ms: nil,
693
- request_options: {}
694
- )
695
- end
696
-
697
- # Scrapes the given URL into LLM usable Markdown. Inspect key_metadata on JSON
698
- # responses from a recognized API key; use error_code to distinguish stable
699
- # failure categories.
700
- #
701
- # ### YouTube
702
- #
703
- # YouTube URLs return the video or channel itself rather than the surrounding
704
- # player and navigation chrome. A URL addressing a single video (`/watch`,
705
- # `youtu.be`, `/shorts`, `/embed`, `/live`) returns its title, channel, duration,
706
- # view count, keywords, full description, and the transcript when the video has
707
- # captions that can be retrieved; videos without captions return everything except
708
- # the transcript. A channel URL (`/channel/UC…`, `/@handle`, `/c/…`, `/user/…`)
709
- # returns its name, handle, subscriber count, video count, and full description.
710
- # When `includeImages=true`, video responses also include the thumbnail and
711
- # channel responses include the avatar. Costs the same as any other scrape.
712
- #
713
- # ### Billing & errors
714
- #
715
- # | HTTP status | Billed? | Meaning |
716
- # | ----------- | ----------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
717
- # | 200 | Yes — 1 credit, or 2 credits with actions | Successful scrape, including a zero-length result when includeSelectors matched nothing. A partial result (`finalDOMState: "still-loading"`, only with timeoutOpts.behavior=return-partial) is billed at the base 1 credit with no OCR or actions surcharge |
718
- # | 400 | No | Invalid input, skipped PDF, or the page could not be scraped. error_code WEBSITE_BLOCKED specifically means the site answered with an anti-bot challenge, CAPTCHA wall, or login shell instead of the page (even when the site returned HTTP 200) — retrying later or from another country sometimes succeeds |
719
- # | 401 / 403 | No | Invalid/disabled key, insufficient permissions, or credits exhausted; inspect error_code |
720
- # | 404 | No | Target page returned or fingerprinted as not found |
721
- # | 408 | No | Request timed out. With timeoutOpts.behavior=return-partial this only happens when nothing usable had rendered by the deadline |
722
- # | 413 | No | Target content exceeds the maximum supported size (20 MB) |
723
- # | 415 | No | Unsupported content type |
724
- # | 429 | No | Per-minute rate limit exceeded; honor Retry-After |
725
- # | 500 | No | Internal error |
726
- sig do
727
- params(
728
- url: String,
729
- actions:
730
- T.nilable(
731
- T::Array[
732
- T.any(
733
- ContextDev::WebWebScrapeMdParams::Action::Wait::OrHash,
734
- ContextDev::WebWebScrapeMdParams::Action::Perform::OrHash,
735
- ContextDev::WebWebScrapeMdParams::Action::Scroll::OrHash
736
- )
737
- ]
738
- ),
739
- country: ContextDev::WebWebScrapeMdParams::Country::OrSymbol,
740
- exclude_selectors: T.nilable(T::Array[String]),
741
- headers: T::Hash[Symbol, String],
742
- include_frames: T::Boolean,
743
- include_html: T::Boolean,
744
- include_images: T::Boolean,
745
- include_links: T::Boolean,
746
- include_selectors: T.nilable(T::Array[String]),
747
- max_age_ms: T.nilable(Integer),
748
- pdf: ContextDev::WebWebScrapeMdParams::Pdf::OrHash,
749
- settle_animations: T::Boolean,
750
- shorten_base64_images: T::Boolean,
751
- tags: T::Array[String],
752
- timeout_opts: ContextDev::WebWebScrapeMdParams::TimeoutOpts::OrHash,
753
- use_main_content_only: T::Boolean,
754
- wait_for_ms: T.nilable(Integer),
755
- zdr: ContextDev::WebWebScrapeMdParams::Zdr::OrSymbol,
756
- request_options: ContextDev::RequestOptions::OrHash
757
- ).returns(ContextDev::Models::WebWebScrapeMdResponse)
758
- end
759
- def web_scrape_md(
760
- # Full URL to scrape into LLM usable Markdown (must include http:// or https://
761
- # protocol)
762
- url:,
763
- # Optional browser actions executed in array order after the page loads and before
764
- # content is captured. Requires a paid plan. Send a JSON array in the query
765
- # parameter. Maximum: 5 actions.
766
- actions: nil,
767
- # Fetch the target page through a residential proxy in this country (ISO 3166-1
768
- # alpha-2).
769
- country: nil,
770
- # CSS selectors to remove before conversion to Markdown. Applied after
771
- # includeSelectors. Exclusion takes precedence: an element matching both is
772
- # removed. Examples: "nav", "footer", ".ad-banner", "[aria-hidden=true]".
773
- exclude_selectors: nil,
774
- # Optional outbound HTTP headers forwarded only to the target URL, sent as
775
- # deep-object query params such as headers[X-Custom]=value. When provided, caching
776
- # is bypassed: the result is neither read from nor written to cache.
777
- headers: nil,
778
- # When true, the contents of iframes are rendered to Markdown.
779
- include_frames: nil,
780
- # When true, the response also includes an `html` field with the page HTML the
781
- # Markdown was converted from — the same body the Scrape HTML endpoint returns for
782
- # the equivalent request.
783
- include_html: nil,
784
- # Include image references in Markdown output
785
- include_images: nil,
786
- # Preserve hyperlinks in Markdown output
787
- include_links: nil,
788
- # CSS selectors. When provided, only matching HTML subtrees (and their
789
- # descendants) are kept before conversion to Markdown. When omitted, the entire
790
- # document is kept. Examples: "article.main", "#content", "[role=main]".
791
- include_selectors: nil,
792
- # Return a cached result if a prior scrape for the same parameters exists and is
793
- # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
794
- # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
795
- max_age_ms: nil,
796
- # PDF parsing controls. Use start/end to limit text extraction and embedded-image
797
- # detection/OCR to an inclusive 1-based page range.
798
- pdf: nil,
799
- # When true, waits briefly for CSS and transition animations to settle before
800
- # converting to Markdown. Defaults to false. This adds a bit of latency in
801
- # exchange for more stable output on animated pages.
802
- settle_animations: nil,
803
- # Shorten base64-encoded image data in the Markdown output
804
- shorten_base64_images: nil,
805
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
806
- # characters.
807
- tags: nil,
808
- # Optional request deadline and behavior on timeout. For GET requests, use
809
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
810
- # timeoutOpts object.
811
- timeout_opts: nil,
812
- # Extract only the main content of the page, excluding headers, footers, sidebars,
813
- # and navigation
814
- use_main_content_only: nil,
815
- # Optional browser wait time in milliseconds after initial page load before
816
- # converting the page to Markdown. Min: 0. Max: 30000 (30 seconds). When combined
817
- # with timeoutOpts, timeoutOpts.milliseconds must be at least waitForMs + 10000
818
- # ms; a shorter deadline is rejected with 400 TIMEOUT_TOO_SHORT_FOR_WAIT.
819
- wait_for_ms: nil,
820
- # Set to enabled to bypass shared caches and omit request and response content
821
- # from retained usage logs. Requires zero data retention to be enabled for your
822
- # organization (contact support@context.dev), otherwise the request fails with
823
- # ZDR_NOT_ENABLED. Successful ZDR responses include X-Context-ZDR: true.
824
- zdr: nil,
825
- request_options: {}
826
- )
827
- end
828
-
829
- # Crawl an entire website's sitemap and return all discovered page URLs. Set
830
- # `includeSubdomains=true` to also discover public pages and sitemaps on child
831
- # hosts such as `docs.example.com` or `brand.example.com`. Pass `search` to have
832
- # the discovered URLs filtered down to the pages about a phrase (for example
833
- # `pricing and plans` or `api authentication docs`), most relevant first — a
834
- # searched crawl scans the whole sitemap and costs 2 credits instead of 1.
835
- sig do
836
- params(
837
- domain: String,
838
- headers: T::Hash[Symbol, String],
839
- include_subdomains: T::Boolean,
840
- max_links: Integer,
841
- search: String,
842
- sitemap_url: String,
843
- tags: T::Array[String],
844
- timeout_opts:
845
- ContextDev::WebWebScrapeSitemapParams::TimeoutOpts::OrHash,
846
- url_regex: String,
847
- zdr: ContextDev::WebWebScrapeSitemapParams::Zdr::OrSymbol,
848
- request_options: ContextDev::RequestOptions::OrHash
849
- ).returns(ContextDev::Models::WebWebScrapeSitemapResponse)
850
- end
851
- def web_scrape_sitemap(
852
- # Domain to build a sitemap for
853
- domain:,
854
- # Optional outbound HTTP headers forwarded only to the target URL, sent as
855
- # deep-object query params such as headers[X-Custom]=value. When provided, caching
856
- # is bypassed: the result is neither read from nor written to cache.
857
- headers: nil,
858
- # When true, discover and include public pages and sitemaps on subdomains of the
859
- # requested domain. Defaults to false.
860
- include_subdomains: nil,
861
- # Maximum number of links to return from the sitemap crawl. Defaults to 10,000.
862
- # Minimum is 1, maximum is 100,000.
863
- max_links: nil,
864
- # Optional search phrase. When provided, the crawled sitemap is filtered to the
865
- # pages whose URLs are about that phrase, most relevant first, and the request
866
- # costs 2 credits instead of 1.
867
- search: nil,
868
- # Optional explicit sitemap URL. When provided, exactly this sitemap is crawled
869
- # instead of discovering the domain's sitemaps.
870
- sitemap_url: nil,
871
- # Comma-separated tags for tracking request usage. Up to 20 tags, each 1-50
872
- # characters.
873
- tags: nil,
874
- # Optional request deadline and behavior on timeout. For GET requests, use
875
- # timeoutOpts[milliseconds]=30000&timeoutOpts[behavior]=fail or a JSON-encoded
876
- # timeoutOpts object.
877
- timeout_opts: nil,
878
- # Optional RE2-compatible regex pattern. Only URLs matching this pattern are
879
- # returned and counted against maxLinks.
880
- url_regex: nil,
881
- # Set to enabled to bypass shared caches and omit request and response content
882
- # from retained usage logs. Requires zero data retention to be enabled for your
883
- # organization (contact support@context.dev), otherwise the request fails with
884
- # ZDR_NOT_ENABLED. Successful ZDR responses include X-Context-ZDR: true.
885
- zdr: nil,
886
- request_options: {}
887
- )
888
- end
889
-
890
523
  # @api private
891
524
  sig { params(client: ContextDev::Client).returns(T.attached_class) }
892
525
  def self.new(client:)