context.dev 2.2.0 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +14 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +7 -2
  5. data/lib/context_dev/internal/type/base_model.rb +5 -5
  6. data/lib/context_dev/models/monitor_create_params.rb +28 -3
  7. data/lib/context_dev/models/monitor_create_response.rb +96 -4
  8. data/lib/context_dev/models/monitor_list_account_runs_response.rb +19 -98
  9. data/lib/context_dev/models/monitor_list_response.rb +100 -4
  10. data/lib/context_dev/models/monitor_list_runs_response.rb +19 -96
  11. data/lib/context_dev/models/monitor_retrieve_change_response.rb +10 -10
  12. data/lib/context_dev/models/monitor_retrieve_response.rb +97 -4
  13. data/lib/context_dev/models/monitor_update_params.rb +28 -3
  14. data/lib/context_dev/models/monitor_update_response.rb +96 -4
  15. data/lib/context_dev/models/parse_handle_params.rb +110 -0
  16. data/lib/context_dev/models/parse_handle_response.rb +132 -0
  17. data/lib/context_dev/models/web_web_crawl_md_params.rb +16 -6
  18. data/lib/context_dev/models/web_web_scrape_html_params.rb +16 -6
  19. data/lib/context_dev/models/web_web_scrape_md_params.rb +16 -6
  20. data/lib/context_dev/models/web_web_scrape_md_response.rb +11 -1
  21. data/lib/context_dev/models/webhook_delivery.rb +109 -0
  22. data/lib/context_dev/models.rb +4 -0
  23. data/lib/context_dev/resources/monitors.rb +3 -2
  24. data/lib/context_dev/resources/parse.rb +63 -0
  25. data/lib/context_dev/resources/web.rb +19 -4
  26. data/lib/context_dev/version.rb +1 -1
  27. data/lib/context_dev.rb +4 -0
  28. data/rbi/context_dev/client.rbi +6 -2
  29. data/rbi/context_dev/models/monitor_create_params.rbi +85 -5
  30. data/rbi/context_dev/models/monitor_create_response.rbi +234 -6
  31. data/rbi/context_dev/models/monitor_list_account_runs_response.rbi +27 -192
  32. data/rbi/context_dev/models/monitor_list_response.rbi +235 -5
  33. data/rbi/context_dev/models/monitor_list_runs_response.rbi +27 -192
  34. data/rbi/context_dev/models/monitor_retrieve_change_response.rbi +12 -15
  35. data/rbi/context_dev/models/monitor_retrieve_response.rbi +234 -6
  36. data/rbi/context_dev/models/monitor_update_params.rbi +85 -5
  37. data/rbi/context_dev/models/monitor_update_response.rbi +234 -6
  38. data/rbi/context_dev/models/parse_handle_params.rbi +163 -0
  39. data/rbi/context_dev/models/parse_handle_response.rbi +377 -0
  40. data/rbi/context_dev/models/web_web_crawl_md_params.rbi +26 -7
  41. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +26 -7
  42. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +26 -7
  43. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +12 -0
  44. data/rbi/context_dev/models/webhook_delivery.rbi +172 -0
  45. data/rbi/context_dev/models.rbi +4 -0
  46. data/rbi/context_dev/resources/monitors.rbi +3 -2
  47. data/rbi/context_dev/resources/parse.rbi +65 -0
  48. data/rbi/context_dev/resources/web.rbi +22 -7
  49. data/sig/context_dev/client.rbs +2 -0
  50. data/sig/context_dev/models/monitor_create_params.rbs +32 -3
  51. data/sig/context_dev/models/monitor_create_response.rbs +85 -6
  52. data/sig/context_dev/models/monitor_list_account_runs_response.rbs +15 -68
  53. data/sig/context_dev/models/monitor_list_response.rbs +85 -6
  54. data/sig/context_dev/models/monitor_list_runs_response.rbs +15 -68
  55. data/sig/context_dev/models/monitor_retrieve_change_response.rbs +8 -10
  56. data/sig/context_dev/models/monitor_retrieve_response.rbs +85 -6
  57. data/sig/context_dev/models/monitor_update_params.rbs +32 -3
  58. data/sig/context_dev/models/monitor_update_response.rbs +85 -6
  59. data/sig/context_dev/models/parse_handle_params.rbs +96 -0
  60. data/sig/context_dev/models/parse_handle_response.rbs +159 -0
  61. data/sig/context_dev/models/web_web_crawl_md_params.rbs +13 -2
  62. data/sig/context_dev/models/web_web_scrape_html_params.rbs +13 -2
  63. data/sig/context_dev/models/web_web_scrape_md_params.rbs +13 -2
  64. data/sig/context_dev/models/web_web_scrape_md_response.rbs +5 -0
  65. data/sig/context_dev/models/webhook_delivery.rbs +81 -0
  66. data/sig/context_dev/models.rbs +4 -0
  67. data/sig/context_dev/resources/parse.rbs +22 -0
  68. metadata +14 -2
@@ -0,0 +1,163 @@
1
+ # typed: strong
2
+
3
+ module ContextDev
4
+ module Models
5
+ class ParseHandleParams < ContextDev::Internal::Type::BaseModel
6
+ extend ContextDev::Internal::Type::RequestParameters::Converter
7
+ include ContextDev::Internal::Type::RequestParameters
8
+
9
+ OrHash =
10
+ T.type_alias do
11
+ T.any(ContextDev::ParseHandleParams, ContextDev::Internal::AnyHash)
12
+ end
13
+
14
+ sig { returns(ContextDev::Internal::FileInput) }
15
+ attr_accessor :body
16
+
17
+ # Optional HTTP(S) source document URL used to resolve relative links and image
18
+ # references. Relative references remain relative when omitted.
19
+ sig { returns(T.nilable(String)) }
20
+ attr_reader :base_url
21
+
22
+ sig { params(base_url: String).void }
23
+ attr_writer :base_url
24
+
25
+ # Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv,
26
+ # md, py, rtf, jpg, png, or txt.
27
+ sig { returns(T.nilable(String)) }
28
+ attr_reader :extension
29
+
30
+ sig { params(extension: String).void }
31
+ attr_writer :extension
32
+
33
+ # Optional filename hint used to infer the extension when extension is omitted.
34
+ sig { returns(T.nilable(String)) }
35
+ attr_reader :filename
36
+
37
+ sig { params(filename: String).void }
38
+ attr_writer :filename
39
+
40
+ # Include image references in Markdown output
41
+ sig { returns(T.nilable(T::Boolean)) }
42
+ attr_reader :include_images
43
+
44
+ sig { params(include_images: T::Boolean).void }
45
+ attr_writer :include_images
46
+
47
+ # Preserve hyperlinks in Markdown output
48
+ sig { returns(T.nilable(T::Boolean)) }
49
+ attr_reader :include_links
50
+
51
+ sig { params(include_links: T::Boolean).void }
52
+ attr_writer :include_links
53
+
54
+ # When true for PDF inputs, detect and OCR images embedded in the selected pages,
55
+ # inserting recognized text at each image's position in page reading order while
56
+ # preserving the PDF text layer. pdfStart/pdfEnd limit the inclusive page range.
57
+ # This is separate from automatic scanned-PDF OCR fallback.
58
+ sig { returns(T.nilable(T::Boolean)) }
59
+ attr_reader :ocr
60
+
61
+ sig { params(ocr: T::Boolean).void }
62
+ attr_writer :ocr
63
+
64
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
65
+ # Must be greater than or equal to pdfStart when both are provided.
66
+ sig { returns(T.nilable(Integer)) }
67
+ attr_reader :pdf_end
68
+
69
+ sig { params(pdf_end: Integer).void }
70
+ attr_writer :pdf_end
71
+
72
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
73
+ sig { returns(T.nilable(Integer)) }
74
+ attr_reader :pdf_start
75
+
76
+ sig { params(pdf_start: Integer).void }
77
+ attr_writer :pdf_start
78
+
79
+ # Shorten base64-encoded image data in the Markdown output
80
+ sig { returns(T.nilable(T::Boolean)) }
81
+ attr_reader :shorten_base64_images
82
+
83
+ sig { params(shorten_base64_images: T::Boolean).void }
84
+ attr_writer :shorten_base64_images
85
+
86
+ # Extract only the main content from HTML-like inputs
87
+ sig { returns(T.nilable(T::Boolean)) }
88
+ attr_reader :use_main_content_only
89
+
90
+ sig { params(use_main_content_only: T::Boolean).void }
91
+ attr_writer :use_main_content_only
92
+
93
+ sig do
94
+ params(
95
+ body: ContextDev::Internal::FileInput,
96
+ base_url: String,
97
+ extension: String,
98
+ filename: String,
99
+ include_images: T::Boolean,
100
+ include_links: T::Boolean,
101
+ ocr: T::Boolean,
102
+ pdf_end: Integer,
103
+ pdf_start: Integer,
104
+ shorten_base64_images: T::Boolean,
105
+ use_main_content_only: T::Boolean,
106
+ request_options: ContextDev::RequestOptions::OrHash
107
+ ).returns(T.attached_class)
108
+ end
109
+ def self.new(
110
+ body:,
111
+ # Optional HTTP(S) source document URL used to resolve relative links and image
112
+ # references. Relative references remain relative when omitted.
113
+ base_url: nil,
114
+ # Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv,
115
+ # md, py, rtf, jpg, png, or txt.
116
+ extension: nil,
117
+ # Optional filename hint used to infer the extension when extension is omitted.
118
+ filename: nil,
119
+ # Include image references in Markdown output
120
+ include_images: nil,
121
+ # Preserve hyperlinks in Markdown output
122
+ include_links: nil,
123
+ # When true for PDF inputs, detect and OCR images embedded in the selected pages,
124
+ # inserting recognized text at each image's position in page reading order while
125
+ # preserving the PDF text layer. pdfStart/pdfEnd limit the inclusive page range.
126
+ # This is separate from automatic scanned-PDF OCR fallback.
127
+ ocr: nil,
128
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
129
+ # Must be greater than or equal to pdfStart when both are provided.
130
+ pdf_end: nil,
131
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
132
+ pdf_start: nil,
133
+ # Shorten base64-encoded image data in the Markdown output
134
+ shorten_base64_images: nil,
135
+ # Extract only the main content from HTML-like inputs
136
+ use_main_content_only: nil,
137
+ request_options: {}
138
+ )
139
+ end
140
+
141
+ sig do
142
+ override.returns(
143
+ {
144
+ body: ContextDev::Internal::FileInput,
145
+ base_url: String,
146
+ extension: String,
147
+ filename: String,
148
+ include_images: T::Boolean,
149
+ include_links: T::Boolean,
150
+ ocr: T::Boolean,
151
+ pdf_end: Integer,
152
+ pdf_start: Integer,
153
+ shorten_base64_images: T::Boolean,
154
+ use_main_content_only: T::Boolean,
155
+ request_options: ContextDev::RequestOptions
156
+ }
157
+ )
158
+ end
159
+ def to_hash
160
+ end
161
+ end
162
+ end
163
+ end
@@ -0,0 +1,377 @@
1
+ # typed: strong
2
+
3
+ module ContextDev
4
+ module Models
5
+ class ParseHandleResponse < ContextDev::Internal::Type::BaseModel
6
+ OrHash =
7
+ T.type_alias do
8
+ T.any(
9
+ ContextDev::Models::ParseHandleResponse,
10
+ ContextDev::Internal::AnyHash
11
+ )
12
+ end
13
+
14
+ # Input bytes converted to GitHub Flavored Markdown
15
+ sig { returns(String) }
16
+ attr_accessor :markdown
17
+
18
+ # Indicates success
19
+ sig do
20
+ returns(ContextDev::Models::ParseHandleResponse::Success::TaggedBoolean)
21
+ end
22
+ attr_accessor :success
23
+
24
+ # Detected content type used for parsing
25
+ sig do
26
+ returns(ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol)
27
+ end
28
+ attr_accessor :type
29
+
30
+ # Metadata about the API key used for the request. Included in every response
31
+ # whenever a valid API key is provided, even when the response status is not 200.
32
+ sig do
33
+ returns(T.nilable(ContextDev::Models::ParseHandleResponse::KeyMetadata))
34
+ end
35
+ attr_reader :key_metadata
36
+
37
+ sig do
38
+ params(
39
+ key_metadata:
40
+ ContextDev::Models::ParseHandleResponse::KeyMetadata::OrHash
41
+ ).void
42
+ end
43
+ attr_writer :key_metadata
44
+
45
+ sig do
46
+ params(
47
+ markdown: String,
48
+ success: ContextDev::Models::ParseHandleResponse::Success::OrBoolean,
49
+ type: ContextDev::Models::ParseHandleResponse::Type::OrSymbol,
50
+ key_metadata:
51
+ ContextDev::Models::ParseHandleResponse::KeyMetadata::OrHash
52
+ ).returns(T.attached_class)
53
+ end
54
+ def self.new(
55
+ # Input bytes converted to GitHub Flavored Markdown
56
+ markdown:,
57
+ # Indicates success
58
+ success:,
59
+ # Detected content type used for parsing
60
+ type:,
61
+ # Metadata about the API key used for the request. Included in every response
62
+ # whenever a valid API key is provided, even when the response status is not 200.
63
+ key_metadata: nil
64
+ )
65
+ end
66
+
67
+ sig do
68
+ override.returns(
69
+ {
70
+ markdown: String,
71
+ success:
72
+ ContextDev::Models::ParseHandleResponse::Success::TaggedBoolean,
73
+ type: ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol,
74
+ key_metadata: ContextDev::Models::ParseHandleResponse::KeyMetadata
75
+ }
76
+ )
77
+ end
78
+ def to_hash
79
+ end
80
+
81
+ # Indicates success
82
+ module Success
83
+ extend ContextDev::Internal::Type::Enum
84
+
85
+ TaggedBoolean =
86
+ T.type_alias do
87
+ T.all(T::Boolean, ContextDev::Models::ParseHandleResponse::Success)
88
+ end
89
+ OrBoolean = T.type_alias { T::Boolean }
90
+
91
+ TRUE =
92
+ T.let(
93
+ true,
94
+ ContextDev::Models::ParseHandleResponse::Success::TaggedBoolean
95
+ )
96
+
97
+ sig do
98
+ override.returns(
99
+ T::Array[
100
+ ContextDev::Models::ParseHandleResponse::Success::TaggedBoolean
101
+ ]
102
+ )
103
+ end
104
+ def self.values
105
+ end
106
+ end
107
+
108
+ # Detected content type used for parsing
109
+ module Type
110
+ extend ContextDev::Internal::Type::Enum
111
+
112
+ TaggedSymbol =
113
+ T.type_alias do
114
+ T.all(Symbol, ContextDev::Models::ParseHandleResponse::Type)
115
+ end
116
+ OrSymbol = T.type_alias { T.any(Symbol, String) }
117
+
118
+ HTML =
119
+ T.let(
120
+ :html,
121
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
122
+ )
123
+ XML =
124
+ T.let(
125
+ :xml,
126
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
127
+ )
128
+ JSON =
129
+ T.let(
130
+ :json,
131
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
132
+ )
133
+ JSONL =
134
+ T.let(
135
+ :jsonl,
136
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
137
+ )
138
+ TEXT =
139
+ T.let(
140
+ :text,
141
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
142
+ )
143
+ CSV =
144
+ T.let(
145
+ :csv,
146
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
147
+ )
148
+ TSV =
149
+ T.let(
150
+ :tsv,
151
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
152
+ )
153
+ MARKDOWN =
154
+ T.let(
155
+ :markdown,
156
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
157
+ )
158
+ YAML =
159
+ T.let(
160
+ :yaml,
161
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
162
+ )
163
+ PYTHON =
164
+ T.let(
165
+ :python,
166
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
167
+ )
168
+ JAVA =
169
+ T.let(
170
+ :java,
171
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
172
+ )
173
+ JAVASCRIPT =
174
+ T.let(
175
+ :javascript,
176
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
177
+ )
178
+ PHP =
179
+ T.let(
180
+ :php,
181
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
182
+ )
183
+ SHELL =
184
+ T.let(
185
+ :shell,
186
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
187
+ )
188
+ RUBY =
189
+ T.let(
190
+ :ruby,
191
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
192
+ )
193
+ TYPESCRIPT =
194
+ T.let(
195
+ :typescript,
196
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
197
+ )
198
+ RTF =
199
+ T.let(
200
+ :rtf,
201
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
202
+ )
203
+ SRT =
204
+ T.let(
205
+ :srt,
206
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
207
+ )
208
+ CSS =
209
+ T.let(
210
+ :css,
211
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
212
+ )
213
+ SCSS =
214
+ T.let(
215
+ :scss,
216
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
217
+ )
218
+ LESS =
219
+ T.let(
220
+ :less,
221
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
222
+ )
223
+ STYLUS =
224
+ T.let(
225
+ :stylus,
226
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
227
+ )
228
+ SASS =
229
+ T.let(
230
+ :sass,
231
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
232
+ )
233
+ SVG =
234
+ T.let(
235
+ :svg,
236
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
237
+ )
238
+ PDF =
239
+ T.let(
240
+ :pdf,
241
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
242
+ )
243
+ DOCX =
244
+ T.let(
245
+ :docx,
246
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
247
+ )
248
+ DOC =
249
+ T.let(
250
+ :doc,
251
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
252
+ )
253
+ XLSX =
254
+ T.let(
255
+ :xlsx,
256
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
257
+ )
258
+ XLS =
259
+ T.let(
260
+ :xls,
261
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
262
+ )
263
+ PPTX =
264
+ T.let(
265
+ :pptx,
266
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
267
+ )
268
+ PPT =
269
+ T.let(
270
+ :ppt,
271
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
272
+ )
273
+ JPG =
274
+ T.let(
275
+ :jpg,
276
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
277
+ )
278
+ PNG =
279
+ T.let(
280
+ :png,
281
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
282
+ )
283
+ GIF =
284
+ T.let(
285
+ :gif,
286
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
287
+ )
288
+ BMP =
289
+ T.let(
290
+ :bmp,
291
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
292
+ )
293
+ TIFF =
294
+ T.let(
295
+ :tiff,
296
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
297
+ )
298
+ WEBP =
299
+ T.let(
300
+ :webp,
301
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
302
+ )
303
+ PPM =
304
+ T.let(
305
+ :ppm,
306
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
307
+ )
308
+ PBM =
309
+ T.let(
310
+ :pbm,
311
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
312
+ )
313
+ PGM =
314
+ T.let(
315
+ :pgm,
316
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
317
+ )
318
+ PNM =
319
+ T.let(
320
+ :pnm,
321
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
322
+ )
323
+
324
+ sig do
325
+ override.returns(
326
+ T::Array[
327
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
328
+ ]
329
+ )
330
+ end
331
+ def self.values
332
+ end
333
+ end
334
+
335
+ class KeyMetadata < ContextDev::Internal::Type::BaseModel
336
+ OrHash =
337
+ T.type_alias do
338
+ T.any(
339
+ ContextDev::Models::ParseHandleResponse::KeyMetadata,
340
+ ContextDev::Internal::AnyHash
341
+ )
342
+ end
343
+
344
+ # The number of credits consumed by this request.
345
+ sig { returns(Integer) }
346
+ attr_accessor :credits_consumed
347
+
348
+ # The number of credits remaining for your organization after this request.
349
+ sig { returns(Integer) }
350
+ attr_accessor :credits_remaining
351
+
352
+ # Metadata about the API key used for the request. Included in every response
353
+ # whenever a valid API key is provided, even when the response status is not 200.
354
+ sig do
355
+ params(credits_consumed: Integer, credits_remaining: Integer).returns(
356
+ T.attached_class
357
+ )
358
+ end
359
+ def self.new(
360
+ # The number of credits consumed by this request.
361
+ credits_consumed:,
362
+ # The number of credits remaining for your organization after this request.
363
+ credits_remaining:
364
+ )
365
+ end
366
+
367
+ sig do
368
+ override.returns(
369
+ { credits_consumed: Integer, credits_remaining: Integer }
370
+ )
371
+ end
372
+ def to_hash
373
+ end
374
+ end
375
+ end
376
+ end
377
+ end
@@ -101,8 +101,8 @@ module ContextDev
101
101
  sig { params(max_pages: Integer).void }
102
102
  attr_writer :max_pages
103
103
 
104
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
105
- # inclusive 1-based page range.
104
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
105
+ # detection/OCR to an inclusive 1-based page range.
106
106
  sig { returns(T.nilable(ContextDev::WebWebCrawlMdParams::Pdf)) }
107
107
  attr_reader :pdf
108
108
 
@@ -226,8 +226,8 @@ module ContextDev
226
226
  max_depth: nil,
227
227
  # Maximum number of pages to crawl. Hard cap: 500.
228
228
  max_pages: nil,
229
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
230
- # inclusive 1-based page range.
229
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
230
+ # detection/OCR to an inclusive 1-based page range.
231
231
  pdf: nil,
232
232
  # When true, waits briefly for CSS and transition animations to settle before
233
233
  # extracting each crawled page. Defaults to false. This adds a bit of latency in
@@ -528,6 +528,15 @@ module ContextDev
528
528
  sig { params(end_: Integer).void }
529
529
  attr_writer :end_
530
530
 
531
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
532
+ # recognized text at each image's position in page reading order while preserving
533
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
534
+ sig { returns(T.nilable(T::Boolean)) }
535
+ attr_reader :ocr
536
+
537
+ sig { params(ocr: T::Boolean).void }
538
+ attr_writer :ocr
539
+
531
540
  # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
532
541
  # entirely (not included in results and not counted as failures).
533
542
  sig { returns(T.nilable(T::Boolean)) }
@@ -543,11 +552,12 @@ module ContextDev
543
552
  sig { params(start: Integer).void }
544
553
  attr_writer :start
545
554
 
546
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
547
- # inclusive 1-based page range.
555
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
556
+ # detection/OCR to an inclusive 1-based page range.
548
557
  sig do
549
558
  params(
550
559
  end_: Integer,
560
+ ocr: T::Boolean,
551
561
  should_parse: T::Boolean,
552
562
  start: Integer
553
563
  ).returns(T.attached_class)
@@ -556,6 +566,10 @@ module ContextDev
556
566
  # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
557
567
  # Must be greater than or equal to start when both are provided.
558
568
  end_: nil,
569
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
570
+ # recognized text at each image's position in page reading order while preserving
571
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
572
+ ocr: nil,
559
573
  # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
560
574
  # entirely (not included in results and not counted as failures).
561
575
  should_parse: nil,
@@ -566,7 +580,12 @@ module ContextDev
566
580
 
567
581
  sig do
568
582
  override.returns(
569
- { end_: Integer, should_parse: T::Boolean, start: Integer }
583
+ {
584
+ end_: Integer,
585
+ ocr: T::Boolean,
586
+ should_parse: T::Boolean,
587
+ start: Integer
588
+ }
570
589
  )
571
590
  end
572
591
  def to_hash
@@ -77,8 +77,8 @@ module ContextDev
77
77
  sig { params(max_age_ms: Integer).void }
78
78
  attr_writer :max_age_ms
79
79
 
80
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
81
- # inclusive 1-based page range.
80
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
81
+ # detection/OCR to an inclusive 1-based page range.
82
82
  sig { returns(T.nilable(ContextDev::WebWebScrapeHTMLParams::Pdf)) }
83
83
  attr_reader :pdf
84
84
 
@@ -160,8 +160,8 @@ module ContextDev
160
160
  # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
161
161
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
162
162
  max_age_ms: nil,
163
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
164
- # inclusive 1-based page range.
163
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
164
+ # detection/OCR to an inclusive 1-based page range.
165
165
  pdf: nil,
166
166
  # When true, waits briefly for CSS and transition animations to settle before
167
167
  # extracting HTML. Defaults to false. This adds a bit of latency in exchange for
@@ -649,6 +649,15 @@ module ContextDev
649
649
  sig { params(end_: Integer).void }
650
650
  attr_writer :end_
651
651
 
652
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
653
+ # recognized text at each image's position in page reading order while preserving
654
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
655
+ sig { returns(T.nilable(T::Boolean)) }
656
+ attr_reader :ocr
657
+
658
+ sig { params(ocr: T::Boolean).void }
659
+ attr_writer :ocr
660
+
652
661
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
653
662
  # a 400 WEBSITE_ACCESS_ERROR is returned.
654
663
  sig { returns(T.nilable(T::Boolean)) }
@@ -664,11 +673,12 @@ module ContextDev
664
673
  sig { params(start: Integer).void }
665
674
  attr_writer :start
666
675
 
667
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
668
- # inclusive 1-based page range.
676
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
677
+ # detection/OCR to an inclusive 1-based page range.
669
678
  sig do
670
679
  params(
671
680
  end_: Integer,
681
+ ocr: T::Boolean,
672
682
  should_parse: T::Boolean,
673
683
  start: Integer
674
684
  ).returns(T.attached_class)
@@ -677,6 +687,10 @@ module ContextDev
677
687
  # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
678
688
  # Must be greater than or equal to start when both are provided.
679
689
  end_: nil,
690
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
691
+ # recognized text at each image's position in page reading order while preserving
692
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
693
+ ocr: nil,
680
694
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
681
695
  # a 400 WEBSITE_ACCESS_ERROR is returned.
682
696
  should_parse: nil,
@@ -687,7 +701,12 @@ module ContextDev
687
701
 
688
702
  sig do
689
703
  override.returns(
690
- { end_: Integer, should_parse: T::Boolean, start: Integer }
704
+ {
705
+ end_: Integer,
706
+ ocr: T::Boolean,
707
+ should_parse: T::Boolean,
708
+ start: Integer
709
+ }
691
710
  )
692
711
  end
693
712
  def to_hash