context.dev 2.1.0 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +23 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +7 -2
  5. data/lib/context_dev/internal/type/base_model.rb +5 -5
  6. data/lib/context_dev/models/monitor_create_params.rb +44 -14
  7. data/lib/context_dev/models/monitor_create_response.rb +112 -15
  8. data/lib/context_dev/models/monitor_list_account_runs_response.rb +23 -1
  9. data/lib/context_dev/models/monitor_list_response.rb +116 -15
  10. data/lib/context_dev/models/monitor_list_runs_response.rb +23 -1
  11. data/lib/context_dev/models/monitor_retrieve_change_response.rb +10 -10
  12. data/lib/context_dev/models/monitor_retrieve_response.rb +113 -15
  13. data/lib/context_dev/models/monitor_update_params.rb +44 -14
  14. data/lib/context_dev/models/monitor_update_response.rb +112 -15
  15. data/lib/context_dev/models/parse_handle_params.rb +110 -0
  16. data/lib/context_dev/models/parse_handle_response.rb +132 -0
  17. data/lib/context_dev/models/web_web_crawl_md_params.rb +16 -6
  18. data/lib/context_dev/models/web_web_scrape_html_params.rb +16 -6
  19. data/lib/context_dev/models/web_web_scrape_md_params.rb +16 -6
  20. data/lib/context_dev/models/web_web_scrape_md_response.rb +11 -1
  21. data/lib/context_dev/models/webhook_delivery.rb +109 -0
  22. data/lib/context_dev/models.rb +4 -0
  23. data/lib/context_dev/resources/monitors.rb +3 -2
  24. data/lib/context_dev/resources/parse.rb +63 -0
  25. data/lib/context_dev/resources/web.rb +19 -4
  26. data/lib/context_dev/version.rb +1 -1
  27. data/lib/context_dev.rb +4 -0
  28. data/rbi/context_dev/client.rbi +6 -2
  29. data/rbi/context_dev/models/monitor_create_params.rbi +106 -16
  30. data/rbi/context_dev/models/monitor_create_response.rbi +255 -17
  31. data/rbi/context_dev/models/monitor_list_account_runs_response.rbi +39 -3
  32. data/rbi/context_dev/models/monitor_list_response.rbi +256 -16
  33. data/rbi/context_dev/models/monitor_list_runs_response.rbi +39 -3
  34. data/rbi/context_dev/models/monitor_retrieve_change_response.rbi +12 -15
  35. data/rbi/context_dev/models/monitor_retrieve_response.rbi +255 -17
  36. data/rbi/context_dev/models/monitor_update_params.rbi +106 -16
  37. data/rbi/context_dev/models/monitor_update_response.rbi +255 -17
  38. data/rbi/context_dev/models/parse_handle_params.rbi +163 -0
  39. data/rbi/context_dev/models/parse_handle_response.rbi +377 -0
  40. data/rbi/context_dev/models/web_web_crawl_md_params.rbi +26 -7
  41. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +26 -7
  42. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +26 -7
  43. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +12 -0
  44. data/rbi/context_dev/models/webhook_delivery.rbi +172 -0
  45. data/rbi/context_dev/models.rbi +4 -0
  46. data/rbi/context_dev/resources/monitors.rbi +3 -2
  47. data/rbi/context_dev/resources/parse.rbi +65 -0
  48. data/rbi/context_dev/resources/web.rbi +22 -7
  49. data/sig/context_dev/client.rbs +2 -0
  50. data/sig/context_dev/models/monitor_create_params.rbs +32 -3
  51. data/sig/context_dev/models/monitor_create_response.rbs +85 -6
  52. data/sig/context_dev/models/monitor_list_account_runs_response.rbs +21 -3
  53. data/sig/context_dev/models/monitor_list_response.rbs +85 -6
  54. data/sig/context_dev/models/monitor_list_runs_response.rbs +21 -3
  55. data/sig/context_dev/models/monitor_retrieve_change_response.rbs +8 -10
  56. data/sig/context_dev/models/monitor_retrieve_response.rbs +85 -6
  57. data/sig/context_dev/models/monitor_update_params.rbs +32 -3
  58. data/sig/context_dev/models/monitor_update_response.rbs +85 -6
  59. data/sig/context_dev/models/parse_handle_params.rbs +96 -0
  60. data/sig/context_dev/models/parse_handle_response.rbs +159 -0
  61. data/sig/context_dev/models/web_web_crawl_md_params.rbs +13 -2
  62. data/sig/context_dev/models/web_web_scrape_html_params.rbs +13 -2
  63. data/sig/context_dev/models/web_web_scrape_md_params.rbs +13 -2
  64. data/sig/context_dev/models/web_web_scrape_md_response.rbs +5 -0
  65. data/sig/context_dev/models/webhook_delivery.rbs +81 -0
  66. data/sig/context_dev/models.rbs +4 -0
  67. data/sig/context_dev/resources/parse.rbs +22 -0
  68. metadata +14 -2
@@ -0,0 +1,377 @@
1
+ # typed: strong
2
+
3
+ module ContextDev
4
+ module Models
5
+ class ParseHandleResponse < ContextDev::Internal::Type::BaseModel
6
+ OrHash =
7
+ T.type_alias do
8
+ T.any(
9
+ ContextDev::Models::ParseHandleResponse,
10
+ ContextDev::Internal::AnyHash
11
+ )
12
+ end
13
+
14
+ # Input bytes converted to GitHub Flavored Markdown
15
+ sig { returns(String) }
16
+ attr_accessor :markdown
17
+
18
+ # Indicates success
19
+ sig do
20
+ returns(ContextDev::Models::ParseHandleResponse::Success::TaggedBoolean)
21
+ end
22
+ attr_accessor :success
23
+
24
+ # Detected content type used for parsing
25
+ sig do
26
+ returns(ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol)
27
+ end
28
+ attr_accessor :type
29
+
30
+ # Metadata about the API key used for the request. Included in every response
31
+ # whenever a valid API key is provided, even when the response status is not 200.
32
+ sig do
33
+ returns(T.nilable(ContextDev::Models::ParseHandleResponse::KeyMetadata))
34
+ end
35
+ attr_reader :key_metadata
36
+
37
+ sig do
38
+ params(
39
+ key_metadata:
40
+ ContextDev::Models::ParseHandleResponse::KeyMetadata::OrHash
41
+ ).void
42
+ end
43
+ attr_writer :key_metadata
44
+
45
+ sig do
46
+ params(
47
+ markdown: String,
48
+ success: ContextDev::Models::ParseHandleResponse::Success::OrBoolean,
49
+ type: ContextDev::Models::ParseHandleResponse::Type::OrSymbol,
50
+ key_metadata:
51
+ ContextDev::Models::ParseHandleResponse::KeyMetadata::OrHash
52
+ ).returns(T.attached_class)
53
+ end
54
+ def self.new(
55
+ # Input bytes converted to GitHub Flavored Markdown
56
+ markdown:,
57
+ # Indicates success
58
+ success:,
59
+ # Detected content type used for parsing
60
+ type:,
61
+ # Metadata about the API key used for the request. Included in every response
62
+ # whenever a valid API key is provided, even when the response status is not 200.
63
+ key_metadata: nil
64
+ )
65
+ end
66
+
67
+ sig do
68
+ override.returns(
69
+ {
70
+ markdown: String,
71
+ success:
72
+ ContextDev::Models::ParseHandleResponse::Success::TaggedBoolean,
73
+ type: ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol,
74
+ key_metadata: ContextDev::Models::ParseHandleResponse::KeyMetadata
75
+ }
76
+ )
77
+ end
78
+ def to_hash
79
+ end
80
+
81
+ # Indicates success
82
+ module Success
83
+ extend ContextDev::Internal::Type::Enum
84
+
85
+ TaggedBoolean =
86
+ T.type_alias do
87
+ T.all(T::Boolean, ContextDev::Models::ParseHandleResponse::Success)
88
+ end
89
+ OrBoolean = T.type_alias { T::Boolean }
90
+
91
+ TRUE =
92
+ T.let(
93
+ true,
94
+ ContextDev::Models::ParseHandleResponse::Success::TaggedBoolean
95
+ )
96
+
97
+ sig do
98
+ override.returns(
99
+ T::Array[
100
+ ContextDev::Models::ParseHandleResponse::Success::TaggedBoolean
101
+ ]
102
+ )
103
+ end
104
+ def self.values
105
+ end
106
+ end
107
+
108
+ # Detected content type used for parsing
109
+ module Type
110
+ extend ContextDev::Internal::Type::Enum
111
+
112
+ TaggedSymbol =
113
+ T.type_alias do
114
+ T.all(Symbol, ContextDev::Models::ParseHandleResponse::Type)
115
+ end
116
+ OrSymbol = T.type_alias { T.any(Symbol, String) }
117
+
118
+ HTML =
119
+ T.let(
120
+ :html,
121
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
122
+ )
123
+ XML =
124
+ T.let(
125
+ :xml,
126
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
127
+ )
128
+ JSON =
129
+ T.let(
130
+ :json,
131
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
132
+ )
133
+ JSONL =
134
+ T.let(
135
+ :jsonl,
136
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
137
+ )
138
+ TEXT =
139
+ T.let(
140
+ :text,
141
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
142
+ )
143
+ CSV =
144
+ T.let(
145
+ :csv,
146
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
147
+ )
148
+ TSV =
149
+ T.let(
150
+ :tsv,
151
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
152
+ )
153
+ MARKDOWN =
154
+ T.let(
155
+ :markdown,
156
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
157
+ )
158
+ YAML =
159
+ T.let(
160
+ :yaml,
161
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
162
+ )
163
+ PYTHON =
164
+ T.let(
165
+ :python,
166
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
167
+ )
168
+ JAVA =
169
+ T.let(
170
+ :java,
171
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
172
+ )
173
+ JAVASCRIPT =
174
+ T.let(
175
+ :javascript,
176
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
177
+ )
178
+ PHP =
179
+ T.let(
180
+ :php,
181
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
182
+ )
183
+ SHELL =
184
+ T.let(
185
+ :shell,
186
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
187
+ )
188
+ RUBY =
189
+ T.let(
190
+ :ruby,
191
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
192
+ )
193
+ TYPESCRIPT =
194
+ T.let(
195
+ :typescript,
196
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
197
+ )
198
+ RTF =
199
+ T.let(
200
+ :rtf,
201
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
202
+ )
203
+ SRT =
204
+ T.let(
205
+ :srt,
206
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
207
+ )
208
+ CSS =
209
+ T.let(
210
+ :css,
211
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
212
+ )
213
+ SCSS =
214
+ T.let(
215
+ :scss,
216
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
217
+ )
218
+ LESS =
219
+ T.let(
220
+ :less,
221
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
222
+ )
223
+ STYLUS =
224
+ T.let(
225
+ :stylus,
226
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
227
+ )
228
+ SASS =
229
+ T.let(
230
+ :sass,
231
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
232
+ )
233
+ SVG =
234
+ T.let(
235
+ :svg,
236
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
237
+ )
238
+ PDF =
239
+ T.let(
240
+ :pdf,
241
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
242
+ )
243
+ DOCX =
244
+ T.let(
245
+ :docx,
246
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
247
+ )
248
+ DOC =
249
+ T.let(
250
+ :doc,
251
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
252
+ )
253
+ XLSX =
254
+ T.let(
255
+ :xlsx,
256
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
257
+ )
258
+ XLS =
259
+ T.let(
260
+ :xls,
261
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
262
+ )
263
+ PPTX =
264
+ T.let(
265
+ :pptx,
266
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
267
+ )
268
+ PPT =
269
+ T.let(
270
+ :ppt,
271
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
272
+ )
273
+ JPG =
274
+ T.let(
275
+ :jpg,
276
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
277
+ )
278
+ PNG =
279
+ T.let(
280
+ :png,
281
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
282
+ )
283
+ GIF =
284
+ T.let(
285
+ :gif,
286
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
287
+ )
288
+ BMP =
289
+ T.let(
290
+ :bmp,
291
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
292
+ )
293
+ TIFF =
294
+ T.let(
295
+ :tiff,
296
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
297
+ )
298
+ WEBP =
299
+ T.let(
300
+ :webp,
301
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
302
+ )
303
+ PPM =
304
+ T.let(
305
+ :ppm,
306
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
307
+ )
308
+ PBM =
309
+ T.let(
310
+ :pbm,
311
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
312
+ )
313
+ PGM =
314
+ T.let(
315
+ :pgm,
316
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
317
+ )
318
+ PNM =
319
+ T.let(
320
+ :pnm,
321
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
322
+ )
323
+
324
+ sig do
325
+ override.returns(
326
+ T::Array[
327
+ ContextDev::Models::ParseHandleResponse::Type::TaggedSymbol
328
+ ]
329
+ )
330
+ end
331
+ def self.values
332
+ end
333
+ end
334
+
335
+ class KeyMetadata < ContextDev::Internal::Type::BaseModel
336
+ OrHash =
337
+ T.type_alias do
338
+ T.any(
339
+ ContextDev::Models::ParseHandleResponse::KeyMetadata,
340
+ ContextDev::Internal::AnyHash
341
+ )
342
+ end
343
+
344
+ # The number of credits consumed by this request.
345
+ sig { returns(Integer) }
346
+ attr_accessor :credits_consumed
347
+
348
+ # The number of credits remaining for your organization after this request.
349
+ sig { returns(Integer) }
350
+ attr_accessor :credits_remaining
351
+
352
+ # Metadata about the API key used for the request. Included in every response
353
+ # whenever a valid API key is provided, even when the response status is not 200.
354
+ sig do
355
+ params(credits_consumed: Integer, credits_remaining: Integer).returns(
356
+ T.attached_class
357
+ )
358
+ end
359
+ def self.new(
360
+ # The number of credits consumed by this request.
361
+ credits_consumed:,
362
+ # The number of credits remaining for your organization after this request.
363
+ credits_remaining:
364
+ )
365
+ end
366
+
367
+ sig do
368
+ override.returns(
369
+ { credits_consumed: Integer, credits_remaining: Integer }
370
+ )
371
+ end
372
+ def to_hash
373
+ end
374
+ end
375
+ end
376
+ end
377
+ end
@@ -101,8 +101,8 @@ module ContextDev
101
101
  sig { params(max_pages: Integer).void }
102
102
  attr_writer :max_pages
103
103
 
104
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
105
- # inclusive 1-based page range.
104
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
105
+ # detection/OCR to an inclusive 1-based page range.
106
106
  sig { returns(T.nilable(ContextDev::WebWebCrawlMdParams::Pdf)) }
107
107
  attr_reader :pdf
108
108
 
@@ -226,8 +226,8 @@ module ContextDev
226
226
  max_depth: nil,
227
227
  # Maximum number of pages to crawl. Hard cap: 500.
228
228
  max_pages: nil,
229
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
230
- # inclusive 1-based page range.
229
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
230
+ # detection/OCR to an inclusive 1-based page range.
231
231
  pdf: nil,
232
232
  # When true, waits briefly for CSS and transition animations to settle before
233
233
  # extracting each crawled page. Defaults to false. This adds a bit of latency in
@@ -528,6 +528,15 @@ module ContextDev
528
528
  sig { params(end_: Integer).void }
529
529
  attr_writer :end_
530
530
 
531
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
532
+ # recognized text at each image's position in page reading order while preserving
533
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
534
+ sig { returns(T.nilable(T::Boolean)) }
535
+ attr_reader :ocr
536
+
537
+ sig { params(ocr: T::Boolean).void }
538
+ attr_writer :ocr
539
+
531
540
  # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
532
541
  # entirely (not included in results and not counted as failures).
533
542
  sig { returns(T.nilable(T::Boolean)) }
@@ -543,11 +552,12 @@ module ContextDev
543
552
  sig { params(start: Integer).void }
544
553
  attr_writer :start
545
554
 
546
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
547
- # inclusive 1-based page range.
555
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
556
+ # detection/OCR to an inclusive 1-based page range.
548
557
  sig do
549
558
  params(
550
559
  end_: Integer,
560
+ ocr: T::Boolean,
551
561
  should_parse: T::Boolean,
552
562
  start: Integer
553
563
  ).returns(T.attached_class)
@@ -556,6 +566,10 @@ module ContextDev
556
566
  # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
557
567
  # Must be greater than or equal to start when both are provided.
558
568
  end_: nil,
569
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
570
+ # recognized text at each image's position in page reading order while preserving
571
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
572
+ ocr: nil,
559
573
  # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
560
574
  # entirely (not included in results and not counted as failures).
561
575
  should_parse: nil,
@@ -566,7 +580,12 @@ module ContextDev
566
580
 
567
581
  sig do
568
582
  override.returns(
569
- { end_: Integer, should_parse: T::Boolean, start: Integer }
583
+ {
584
+ end_: Integer,
585
+ ocr: T::Boolean,
586
+ should_parse: T::Boolean,
587
+ start: Integer
588
+ }
570
589
  )
571
590
  end
572
591
  def to_hash
@@ -77,8 +77,8 @@ module ContextDev
77
77
  sig { params(max_age_ms: Integer).void }
78
78
  attr_writer :max_age_ms
79
79
 
80
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
81
- # inclusive 1-based page range.
80
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
81
+ # detection/OCR to an inclusive 1-based page range.
82
82
  sig { returns(T.nilable(ContextDev::WebWebScrapeHTMLParams::Pdf)) }
83
83
  attr_reader :pdf
84
84
 
@@ -160,8 +160,8 @@ module ContextDev
160
160
  # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
161
161
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
162
162
  max_age_ms: nil,
163
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
164
- # inclusive 1-based page range.
163
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
164
+ # detection/OCR to an inclusive 1-based page range.
165
165
  pdf: nil,
166
166
  # When true, waits briefly for CSS and transition animations to settle before
167
167
  # extracting HTML. Defaults to false. This adds a bit of latency in exchange for
@@ -649,6 +649,15 @@ module ContextDev
649
649
  sig { params(end_: Integer).void }
650
650
  attr_writer :end_
651
651
 
652
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
653
+ # recognized text at each image's position in page reading order while preserving
654
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
655
+ sig { returns(T.nilable(T::Boolean)) }
656
+ attr_reader :ocr
657
+
658
+ sig { params(ocr: T::Boolean).void }
659
+ attr_writer :ocr
660
+
652
661
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
653
662
  # a 400 WEBSITE_ACCESS_ERROR is returned.
654
663
  sig { returns(T.nilable(T::Boolean)) }
@@ -664,11 +673,12 @@ module ContextDev
664
673
  sig { params(start: Integer).void }
665
674
  attr_writer :start
666
675
 
667
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
668
- # inclusive 1-based page range.
676
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
677
+ # detection/OCR to an inclusive 1-based page range.
669
678
  sig do
670
679
  params(
671
680
  end_: Integer,
681
+ ocr: T::Boolean,
672
682
  should_parse: T::Boolean,
673
683
  start: Integer
674
684
  ).returns(T.attached_class)
@@ -677,6 +687,10 @@ module ContextDev
677
687
  # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
678
688
  # Must be greater than or equal to start when both are provided.
679
689
  end_: nil,
690
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
691
+ # recognized text at each image's position in page reading order while preserving
692
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
693
+ ocr: nil,
680
694
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
681
695
  # a 400 WEBSITE_ACCESS_ERROR is returned.
682
696
  should_parse: nil,
@@ -687,7 +701,12 @@ module ContextDev
687
701
 
688
702
  sig do
689
703
  override.returns(
690
- { end_: Integer, should_parse: T::Boolean, start: Integer }
704
+ {
705
+ end_: Integer,
706
+ ocr: T::Boolean,
707
+ should_parse: T::Boolean,
708
+ start: Integer
709
+ }
691
710
  )
692
711
  end
693
712
  def to_hash
@@ -87,8 +87,8 @@ module ContextDev
87
87
  sig { params(max_age_ms: Integer).void }
88
88
  attr_writer :max_age_ms
89
89
 
90
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
91
- # inclusive 1-based page range.
90
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
91
+ # detection/OCR to an inclusive 1-based page range.
92
92
  sig { returns(T.nilable(ContextDev::WebWebScrapeMdParams::Pdf)) }
93
93
  attr_reader :pdf
94
94
 
@@ -185,8 +185,8 @@ module ContextDev
185
185
  # younger than this many milliseconds. Defaults to 1 day (86400000 ms) when
186
186
  # omitted. Max is 30 days (2592000000 ms). Set to 0 to always scrape fresh.
187
187
  max_age_ms: nil,
188
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
189
- # inclusive 1-based page range.
188
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
189
+ # detection/OCR to an inclusive 1-based page range.
190
190
  pdf: nil,
191
191
  # When true, waits briefly for CSS and transition animations to settle before
192
192
  # converting to Markdown. Defaults to false. This adds a bit of latency in
@@ -475,6 +475,15 @@ module ContextDev
475
475
  sig { params(end_: Integer).void }
476
476
  attr_writer :end_
477
477
 
478
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
479
+ # recognized text at each image's position in page reading order while preserving
480
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
481
+ sig { returns(T.nilable(T::Boolean)) }
482
+ attr_reader :ocr
483
+
484
+ sig { params(ocr: T::Boolean).void }
485
+ attr_writer :ocr
486
+
478
487
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
479
488
  # a 400 WEBSITE_ACCESS_ERROR is returned.
480
489
  sig { returns(T.nilable(T::Boolean)) }
@@ -490,11 +499,12 @@ module ContextDev
490
499
  sig { params(start: Integer).void }
491
500
  attr_writer :start
492
501
 
493
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
494
- # inclusive 1-based page range.
502
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
503
+ # detection/OCR to an inclusive 1-based page range.
495
504
  sig do
496
505
  params(
497
506
  end_: Integer,
507
+ ocr: T::Boolean,
498
508
  should_parse: T::Boolean,
499
509
  start: Integer
500
510
  ).returns(T.attached_class)
@@ -503,6 +513,10 @@ module ContextDev
503
513
  # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
504
514
  # Must be greater than or equal to start when both are provided.
505
515
  end_: nil,
516
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
517
+ # recognized text at each image's position in page reading order while preserving
518
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
519
+ ocr: nil,
506
520
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
507
521
  # a 400 WEBSITE_ACCESS_ERROR is returned.
508
522
  should_parse: nil,
@@ -513,7 +527,12 @@ module ContextDev
513
527
 
514
528
  sig do
515
529
  override.returns(
516
- { end_: Integer, should_parse: T::Boolean, start: Integer }
530
+ {
531
+ end_: Integer,
532
+ ocr: T::Boolean,
533
+ should_parse: T::Boolean,
534
+ start: Integer
535
+ }
517
536
  )
518
537
  end
519
538
  def to_hash
@@ -11,6 +11,12 @@ module ContextDev
11
11
  )
12
12
  end
13
13
 
14
+ # UTF-8 byte length of the returned Markdown. Use 0 to identify an empty result
15
+ # and compare small values against your workload's minimum useful-content
16
+ # threshold.
17
+ sig { returns(Integer) }
18
+ attr_accessor :content_length
19
+
14
20
  # Page content converted to GitHub Flavored Markdown
15
21
  sig { returns(String) }
16
22
  attr_accessor :markdown
@@ -57,6 +63,7 @@ module ContextDev
57
63
 
58
64
  sig do
59
65
  params(
66
+ content_length: Integer,
60
67
  markdown: String,
61
68
  metadata:
62
69
  ContextDev::Models::WebWebScrapeMdResponse::Metadata::OrHash,
@@ -68,6 +75,10 @@ module ContextDev
68
75
  ).returns(T.attached_class)
69
76
  end
70
77
  def self.new(
78
+ # UTF-8 byte length of the returned Markdown. Use 0 to identify an empty result
79
+ # and compare small values against your workload's minimum useful-content
80
+ # threshold.
81
+ content_length:,
71
82
  # Page content converted to GitHub Flavored Markdown
72
83
  markdown:,
73
84
  # Metadata extracted from the scraped page HTML.
@@ -85,6 +96,7 @@ module ContextDev
85
96
  sig do
86
97
  override.returns(
87
98
  {
99
+ content_length: Integer,
88
100
  markdown: String,
89
101
  metadata: ContextDev::Models::WebWebScrapeMdResponse::Metadata,
90
102
  success: