context.dev 2.1.0 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +23 -0
  3. data/README.md +1 -1
  4. data/lib/context_dev/client.rb +7 -2
  5. data/lib/context_dev/internal/type/base_model.rb +5 -5
  6. data/lib/context_dev/models/monitor_create_params.rb +44 -14
  7. data/lib/context_dev/models/monitor_create_response.rb +112 -15
  8. data/lib/context_dev/models/monitor_list_account_runs_response.rb +23 -1
  9. data/lib/context_dev/models/monitor_list_response.rb +116 -15
  10. data/lib/context_dev/models/monitor_list_runs_response.rb +23 -1
  11. data/lib/context_dev/models/monitor_retrieve_change_response.rb +10 -10
  12. data/lib/context_dev/models/monitor_retrieve_response.rb +113 -15
  13. data/lib/context_dev/models/monitor_update_params.rb +44 -14
  14. data/lib/context_dev/models/monitor_update_response.rb +112 -15
  15. data/lib/context_dev/models/parse_handle_params.rb +110 -0
  16. data/lib/context_dev/models/parse_handle_response.rb +132 -0
  17. data/lib/context_dev/models/web_web_crawl_md_params.rb +16 -6
  18. data/lib/context_dev/models/web_web_scrape_html_params.rb +16 -6
  19. data/lib/context_dev/models/web_web_scrape_md_params.rb +16 -6
  20. data/lib/context_dev/models/web_web_scrape_md_response.rb +11 -1
  21. data/lib/context_dev/models/webhook_delivery.rb +109 -0
  22. data/lib/context_dev/models.rb +4 -0
  23. data/lib/context_dev/resources/monitors.rb +3 -2
  24. data/lib/context_dev/resources/parse.rb +63 -0
  25. data/lib/context_dev/resources/web.rb +19 -4
  26. data/lib/context_dev/version.rb +1 -1
  27. data/lib/context_dev.rb +4 -0
  28. data/rbi/context_dev/client.rbi +6 -2
  29. data/rbi/context_dev/models/monitor_create_params.rbi +106 -16
  30. data/rbi/context_dev/models/monitor_create_response.rbi +255 -17
  31. data/rbi/context_dev/models/monitor_list_account_runs_response.rbi +39 -3
  32. data/rbi/context_dev/models/monitor_list_response.rbi +256 -16
  33. data/rbi/context_dev/models/monitor_list_runs_response.rbi +39 -3
  34. data/rbi/context_dev/models/monitor_retrieve_change_response.rbi +12 -15
  35. data/rbi/context_dev/models/monitor_retrieve_response.rbi +255 -17
  36. data/rbi/context_dev/models/monitor_update_params.rbi +106 -16
  37. data/rbi/context_dev/models/monitor_update_response.rbi +255 -17
  38. data/rbi/context_dev/models/parse_handle_params.rbi +163 -0
  39. data/rbi/context_dev/models/parse_handle_response.rbi +377 -0
  40. data/rbi/context_dev/models/web_web_crawl_md_params.rbi +26 -7
  41. data/rbi/context_dev/models/web_web_scrape_html_params.rbi +26 -7
  42. data/rbi/context_dev/models/web_web_scrape_md_params.rbi +26 -7
  43. data/rbi/context_dev/models/web_web_scrape_md_response.rbi +12 -0
  44. data/rbi/context_dev/models/webhook_delivery.rbi +172 -0
  45. data/rbi/context_dev/models.rbi +4 -0
  46. data/rbi/context_dev/resources/monitors.rbi +3 -2
  47. data/rbi/context_dev/resources/parse.rbi +65 -0
  48. data/rbi/context_dev/resources/web.rbi +22 -7
  49. data/sig/context_dev/client.rbs +2 -0
  50. data/sig/context_dev/models/monitor_create_params.rbs +32 -3
  51. data/sig/context_dev/models/monitor_create_response.rbs +85 -6
  52. data/sig/context_dev/models/monitor_list_account_runs_response.rbs +21 -3
  53. data/sig/context_dev/models/monitor_list_response.rbs +85 -6
  54. data/sig/context_dev/models/monitor_list_runs_response.rbs +21 -3
  55. data/sig/context_dev/models/monitor_retrieve_change_response.rbs +8 -10
  56. data/sig/context_dev/models/monitor_retrieve_response.rbs +85 -6
  57. data/sig/context_dev/models/monitor_update_params.rbs +32 -3
  58. data/sig/context_dev/models/monitor_update_response.rbs +85 -6
  59. data/sig/context_dev/models/parse_handle_params.rbs +96 -0
  60. data/sig/context_dev/models/parse_handle_response.rbs +159 -0
  61. data/sig/context_dev/models/web_web_crawl_md_params.rbs +13 -2
  62. data/sig/context_dev/models/web_web_scrape_html_params.rbs +13 -2
  63. data/sig/context_dev/models/web_web_scrape_md_params.rbs +13 -2
  64. data/sig/context_dev/models/web_web_scrape_md_response.rbs +5 -0
  65. data/sig/context_dev/models/webhook_delivery.rbs +81 -0
  66. data/sig/context_dev/models.rbs +4 -0
  67. data/sig/context_dev/resources/parse.rbs +22 -0
  68. metadata +14 -2
@@ -103,7 +103,15 @@ module ContextDev
103
103
  # @return [ContextDev::Models::MonitorUpdateResponse::Webhook, nil]
104
104
  optional :webhook, -> { ContextDev::Models::MonitorUpdateResponse::Webhook }, nil?: true
105
105
 
106
- # @!method initialize(id:, change_detection:, created_at:, mode:, name:, schedule:, status:, target:, updated_at:, baseline: nil, last_change_at: nil, last_error: nil, last_run_at: nil, next_run_at: nil, tags: nil, webhook: nil)
106
+ # @!attribute webhook_failure
107
+ # Present while webhook deliveries are failing consecutively; null when deliveries
108
+ # are healthy or no webhook is configured. Cleared on the next successful delivery
109
+ # and when the webhook URL changes.
110
+ #
111
+ # @return [ContextDev::Models::MonitorUpdateResponse::WebhookFailure, nil]
112
+ optional :webhook_failure, -> { ContextDev::Models::MonitorUpdateResponse::WebhookFailure }, nil?: true
113
+
114
+ # @!method initialize(id:, change_detection:, created_at:, mode:, name:, schedule:, status:, target:, updated_at:, baseline: nil, last_change_at: nil, last_error: nil, last_run_at: nil, next_run_at: nil, tags: nil, webhook: nil, webhook_failure: nil)
107
115
  # Some parameter documentations has been truncated, see
108
116
  # {ContextDev::Models::MonitorUpdateResponse} for more details.
109
117
  #
@@ -141,6 +149,8 @@ module ContextDev
141
149
  # @param tags [Array<String>] User-defined tags for grouping and filtering monitors and their changes.
142
150
  #
143
151
  # @param webhook [ContextDev::Models::MonitorUpdateResponse::Webhook, nil]
152
+ #
153
+ # @param webhook_failure [ContextDev::Models::MonitorUpdateResponse::WebhookFailure, nil] Present while webhook deliveries are failing consecutively; null when deliveries
144
154
 
145
155
  # Discriminated union describing how changes are detected.
146
156
  #
@@ -153,7 +163,7 @@ module ContextDev
153
163
  # Detect exact changes. For page targets, this means visible text diffs. For sitemap targets, this means URL additions and removals.
154
164
  variant :exact, -> { ContextDev::Models::MonitorUpdateResponse::ChangeDetection::Exact }
155
165
 
156
- # Detect meaning-level changes to the extracted data, ignoring cosmetic or paraphrase-only differences. What is watched is determined by the extract target's `schema` and `instructions`.
166
+ # Detect meaning-level changes to tracked page content, ignoring cosmetic or paraphrase-only differences. Which changes are meaningful is judged against the extract target's `instructions` (and `schema`, when provided).
157
167
  variant :semantic, -> { ContextDev::Models::MonitorUpdateResponse::ChangeDetection::Semantic }
158
168
 
159
169
  class Exact < ContextDev::Internal::Type::BaseModel
@@ -181,9 +191,9 @@ module ContextDev
181
191
  optional :confidence_threshold, Float
182
192
 
183
193
  # @!method initialize(confidence_threshold: nil, type: :semantic)
184
- # Detect meaning-level changes to the extracted data, ignoring cosmetic or
185
- # paraphrase-only differences. What is watched is determined by the extract
186
- # target's `schema` and `instructions`.
194
+ # Detect meaning-level changes to tracked page content, ignoring cosmetic or
195
+ # paraphrase-only differences. Which changes are meaningful is judged against the
196
+ # extract target's `instructions` (and `schema`, when provided).
187
197
  #
188
198
  # @param confidence_threshold [Float]
189
199
  # @param type [Symbol, :semantic]
@@ -295,7 +305,7 @@ module ContextDev
295
305
  # Watch a sitemap for URL additions and removals. Crawled URLs are normalized (lowercased host, no trailing slash/fragment) and scoped to the monitored site and its subdomains before comparison. On a detected difference the sitemap is re-fetched within the same run and only URLs both observations agree on are reported, suppressing transient crawl flaps.
296
306
  variant :sitemap, -> { ContextDev::Models::MonitorUpdateResponse::Target::Sitemap }
297
307
 
298
- # Watch the monitor-relevant pages of a site for meaningful changes. A crawl guided by `schema`/`instructions` selects up to `max_pages` relevant pages to track; each run re-checks exactly those pages, and confirmed content changes are judged against the monitor's instructions. The tracked page set is refreshed by a periodic re-discovery crawl.
308
+ # Watch the monitor-relevant pages of a site for meaningful changes. A crawl guided by `schema`/`instructions` selects up to `max_pages` relevant pages to track; each run re-checks exactly those pages, and confirmed content changes are judged for relevance against the monitor's `instructions` (and `schema`, when provided). The tracked page set is refreshed by a periodic re-discovery crawl.
299
309
  variant :extract, -> { ContextDev::Models::MonitorUpdateResponse::Target::Extract }
300
310
 
301
311
  class Page < ContextDev::Internal::Type::BaseModel
@@ -410,9 +420,14 @@ module ContextDev
410
420
  optional :max_pages, Integer
411
421
 
412
422
  # @!attribute schema
413
- # JSON Schema describing the data you care about. It guides which pages are
414
- # selected for tracking and gives the change judge context on what matters. If
415
- # omitted, a default summary + key-points schema is used.
423
+ # JSON Schema describing the data you care about. It is used three ways: it guides
424
+ # which pages are selected for tracking, it gives the change judge extra context
425
+ # on which changes matter (alongside `instructions`), and it defines the shape of
426
+ # the baseline `data` snapshot on GET /monitors/{monitor_id} (refreshed at most
427
+ # about once a day). It is not a response format for changes: change events and
428
+ # webhook payloads always contain diffs, summaries, and evidence excerpts — never
429
+ # data in this schema's shape. If omitted, a default summary + key-points schema
430
+ # is used.
416
431
  #
417
432
  # @return [Hash{Symbol=>Object}, nil]
418
433
  optional :schema, ContextDev::Internal::Type::HashOf[ContextDev::Internal::Type::Unknown]
@@ -424,8 +439,8 @@ module ContextDev
424
439
  # Watch the monitor-relevant pages of a site for meaningful changes. A crawl
425
440
  # guided by `schema`/`instructions` selects up to `max_pages` relevant pages to
426
441
  # track; each run re-checks exactly those pages, and confirmed content changes are
427
- # judged against the monitor's instructions. The tracked page set is refreshed by
428
- # a periodic re-discovery crawl.
442
+ # judged for relevance against the monitor's `instructions` (and `schema`, when
443
+ # provided). The tracked page set is refreshed by a periodic re-discovery crawl.
429
444
  #
430
445
  # @param instructions [String] Natural-language instructions guiding which pages and facts to track and which c
431
446
  #
@@ -437,7 +452,7 @@ module ContextDev
437
452
  #
438
453
  # @param max_pages [Integer] Maximum number of pages to track.
439
454
  #
440
- # @param schema [Hash{Symbol=>Object}] JSON Schema describing the data you care about. It guides which pages are select
455
+ # @param schema [Hash{Symbol=>Object}] JSON Schema describing the data you care about. It is used three ways: it guides
441
456
  #
442
457
  # @param type [Symbol, :extract]
443
458
  end
@@ -578,11 +593,21 @@ module ContextDev
578
593
  # @see ContextDev::Models::MonitorUpdateResponse#webhook
579
594
  class Webhook < ContextDev::Internal::Type::BaseModel
580
595
  # @!attribute url
581
- # Webhook URL called when a change is detected.
596
+ # Webhook URL events are delivered to.
582
597
  #
583
598
  # @return [String]
584
599
  required :url, String
585
600
 
601
+ # @!attribute events
602
+ # Events delivered to this endpoint. `change.detected` fires only when a run
603
+ # detects a change; `run.completed` fires on every completed run — including runs
604
+ # that detected no change — and embeds the change when one was detected. Defaults
605
+ # to `["change.detected"]` when omitted.
606
+ #
607
+ # @return [Array<Symbol, ContextDev::Models::MonitorUpdateResponse::Webhook::Event>, nil]
608
+ optional :events,
609
+ -> { ContextDev::Internal::Type::ArrayOf[enum: ContextDev::Models::MonitorUpdateResponse::Webhook::Event] }
610
+
586
611
  response_only do
587
612
  # @!attribute secret
588
613
  # Signing secret used to verify webhook authenticity. Each delivery includes an
@@ -595,13 +620,85 @@ module ContextDev
595
620
  optional :secret, String
596
621
  end
597
622
 
598
- # @!method initialize(url:, secret: nil)
623
+ # @!method initialize(url:, events: nil, secret: nil)
599
624
  # Some parameter documentations has been truncated, see
600
625
  # {ContextDev::Models::MonitorUpdateResponse::Webhook} for more details.
601
626
  #
602
- # @param url [String] Webhook URL called when a change is detected.
627
+ # @param url [String] Webhook URL events are delivered to.
628
+ #
629
+ # @param events [Array<Symbol, ContextDev::Models::MonitorUpdateResponse::Webhook::Event>] Events delivered to this endpoint. `change.detected` fires only when a run detec
603
630
  #
604
631
  # @param secret [String] Signing secret used to verify webhook authenticity. Each delivery includes an `X
632
+
633
+ module Event
634
+ extend ContextDev::Internal::Type::Enum
635
+
636
+ CHANGE_DETECTED = :"change.detected"
637
+ RUN_COMPLETED = :"run.completed"
638
+
639
+ # @!method self.values
640
+ # @return [Array<Symbol>]
641
+ end
642
+ end
643
+
644
+ # @see ContextDev::Models::MonitorUpdateResponse#webhook_failure
645
+ class WebhookFailure < ContextDev::Internal::Type::BaseModel
646
+ # @!attribute consecutive_failures
647
+ # Number of consecutive delivery attempts that did not succeed.
648
+ #
649
+ # @return [Integer]
650
+ required :consecutive_failures, Integer
651
+
652
+ # @!attribute last_failed_at
653
+ #
654
+ # @return [Time]
655
+ required :last_failed_at, Time
656
+
657
+ # @!attribute last_message
658
+ # Human-readable description of the most recent failure.
659
+ #
660
+ # @return [String]
661
+ required :last_message, String
662
+
663
+ # @!attribute last_status
664
+ # Outcome of the most recent failed delivery. rejected means a non-2xx response;
665
+ # failed means no HTTP response was received; skipped_unsafe_url means the URL
666
+ # failed the public-endpoint safety check.
667
+ #
668
+ # @return [Symbol, ContextDev::Models::MonitorUpdateResponse::WebhookFailure::LastStatus]
669
+ required :last_status, enum: -> { ContextDev::Models::MonitorUpdateResponse::WebhookFailure::LastStatus }
670
+
671
+ # @!method initialize(consecutive_failures:, last_failed_at:, last_message:, last_status:)
672
+ # Some parameter documentations has been truncated, see
673
+ # {ContextDev::Models::MonitorUpdateResponse::WebhookFailure} for more details.
674
+ #
675
+ # Present while webhook deliveries are failing consecutively; null when deliveries
676
+ # are healthy or no webhook is configured. Cleared on the next successful delivery
677
+ # and when the webhook URL changes.
678
+ #
679
+ # @param consecutive_failures [Integer] Number of consecutive delivery attempts that did not succeed.
680
+ #
681
+ # @param last_failed_at [Time]
682
+ #
683
+ # @param last_message [String] Human-readable description of the most recent failure.
684
+ #
685
+ # @param last_status [Symbol, ContextDev::Models::MonitorUpdateResponse::WebhookFailure::LastStatus] Outcome of the most recent failed delivery. rejected means a non-2xx response; f
686
+
687
+ # Outcome of the most recent failed delivery. rejected means a non-2xx response;
688
+ # failed means no HTTP response was received; skipped_unsafe_url means the URL
689
+ # failed the public-endpoint safety check.
690
+ #
691
+ # @see ContextDev::Models::MonitorUpdateResponse::WebhookFailure#last_status
692
+ module LastStatus
693
+ extend ContextDev::Internal::Type::Enum
694
+
695
+ REJECTED = :rejected
696
+ FAILED = :failed
697
+ SKIPPED_UNSAFE_URL = :skipped_unsafe_url
698
+
699
+ # @!method self.values
700
+ # @return [Array<Symbol>]
701
+ end
605
702
  end
606
703
  end
607
704
  end
@@ -0,0 +1,110 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ContextDev
4
+ module Models
5
+ # @see ContextDev::Resources::Parse#handle
6
+ class ParseHandleParams < ContextDev::Internal::Type::BaseModel
7
+ extend ContextDev::Internal::Type::RequestParameters::Converter
8
+ include ContextDev::Internal::Type::RequestParameters
9
+
10
+ # @!attribute body
11
+ #
12
+ # @return [Pathname, StringIO, IO, String, ContextDev::FilePart]
13
+ required :body, ContextDev::Internal::Type::FileInput
14
+
15
+ # @!attribute base_url
16
+ # Optional HTTP(S) source document URL used to resolve relative links and image
17
+ # references. Relative references remain relative when omitted.
18
+ #
19
+ # @return [String, nil]
20
+ optional :base_url, String
21
+
22
+ # @!attribute extension
23
+ # Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv,
24
+ # md, py, rtf, jpg, png, or txt.
25
+ #
26
+ # @return [String, nil]
27
+ optional :extension, String
28
+
29
+ # @!attribute filename
30
+ # Optional filename hint used to infer the extension when extension is omitted.
31
+ #
32
+ # @return [String, nil]
33
+ optional :filename, String
34
+
35
+ # @!attribute include_images
36
+ # Include image references in Markdown output
37
+ #
38
+ # @return [Boolean, nil]
39
+ optional :include_images, ContextDev::Internal::Type::Boolean
40
+
41
+ # @!attribute include_links
42
+ # Preserve hyperlinks in Markdown output
43
+ #
44
+ # @return [Boolean, nil]
45
+ optional :include_links, ContextDev::Internal::Type::Boolean
46
+
47
+ # @!attribute ocr
48
+ # When true for PDF inputs, detect and OCR images embedded in the selected pages,
49
+ # inserting recognized text at each image's position in page reading order while
50
+ # preserving the PDF text layer. pdfStart/pdfEnd limit the inclusive page range.
51
+ # This is separate from automatic scanned-PDF OCR fallback.
52
+ #
53
+ # @return [Boolean, nil]
54
+ optional :ocr, ContextDev::Internal::Type::Boolean
55
+
56
+ # @!attribute pdf_end
57
+ # Last 1-based PDF page to parse. When omitted, parsing ends at the last page.
58
+ # Must be greater than or equal to pdfStart when both are provided.
59
+ #
60
+ # @return [Integer, nil]
61
+ optional :pdf_end, Integer
62
+
63
+ # @!attribute pdf_start
64
+ # First 1-based PDF page to parse. When omitted, parsing starts at the first page.
65
+ #
66
+ # @return [Integer, nil]
67
+ optional :pdf_start, Integer
68
+
69
+ # @!attribute shorten_base64_images
70
+ # Shorten base64-encoded image data in the Markdown output
71
+ #
72
+ # @return [Boolean, nil]
73
+ optional :shorten_base64_images, ContextDev::Internal::Type::Boolean
74
+
75
+ # @!attribute use_main_content_only
76
+ # Extract only the main content from HTML-like inputs
77
+ #
78
+ # @return [Boolean, nil]
79
+ optional :use_main_content_only, ContextDev::Internal::Type::Boolean
80
+
81
+ # @!method initialize(body:, base_url: nil, extension: nil, filename: nil, include_images: nil, include_links: nil, ocr: nil, pdf_end: nil, pdf_start: nil, shorten_base64_images: nil, use_main_content_only: nil, request_options: {})
82
+ # Some parameter documentations has been truncated, see
83
+ # {ContextDev::Models::ParseHandleParams} for more details.
84
+ #
85
+ # @param body [Pathname, StringIO, IO, String, ContextDev::FilePart]
86
+ #
87
+ # @param base_url [String] Optional HTTP(S) source document URL used to resolve relative links and image re
88
+ #
89
+ # @param extension [String] Optional file extension hint, such as pdf, docx, xlsx, pptx, html, json, csv, md
90
+ #
91
+ # @param filename [String] Optional filename hint used to infer the extension when extension is omitted.
92
+ #
93
+ # @param include_images [Boolean] Include image references in Markdown output
94
+ #
95
+ # @param include_links [Boolean] Preserve hyperlinks in Markdown output
96
+ #
97
+ # @param ocr [Boolean] When true for PDF inputs, detect and OCR images embedded in the selected pages,
98
+ #
99
+ # @param pdf_end [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
100
+ #
101
+ # @param pdf_start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
102
+ #
103
+ # @param shorten_base64_images [Boolean] Shorten base64-encoded image data in the Markdown output
104
+ #
105
+ # @param use_main_content_only [Boolean] Extract only the main content from HTML-like inputs
106
+ #
107
+ # @param request_options [ContextDev::RequestOptions, Hash{Symbol=>Object}]
108
+ end
109
+ end
110
+ end
@@ -0,0 +1,132 @@
1
+ # frozen_string_literal: true
2
+
3
+ module ContextDev
4
+ module Models
5
+ # @see ContextDev::Resources::Parse#handle
6
+ class ParseHandleResponse < ContextDev::Internal::Type::BaseModel
7
+ # @!attribute markdown
8
+ # Input bytes converted to GitHub Flavored Markdown
9
+ #
10
+ # @return [String]
11
+ required :markdown, String
12
+
13
+ # @!attribute success
14
+ # Indicates success
15
+ #
16
+ # @return [Boolean, ContextDev::Models::ParseHandleResponse::Success]
17
+ required :success, enum: -> { ContextDev::Models::ParseHandleResponse::Success }
18
+
19
+ # @!attribute type
20
+ # Detected content type used for parsing
21
+ #
22
+ # @return [Symbol, ContextDev::Models::ParseHandleResponse::Type]
23
+ required :type, enum: -> { ContextDev::Models::ParseHandleResponse::Type }
24
+
25
+ # @!attribute key_metadata
26
+ # Metadata about the API key used for the request. Included in every response
27
+ # whenever a valid API key is provided, even when the response status is not 200.
28
+ #
29
+ # @return [ContextDev::Models::ParseHandleResponse::KeyMetadata, nil]
30
+ optional :key_metadata, -> { ContextDev::Models::ParseHandleResponse::KeyMetadata }
31
+
32
+ # @!method initialize(markdown:, success:, type:, key_metadata: nil)
33
+ # Some parameter documentations has been truncated, see
34
+ # {ContextDev::Models::ParseHandleResponse} for more details.
35
+ #
36
+ # @param markdown [String] Input bytes converted to GitHub Flavored Markdown
37
+ #
38
+ # @param success [Boolean, ContextDev::Models::ParseHandleResponse::Success] Indicates success
39
+ #
40
+ # @param type [Symbol, ContextDev::Models::ParseHandleResponse::Type] Detected content type used for parsing
41
+ #
42
+ # @param key_metadata [ContextDev::Models::ParseHandleResponse::KeyMetadata] Metadata about the API key used for the request. Included in every response when
43
+
44
+ # Indicates success
45
+ #
46
+ # @see ContextDev::Models::ParseHandleResponse#success
47
+ module Success
48
+ extend ContextDev::Internal::Type::Enum
49
+
50
+ TRUE = true
51
+
52
+ # @!method self.values
53
+ # @return [Array<Boolean>]
54
+ end
55
+
56
+ # Detected content type used for parsing
57
+ #
58
+ # @see ContextDev::Models::ParseHandleResponse#type
59
+ module Type
60
+ extend ContextDev::Internal::Type::Enum
61
+
62
+ HTML = :html
63
+ XML = :xml
64
+ JSON = :json
65
+ JSONL = :jsonl
66
+ TEXT = :text
67
+ CSV = :csv
68
+ TSV = :tsv
69
+ MARKDOWN = :markdown
70
+ YAML = :yaml
71
+ PYTHON = :python
72
+ JAVA = :java
73
+ JAVASCRIPT = :javascript
74
+ PHP = :php
75
+ SHELL = :shell
76
+ RUBY = :ruby
77
+ TYPESCRIPT = :typescript
78
+ RTF = :rtf
79
+ SRT = :srt
80
+ CSS = :css
81
+ SCSS = :scss
82
+ LESS = :less
83
+ STYLUS = :stylus
84
+ SASS = :sass
85
+ SVG = :svg
86
+ PDF = :pdf
87
+ DOCX = :docx
88
+ DOC = :doc
89
+ XLSX = :xlsx
90
+ XLS = :xls
91
+ PPTX = :pptx
92
+ PPT = :ppt
93
+ JPG = :jpg
94
+ PNG = :png
95
+ GIF = :gif
96
+ BMP = :bmp
97
+ TIFF = :tiff
98
+ WEBP = :webp
99
+ PPM = :ppm
100
+ PBM = :pbm
101
+ PGM = :pgm
102
+ PNM = :pnm
103
+
104
+ # @!method self.values
105
+ # @return [Array<Symbol>]
106
+ end
107
+
108
+ # @see ContextDev::Models::ParseHandleResponse#key_metadata
109
+ class KeyMetadata < ContextDev::Internal::Type::BaseModel
110
+ # @!attribute credits_consumed
111
+ # The number of credits consumed by this request.
112
+ #
113
+ # @return [Integer]
114
+ required :credits_consumed, Integer
115
+
116
+ # @!attribute credits_remaining
117
+ # The number of credits remaining for your organization after this request.
118
+ #
119
+ # @return [Integer]
120
+ required :credits_remaining, Integer
121
+
122
+ # @!method initialize(credits_consumed:, credits_remaining:)
123
+ # Metadata about the API key used for the request. Included in every response
124
+ # whenever a valid API key is provided, even when the response status is not 200.
125
+ #
126
+ # @param credits_consumed [Integer] The number of credits consumed by this request.
127
+ #
128
+ # @param credits_remaining [Integer] The number of credits remaining for your organization after this request.
129
+ end
130
+ end
131
+ end
132
+ end
@@ -86,8 +86,8 @@ module ContextDev
86
86
  optional :max_pages, Integer, api_name: :maxPages
87
87
 
88
88
  # @!attribute pdf
89
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
90
- # inclusive 1-based page range.
89
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
90
+ # detection/OCR to an inclusive 1-based page range.
91
91
  #
92
92
  # @return [ContextDev::Models::WebWebCrawlMdParams::Pdf, nil]
93
93
  optional :pdf, -> { ContextDev::WebWebCrawlMdParams::Pdf }
@@ -169,7 +169,7 @@ module ContextDev
169
169
  #
170
170
  # @param max_pages [Integer] Maximum number of pages to crawl. Hard cap: 500.
171
171
  #
172
- # @param pdf [ContextDev::Models::WebWebCrawlMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
172
+ # @param pdf [ContextDev::Models::WebWebCrawlMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
173
173
  #
174
174
  # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
175
175
  #
@@ -410,6 +410,14 @@ module ContextDev
410
410
  # @return [Integer, nil]
411
411
  optional :end_, Integer, api_name: :end
412
412
 
413
+ # @!attribute ocr
414
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
415
+ # recognized text at each image's position in page reading order while preserving
416
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
417
+ #
418
+ # @return [Boolean, nil]
419
+ optional :ocr, ContextDev::Internal::Type::Boolean
420
+
413
421
  # @!attribute should_parse
414
422
  # When true, PDF pages are fetched and parsed. When false, PDF pages are skipped
415
423
  # entirely (not included in results and not counted as failures).
@@ -423,15 +431,17 @@ module ContextDev
423
431
  # @return [Integer, nil]
424
432
  optional :start, Integer
425
433
 
426
- # @!method initialize(end_: nil, should_parse: nil, start: nil)
434
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
427
435
  # Some parameter documentations has been truncated, see
428
436
  # {ContextDev::Models::WebWebCrawlMdParams::Pdf} for more details.
429
437
  #
430
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
431
- # inclusive 1-based page range.
438
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
439
+ # detection/OCR to an inclusive 1-based page range.
432
440
  #
433
441
  # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
434
442
  #
443
+ # @param ocr [Boolean] When true, detect and OCR images embedded in the selected PDF pages, inserting r
444
+ #
435
445
  # @param should_parse [Boolean] When true, PDF pages are fetched and parsed. When false, PDF pages are skipped e
436
446
  #
437
447
  # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -59,8 +59,8 @@ module ContextDev
59
59
  optional :max_age_ms, Integer
60
60
 
61
61
  # @!attribute pdf
62
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
63
- # inclusive 1-based page range.
62
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
63
+ # detection/OCR to an inclusive 1-based page range.
64
64
  #
65
65
  # @return [ContextDev::Models::WebWebScrapeHTMLParams::Pdf, nil]
66
66
  optional :pdf, -> { ContextDev::WebWebScrapeHTMLParams::Pdf }
@@ -113,7 +113,7 @@ module ContextDev
113
113
  #
114
114
  # @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
115
115
  #
116
- # @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
116
+ # @param pdf [ContextDev::Models::WebWebScrapeHTMLParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
117
117
  #
118
118
  # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before extr
119
119
  #
@@ -347,6 +347,14 @@ module ContextDev
347
347
  # @return [Integer, nil]
348
348
  optional :end_, Integer, api_name: :end
349
349
 
350
+ # @!attribute ocr
351
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
352
+ # recognized text at each image's position in page reading order while preserving
353
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
354
+ #
355
+ # @return [Boolean, nil]
356
+ optional :ocr, ContextDev::Internal::Type::Boolean
357
+
350
358
  # @!attribute should_parse
351
359
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
352
360
  # a 400 WEBSITE_ACCESS_ERROR is returned.
@@ -360,15 +368,17 @@ module ContextDev
360
368
  # @return [Integer, nil]
361
369
  optional :start, Integer
362
370
 
363
- # @!method initialize(end_: nil, should_parse: nil, start: nil)
371
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
364
372
  # Some parameter documentations has been truncated, see
365
373
  # {ContextDev::Models::WebWebScrapeHTMLParams::Pdf} for more details.
366
374
  #
367
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
368
- # inclusive 1-based page range.
375
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
376
+ # detection/OCR to an inclusive 1-based page range.
369
377
  #
370
378
  # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
371
379
  #
380
+ # @param ocr [Boolean] When true, detect and OCR images embedded in the selected PDF pages, inserting r
381
+ #
372
382
  # @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
373
383
  #
374
384
  # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -72,8 +72,8 @@ module ContextDev
72
72
  optional :max_age_ms, Integer
73
73
 
74
74
  # @!attribute pdf
75
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
76
- # inclusive 1-based page range.
75
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
76
+ # detection/OCR to an inclusive 1-based page range.
77
77
  #
78
78
  # @return [ContextDev::Models::WebWebScrapeMdParams::Pdf, nil]
79
79
  optional :pdf, -> { ContextDev::WebWebScrapeMdParams::Pdf }
@@ -136,7 +136,7 @@ module ContextDev
136
136
  #
137
137
  # @param max_age_ms [Integer] Return a cached result if a prior scrape for the same parameters exists and is y
138
138
  #
139
- # @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and OCR to an inclu
139
+ # @param pdf [ContextDev::Models::WebWebScrapeMdParams::Pdf] PDF parsing controls. Use start/end to limit text extraction and embedded-image
140
140
  #
141
141
  # @param settle_animations [Boolean] When true, waits briefly for CSS and transition animations to settle before conv
142
142
  #
@@ -372,6 +372,14 @@ module ContextDev
372
372
  # @return [Integer, nil]
373
373
  optional :end_, Integer, api_name: :end
374
374
 
375
+ # @!attribute ocr
376
+ # When true, detect and OCR images embedded in the selected PDF pages, inserting
377
+ # recognized text at each image's position in page reading order while preserving
378
+ # the PDF text layer. This is separate from automatic scanned-PDF OCR fallback.
379
+ #
380
+ # @return [Boolean, nil]
381
+ optional :ocr, ContextDev::Internal::Type::Boolean
382
+
375
383
  # @!attribute should_parse
376
384
  # When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
377
385
  # a 400 WEBSITE_ACCESS_ERROR is returned.
@@ -385,15 +393,17 @@ module ContextDev
385
393
  # @return [Integer, nil]
386
394
  optional :start, Integer
387
395
 
388
- # @!method initialize(end_: nil, should_parse: nil, start: nil)
396
+ # @!method initialize(end_: nil, ocr: nil, should_parse: nil, start: nil)
389
397
  # Some parameter documentations has been truncated, see
390
398
  # {ContextDev::Models::WebWebScrapeMdParams::Pdf} for more details.
391
399
  #
392
- # PDF parsing controls. Use start/end to limit text extraction and OCR to an
393
- # inclusive 1-based page range.
400
+ # PDF parsing controls. Use start/end to limit text extraction and embedded-image
401
+ # detection/OCR to an inclusive 1-based page range.
394
402
  #
395
403
  # @param end_ [Integer] Last 1-based PDF page to parse. When omitted, parsing ends at the last page. Mus
396
404
  #
405
+ # @param ocr [Boolean] When true, detect and OCR images embedded in the selected PDF pages, inserting r
406
+ #
397
407
  # @param should_parse [Boolean] When true, PDF URLs are fetched and parsed. When false, PDF URLs are skipped and
398
408
  #
399
409
  # @param start [Integer] First 1-based PDF page to parse. When omitted, parsing starts at the first page.
@@ -4,6 +4,14 @@ module ContextDev
4
4
  module Models
5
5
  # @see ContextDev::Resources::Web#web_scrape_md
6
6
  class WebWebScrapeMdResponse < ContextDev::Internal::Type::BaseModel
7
+ # @!attribute content_length
8
+ # UTF-8 byte length of the returned Markdown. Use 0 to identify an empty result
9
+ # and compare small values against your workload's minimum useful-content
10
+ # threshold.
11
+ #
12
+ # @return [Integer]
13
+ required :content_length, Integer, api_name: :contentLength
14
+
7
15
  # @!attribute markdown
8
16
  # Page content converted to GitHub Flavored Markdown
9
17
  #
@@ -35,10 +43,12 @@ module ContextDev
35
43
  # @return [ContextDev::Models::WebWebScrapeMdResponse::KeyMetadata, nil]
36
44
  optional :key_metadata, -> { ContextDev::Models::WebWebScrapeMdResponse::KeyMetadata }
37
45
 
38
- # @!method initialize(markdown:, metadata:, success:, url:, key_metadata: nil)
46
+ # @!method initialize(content_length:, markdown:, metadata:, success:, url:, key_metadata: nil)
39
47
  # Some parameter documentations has been truncated, see
40
48
  # {ContextDev::Models::WebWebScrapeMdResponse} for more details.
41
49
  #
50
+ # @param content_length [Integer] UTF-8 byte length of the returned Markdown. Use 0 to identify an empty result an
51
+ #
42
52
  # @param markdown [String] Page content converted to GitHub Flavored Markdown
43
53
  #
44
54
  # @param metadata [ContextDev::Models::WebWebScrapeMdResponse::Metadata] Metadata extracted from the scraped page HTML.