parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,882 @@
1
+ """Provider for Anthropic Claude vision-based PARSE."""
2
+
3
+ import base64
4
+ import io
5
+ import math
6
+ import os
7
+ from datetime import date, datetime
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ import httpx
12
+ from PIL import Image
13
+
14
+ from parse_bench.inference.providers.base import (
15
+ Provider,
16
+ ProviderConfigError,
17
+ ProviderPermanentError,
18
+ ProviderTransientError,
19
+ )
20
+ from parse_bench.inference.providers.parse._layout_utils import (
21
+ build_layout_pages,
22
+ items_to_markdown,
23
+ parse_layout_blocks,
24
+ resolve_layout_prompts,
25
+ split_pdf_to_pages,
26
+ )
27
+ from parse_bench.inference.providers.registry import register_provider
28
+ from parse_bench.schemas.parse_output import PageIR, ParseLayoutPageIR, ParseOutput
29
+ from parse_bench.schemas.pipeline import PipelineSpec
30
+ from parse_bench.schemas.pipeline_io import (
31
+ InferenceRequest,
32
+ InferenceResult,
33
+ RawInferenceResult,
34
+ )
35
+ from parse_bench.schemas.product import ProductType
36
+
37
+ SYSTEM_PROMPT = (
38
+ "You are a document parser. Your task is to convert "
39
+ "document images to clean, well-structured markdown."
40
+ "\n\nGuidelines:\n"
41
+ "- Preserve the document structure "
42
+ "(headings, paragraphs, lists, tables)\n"
43
+ "- Convert tables to HTML format "
44
+ "(<table>, <tr>, <th>, <td>)\n"
45
+ "- For existing tables in the document: use colspan "
46
+ "and rowspan attributes to preserve merged cells "
47
+ "and hierarchical headers\n"
48
+ "- For charts/graphs being converted to tables: use "
49
+ "flat combined column headers (e.g., "
50
+ '"Primary 2015" not separate rows) so each data '
51
+ "cell's row contains all its labels\n"
52
+ "- Describe images/figures briefly in square brackets "
53
+ "like [Figure: description]\n"
54
+ "- Preserve any code blocks with appropriate syntax "
55
+ "highlighting\n"
56
+ "- Maintain reading order (left-to-right, "
57
+ "top-to-bottom for Western documents)\n"
58
+ "- Do not add commentary or explanations "
59
+ "- only output the parsed content"
60
+ )
61
+
62
+ USER_PROMPT = (
63
+ "Parse this document page and output its content as "
64
+ "clean markdown. Use HTML tables for any tabular "
65
+ "data. For charts/graphs, use flat combined column "
66
+ "headers. Output ONLY the parsed content, "
67
+ "no explanations."
68
+ )
69
+
70
+
71
+ # Anthropic pricing: USD per million tokens (input, output)
72
+ # Source: https://platform.claude.com/docs/en/about-claude/pricing (2026-03-25)
73
+ _ANTHROPIC_PRICING_PER_M: dict[str, tuple[float, float]] = {
74
+ # model-prefix: (input_per_M, output_per_M)
75
+ "claude-fable-5-1": (10.00, 50.00),
76
+ "claude-fable-5": (10.00, 50.00),
77
+ # Sonnet 5 has introductory pricing through 2026-08-31; handled in
78
+ # _get_pricing so benchmark costs switch to standard pricing on time.
79
+ "claude-sonnet-5": (3.00, 15.00),
80
+ "claude-haiku-4-5": (1.00, 5.00),
81
+ "claude-haiku-3-5": (0.80, 4.00),
82
+ "claude-haiku-3": (0.25, 1.25),
83
+ "claude-sonnet-4": (3.00, 15.00),
84
+ "claude-sonnet-3": (3.00, 15.00),
85
+ "claude-opus-4-8": (5.00, 25.00),
86
+ "claude-opus-4-7": (5.00, 25.00),
87
+ "claude-opus-4-6": (5.00, 25.00),
88
+ "claude-opus-4-5": (5.00, 25.00),
89
+ "claude-opus-4-1": (15.00, 75.00),
90
+ "claude-opus-4": (15.00, 75.00),
91
+ }
92
+
93
+
94
+ def _count_image_tokens(width: int, height: int) -> int:
95
+ """Visual tokens consumed by an image: one token per 28x28 pixel patch."""
96
+ return math.ceil(width / 28) * math.ceil(height / 28)
97
+
98
+
99
+ def _resized_size(width: int, height: int, max_edge: int, max_tokens: int) -> tuple[int, int]:
100
+ """The size the Claude API resizes an image to before padding.
101
+
102
+ Reference implementation from the vision docs (see
103
+ ``_resize_to_perceived_size``); images already within the limits are
104
+ returned unchanged.
105
+ """
106
+
107
+ def fits(w: int, h: int) -> bool:
108
+ return (
109
+ math.ceil(w / 28) * 28 <= max_edge
110
+ and math.ceil(h / 28) * 28 <= max_edge
111
+ and _count_image_tokens(w, h) <= max_tokens
112
+ )
113
+
114
+ if fits(width, height):
115
+ return (width, height)
116
+ if height > width:
117
+ resized_h, resized_w = _resized_size(height, width, max_edge, max_tokens)
118
+ return (resized_w, resized_h)
119
+
120
+ # Binary search along the long edge for the largest aspect-preserving
121
+ # size that fits.
122
+ aspect_ratio = width / height
123
+ lo, hi = 1, width # lo always fits; hi never fits
124
+ while lo + 1 < hi:
125
+ mid = (lo + hi) // 2
126
+ if fits(mid, max(round(mid / aspect_ratio), 1)):
127
+ lo = mid
128
+ else:
129
+ hi = mid
130
+ return (lo, max(round(lo / aspect_ratio), 1))
131
+
132
+
133
+ def anthropic_cache_aware_cost_usd(usage: dict[str, int], input_rate: float, output_rate: float) -> float:
134
+ """USD cost using Anthropic cache multipliers (write 1.25x, read 0.1x)."""
135
+ n_in = float(usage.get("input", 0) or 0)
136
+ n_out = float(usage.get("output", 0) or 0)
137
+ read = float(usage.get("cache_read", 0) or 0)
138
+ write = float(usage.get("cache_write", 0) or 0)
139
+ in_cost = (n_in + 1.25 * write + 0.1 * read) / 1_000_000.0 * input_rate
140
+ out_cost = n_out / 1_000_000.0 * output_rate
141
+ return in_cost + out_cost
142
+
143
+
144
+ @register_provider("anthropic")
145
+ class AnthropicProvider(Provider):
146
+ """
147
+ Provider for Anthropic Claude vision-based document parsing.
148
+
149
+ Renders PDF pages to images and uses Claude's vision
150
+ capabilities to parse document content to markdown.
151
+ """
152
+
153
+ def __init__(self, provider_name: str, base_config: dict[str, Any] | None = None):
154
+ """
155
+ Initialize the provider.
156
+
157
+ :param provider_name: Name of the provider
158
+ :param base_config: Optional configuration with:
159
+ - `model`: Claude model to use (default: "claude-haiku-4-5-20250514")
160
+ - `dpi`: DPI for PDF to image conversion (default: 150)
161
+ - `max_tokens`: Max tokens per response (default: 8192)
162
+ - `timeout`: Request timeout in seconds (default: 120)
163
+ - `mode`: "image" (default) to send page screenshots, or "file" to send raw PDF
164
+ """
165
+ super().__init__(provider_name, base_config)
166
+
167
+ # Get API key from environment
168
+ self._api_key = os.environ.get("ANTHROPIC_API_KEY")
169
+ if not self._api_key:
170
+ raise ProviderConfigError("ANTHROPIC_API_KEY environment variable not set")
171
+
172
+ # Configuration
173
+ self._model = self.base_config.get("model", "claude-haiku-4-5-20251001")
174
+ self._dpi = self.base_config.get("dpi", 150)
175
+ self._max_tokens = self.base_config.get("max_tokens", 8192)
176
+ self._timeout = self.base_config.get("timeout", 120)
177
+ self._mode = self.base_config.get("mode", "image") # "image", "file", or "parse_with_layout"
178
+ self._thinking = self.base_config.get("thinking") # e.g. {"type": "enabled", "budget_tokens": 32768}
179
+ self._effort = self.base_config.get("effort") # e.g. "high", "xhigh" — for Opus 4.7+
180
+ # Opus 4.7+ and Fable 5 reject temperature/top_p/top_k at non-default values (400 error)
181
+ self._supports_temperature = not self._model.startswith(
182
+ ("claude-opus-4-7", "claude-opus-4-8", "claude-fable-5")
183
+ )
184
+
185
+ if self._mode not in ("image", "file", "parse_with_layout", "parse_with_layout_file"):
186
+ raise ProviderConfigError(
187
+ f"Invalid mode '{self._mode}'. "
188
+ "Must be 'image', 'file', 'parse_with_layout', or 'parse_with_layout_file'."
189
+ )
190
+
191
+ # Grid the layout-mode bboxes are on: 1000 (the 0-1000 prompt, the
192
+ # default) or None (absolute pixels of the sent image, for models
193
+ # that ground well but re-normalize unreliably).
194
+ self._bbox_scale = self.base_config.get("bbox_scale", 1000)
195
+ # Pixel mode is safe here because _resize_to_perceived_size() pins the
196
+ # sent image to the size the model perceives, so the recorded page
197
+ # dimensions and the model's coordinate frame are the same.
198
+ self._layout_system_prompt, self._layout_user_prompt = resolve_layout_prompts(
199
+ self._bbox_scale, self._mode, pixel_frame_supported=True
200
+ )
201
+ # Image limits used to pre-resize pages in pixel-coordinate mode.
202
+ # Defaults are the standard resolution tier, which is safe for every
203
+ # model (an image within standard limits is never downscaled
204
+ # server-side on any tier); high-resolution-tier models can pass
205
+ # vision_max_edge=2576, vision_max_tokens=4784 to keep fidelity.
206
+ self._vision_max_edge = int(self.base_config.get("vision_max_edge", 1568))
207
+ self._vision_max_tokens = int(self.base_config.get("vision_max_tokens", 1568))
208
+
209
+ # Initialize Anthropic client
210
+ try:
211
+ import anthropic
212
+
213
+ self._client = anthropic.Anthropic(api_key=self._api_key)
214
+ except ImportError as e:
215
+ raise ProviderConfigError("anthropic package not installed. Run: pip install anthropic") from e
216
+
217
+ # Claude API limits
218
+ MAX_IMAGE_DIMENSION = 8000 # pixels
219
+ # API limit is 5MB for base64 data; base64 adds ~33% overhead, so raw limit is 5MB * 3/4
220
+ MAX_IMAGE_SIZE_BYTES = int(5 * 1024 * 1024 * 3 / 4) # ~3.75 MB raw -> ~5 MB base64
221
+
222
+ def _get_pricing(self) -> tuple[float, float]:
223
+ """Return (input_rate, output_rate) in USD per million tokens.
224
+
225
+ Uses longest-prefix matching to avoid ambiguity when one model
226
+ prefix is a substring of another.
227
+ """
228
+ if self._model.startswith("claude-sonnet-5") and date.today() <= date(2026, 8, 31):
229
+ return (2.00, 10.00)
230
+
231
+ matches = [(p, r) for p, r in _ANTHROPIC_PRICING_PER_M.items() if self._model.startswith(p)]
232
+ return max(matches, key=lambda x: len(x[0]))[1] if matches else (0.0, 0.0)
233
+
234
+ @staticmethod
235
+ def _extract_text(response) -> str: # type: ignore[no-untyped-def]
236
+ """Extract text content from response, skipping any thinking blocks."""
237
+ for block in response.content or []:
238
+ if getattr(block, "type", None) == "text":
239
+ return getattr(block, "text", "")
240
+ return ""
241
+
242
+ @staticmethod
243
+ def _extract_usage(response) -> dict[str, int]: # type: ignore[no-untyped-def]
244
+ """Extract token counts from an Anthropic API response."""
245
+ usage = getattr(response, "usage", None)
246
+ if usage is None:
247
+ return {
248
+ "input_tokens": 0,
249
+ "output_tokens": 0,
250
+ "cache_read_tokens": 0,
251
+ "cache_write_tokens": 0,
252
+ "thinking_tokens": 0,
253
+ "total_tokens": 0,
254
+ }
255
+ input_tok = getattr(usage, "input_tokens", 0) or 0
256
+ output_tok = getattr(usage, "output_tokens", 0) or 0
257
+ cache_read = getattr(usage, "cache_read_input_tokens", 0) or 0
258
+ cache_write = getattr(usage, "cache_creation_input_tokens", 0) or 0
259
+ # With extended thinking, output_tokens includes thinking tokens.
260
+ # Try to count thinking tokens from content blocks for reporting.
261
+ thinking_tok = 0
262
+ for block in response.content or []:
263
+ if getattr(block, "type", None) == "thinking":
264
+ # Token count not directly available; use output_tokens as-is.
265
+ break
266
+ total_tok = input_tok + output_tok + cache_read + cache_write
267
+ return {
268
+ "input_tokens": input_tok,
269
+ "output_tokens": output_tok,
270
+ "cache_read_tokens": cache_read,
271
+ "cache_write_tokens": cache_write,
272
+ "thinking_tokens": thinking_tok,
273
+ "total_tokens": total_tok,
274
+ }
275
+
276
+ def _resize_to_perceived_size(self, image: Image.Image) -> Image.Image:
277
+ """Pre-resize so the image Claude perceives is the image we recorded.
278
+
279
+ The API downscales images that exceed the model's native limits, and
280
+ with ``bbox_scale=None`` the model's pixel coordinates are in that
281
+ downscaled frame — while normalize() divides by the recorded dims.
282
+ Resizing up front (per the published resize rule) keeps the two
283
+ identical, as the vision docs recommend:
284
+ https://platform.claude.com/docs/en/docs/build-with-claude/vision-coordinates
285
+ """
286
+ target = _resized_size(image.width, image.height, self._vision_max_edge, self._vision_max_tokens)
287
+ if target == (image.width, image.height):
288
+ return image
289
+ return image.resize(target, Image.Resampling.LANCZOS)
290
+
291
+ def _prepare_image_for_api(self, image: Image.Image) -> Image.Image:
292
+ """
293
+ Resize image if it exceeds Claude API dimension limits.
294
+
295
+ :param image: PIL Image to prepare
296
+ :return: Resized image if needed, otherwise original
297
+ """
298
+ width, height = image.size
299
+ max_dim = max(width, height)
300
+
301
+ if max_dim <= self.MAX_IMAGE_DIMENSION:
302
+ return image
303
+
304
+ # Calculate scale factor to fit within limits
305
+ scale = self.MAX_IMAGE_DIMENSION / max_dim
306
+ new_width = int(width * scale)
307
+ new_height = int(height * scale)
308
+
309
+ return image.resize((new_width, new_height), Image.Resampling.LANCZOS)
310
+
311
+ def _image_to_base64(self, image: Image.Image) -> str:
312
+ """
313
+ Convert PIL Image to base64 string, respecting Claude API limits.
314
+
315
+ Handles:
316
+ - Images with dimensions exceeding 8000 pixels (resizes proportionally)
317
+ - Images exceeding 5MB after encoding (reduces quality iteratively)
318
+ """
319
+ # Resize if dimensions exceed limit
320
+ image = self._prepare_image_for_api(image)
321
+
322
+ # Convert to RGB if necessary (e.g., RGBA images)
323
+ if image.mode in ("RGBA", "P"):
324
+ image = image.convert("RGB")
325
+
326
+ # Try encoding with decreasing quality until under size limit
327
+ quality = 85
328
+ min_quality = 20
329
+
330
+ while quality >= min_quality:
331
+ buffer = io.BytesIO()
332
+ image.save(buffer, format="JPEG", quality=quality)
333
+ buffer.seek(0)
334
+ data = buffer.getvalue()
335
+
336
+ if len(data) <= self.MAX_IMAGE_SIZE_BYTES:
337
+ return base64.standard_b64encode(data).decode("utf-8")
338
+
339
+ quality -= 10
340
+
341
+ # If still too large after quality reduction, resize the image
342
+ while True:
343
+ width, height = image.size
344
+ new_width = int(width * 0.8)
345
+ new_height = int(height * 0.8)
346
+
347
+ if new_width < 100 or new_height < 100:
348
+ # Give up - image is too complex to fit in limits
349
+ break
350
+
351
+ image = image.resize((new_width, new_height), Image.Resampling.LANCZOS)
352
+
353
+ buffer = io.BytesIO()
354
+ image.save(buffer, format="JPEG", quality=min_quality)
355
+ buffer.seek(0)
356
+ data = buffer.getvalue()
357
+
358
+ if len(data) <= self.MAX_IMAGE_SIZE_BYTES:
359
+ return base64.standard_b64encode(data).decode("utf-8")
360
+
361
+ # Final fallback - return what we have
362
+ buffer = io.BytesIO()
363
+ image.save(buffer, format="JPEG", quality=min_quality)
364
+ buffer.seek(0)
365
+ return base64.standard_b64encode(buffer.getvalue()).decode("utf-8")
366
+
367
+ def _pdf_to_images(self, pdf_path: str) -> list[Image.Image]:
368
+ """
369
+ Convert PDF pages to images.
370
+
371
+ :param pdf_path: Path to the PDF file
372
+ :return: List of PIL Images, one per page
373
+ """
374
+ try:
375
+ from pdf2image import convert_from_path
376
+ except ImportError as e:
377
+ raise ProviderConfigError("pdf2image package not installed. Run: pip install pdf2image") from e
378
+
379
+ try:
380
+ images = convert_from_path(pdf_path, dpi=self._dpi)
381
+ return images
382
+ except Exception as e:
383
+ raise ProviderPermanentError(f"Failed to convert PDF to images: {e}") from e
384
+
385
+ def _parse_image(self, image: Image.Image) -> tuple[str, dict[str, int]]:
386
+ """
387
+ Send image to Claude and get markdown response.
388
+
389
+ :param image: PIL Image to parse
390
+ :return: Tuple of (markdown content, usage dict)
391
+ """
392
+ img_base64 = self._image_to_base64(image)
393
+
394
+ try:
395
+ extra_kwargs: dict[str, Any] = {}
396
+ if self._thinking:
397
+ extra_kwargs["thinking"] = self._thinking
398
+ elif self._supports_temperature:
399
+ extra_kwargs["temperature"] = 0
400
+ if self._effort:
401
+ extra_kwargs["output_config"] = {"effort": self._effort}
402
+
403
+ response = self._client.messages.create(
404
+ model=self._model,
405
+ max_tokens=self._max_tokens,
406
+ system=SYSTEM_PROMPT,
407
+ timeout=httpx.Timeout(3600.0, connect=5.0),
408
+ messages=[
409
+ {
410
+ "role": "user",
411
+ "content": [
412
+ {
413
+ "type": "image",
414
+ "source": {
415
+ "type": "base64",
416
+ "media_type": "image/jpeg",
417
+ "data": img_base64,
418
+ },
419
+ },
420
+ {
421
+ "type": "text",
422
+ "text": USER_PROMPT,
423
+ },
424
+ ],
425
+ }
426
+ ],
427
+ **extra_kwargs,
428
+ )
429
+
430
+ usage = self._extract_usage(response)
431
+ content = self._extract_text(response)
432
+ return content, usage
433
+
434
+ except Exception as e:
435
+ error_str = str(e).lower()
436
+ if any(kw in error_str for kw in ["timeout", "connection", "network"]):
437
+ raise ProviderTransientError(f"Transient error calling Claude API: {e}") from e
438
+ if any(kw in error_str for kw in ["rate_limit", "rate limit", "429"]):
439
+ raise ProviderTransientError(f"Rate limited: {e}") from e
440
+ raise ProviderPermanentError(f"Error calling Claude API: {e}") from e
441
+
442
+ def _parse_image_with_layout(self, image: Image.Image) -> tuple[list[dict[str, Any]], str, dict[str, int]]:
443
+ """Send image to Claude with layout prompt and get annotated response.
444
+
445
+ :param image: PIL Image to parse
446
+ :return: Tuple of (parsed layout items, raw content, usage dict)
447
+ """
448
+ img_base64 = self._image_to_base64(image)
449
+
450
+ try:
451
+ extra_kwargs: dict[str, Any] = {}
452
+ if self._thinking:
453
+ extra_kwargs["thinking"] = self._thinking
454
+ elif self._supports_temperature:
455
+ extra_kwargs["temperature"] = 0
456
+ if self._effort:
457
+ extra_kwargs["output_config"] = {"effort": self._effort}
458
+
459
+ response = self._client.messages.create(
460
+ model=self._model,
461
+ max_tokens=self._max_tokens,
462
+ system=self._layout_system_prompt,
463
+ timeout=httpx.Timeout(3600.0, connect=5.0),
464
+ messages=[
465
+ {
466
+ "role": "user",
467
+ "content": [
468
+ {
469
+ "type": "image",
470
+ "source": {
471
+ "type": "base64",
472
+ "media_type": "image/jpeg",
473
+ "data": img_base64,
474
+ },
475
+ },
476
+ {
477
+ "type": "text",
478
+ "text": self._layout_user_prompt,
479
+ },
480
+ ],
481
+ }
482
+ ],
483
+ **extra_kwargs,
484
+ )
485
+
486
+ usage = self._extract_usage(response)
487
+ text = self._extract_text(response)
488
+
489
+ items = parse_layout_blocks(text)
490
+ return items, text, usage
491
+
492
+ except Exception as e:
493
+ error_str = str(e).lower()
494
+ if any(kw in error_str for kw in ["timeout", "connection", "network"]):
495
+ raise ProviderTransientError(f"Transient error calling Claude API: {e}") from e
496
+ if any(kw in error_str for kw in ["rate_limit", "rate limit", "429"]):
497
+ raise ProviderTransientError(f"Rate limited: {e}") from e
498
+ raise ProviderPermanentError(f"Error calling Claude API: {e}") from e
499
+
500
+ def _parse_pdf_file(self, pdf_path: str) -> tuple[str, dict[str, int]]:
501
+ """
502
+ Send raw PDF file to Claude using the Files API (beta).
503
+
504
+ Uses the Anthropic Files API to upload the PDF and reference it
505
+ in the message as a document content block.
506
+
507
+ :param pdf_path: Path to the PDF file
508
+ :return: Tuple of (markdown content, usage dict)
509
+ """
510
+ try:
511
+ # Read PDF file
512
+ with open(pdf_path, "rb") as f:
513
+ pdf_data = f.read()
514
+
515
+ pdf_base64 = base64.standard_b64encode(pdf_data).decode("utf-8")
516
+
517
+ # Use the beta messages API with PDF support
518
+ extra_kwargs: dict[str, Any] = {}
519
+ if self._thinking:
520
+ extra_kwargs["thinking"] = self._thinking
521
+ elif self._supports_temperature:
522
+ extra_kwargs["temperature"] = 0
523
+ if self._effort:
524
+ extra_kwargs["output_config"] = {"effort": self._effort}
525
+
526
+ response = self._client.beta.messages.create(
527
+ model=self._model,
528
+ max_tokens=self._max_tokens,
529
+ betas=["pdfs-2024-09-25"],
530
+ system=SYSTEM_PROMPT,
531
+ timeout=httpx.Timeout(3600.0, connect=5.0),
532
+ messages=[
533
+ {
534
+ "role": "user",
535
+ "content": [
536
+ {
537
+ "type": "document",
538
+ "source": {
539
+ "type": "base64",
540
+ "media_type": "application/pdf",
541
+ "data": pdf_base64,
542
+ },
543
+ },
544
+ {
545
+ "type": "text",
546
+ "text": USER_PROMPT,
547
+ },
548
+ ],
549
+ }
550
+ ],
551
+ **extra_kwargs,
552
+ )
553
+
554
+ usage = self._extract_usage(response)
555
+ content = self._extract_text(response)
556
+ return content, usage
557
+
558
+ except Exception as e:
559
+ error_str = str(e).lower()
560
+ if any(kw in error_str for kw in ["timeout", "connection", "network"]):
561
+ raise ProviderTransientError(f"Transient error calling Claude API: {e}") from e
562
+ if any(kw in error_str for kw in ["rate_limit", "rate limit", "429"]):
563
+ raise ProviderTransientError(f"Rate limited: {e}") from e
564
+ raise ProviderPermanentError(f"Error calling Claude API: {e}") from e
565
+
566
+ def _parse_pdf_page_with_layout(self, pdf_bytes: bytes) -> tuple[list[dict[str, Any]], str, dict[str, int]]:
567
+ """Send a single-page PDF to Claude with layout prompt.
568
+
569
+ :param pdf_bytes: Raw bytes of a single-page PDF
570
+ :return: Tuple of (parsed layout items, raw content, usage dict)
571
+ """
572
+ try:
573
+ pdf_base64 = base64.standard_b64encode(pdf_bytes).decode("utf-8")
574
+
575
+ extra_kwargs: dict[str, Any] = {}
576
+ if self._thinking:
577
+ extra_kwargs["thinking"] = self._thinking
578
+ elif self._supports_temperature:
579
+ extra_kwargs["temperature"] = 0
580
+ if self._effort:
581
+ extra_kwargs["output_config"] = {"effort": self._effort}
582
+
583
+ response = self._client.beta.messages.create(
584
+ model=self._model,
585
+ max_tokens=self._max_tokens,
586
+ betas=["pdfs-2024-09-25"],
587
+ system=self._layout_system_prompt,
588
+ timeout=httpx.Timeout(3600.0, connect=5.0),
589
+ messages=[
590
+ {
591
+ "role": "user",
592
+ "content": [
593
+ {
594
+ "type": "document",
595
+ "source": {
596
+ "type": "base64",
597
+ "media_type": "application/pdf",
598
+ "data": pdf_base64,
599
+ },
600
+ },
601
+ {
602
+ "type": "text",
603
+ "text": self._layout_user_prompt,
604
+ },
605
+ ],
606
+ }
607
+ ],
608
+ **extra_kwargs,
609
+ )
610
+
611
+ usage = self._extract_usage(response)
612
+ text = self._extract_text(response)
613
+
614
+ items = parse_layout_blocks(text)
615
+ return items, text, usage
616
+
617
+ except Exception as e:
618
+ error_str = str(e).lower()
619
+ if any(kw in error_str for kw in ["timeout", "connection", "network"]):
620
+ raise ProviderTransientError(f"Transient error calling Claude API: {e}") from e
621
+ if any(kw in error_str for kw in ["rate_limit", "rate limit", "429"]):
622
+ raise ProviderTransientError(f"Rate limited: {e}") from e
623
+ raise ProviderPermanentError(f"Error calling Claude API: {e}") from e
624
+
625
+ def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
626
+ """
627
+ Run inference and return raw results.
628
+
629
+ :param pipeline: Pipeline specification
630
+ :param request: Inference request
631
+ :return: Raw inference result
632
+ """
633
+ if request.product_type != ProductType.PARSE:
634
+ raise ProviderPermanentError(
635
+ f"AnthropicProvider only supports PARSE product type, got {request.product_type}"
636
+ )
637
+
638
+ source_path = Path(request.source_file_path)
639
+ if not source_path.exists():
640
+ raise ProviderPermanentError(f"Source file not found: {source_path}")
641
+
642
+ # Check file extension
643
+ supported_extensions = {".pdf", ".png", ".jpg", ".jpeg"}
644
+ if source_path.suffix.lower() not in supported_extensions:
645
+ raise ProviderPermanentError(f"AnthropicProvider supports {supported_extensions}, got {source_path.suffix}")
646
+
647
+ started_at = datetime.now()
648
+
649
+ try:
650
+ page_usages: list[dict[str, int]] = []
651
+
652
+ if self._mode == "file":
653
+ if source_path.suffix.lower() == ".pdf":
654
+ # File mode: send raw PDF to API
655
+ markdown, usage = self._parse_pdf_file(str(source_path))
656
+ page_usages.append(usage)
657
+ # In file mode, we get one response for the entire document
658
+ # We don't have page-level info, so we treat it as a single "page"
659
+ pages = [
660
+ {
661
+ "page_index": 0,
662
+ "markdown": markdown,
663
+ "width": None,
664
+ "height": None,
665
+ }
666
+ ]
667
+ num_pages = 1 # We don't know actual page count in file mode
668
+ else:
669
+ # Non-PDF: fall back to image-based parsing
670
+ image = Image.open(source_path)
671
+ markdown, usage = self._parse_image(image)
672
+ page_usages.append(usage)
673
+ pages = [
674
+ {
675
+ "page_index": 0,
676
+ "markdown": markdown,
677
+ "width": image.width,
678
+ "height": image.height,
679
+ }
680
+ ]
681
+ num_pages = 1
682
+ elif self._mode == "parse_with_layout_file":
683
+ if source_path.suffix.lower() == ".pdf":
684
+ # Split PDF into single-page PDFs, send each with layout prompt
685
+ pdf_pages = split_pdf_to_pages(str(source_path))
686
+ pages = []
687
+ for page_index, (pdf_bytes, w, h) in enumerate(pdf_pages):
688
+ items, raw_content, usage = self._parse_pdf_page_with_layout(pdf_bytes)
689
+ page_usages.append(usage)
690
+ pages.append(
691
+ {
692
+ "page_index": page_index,
693
+ "items": items,
694
+ "raw_content": raw_content,
695
+ "width": w,
696
+ "height": h,
697
+ }
698
+ )
699
+ num_pages = len(pdf_pages)
700
+ else:
701
+ # Non-PDF: fall back to image-based layout parsing
702
+ image = Image.open(source_path)
703
+ items, raw_content, usage = self._parse_image_with_layout(image)
704
+ page_usages.append(usage)
705
+ pages = [
706
+ {
707
+ "page_index": 0,
708
+ "items": items,
709
+ "raw_content": raw_content,
710
+ "width": image.width,
711
+ "height": image.height,
712
+ }
713
+ ]
714
+ num_pages = 1
715
+ else:
716
+ # Image mode (both "image" and "parse_with_layout"):
717
+ # convert PDF to images and process each page
718
+ if source_path.suffix.lower() == ".pdf":
719
+ images = self._pdf_to_images(str(source_path))
720
+ else:
721
+ images = [Image.open(source_path)]
722
+
723
+ # Parse each page
724
+ pages = []
725
+ for page_index, image in enumerate(images): # type: ignore[assignment]
726
+ if self._mode == "parse_with_layout":
727
+ # In pixel mode, send the pre-resized image so the recorded
728
+ # dims are exactly what the model perceives.
729
+ page = self._resize_to_perceived_size(image) if self._bbox_scale is None else image
730
+ items, raw_content, usage = self._parse_image_with_layout(page)
731
+ page_usages.append(usage)
732
+ pages.append(
733
+ {
734
+ "page_index": page_index,
735
+ "items": items,
736
+ "raw_content": raw_content,
737
+ "width": page.width,
738
+ "height": page.height,
739
+ }
740
+ )
741
+ else:
742
+ markdown, usage = self._parse_image(image)
743
+ page_usages.append(usage)
744
+ pages.append(
745
+ {
746
+ "page_index": page_index,
747
+ "markdown": markdown,
748
+ "width": image.width,
749
+ "height": image.height,
750
+ }
751
+ )
752
+ num_pages = len(images)
753
+
754
+ completed_at = datetime.now()
755
+ latency_ms = int((completed_at - started_at).total_seconds() * 1000)
756
+
757
+ # Aggregate token usage across pages
758
+ total_input = sum(u.get("input_tokens", 0) for u in page_usages)
759
+ total_output = sum(u.get("output_tokens", 0) for u in page_usages)
760
+ total_cache_read = sum(u.get("cache_read_tokens", 0) for u in page_usages)
761
+ total_cache_write = sum(u.get("cache_write_tokens", 0) for u in page_usages)
762
+ total_thinking = sum(u.get("thinking_tokens", 0) for u in page_usages)
763
+ total_all = sum(u.get("total_tokens", 0) for u in page_usages)
764
+
765
+ # Compute cost (Anthropic cache multipliers when cache tokens present)
766
+ input_rate, output_rate = self._get_pricing()
767
+ cost = anthropic_cache_aware_cost_usd(
768
+ {
769
+ "input": total_input,
770
+ "output": total_output + total_thinking,
771
+ "cache_read": total_cache_read,
772
+ "cache_write": total_cache_write,
773
+ },
774
+ input_rate,
775
+ output_rate,
776
+ )
777
+
778
+ raw_output = {
779
+ "pages": pages,
780
+ "num_pages": num_pages,
781
+ "model": self._model,
782
+ "mode": self._mode,
783
+ "bbox_scale": self._bbox_scale,
784
+ "config": {
785
+ "dpi": self._dpi,
786
+ "max_tokens": self._max_tokens,
787
+ "mode": self._mode,
788
+ },
789
+ "input_tokens": total_input,
790
+ "output_tokens": total_output,
791
+ "cache_read_tokens": total_cache_read,
792
+ "cache_write_tokens": total_cache_write,
793
+ "thinking_tokens": total_thinking,
794
+ "total_tokens": total_all,
795
+ "cost_usd": cost,
796
+ "cost_per_page_usd": cost / num_pages if num_pages > 0 else 0.0,
797
+ "input_tokens_per_page": total_input / num_pages if num_pages > 0 else 0.0,
798
+ "output_tokens_per_page": total_output / num_pages if num_pages > 0 else 0.0,
799
+ }
800
+
801
+ return RawInferenceResult(
802
+ request=request,
803
+ pipeline=pipeline,
804
+ pipeline_name=pipeline.pipeline_name,
805
+ product_type=request.product_type,
806
+ raw_output=raw_output,
807
+ started_at=started_at,
808
+ completed_at=completed_at,
809
+ latency_in_ms=latency_ms,
810
+ )
811
+
812
+ except (ProviderPermanentError, ProviderTransientError, ProviderConfigError):
813
+ raise
814
+ except Exception as e:
815
+ raise ProviderPermanentError(f"Unexpected error during inference: {e}") from e
816
+
817
+ def normalize(self, raw_result: RawInferenceResult) -> InferenceResult:
818
+ """
819
+ Normalize raw inference result to produce ParseOutput.
820
+
821
+ :param raw_result: Raw inference result from run_inference()
822
+ :return: Inference result with both raw and normalized outputs
823
+ """
824
+ if raw_result.product_type != ProductType.PARSE:
825
+ raise ProviderPermanentError(
826
+ f"AnthropicProvider only supports PARSE product type, got {raw_result.product_type}"
827
+ )
828
+
829
+ mode = raw_result.raw_output.get("mode", "image")
830
+
831
+ # Build page-level output
832
+ pages: list[PageIR] = []
833
+ page_markdowns: list[str] = []
834
+ layout_pages: list[ParseLayoutPageIR] = []
835
+
836
+ for page_data in raw_result.raw_output.get("pages", []):
837
+ page_index = page_data.get("page_index", 0)
838
+
839
+ if mode in ("parse_with_layout", "parse_with_layout_file"):
840
+ items = page_data.get("items", [])
841
+ image_width = page_data.get("width", 0)
842
+ image_height = page_data.get("height", 0)
843
+ markdown = items_to_markdown(items)
844
+ layout_pages.extend(
845
+ build_layout_pages(
846
+ items,
847
+ image_width,
848
+ image_height,
849
+ markdown,
850
+ page_number=page_index + 1,
851
+ bbox_scale=raw_result.raw_output.get("bbox_scale", 1000),
852
+ )
853
+ )
854
+ else:
855
+ markdown = page_data.get("markdown", "")
856
+
857
+ pages.append(PageIR(page_index=page_index, markdown=markdown))
858
+ page_markdowns.append(markdown)
859
+
860
+ # Sort by page index and concatenate
861
+ pages.sort(key=lambda p: p.page_index)
862
+ full_markdown = "\n\n".join(page_markdowns)
863
+
864
+ output = ParseOutput(
865
+ task_type="parse",
866
+ example_id=raw_result.request.example_id,
867
+ pipeline_name=raw_result.pipeline_name,
868
+ pages=pages,
869
+ markdown=full_markdown,
870
+ layout_pages=layout_pages,
871
+ )
872
+
873
+ return InferenceResult(
874
+ request=raw_result.request,
875
+ pipeline_name=raw_result.pipeline_name,
876
+ product_type=raw_result.product_type,
877
+ raw_output=raw_result.raw_output,
878
+ output=output,
879
+ started_at=raw_result.started_at,
880
+ completed_at=raw_result.completed_at,
881
+ latency_in_ms=raw_result.latency_in_ms,
882
+ )