parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,229 @@
1
+ """Provider for GLM (z.ai) vision-based PARSE.
2
+
3
+ GLM-5.3-flash is served through z.ai's OpenAI-compatible chat completions
4
+ endpoint (``https://api.z.ai/api/paas/v4``). It is a vision-language model that
5
+ accepts documents directly: PDFs ride in a ``file_url`` content block and raw
6
+ images in an ``image_url`` block, both as base64 data URLs — z.ai exposes no
7
+ Files API, so nothing is uploaded first.
8
+
9
+ This subclasses :class:`OpenAIProvider` to reuse its ``parse_with_layout_file``
10
+ plumbing (per-page PDF splitting, the ``<div data-bbox data-label>`` layout
11
+ prompt/parse machinery, and ``normalize``). Only the pieces that are genuinely
12
+ z.ai-specific are overridden: the client/auth, the per-page API calls (z.ai uses
13
+ ``file_url`` / ``image_url`` blocks rather than OpenAI's ``type: file`` blocks),
14
+ token accounting, pricing, and error wording.
15
+
16
+ Thinking is always on for GLM-5.3-flash and cannot be disabled, so the layout
17
+ pipeline uses the model's default reasoning; ``reasoning_tokens`` are reported as
18
+ part of ``completion_tokens`` and billed at the output rate.
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import base64
24
+ import os
25
+ import threading
26
+ from typing import Any, NoReturn
27
+
28
+ from PIL import Image
29
+
30
+ from parse_bench.inference.providers.base import (
31
+ Provider,
32
+ ProviderConfigError,
33
+ ProviderPermanentError,
34
+ ProviderTransientError,
35
+ )
36
+ from parse_bench.inference.providers.parse._layout_utils import (
37
+ SYSTEM_PROMPT_LAYOUT,
38
+ USER_PROMPT_LAYOUT,
39
+ parse_layout_blocks,
40
+ )
41
+ from parse_bench.inference.providers.parse.openai import OpenAIProvider
42
+ from parse_bench.inference.providers.registry import register_provider
43
+ from parse_bench.schemas.pipeline import PipelineSpec
44
+ from parse_bench.schemas.pipeline_io import InferenceRequest, RawInferenceResult
45
+
46
+ # z.ai list pricing: USD per million tokens (input, cached_input, output).
47
+ # Cached reads are credited in run_inference (the inherited OpenAIProvider cost
48
+ # formula only has input/output terms). A 50%-off promo (0.075 / 0.015 / 0.25)
49
+ # runs through 2026-09-09; list price is used so the benchmark cost stays stable
50
+ # after it ends. Source: https://docs.z.ai/guides/overview/pricing (verified 2026-08-26)
51
+ _GLM_ZAI_PARSE_PRICING_PER_M: dict[str, tuple[float, float, float]] = {
52
+ "glm-5.3-flash": (0.15, 0.03, 0.50),
53
+ }
54
+
55
+ _ZAI_BASE_URL = "https://api.z.ai/api/paas/v4"
56
+
57
+
58
+ @register_provider("glm_zai")
59
+ class GLMZaiParseProvider(OpenAIProvider):
60
+ """GLM-5.3-flash document parsing through z.ai's OpenAI-compatible API."""
61
+
62
+ DEFAULT_MODEL = "glm-5.3-flash"
63
+
64
+ def __init__(self, provider_name: str, base_config: dict[str, Any] | None = None):
65
+ # Skip OpenAIProvider.__init__ (it demands OPENAI_API_KEY and an OpenAI
66
+ # client); wire the z.ai client and the fields run_inference/normalize use.
67
+ Provider.__init__(self, provider_name, base_config)
68
+
69
+ self._api_key = self.base_config.get("api_key") or os.environ.get("GLM_ZAI_API_KEY")
70
+ if not self._api_key:
71
+ raise ProviderConfigError(
72
+ "GLM z.ai API key is required. Set GLM_ZAI_API_KEY or pass api_key in base_config."
73
+ )
74
+
75
+ self._model = self.base_config.get("model", self.DEFAULT_MODEL)
76
+ self._dpi = self.base_config.get("dpi", 150)
77
+ self._max_tokens = self.base_config.get("max_tokens", 32768)
78
+ # Thinking is always on, so a page can take a while — give it more room
79
+ # than the OpenAI default of 120s.
80
+ self._timeout = self.base_config.get("timeout", 600)
81
+ self._reasoning_effort = self.base_config.get("reasoning_effort", None)
82
+ self._temperature = self.base_config.get("temperature", 0)
83
+ self._base_url = self.base_config.get("base_url", _ZAI_BASE_URL)
84
+ self._mode = self.base_config.get("mode", "parse_with_layout_file")
85
+ # The shared layout prompt requests normalized 0-1000 coordinates, and
86
+ # inherited run/normalize code records and consumes this scale.
87
+ self._bbox_scale = self.base_config.get("bbox_scale", 1000)
88
+ self._cached_input_price_per_1m = float(self.base_config.get("cached_input_price_per_1m", self._pricing3()[1]))
89
+ # Per-thread tally of cache-read tokens across a request's per-page API
90
+ # calls. The runner shares one provider instance across a thread pool, so
91
+ # a plain attribute would race between concurrent documents; thread-local
92
+ # state is private to the thread running a single run_inference call.
93
+ self._cache_tls = threading.local()
94
+
95
+ if self._mode not in ("image", "file", "parse_with_layout", "parse_with_layout_file"):
96
+ raise ProviderConfigError(
97
+ f"Invalid mode '{self._mode}'. "
98
+ "Must be 'image', 'file', 'parse_with_layout', or 'parse_with_layout_file'."
99
+ )
100
+
101
+ try:
102
+ from openai import OpenAI
103
+
104
+ self._client = OpenAI(api_key=self._api_key, base_url=self._base_url, timeout=self._timeout)
105
+ except ImportError as e:
106
+ raise ProviderConfigError("openai package not installed. Run: pip install openai") from e
107
+
108
+ def _pricing3(self) -> tuple[float, float, float]:
109
+ """Longest-prefix (input, cached_input, output) rate per 1M tokens."""
110
+ matches = [(p, r) for p, r in _GLM_ZAI_PARSE_PRICING_PER_M.items() if self._model.startswith(p)]
111
+ return max(matches, key=lambda x: len(x[0]))[1] if matches else (0.0, 0.0, 0.0)
112
+
113
+ def _get_pricing(self) -> tuple[float, float]:
114
+ # The inherited cost formula bills (input, output); the cached-read
115
+ # discount is applied separately in run_inference.
116
+ in_rate, _cached_rate, out_rate = self._pricing3()
117
+ return in_rate, out_rate
118
+
119
+ @staticmethod
120
+ def _read_cached_tokens(response) -> int: # type: ignore[no-untyped-def]
121
+ """Cache-read (hit) tokens the API reports for this call, 0 if none."""
122
+ usage = getattr(response, "usage", None)
123
+ details = getattr(usage, "prompt_tokens_details", None) if usage is not None else None
124
+ return int(getattr(details, "cached_tokens", 0) or 0) if details is not None else 0
125
+
126
+ def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
127
+ # Tally cache-read tokens across this request's per-page calls, run the
128
+ # inherited parse/normalize path (which bills every input token at the
129
+ # full input rate), then credit the cache-read tokens down to the cheaper
130
+ # cached rate. z.ai returns the cache-hit count, so it should not be
131
+ # billed as fresh input.
132
+ self._cache_tls.value = 0
133
+ result = super().run_inference(pipeline, request)
134
+ cached = int(getattr(self._cache_tls, "value", 0) or 0)
135
+ raw = result.raw_output
136
+ raw["cached_input_tokens"] = cached
137
+ if cached > 0:
138
+ in_rate, _out_rate = self._get_pricing()
139
+ credit = cached * (in_rate - self._cached_input_price_per_1m) / 1_000_000
140
+ raw["cost_usd"] = max(0.0, float(raw.get("cost_usd", 0.0)) - credit)
141
+ num_pages = raw.get("num_pages") or 0
142
+ if num_pages > 0:
143
+ raw["cost_per_page_usd"] = raw["cost_usd"] / num_pages
144
+ return result
145
+
146
+ def _raise_glm_error(self, e: Exception) -> NoReturn:
147
+ """Classify a z.ai/GLM SDK exception as transient (retried) or permanent."""
148
+ status_code = getattr(e, "status_code", None)
149
+ is_retryable_status = isinstance(status_code, int) and (
150
+ status_code in {408, 409, 429} or 500 <= status_code < 600
151
+ )
152
+ is_retryable_type = isinstance(e, (TimeoutError, ConnectionError)) or type(e).__name__ in {
153
+ "APIConnectionError",
154
+ "APITimeoutError",
155
+ "InternalServerError",
156
+ "RateLimitError",
157
+ }
158
+ if is_retryable_status or is_retryable_type:
159
+ raise ProviderTransientError(f"Transient error calling z.ai GLM API: {e}") from e
160
+ raise ProviderPermanentError(f"Error calling z.ai GLM API: {e}") from e
161
+
162
+ # OpenAI's chat-completions usage reports the full ``completion_tokens``
163
+ # (visible output *plus* reasoning), with the reasoning count broken out in
164
+ # ``completion_tokens_details``. Splitting them here — output = visible,
165
+ # thinking = reasoning — lets the inherited cost formula bill
166
+ # ``(output + thinking)`` = the full completion at the output rate exactly
167
+ # once, while still recording the reasoning token count.
168
+ @staticmethod
169
+ def _extract_usage(response) -> dict[str, int]: # type: ignore[no-untyped-def]
170
+ usage = getattr(response, "usage", None)
171
+ if usage is None:
172
+ return {"input_tokens": 0, "output_tokens": 0, "thinking_tokens": 0, "total_tokens": 0}
173
+ input_tok = getattr(usage, "prompt_tokens", 0) or 0
174
+ completion_tok = getattr(usage, "completion_tokens", 0) or 0
175
+ total_tok = getattr(usage, "total_tokens", 0) or 0
176
+ details = getattr(usage, "completion_tokens_details", None)
177
+ thinking_tok = (getattr(details, "reasoning_tokens", 0) or 0) if details else 0
178
+ visible_tok = max(0, completion_tok - thinking_tok)
179
+ return {
180
+ "input_tokens": input_tok,
181
+ "output_tokens": visible_tok,
182
+ "thinking_tokens": thinking_tok,
183
+ "total_tokens": total_tok,
184
+ }
185
+
186
+ def _layout_request_kwargs(self, file_block: dict[str, Any]) -> dict[str, Any]:
187
+ kwargs: dict[str, Any] = {
188
+ "model": self._model,
189
+ "max_tokens": self._max_tokens,
190
+ "temperature": self._temperature,
191
+ "messages": [
192
+ {"role": "system", "content": SYSTEM_PROMPT_LAYOUT},
193
+ {
194
+ "role": "user",
195
+ "content": [file_block, {"type": "text", "text": USER_PROMPT_LAYOUT}],
196
+ },
197
+ ],
198
+ }
199
+ if self._reasoning_effort is not None:
200
+ kwargs["reasoning_effort"] = self._reasoning_effort
201
+ return kwargs
202
+
203
+ def _parse_pdf_page_with_layout(self, pdf_bytes: bytes) -> tuple[list[dict[str, Any]], str, dict[str, int]]:
204
+ """Send a single-page PDF to GLM with the layout prompt via a file_url block."""
205
+ pdf_base64 = base64.standard_b64encode(pdf_bytes).decode("utf-8")
206
+ file_block = {"type": "file_url", "file_url": {"url": f"data:application/pdf;base64,{pdf_base64}"}}
207
+ try:
208
+ response = self._client.chat.completions.create(**self._layout_request_kwargs(file_block))
209
+ self._cache_tls.value = getattr(self._cache_tls, "value", 0) + self._read_cached_tokens(response)
210
+ usage = self._extract_usage(response)
211
+ content = response.choices[0].message.content if response.choices else ""
212
+ text = content or ""
213
+ return parse_layout_blocks(text), text, usage
214
+ except Exception as e:
215
+ self._raise_glm_error(e)
216
+
217
+ def _parse_image_with_layout(self, image: Image.Image) -> tuple[list[dict[str, Any]], str, dict[str, int]]:
218
+ """Send a page image to GLM with the layout prompt via an image_url block."""
219
+ img_base64 = self._image_to_base64(image)
220
+ file_block = {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{img_base64}"}}
221
+ try:
222
+ response = self._client.chat.completions.create(**self._layout_request_kwargs(file_block))
223
+ self._cache_tls.value = getattr(self._cache_tls, "value", 0) + self._read_cached_tokens(response)
224
+ usage = self._extract_usage(response)
225
+ content = response.choices[0].message.content if response.choices else ""
226
+ text = content or ""
227
+ return parse_layout_blocks(text), text, usage
228
+ except Exception as e:
229
+ self._raise_glm_error(e)