parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
"""Provider for GLM (z.ai) vision-based PARSE.
|
|
2
|
+
|
|
3
|
+
GLM-5.3-flash is served through z.ai's OpenAI-compatible chat completions
|
|
4
|
+
endpoint (``https://api.z.ai/api/paas/v4``). It is a vision-language model that
|
|
5
|
+
accepts documents directly: PDFs ride in a ``file_url`` content block and raw
|
|
6
|
+
images in an ``image_url`` block, both as base64 data URLs — z.ai exposes no
|
|
7
|
+
Files API, so nothing is uploaded first.
|
|
8
|
+
|
|
9
|
+
This subclasses :class:`OpenAIProvider` to reuse its ``parse_with_layout_file``
|
|
10
|
+
plumbing (per-page PDF splitting, the ``<div data-bbox data-label>`` layout
|
|
11
|
+
prompt/parse machinery, and ``normalize``). Only the pieces that are genuinely
|
|
12
|
+
z.ai-specific are overridden: the client/auth, the per-page API calls (z.ai uses
|
|
13
|
+
``file_url`` / ``image_url`` blocks rather than OpenAI's ``type: file`` blocks),
|
|
14
|
+
token accounting, pricing, and error wording.
|
|
15
|
+
|
|
16
|
+
Thinking is always on for GLM-5.3-flash and cannot be disabled, so the layout
|
|
17
|
+
pipeline uses the model's default reasoning; ``reasoning_tokens`` are reported as
|
|
18
|
+
part of ``completion_tokens`` and billed at the output rate.
|
|
19
|
+
"""
|
|
20
|
+
|
|
21
|
+
from __future__ import annotations
|
|
22
|
+
|
|
23
|
+
import base64
|
|
24
|
+
import os
|
|
25
|
+
import threading
|
|
26
|
+
from typing import Any, NoReturn
|
|
27
|
+
|
|
28
|
+
from PIL import Image
|
|
29
|
+
|
|
30
|
+
from parse_bench.inference.providers.base import (
|
|
31
|
+
Provider,
|
|
32
|
+
ProviderConfigError,
|
|
33
|
+
ProviderPermanentError,
|
|
34
|
+
ProviderTransientError,
|
|
35
|
+
)
|
|
36
|
+
from parse_bench.inference.providers.parse._layout_utils import (
|
|
37
|
+
SYSTEM_PROMPT_LAYOUT,
|
|
38
|
+
USER_PROMPT_LAYOUT,
|
|
39
|
+
parse_layout_blocks,
|
|
40
|
+
)
|
|
41
|
+
from parse_bench.inference.providers.parse.openai import OpenAIProvider
|
|
42
|
+
from parse_bench.inference.providers.registry import register_provider
|
|
43
|
+
from parse_bench.schemas.pipeline import PipelineSpec
|
|
44
|
+
from parse_bench.schemas.pipeline_io import InferenceRequest, RawInferenceResult
|
|
45
|
+
|
|
46
|
+
# z.ai list pricing: USD per million tokens (input, cached_input, output).
|
|
47
|
+
# Cached reads are credited in run_inference (the inherited OpenAIProvider cost
|
|
48
|
+
# formula only has input/output terms). A 50%-off promo (0.075 / 0.015 / 0.25)
|
|
49
|
+
# runs through 2026-09-09; list price is used so the benchmark cost stays stable
|
|
50
|
+
# after it ends. Source: https://docs.z.ai/guides/overview/pricing (verified 2026-08-26)
|
|
51
|
+
_GLM_ZAI_PARSE_PRICING_PER_M: dict[str, tuple[float, float, float]] = {
|
|
52
|
+
"glm-5.3-flash": (0.15, 0.03, 0.50),
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
_ZAI_BASE_URL = "https://api.z.ai/api/paas/v4"
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@register_provider("glm_zai")
|
|
59
|
+
class GLMZaiParseProvider(OpenAIProvider):
|
|
60
|
+
"""GLM-5.3-flash document parsing through z.ai's OpenAI-compatible API."""
|
|
61
|
+
|
|
62
|
+
DEFAULT_MODEL = "glm-5.3-flash"
|
|
63
|
+
|
|
64
|
+
def __init__(self, provider_name: str, base_config: dict[str, Any] | None = None):
|
|
65
|
+
# Skip OpenAIProvider.__init__ (it demands OPENAI_API_KEY and an OpenAI
|
|
66
|
+
# client); wire the z.ai client and the fields run_inference/normalize use.
|
|
67
|
+
Provider.__init__(self, provider_name, base_config)
|
|
68
|
+
|
|
69
|
+
self._api_key = self.base_config.get("api_key") or os.environ.get("GLM_ZAI_API_KEY")
|
|
70
|
+
if not self._api_key:
|
|
71
|
+
raise ProviderConfigError(
|
|
72
|
+
"GLM z.ai API key is required. Set GLM_ZAI_API_KEY or pass api_key in base_config."
|
|
73
|
+
)
|
|
74
|
+
|
|
75
|
+
self._model = self.base_config.get("model", self.DEFAULT_MODEL)
|
|
76
|
+
self._dpi = self.base_config.get("dpi", 150)
|
|
77
|
+
self._max_tokens = self.base_config.get("max_tokens", 32768)
|
|
78
|
+
# Thinking is always on, so a page can take a while — give it more room
|
|
79
|
+
# than the OpenAI default of 120s.
|
|
80
|
+
self._timeout = self.base_config.get("timeout", 600)
|
|
81
|
+
self._reasoning_effort = self.base_config.get("reasoning_effort", None)
|
|
82
|
+
self._temperature = self.base_config.get("temperature", 0)
|
|
83
|
+
self._base_url = self.base_config.get("base_url", _ZAI_BASE_URL)
|
|
84
|
+
self._mode = self.base_config.get("mode", "parse_with_layout_file")
|
|
85
|
+
# The shared layout prompt requests normalized 0-1000 coordinates, and
|
|
86
|
+
# inherited run/normalize code records and consumes this scale.
|
|
87
|
+
self._bbox_scale = self.base_config.get("bbox_scale", 1000)
|
|
88
|
+
self._cached_input_price_per_1m = float(self.base_config.get("cached_input_price_per_1m", self._pricing3()[1]))
|
|
89
|
+
# Per-thread tally of cache-read tokens across a request's per-page API
|
|
90
|
+
# calls. The runner shares one provider instance across a thread pool, so
|
|
91
|
+
# a plain attribute would race between concurrent documents; thread-local
|
|
92
|
+
# state is private to the thread running a single run_inference call.
|
|
93
|
+
self._cache_tls = threading.local()
|
|
94
|
+
|
|
95
|
+
if self._mode not in ("image", "file", "parse_with_layout", "parse_with_layout_file"):
|
|
96
|
+
raise ProviderConfigError(
|
|
97
|
+
f"Invalid mode '{self._mode}'. "
|
|
98
|
+
"Must be 'image', 'file', 'parse_with_layout', or 'parse_with_layout_file'."
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
try:
|
|
102
|
+
from openai import OpenAI
|
|
103
|
+
|
|
104
|
+
self._client = OpenAI(api_key=self._api_key, base_url=self._base_url, timeout=self._timeout)
|
|
105
|
+
except ImportError as e:
|
|
106
|
+
raise ProviderConfigError("openai package not installed. Run: pip install openai") from e
|
|
107
|
+
|
|
108
|
+
def _pricing3(self) -> tuple[float, float, float]:
|
|
109
|
+
"""Longest-prefix (input, cached_input, output) rate per 1M tokens."""
|
|
110
|
+
matches = [(p, r) for p, r in _GLM_ZAI_PARSE_PRICING_PER_M.items() if self._model.startswith(p)]
|
|
111
|
+
return max(matches, key=lambda x: len(x[0]))[1] if matches else (0.0, 0.0, 0.0)
|
|
112
|
+
|
|
113
|
+
def _get_pricing(self) -> tuple[float, float]:
|
|
114
|
+
# The inherited cost formula bills (input, output); the cached-read
|
|
115
|
+
# discount is applied separately in run_inference.
|
|
116
|
+
in_rate, _cached_rate, out_rate = self._pricing3()
|
|
117
|
+
return in_rate, out_rate
|
|
118
|
+
|
|
119
|
+
@staticmethod
|
|
120
|
+
def _read_cached_tokens(response) -> int: # type: ignore[no-untyped-def]
|
|
121
|
+
"""Cache-read (hit) tokens the API reports for this call, 0 if none."""
|
|
122
|
+
usage = getattr(response, "usage", None)
|
|
123
|
+
details = getattr(usage, "prompt_tokens_details", None) if usage is not None else None
|
|
124
|
+
return int(getattr(details, "cached_tokens", 0) or 0) if details is not None else 0
|
|
125
|
+
|
|
126
|
+
def run_inference(self, pipeline: PipelineSpec, request: InferenceRequest) -> RawInferenceResult:
|
|
127
|
+
# Tally cache-read tokens across this request's per-page calls, run the
|
|
128
|
+
# inherited parse/normalize path (which bills every input token at the
|
|
129
|
+
# full input rate), then credit the cache-read tokens down to the cheaper
|
|
130
|
+
# cached rate. z.ai returns the cache-hit count, so it should not be
|
|
131
|
+
# billed as fresh input.
|
|
132
|
+
self._cache_tls.value = 0
|
|
133
|
+
result = super().run_inference(pipeline, request)
|
|
134
|
+
cached = int(getattr(self._cache_tls, "value", 0) or 0)
|
|
135
|
+
raw = result.raw_output
|
|
136
|
+
raw["cached_input_tokens"] = cached
|
|
137
|
+
if cached > 0:
|
|
138
|
+
in_rate, _out_rate = self._get_pricing()
|
|
139
|
+
credit = cached * (in_rate - self._cached_input_price_per_1m) / 1_000_000
|
|
140
|
+
raw["cost_usd"] = max(0.0, float(raw.get("cost_usd", 0.0)) - credit)
|
|
141
|
+
num_pages = raw.get("num_pages") or 0
|
|
142
|
+
if num_pages > 0:
|
|
143
|
+
raw["cost_per_page_usd"] = raw["cost_usd"] / num_pages
|
|
144
|
+
return result
|
|
145
|
+
|
|
146
|
+
def _raise_glm_error(self, e: Exception) -> NoReturn:
|
|
147
|
+
"""Classify a z.ai/GLM SDK exception as transient (retried) or permanent."""
|
|
148
|
+
status_code = getattr(e, "status_code", None)
|
|
149
|
+
is_retryable_status = isinstance(status_code, int) and (
|
|
150
|
+
status_code in {408, 409, 429} or 500 <= status_code < 600
|
|
151
|
+
)
|
|
152
|
+
is_retryable_type = isinstance(e, (TimeoutError, ConnectionError)) or type(e).__name__ in {
|
|
153
|
+
"APIConnectionError",
|
|
154
|
+
"APITimeoutError",
|
|
155
|
+
"InternalServerError",
|
|
156
|
+
"RateLimitError",
|
|
157
|
+
}
|
|
158
|
+
if is_retryable_status or is_retryable_type:
|
|
159
|
+
raise ProviderTransientError(f"Transient error calling z.ai GLM API: {e}") from e
|
|
160
|
+
raise ProviderPermanentError(f"Error calling z.ai GLM API: {e}") from e
|
|
161
|
+
|
|
162
|
+
# OpenAI's chat-completions usage reports the full ``completion_tokens``
|
|
163
|
+
# (visible output *plus* reasoning), with the reasoning count broken out in
|
|
164
|
+
# ``completion_tokens_details``. Splitting them here — output = visible,
|
|
165
|
+
# thinking = reasoning — lets the inherited cost formula bill
|
|
166
|
+
# ``(output + thinking)`` = the full completion at the output rate exactly
|
|
167
|
+
# once, while still recording the reasoning token count.
|
|
168
|
+
@staticmethod
|
|
169
|
+
def _extract_usage(response) -> dict[str, int]: # type: ignore[no-untyped-def]
|
|
170
|
+
usage = getattr(response, "usage", None)
|
|
171
|
+
if usage is None:
|
|
172
|
+
return {"input_tokens": 0, "output_tokens": 0, "thinking_tokens": 0, "total_tokens": 0}
|
|
173
|
+
input_tok = getattr(usage, "prompt_tokens", 0) or 0
|
|
174
|
+
completion_tok = getattr(usage, "completion_tokens", 0) or 0
|
|
175
|
+
total_tok = getattr(usage, "total_tokens", 0) or 0
|
|
176
|
+
details = getattr(usage, "completion_tokens_details", None)
|
|
177
|
+
thinking_tok = (getattr(details, "reasoning_tokens", 0) or 0) if details else 0
|
|
178
|
+
visible_tok = max(0, completion_tok - thinking_tok)
|
|
179
|
+
return {
|
|
180
|
+
"input_tokens": input_tok,
|
|
181
|
+
"output_tokens": visible_tok,
|
|
182
|
+
"thinking_tokens": thinking_tok,
|
|
183
|
+
"total_tokens": total_tok,
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
def _layout_request_kwargs(self, file_block: dict[str, Any]) -> dict[str, Any]:
|
|
187
|
+
kwargs: dict[str, Any] = {
|
|
188
|
+
"model": self._model,
|
|
189
|
+
"max_tokens": self._max_tokens,
|
|
190
|
+
"temperature": self._temperature,
|
|
191
|
+
"messages": [
|
|
192
|
+
{"role": "system", "content": SYSTEM_PROMPT_LAYOUT},
|
|
193
|
+
{
|
|
194
|
+
"role": "user",
|
|
195
|
+
"content": [file_block, {"type": "text", "text": USER_PROMPT_LAYOUT}],
|
|
196
|
+
},
|
|
197
|
+
],
|
|
198
|
+
}
|
|
199
|
+
if self._reasoning_effort is not None:
|
|
200
|
+
kwargs["reasoning_effort"] = self._reasoning_effort
|
|
201
|
+
return kwargs
|
|
202
|
+
|
|
203
|
+
def _parse_pdf_page_with_layout(self, pdf_bytes: bytes) -> tuple[list[dict[str, Any]], str, dict[str, int]]:
|
|
204
|
+
"""Send a single-page PDF to GLM with the layout prompt via a file_url block."""
|
|
205
|
+
pdf_base64 = base64.standard_b64encode(pdf_bytes).decode("utf-8")
|
|
206
|
+
file_block = {"type": "file_url", "file_url": {"url": f"data:application/pdf;base64,{pdf_base64}"}}
|
|
207
|
+
try:
|
|
208
|
+
response = self._client.chat.completions.create(**self._layout_request_kwargs(file_block))
|
|
209
|
+
self._cache_tls.value = getattr(self._cache_tls, "value", 0) + self._read_cached_tokens(response)
|
|
210
|
+
usage = self._extract_usage(response)
|
|
211
|
+
content = response.choices[0].message.content if response.choices else ""
|
|
212
|
+
text = content or ""
|
|
213
|
+
return parse_layout_blocks(text), text, usage
|
|
214
|
+
except Exception as e:
|
|
215
|
+
self._raise_glm_error(e)
|
|
216
|
+
|
|
217
|
+
def _parse_image_with_layout(self, image: Image.Image) -> tuple[list[dict[str, Any]], str, dict[str, int]]:
|
|
218
|
+
"""Send a page image to GLM with the layout prompt via an image_url block."""
|
|
219
|
+
img_base64 = self._image_to_base64(image)
|
|
220
|
+
file_block = {"type": "image_url", "image_url": {"url": f"data:image/jpeg;base64,{img_base64}"}}
|
|
221
|
+
try:
|
|
222
|
+
response = self._client.chat.completions.create(**self._layout_request_kwargs(file_block))
|
|
223
|
+
self._cache_tls.value = getattr(self._cache_tls, "value", 0) + self._read_cached_tokens(response)
|
|
224
|
+
usage = self._extract_usage(response)
|
|
225
|
+
content = response.choices[0].message.content if response.choices else ""
|
|
226
|
+
text = content or ""
|
|
227
|
+
return parse_layout_blocks(text), text, usage
|
|
228
|
+
except Exception as e:
|
|
229
|
+
self._raise_glm_error(e)
|