parse-bench 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- parse_bench/__init__.py +3 -0
- parse_bench/analysis/__init__.py +6 -0
- parse_bench/analysis/aggregation_report.py +582 -0
- parse_bench/analysis/cli.py +472 -0
- parse_bench/analysis/comparison.py +382 -0
- parse_bench/analysis/comparison_core.py +357 -0
- parse_bench/analysis/comparison_report.py +2066 -0
- parse_bench/analysis/detailed_report.py +2254 -0
- parse_bench/analysis/leaderboard_report.py +852 -0
- parse_bench/analysis/metric_definitions.py +771 -0
- parse_bench/cli.py +267 -0
- parse_bench/data/__init__.py +1 -0
- parse_bench/data/cli.py +118 -0
- parse_bench/data/download.py +127 -0
- parse_bench/evaluation/__init__.py +11 -0
- parse_bench/evaluation/cli.py +435 -0
- parse_bench/evaluation/evaluators/__init__.py +17 -0
- parse_bench/evaluation/evaluators/base.py +34 -0
- parse_bench/evaluation/evaluators/extract.py +429 -0
- parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
- parse_bench/evaluation/evaluators/parse.py +1353 -0
- parse_bench/evaluation/evaluators/qa.py +199 -0
- parse_bench/evaluation/layout_adapters/__init__.py +21 -0
- parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
- parse_bench/evaluation/layout_adapters/base.py +105 -0
- parse_bench/evaluation/layout_adapters/registry.py +109 -0
- parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
- parse_bench/evaluation/layout_label_mappers/base.py +66 -0
- parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
- parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
- parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
- parse_bench/evaluation/metric_aggregation.py +56 -0
- parse_bench/evaluation/metrics/__init__.py +5 -0
- parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
- parse_bench/evaluation/metrics/attribution/constants.py +12 -0
- parse_bench/evaluation/metrics/attribution/core.py +1108 -0
- parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
- parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
- parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
- parse_bench/evaluation/metrics/base.py +33 -0
- parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
- parse_bench/evaluation/metrics/extract/__init__.py +29 -0
- parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
- parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
- parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
- parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
- parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
- parse_bench/evaluation/metrics/extract/test_types.py +11 -0
- parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
- parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
- parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
- parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
- parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
- parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
- parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
- parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
- parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
- parse_bench/evaluation/metrics/parse/__init__.py +5 -0
- parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
- parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
- parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
- parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
- parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
- parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
- parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
- parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
- parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
- parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
- parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
- parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
- parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
- parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
- parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
- parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
- parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
- parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
- parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
- parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
- parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
- parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
- parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
- parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
- parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
- parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
- parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
- parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
- parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
- parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
- parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
- parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
- parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
- parse_bench/evaluation/metrics/parse/test_types.py +103 -0
- parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
- parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
- parse_bench/evaluation/metrics/parse/utils.py +885 -0
- parse_bench/evaluation/metrics/qa/__init__.py +5 -0
- parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
- parse_bench/evaluation/qa/__init__.py +5 -0
- parse_bench/evaluation/qa/llm_service.py +335 -0
- parse_bench/evaluation/reports/__init__.py +8 -0
- parse_bench/evaluation/reports/csv.py +64 -0
- parse_bench/evaluation/reports/html.py +338 -0
- parse_bench/evaluation/reports/markdown.py +98 -0
- parse_bench/evaluation/reports/rule_csv.py +22 -0
- parse_bench/evaluation/runner.py +1864 -0
- parse_bench/evaluation/stats.py +104 -0
- parse_bench/extensions.py +72 -0
- parse_bench/inference/__init__.py +33 -0
- parse_bench/inference/chunkr_layout_extraction.py +160 -0
- parse_bench/inference/cli.py +484 -0
- parse_bench/inference/layout_extraction.py +422 -0
- parse_bench/inference/pipelines/__init__.py +59 -0
- parse_bench/inference/pipelines/extract.py +39 -0
- parse_bench/inference/pipelines/layout.py +142 -0
- parse_bench/inference/pipelines/parse.py +2603 -0
- parse_bench/inference/pipelines.py +0 -0
- parse_bench/inference/providers/__init__.py +28 -0
- parse_bench/inference/providers/base.py +196 -0
- parse_bench/inference/providers/cancellation.py +137 -0
- parse_bench/inference/providers/extract/__init__.py +22 -0
- parse_bench/inference/providers/extract/citations.py +549 -0
- parse_bench/inference/providers/extract/extend.py +851 -0
- parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
- parse_bench/inference/providers/layoutdet/__init__.py +25 -0
- parse_bench/inference/providers/layoutdet/adapters.py +946 -0
- parse_bench/inference/providers/layoutdet/base.py +203 -0
- parse_bench/inference/providers/layoutdet/chandra.py +449 -0
- parse_bench/inference/providers/layoutdet/docling.py +125 -0
- parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
- parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
- parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
- parse_bench/inference/providers/layoutdet/paddle.py +117 -0
- parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
- parse_bench/inference/providers/layoutdet/surya.py +250 -0
- parse_bench/inference/providers/layoutdet/yolo.py +109 -0
- parse_bench/inference/providers/parse/__init__.py +64 -0
- parse_bench/inference/providers/parse/_docling_common.py +233 -0
- parse_bench/inference/providers/parse/_layout_utils.py +611 -0
- parse_bench/inference/providers/parse/amazon_nova.py +515 -0
- parse_bench/inference/providers/parse/anthropic.py +882 -0
- parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
- parse_bench/inference/providers/parse/chandra2.py +633 -0
- parse_bench/inference/providers/parse/chunkr.py +268 -0
- parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
- parse_bench/inference/providers/parse/datalab.py +370 -0
- parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
- parse_bench/inference/providers/parse/docling.py +281 -0
- parse_bench/inference/providers/parse/docling_serve.py +289 -0
- parse_bench/inference/providers/parse/dots_ocr.py +574 -0
- parse_bench/inference/providers/parse/extend_parse.py +710 -0
- parse_bench/inference/providers/parse/falconocr.py +436 -0
- parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
- parse_bench/inference/providers/parse/gemma4.py +472 -0
- parse_bench/inference/providers/parse/glm_zai.py +229 -0
- parse_bench/inference/providers/parse/google.py +1125 -0
- parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
- parse_bench/inference/providers/parse/google_docai.py +776 -0
- parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
- parse_bench/inference/providers/parse/granite_vision.py +515 -0
- parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
- parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
- parse_bench/inference/providers/parse/landingai.py +452 -0
- parse_bench/inference/providers/parse/liteparse.py +350 -0
- parse_bench/inference/providers/parse/llamaparse.py +677 -0
- parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
- parse_bench/inference/providers/parse/markitdown.py +138 -0
- parse_bench/inference/providers/parse/mineru25.py +405 -0
- parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
- parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
- parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
- parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
- parse_bench/inference/providers/parse/oi_parser.py +222 -0
- parse_bench/inference/providers/parse/openai.py +740 -0
- parse_bench/inference/providers/parse/opendataloader.py +152 -0
- parse_bench/inference/providers/parse/paddleocr.py +624 -0
- parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
- parse_bench/inference/providers/parse/pulse.py +785 -0
- parse_bench/inference/providers/parse/pymupdf.py +207 -0
- parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
- parse_bench/inference/providers/parse/pypdf.py +179 -0
- parse_bench/inference/providers/parse/qwen.py +678 -0
- parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
- parse_bench/inference/providers/parse/reducto.py +546 -0
- parse_bench/inference/providers/parse/surya2.py +372 -0
- parse_bench/inference/providers/parse/tesseract.py +301 -0
- parse_bench/inference/providers/parse/textract.py +694 -0
- parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
- parse_bench/inference/providers/parse/unstructured.py +485 -0
- parse_bench/inference/providers/parse/warp_ingest.py +199 -0
- parse_bench/inference/providers/registry.py +49 -0
- parse_bench/inference/renormalize.py +170 -0
- parse_bench/inference/runner.py +2023 -0
- parse_bench/layout_label_mapping.py +424 -0
- parse_bench/layout_projection.py +179 -0
- parse_bench/pipeline/__init__.py +1 -0
- parse_bench/pipeline/cli.py +549 -0
- parse_bench/schemas/__init__.py +33 -0
- parse_bench/schemas/evaluation.py +93 -0
- parse_bench/schemas/extract_output.py +36 -0
- parse_bench/schemas/layout_detection_output.py +545 -0
- parse_bench/schemas/layout_ontology.py +315 -0
- parse_bench/schemas/metrics.py +69 -0
- parse_bench/schemas/parse_output.py +152 -0
- parse_bench/schemas/pipeline.py +22 -0
- parse_bench/schemas/pipeline_io.py +106 -0
- parse_bench/schemas/product.py +97 -0
- parse_bench/test_cases/__init__.py +25 -0
- parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
- parse_bench/test_cases/extract_field_paths.py +164 -0
- parse_bench/test_cases/layout_attribution_generation.py +287 -0
- parse_bench/test_cases/loader.py +652 -0
- parse_bench/test_cases/parse_rule_schemas.py +1071 -0
- parse_bench/test_cases/rule_filters.py +32 -0
- parse_bench/test_cases/rule_ids.py +107 -0
- parse_bench/test_cases/schema.py +427 -0
- parse_bench/utils/__init__.py +15 -0
- parse_bench/utils/gemini_layout_utils.py +670 -0
- parse_bench/utils/text_aggregation.py +100 -0
- parse_bench-1.0.0.dist-info/METADATA +476 -0
- parse_bench-1.0.0.dist-info/RECORD +227 -0
- parse_bench-1.0.0.dist-info/WHEEL +4 -0
- parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
- parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
|
@@ -0,0 +1,819 @@
|
|
|
1
|
+
"""Helpers for Gemini Agentic Vision parse-with-layout flows."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import json
|
|
7
|
+
import logging
|
|
8
|
+
import re
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from datetime import timedelta
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from PIL import Image
|
|
14
|
+
|
|
15
|
+
from parse_bench.inference.providers.base import ProviderPermanentError, ProviderTransientError
|
|
16
|
+
from parse_bench.inference.providers.parse._layout_utils import LABEL_MAP, items_to_markdown
|
|
17
|
+
from parse_bench.schemas.parse_output import LayoutItemIR, LayoutSegmentIR, ParseLayoutPageIR
|
|
18
|
+
|
|
19
|
+
logger = logging.getLogger(__name__)
|
|
20
|
+
|
|
21
|
+
TRANSIENT_ERROR_KEYWORDS = ("timeout", "connection", "network")
|
|
22
|
+
RATE_LIMIT_ERROR_KEYWORDS = ("rate_limit", "rate limit", "429", "resource_exhausted")
|
|
23
|
+
|
|
24
|
+
CORE11_LABELS = [
|
|
25
|
+
"Caption",
|
|
26
|
+
"Footnote",
|
|
27
|
+
"Formula",
|
|
28
|
+
"List-item",
|
|
29
|
+
"Page-footer",
|
|
30
|
+
"Page-header",
|
|
31
|
+
"Picture",
|
|
32
|
+
"Section-header",
|
|
33
|
+
"Table",
|
|
34
|
+
"Text",
|
|
35
|
+
"Title",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
SYSTEM_PROMPT_AGENTIC_VISION = (
|
|
39
|
+
"You are a document parser. Convert a document page image into clean, well-structured markdown "
|
|
40
|
+
"with layout grounding.\n\n"
|
|
41
|
+
"Rules:\n"
|
|
42
|
+
"- Preserve reading order.\n"
|
|
43
|
+
"- Preserve document structure, including headings, lists, formulas, captions, and tables.\n"
|
|
44
|
+
"- Use HTML tables for tabular data.\n"
|
|
45
|
+
"- For figures or pictures, describe them briefly in square brackets like [Figure: description].\n"
|
|
46
|
+
"- Do not add commentary outside the requested wrapped content.\n"
|
|
47
|
+
"- Wrap each layout element in a <div> tag with a data-bbox and data-label attribute.\n"
|
|
48
|
+
'- data-bbox must use Gemini native coordinates: "[y_min, x_min, y_max, x_max]".\n'
|
|
49
|
+
"- Coordinates must be normalized to 0..1000 relative to the full original page image.\n"
|
|
50
|
+
"- data-label must be one of: Caption, Footnote, Formula, List-item, Page-footer, "
|
|
51
|
+
"Page-header, Picture, Section-header, Table, Text, Title.\n"
|
|
52
|
+
"- Every piece of content must be inside exactly one <div> wrapper.\n"
|
|
53
|
+
"- Start from the full page image and preserve reading order from that full-page view.\n"
|
|
54
|
+
"- First try to read and ground content from the full page image.\n"
|
|
55
|
+
"- If you zoom, crop, rotate, or enhance the page using code execution, always convert the final box "
|
|
56
|
+
"back to the full original page image coordinate system before returning data-bbox.\n"
|
|
57
|
+
"- Use code execution only if text is too small, dense, rotated, low-contrast, or ambiguous at full-page "
|
|
58
|
+
"scale, or if the bounding box would otherwise be unreliable. Do not guess.\n"
|
|
59
|
+
"- If only one region is ambiguous, inspect only that region. Prefer the smallest crop or zoom needed.\n"
|
|
60
|
+
"- Do not crop or zoom the whole page by default.\n"
|
|
61
|
+
"- Use code execution only for visual inspection, cropping, rotation, or measurement. Do not use it to "
|
|
62
|
+
"construct Python dictionaries, lists, or JSON for the final answer.\n"
|
|
63
|
+
"- Every returned data-bbox must refer to the original full-page coordinate frame, never the crop frame.\n"
|
|
64
|
+
"- After inspection, return the final wrapped markdown as assistant text. If you must use code for the final "
|
|
65
|
+
"step, print only one raw triple-quoted string containing the wrapped markdown.\n"
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
USER_PROMPT_AGENTIC_VISION_PREFIX = (
|
|
69
|
+
"Parse this document page and output its content as clean markdown, with each layout element wrapped in a "
|
|
70
|
+
'<div data-bbox="[y_min,x_min,y_max,x_max]" data-label="Category"> tag. '
|
|
71
|
+
"Use Gemini native bbox order [y_min, x_min, y_max, x_max], normalized to 0..1000 on the full page image.\n"
|
|
72
|
+
"Use HTML tables for any tabular data.\n"
|
|
73
|
+
"For Title and Section-header items, output only the heading text inside the wrapper, "
|
|
74
|
+
"not markdown heading markers.\n"
|
|
75
|
+
"For Formula items, output only the formula content inside the wrapper, not $$ fences.\n"
|
|
76
|
+
"Start from the full page image and only zoom or crop when needed for a specific ambiguous region.\n"
|
|
77
|
+
"Use code execution only to inspect the page. Do not return screenshots, plots, or other artifacts.\n"
|
|
78
|
+
"Use code execution only if text is too small, dense, rotated, low-contrast, or ambiguous at full-page scale, "
|
|
79
|
+
"or if the bbox would otherwise be unreliable.\n"
|
|
80
|
+
"Prefer the smallest crop or zoom needed and do not zoom the whole page by default.\n"
|
|
81
|
+
"Do not use code execution to build Python dictionaries, lists, or JSON for the final answer.\n"
|
|
82
|
+
"Every returned data-bbox must be mapped back to the original full page image, normalized to 0..1000.\n"
|
|
83
|
+
"After inspection, return the wrapped markdown as assistant text. If you must use code for the final step, "
|
|
84
|
+
"print only one raw triple-quoted string containing the wrapped markdown and nothing else.\n"
|
|
85
|
+
"Output ONLY the wrapped content, no explanations.\n"
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
RETRY_PROMPT_RECITATION = (
|
|
89
|
+
"Retry mode: the previous attempt returned no final wrapped markdown and triggered recitation-style behavior.\n"
|
|
90
|
+
"Do not rely on memorized text, web recall, citations, URLs, or external sources.\n"
|
|
91
|
+
"Read only from the attached page image.\n"
|
|
92
|
+
"If you use code execution, execute the final code and print the complete wrapped markdown as a single raw "
|
|
93
|
+
'triple-quoted string containing <div data-bbox="[y_min,x_min,y_max,x_max]" ...> wrappers.\n'
|
|
94
|
+
"Do not stop after writing planning code. The executed code must print the final wrapped markdown.\n"
|
|
95
|
+
"Do not emit citations, URLs, commentary, or any text outside the wrapped markdown.\n"
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
RETRY_PROMPT_EMPTY_OUTPUT = (
|
|
99
|
+
"Retry mode: the previous attempt returned code or citations but no usable wrapped markdown.\n"
|
|
100
|
+
"You must finish this retry with actual wrapped markdown output, not just planning code.\n"
|
|
101
|
+
"If needed, use code execution to inspect crops, then print the complete wrapped markdown as one raw "
|
|
102
|
+
"triple-quoted string.\n"
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
RETRY_PROMPT_FINAL_ONLY = (
|
|
106
|
+
"Final retry: do not write planning code, helper code, comments, crop definitions, or analysis.\n"
|
|
107
|
+
"Return the final wrapped markdown now.\n"
|
|
108
|
+
"Preferred form: execute exactly one print(r'''...''') statement containing the complete wrapped markdown.\n"
|
|
109
|
+
"Alternative form: return the wrapped markdown directly as assistant text.\n"
|
|
110
|
+
"Do not output anything except the wrapped markdown itself.\n"
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
_PATTERN_BBOX_FIRST = re.compile(
|
|
114
|
+
r'<div\s+[^>]*?data-bbox=["\'](\[[^\]]+\])["\'][^>]*?data-label=["\']([^"\']+)["\'][^>]*?>'
|
|
115
|
+
r"([\s\S]*?)</div>",
|
|
116
|
+
re.IGNORECASE,
|
|
117
|
+
)
|
|
118
|
+
_PATTERN_LABEL_FIRST = re.compile(
|
|
119
|
+
r'<div\s+[^>]*?data-label=["\']([^"\']+)["\'][^>]*?data-bbox=["\'](\[[^\]]+\])["\'][^>]*?>'
|
|
120
|
+
r"([\s\S]*?)</div>",
|
|
121
|
+
re.IGNORECASE,
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
@dataclass(frozen=True)
|
|
126
|
+
class AgenticVisionCacheInfo:
|
|
127
|
+
"""Resolved explicit cache metadata for a run."""
|
|
128
|
+
|
|
129
|
+
name: str
|
|
130
|
+
display_name: str
|
|
131
|
+
token_count: int
|
|
132
|
+
ttl_seconds: int
|
|
133
|
+
storage_cost_usd: float
|
|
134
|
+
created: bool
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
@dataclass(frozen=True)
|
|
138
|
+
class AgenticVisionPageResponse:
|
|
139
|
+
"""Parsed wrapped layout output for one page."""
|
|
140
|
+
|
|
141
|
+
raw_content: str
|
|
142
|
+
items: list[dict[str, Any]]
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
@dataclass
|
|
146
|
+
class AgenticVisionPageResult:
|
|
147
|
+
"""Per-page parse result plus serialized API call traces."""
|
|
148
|
+
|
|
149
|
+
page_index: int
|
|
150
|
+
width: int
|
|
151
|
+
height: int
|
|
152
|
+
image_mime_type: str
|
|
153
|
+
items: list[dict[str, Any]]
|
|
154
|
+
markdown: str
|
|
155
|
+
raw_content: str
|
|
156
|
+
thought_summaries: list[str]
|
|
157
|
+
thought_signatures: list[str]
|
|
158
|
+
generated_code: list[dict[str, Any]]
|
|
159
|
+
code_execution_results: list[dict[str, Any]]
|
|
160
|
+
api_calls: list[dict[str, Any]]
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def estimate_text_tokens(text: str) -> int:
|
|
164
|
+
"""Very rough token estimate for cache gating without extra API calls."""
|
|
165
|
+
return max(1, len(text) // 4)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def build_page_prompt_suffix(page_width: int, page_height: int) -> str:
|
|
169
|
+
"""Return page-specific prompt instructions kept separate from the cached prefix."""
|
|
170
|
+
return (
|
|
171
|
+
f"Page image dimensions: {page_width}x{page_height} pixels.\n"
|
|
172
|
+
"The attached page image is the original full-page reference frame.\n"
|
|
173
|
+
"If you use code execution to zoom or crop, convert final boxes back to the original full page before "
|
|
174
|
+
"returning data-bbox.\n"
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def identify_part_kind(part: Any) -> str:
|
|
179
|
+
"""Return a stable kind string for a Gemini content part."""
|
|
180
|
+
if getattr(part, "executable_code", None) is not None:
|
|
181
|
+
return "executable_code"
|
|
182
|
+
if getattr(part, "code_execution_result", None) is not None:
|
|
183
|
+
return "code_execution_result"
|
|
184
|
+
if getattr(part, "inline_data", None) is not None:
|
|
185
|
+
return "inline_data"
|
|
186
|
+
if getattr(part, "file_data", None) is not None:
|
|
187
|
+
return "file_data"
|
|
188
|
+
if getattr(part, "function_call", None) is not None:
|
|
189
|
+
return "function_call"
|
|
190
|
+
if getattr(part, "function_response", None) is not None:
|
|
191
|
+
return "function_response"
|
|
192
|
+
if getattr(part, "text", None) is not None:
|
|
193
|
+
return "thought_text" if getattr(part, "thought", False) else "text"
|
|
194
|
+
return "unknown"
|
|
195
|
+
|
|
196
|
+
|
|
197
|
+
def summarize_part_for_request(part: Any) -> dict[str, Any]:
|
|
198
|
+
"""Serialize a request part without embedding large binary payloads."""
|
|
199
|
+
kind = identify_part_kind(part)
|
|
200
|
+
summary: dict[str, Any] = {"kind": kind}
|
|
201
|
+
text = getattr(part, "text", None)
|
|
202
|
+
if text is not None:
|
|
203
|
+
summary["text"] = text
|
|
204
|
+
inline_data = getattr(part, "inline_data", None)
|
|
205
|
+
if inline_data is not None:
|
|
206
|
+
summary["inline_data"] = {
|
|
207
|
+
"mime_type": getattr(inline_data, "mime_type", None),
|
|
208
|
+
"data_size_bytes": len(getattr(inline_data, "data", b"") or b""),
|
|
209
|
+
}
|
|
210
|
+
file_data = getattr(part, "file_data", None)
|
|
211
|
+
if file_data is not None:
|
|
212
|
+
summary["file_data"] = {
|
|
213
|
+
"mime_type": getattr(file_data, "mime_type", None),
|
|
214
|
+
"file_uri": getattr(file_data, "file_uri", None),
|
|
215
|
+
}
|
|
216
|
+
if getattr(part, "thought", False):
|
|
217
|
+
summary["thought"] = True
|
|
218
|
+
thought_signature = getattr(part, "thought_signature", None)
|
|
219
|
+
if thought_signature:
|
|
220
|
+
summary["thought_signature"] = normalize_signature(thought_signature)
|
|
221
|
+
return summary
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def serialize_part(part: Any, part_index: int) -> dict[str, Any]:
|
|
225
|
+
"""Serialize a Gemini content part into a JSON-safe dict."""
|
|
226
|
+
serialized = safe_model_dump(part)
|
|
227
|
+
if serialized is None:
|
|
228
|
+
serialized = {}
|
|
229
|
+
if not isinstance(serialized, dict):
|
|
230
|
+
serialized = {"value": serialized}
|
|
231
|
+
serialized["kind"] = identify_part_kind(part)
|
|
232
|
+
serialized["part_index"] = part_index
|
|
233
|
+
thought_signature = getattr(part, "thought_signature", None)
|
|
234
|
+
if thought_signature:
|
|
235
|
+
serialized["thought_signature"] = normalize_signature(thought_signature)
|
|
236
|
+
return serialized
|
|
237
|
+
|
|
238
|
+
|
|
239
|
+
def safe_model_dump(value: Any) -> Any:
|
|
240
|
+
"""Best-effort JSON-safe serializer for SDK objects."""
|
|
241
|
+
if value is None:
|
|
242
|
+
return None
|
|
243
|
+
if hasattr(value, "model_dump"):
|
|
244
|
+
try:
|
|
245
|
+
return value.model_dump(mode="json", exclude_none=True)
|
|
246
|
+
except TypeError:
|
|
247
|
+
return value.model_dump(exclude_none=True)
|
|
248
|
+
if isinstance(value, dict):
|
|
249
|
+
return {k: safe_model_dump(v) for k, v in value.items() if v is not None}
|
|
250
|
+
if isinstance(value, list):
|
|
251
|
+
return [safe_model_dump(v) for v in value]
|
|
252
|
+
if isinstance(value, tuple):
|
|
253
|
+
return [safe_model_dump(v) for v in value]
|
|
254
|
+
if isinstance(value, bytes):
|
|
255
|
+
return value.hex()
|
|
256
|
+
return value
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def normalize_signature(signature: Any) -> str:
|
|
260
|
+
"""Return a stable string representation for a thought signature payload."""
|
|
261
|
+
if isinstance(signature, bytes):
|
|
262
|
+
return signature.hex()
|
|
263
|
+
return str(signature)
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def extract_candidate_parts(response: Any) -> list[Any]:
|
|
267
|
+
"""Extract parts from the first candidate content."""
|
|
268
|
+
candidates = getattr(response, "candidates", None) or []
|
|
269
|
+
if not candidates:
|
|
270
|
+
return []
|
|
271
|
+
content = getattr(candidates[0], "content", None)
|
|
272
|
+
parts = getattr(content, "parts", None)
|
|
273
|
+
return list(parts or [])
|
|
274
|
+
|
|
275
|
+
|
|
276
|
+
def extract_finish_reason(response: Any) -> str | None:
|
|
277
|
+
"""Extract the first candidate finish reason, if any."""
|
|
278
|
+
candidates = getattr(response, "candidates", None) or []
|
|
279
|
+
if not candidates:
|
|
280
|
+
return None
|
|
281
|
+
finish_reason = getattr(candidates[0], "finish_reason", None)
|
|
282
|
+
return str(finish_reason) if finish_reason else None
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def is_recitation_finish_reason(finish_reason: str | None) -> bool:
|
|
286
|
+
"""Return whether a finish reason represents Gemini's RECITATION stop."""
|
|
287
|
+
return bool(finish_reason and "RECITATION" in finish_reason.upper())
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def response_has_citations(response: Any) -> bool:
|
|
291
|
+
"""Return whether the first candidate carries citation metadata."""
|
|
292
|
+
candidates = getattr(response, "candidates", None) or []
|
|
293
|
+
if not candidates:
|
|
294
|
+
return False
|
|
295
|
+
citation_metadata = getattr(candidates[0], "citation_metadata", None)
|
|
296
|
+
citations = getattr(citation_metadata, "citations", None)
|
|
297
|
+
return bool(citations)
|
|
298
|
+
|
|
299
|
+
|
|
300
|
+
def build_retry_instruction(response: Any, last_error: str, *, attempt: int) -> str | None:
|
|
301
|
+
"""Return an adaptive retry instruction for recitation and empty-output failures."""
|
|
302
|
+
finish_reason = extract_finish_reason(response)
|
|
303
|
+
parts = extract_candidate_parts(response)
|
|
304
|
+
has_code = any(getattr(part, "executable_code", None) is not None for part in parts)
|
|
305
|
+
has_code_output = any(getattr(getattr(part, "code_execution_result", None), "output", None) for part in parts)
|
|
306
|
+
|
|
307
|
+
instructions: list[str] = []
|
|
308
|
+
if is_recitation_finish_reason(finish_reason) or response_has_citations(response):
|
|
309
|
+
instructions.append(RETRY_PROMPT_RECITATION)
|
|
310
|
+
if "No wrapped layout payload found" in last_error or "No valid wrapped layout payload found" in last_error:
|
|
311
|
+
instructions.append(RETRY_PROMPT_EMPTY_OUTPUT)
|
|
312
|
+
if has_code and not has_code_output:
|
|
313
|
+
instructions.append(
|
|
314
|
+
"The previous attempt produced executable code but no printed final answer. "
|
|
315
|
+
"This retry must execute code that prints the final wrapped markdown."
|
|
316
|
+
)
|
|
317
|
+
if attempt >= 2 and has_code and not has_code_output:
|
|
318
|
+
instructions.append(RETRY_PROMPT_FINAL_ONLY)
|
|
319
|
+
|
|
320
|
+
if not instructions:
|
|
321
|
+
return None
|
|
322
|
+
return "\n".join(instructions)
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def extract_serialized_response_parts(response: Any) -> list[dict[str, Any]]:
|
|
326
|
+
"""Serialize all first-candidate parts."""
|
|
327
|
+
return [serialize_part(part, idx) for idx, part in enumerate(extract_candidate_parts(response))]
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def extract_thought_summaries(response: Any) -> list[str]:
|
|
331
|
+
"""Extract exposed thought summary text parts."""
|
|
332
|
+
summaries: list[str] = []
|
|
333
|
+
for part in extract_candidate_parts(response):
|
|
334
|
+
if getattr(part, "thought", False) and getattr(part, "text", None):
|
|
335
|
+
summaries.append(str(part.text))
|
|
336
|
+
return summaries
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def extract_thought_signatures(response: Any) -> list[str]:
|
|
340
|
+
"""Extract exposed thought signatures from all parts."""
|
|
341
|
+
signatures: list[str] = []
|
|
342
|
+
for part in extract_candidate_parts(response):
|
|
343
|
+
thought_signature = getattr(part, "thought_signature", None)
|
|
344
|
+
if thought_signature:
|
|
345
|
+
signatures.append(normalize_signature(thought_signature))
|
|
346
|
+
return signatures
|
|
347
|
+
|
|
348
|
+
|
|
349
|
+
def extract_generated_code(response: Any) -> list[dict[str, Any]]:
|
|
350
|
+
"""Extract generated code parts."""
|
|
351
|
+
code_parts: list[dict[str, Any]] = []
|
|
352
|
+
for idx, part in enumerate(extract_candidate_parts(response)):
|
|
353
|
+
executable_code = getattr(part, "executable_code", None)
|
|
354
|
+
if executable_code is not None:
|
|
355
|
+
payload = safe_model_dump(executable_code)
|
|
356
|
+
if isinstance(payload, dict):
|
|
357
|
+
payload["part_index"] = idx
|
|
358
|
+
code_parts.append(payload)
|
|
359
|
+
return code_parts
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def extract_code_execution_results(response: Any) -> list[dict[str, Any]]:
|
|
363
|
+
"""Extract code execution result parts."""
|
|
364
|
+
results: list[dict[str, Any]] = []
|
|
365
|
+
for idx, part in enumerate(extract_candidate_parts(response)):
|
|
366
|
+
execution_result = getattr(part, "code_execution_result", None)
|
|
367
|
+
if execution_result is not None:
|
|
368
|
+
payload = safe_model_dump(execution_result)
|
|
369
|
+
if isinstance(payload, dict):
|
|
370
|
+
payload["part_index"] = idx
|
|
371
|
+
results.append(payload)
|
|
372
|
+
return results
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def extract_final_text(response: Any) -> str:
|
|
376
|
+
"""Extract the last non-thought text part from a response."""
|
|
377
|
+
final_text = ""
|
|
378
|
+
for part in extract_candidate_parts(response):
|
|
379
|
+
text = getattr(part, "text", None)
|
|
380
|
+
if text and not getattr(part, "thought", False):
|
|
381
|
+
final_text = str(text)
|
|
382
|
+
return final_text
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def _candidate_layout_payloads(response: Any) -> list[str]:
|
|
386
|
+
"""Return possible final wrapped-markdown payloads from text and code-execution outputs."""
|
|
387
|
+
payloads: list[str] = []
|
|
388
|
+
final_text = extract_final_text(response)
|
|
389
|
+
if final_text:
|
|
390
|
+
payloads.append(final_text)
|
|
391
|
+
|
|
392
|
+
for part in reversed(extract_candidate_parts(response)):
|
|
393
|
+
execution_result = getattr(part, "code_execution_result", None)
|
|
394
|
+
output = getattr(execution_result, "output", None) if execution_result is not None else None
|
|
395
|
+
if output:
|
|
396
|
+
payloads.append(str(output))
|
|
397
|
+
return payloads
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def _normalize_bbox_2d(value: object) -> list[int]:
|
|
401
|
+
"""Normalize a candidate bbox payload into Gemini-native integer coordinates."""
|
|
402
|
+
if not isinstance(value, list) or len(value) != 4:
|
|
403
|
+
raise ValueError("bbox must be a list of four coordinates")
|
|
404
|
+
coords = [int(round(float(v))) for v in value]
|
|
405
|
+
y_min, x_min, y_max, x_max = coords
|
|
406
|
+
if min(coords) < 0 or max(coords) > 1000:
|
|
407
|
+
raise ValueError("bbox coordinates must be in the 0..1000 range")
|
|
408
|
+
if y_max < y_min or x_max < x_min:
|
|
409
|
+
raise ValueError("bbox must be ordered as [y_min, x_min, y_max, x_max]")
|
|
410
|
+
return coords
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def parse_agentic_layout_blocks(content: str) -> AgenticVisionPageResponse:
|
|
414
|
+
"""Parse wrapped layout blocks using Gemini-native y-first bbox ordering."""
|
|
415
|
+
raw_matches: list[tuple[int, list[int], str, str, str]] = []
|
|
416
|
+
|
|
417
|
+
for match in _PATTERN_BBOX_FIRST.finditer(content):
|
|
418
|
+
try:
|
|
419
|
+
bbox = _normalize_bbox_2d(json.loads(match.group(1)))
|
|
420
|
+
except Exception:
|
|
421
|
+
continue
|
|
422
|
+
raw_matches.append((match.start(), bbox, match.group(2), match.group(3).strip(), match.group(0).strip()))
|
|
423
|
+
|
|
424
|
+
for match in _PATTERN_LABEL_FIRST.finditer(content):
|
|
425
|
+
try:
|
|
426
|
+
bbox = _normalize_bbox_2d(json.loads(match.group(2)))
|
|
427
|
+
except Exception:
|
|
428
|
+
continue
|
|
429
|
+
raw_matches.append((match.start(), bbox, match.group(1), match.group(3).strip(), match.group(0).strip()))
|
|
430
|
+
|
|
431
|
+
raw_matches.sort(key=lambda item: item[0])
|
|
432
|
+
|
|
433
|
+
items: list[dict[str, Any]] = []
|
|
434
|
+
wrapper_blocks: list[str] = []
|
|
435
|
+
seen_positions: set[int] = set()
|
|
436
|
+
for pos, bbox, label, text, full_block in raw_matches:
|
|
437
|
+
if pos in seen_positions:
|
|
438
|
+
continue
|
|
439
|
+
seen_positions.add(pos)
|
|
440
|
+
items.append(
|
|
441
|
+
{
|
|
442
|
+
"bbox_2d": bbox,
|
|
443
|
+
"label": normalize_label(label),
|
|
444
|
+
"text": text,
|
|
445
|
+
}
|
|
446
|
+
)
|
|
447
|
+
wrapper_blocks.append(full_block)
|
|
448
|
+
|
|
449
|
+
return AgenticVisionPageResponse(raw_content="\n\n".join(wrapper_blocks), items=items)
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def parse_page_response(response: Any) -> AgenticVisionPageResponse:
|
|
453
|
+
"""Parse the final wrapped layout response from a Gemini response."""
|
|
454
|
+
errors: list[str] = []
|
|
455
|
+
for payload in _candidate_layout_payloads(response):
|
|
456
|
+
parsed = parse_agentic_layout_blocks(payload)
|
|
457
|
+
if parsed.items:
|
|
458
|
+
return parsed
|
|
459
|
+
errors.append("No wrapped layout blocks found")
|
|
460
|
+
|
|
461
|
+
if errors:
|
|
462
|
+
raise ValueError(f"No valid wrapped layout payload found in Gemini response: {errors[-1]}")
|
|
463
|
+
raise ValueError("No wrapped layout payload found in Gemini response")
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def normalize_label(label: str) -> str:
|
|
467
|
+
"""Canonicalize a raw label string into the benchmark label set."""
|
|
468
|
+
return LABEL_MAP.get(label.lower(), label)
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
def infer_item_type(label: str) -> str:
|
|
472
|
+
"""Infer normalized item type from a Core11 label."""
|
|
473
|
+
norm_label = label.lower()
|
|
474
|
+
if norm_label == "table":
|
|
475
|
+
return "table"
|
|
476
|
+
if norm_label in ("picture", "figure"):
|
|
477
|
+
return "image"
|
|
478
|
+
return "text"
|
|
479
|
+
|
|
480
|
+
|
|
481
|
+
def bbox_2d_to_xyxy(bbox_2d: list[int]) -> list[int]:
|
|
482
|
+
"""Convert Gemini-native [y_min, x_min, y_max, x_max] to x-first [x1, y1, x2, y2]."""
|
|
483
|
+
y_min, x_min, y_max, x_max = bbox_2d
|
|
484
|
+
return [x_min, y_min, x_max, y_max]
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def build_layout_pages_from_agentic_items(
|
|
488
|
+
items_data: list[dict[str, Any]],
|
|
489
|
+
image_width: int,
|
|
490
|
+
image_height: int,
|
|
491
|
+
page_number: int,
|
|
492
|
+
) -> tuple[str, list[ParseLayoutPageIR]]:
|
|
493
|
+
"""Convert wrapped Agentic Vision items to page markdown and ParseLayoutPageIR."""
|
|
494
|
+
if not items_data or not image_width or not image_height:
|
|
495
|
+
return "", []
|
|
496
|
+
|
|
497
|
+
markdown = items_to_markdown(items_data)
|
|
498
|
+
layout_items: list[LayoutItemIR] = []
|
|
499
|
+
for item in items_data:
|
|
500
|
+
try:
|
|
501
|
+
bbox_2d = _normalize_bbox_2d(item.get("bbox_2d", []))
|
|
502
|
+
except (TypeError, ValueError):
|
|
503
|
+
continue
|
|
504
|
+
x1, y1, x2, y2 = bbox_2d_to_xyxy(bbox_2d)
|
|
505
|
+
text = str(item.get("text", ""))
|
|
506
|
+
label = normalize_label(str(item.get("label", "Text")))
|
|
507
|
+
item_type = infer_item_type(label)
|
|
508
|
+
seg = LayoutSegmentIR(
|
|
509
|
+
x=x1 / 1000.0,
|
|
510
|
+
y=y1 / 1000.0,
|
|
511
|
+
w=(x2 - x1) / 1000.0,
|
|
512
|
+
h=(y2 - y1) / 1000.0,
|
|
513
|
+
confidence=1.0,
|
|
514
|
+
label=label,
|
|
515
|
+
)
|
|
516
|
+
layout_items.append(
|
|
517
|
+
LayoutItemIR(
|
|
518
|
+
type=item_type,
|
|
519
|
+
md=text,
|
|
520
|
+
html=text if item_type == "table" else "",
|
|
521
|
+
value=text,
|
|
522
|
+
bbox=seg,
|
|
523
|
+
layout_segments=[seg],
|
|
524
|
+
)
|
|
525
|
+
)
|
|
526
|
+
|
|
527
|
+
return markdown, [
|
|
528
|
+
ParseLayoutPageIR(
|
|
529
|
+
page_number=page_number,
|
|
530
|
+
width=float(image_width),
|
|
531
|
+
height=float(image_height),
|
|
532
|
+
md=markdown,
|
|
533
|
+
items=layout_items,
|
|
534
|
+
)
|
|
535
|
+
]
|
|
536
|
+
|
|
537
|
+
|
|
538
|
+
class GoogleAgenticVisionRunner:
|
|
539
|
+
"""One-call-per-page Agentic Vision runner with optional explicit prefix caching."""
|
|
540
|
+
|
|
541
|
+
def __init__(
|
|
542
|
+
self,
|
|
543
|
+
*,
|
|
544
|
+
client: Any,
|
|
545
|
+
types_module: Any,
|
|
546
|
+
model: str,
|
|
547
|
+
max_output_tokens: int,
|
|
548
|
+
thinking_level: str | None,
|
|
549
|
+
enable_explicit_context_cache: bool,
|
|
550
|
+
context_cache_ttl_seconds: int,
|
|
551
|
+
min_cacheable_tokens: int,
|
|
552
|
+
input_cost_per_million: float,
|
|
553
|
+
cache_hit_cost_per_million: float,
|
|
554
|
+
cache_storage_cost_per_million_token_hour: float,
|
|
555
|
+
expected_page_calls: int,
|
|
556
|
+
) -> None:
|
|
557
|
+
self._client = client
|
|
558
|
+
self._types = types_module
|
|
559
|
+
self._model = model
|
|
560
|
+
self._max_output_tokens = max_output_tokens
|
|
561
|
+
self._thinking_level = thinking_level
|
|
562
|
+
self._enable_explicit_context_cache = enable_explicit_context_cache
|
|
563
|
+
self._context_cache_ttl_seconds = context_cache_ttl_seconds
|
|
564
|
+
self._min_cacheable_tokens = min_cacheable_tokens
|
|
565
|
+
self._input_cost_per_million = input_cost_per_million
|
|
566
|
+
self._cache_hit_cost_per_million = cache_hit_cost_per_million
|
|
567
|
+
self._cache_storage_cost_per_million_token_hour = cache_storage_cost_per_million_token_hour
|
|
568
|
+
self._expected_page_calls = expected_page_calls
|
|
569
|
+
self._cache_info: AgenticVisionCacheInfo | None = None
|
|
570
|
+
self._cache_error: str | None = None
|
|
571
|
+
|
|
572
|
+
@property
|
|
573
|
+
def cache_info(self) -> AgenticVisionCacheInfo | None:
|
|
574
|
+
return self._cache_info
|
|
575
|
+
|
|
576
|
+
@property
|
|
577
|
+
def cache_error(self) -> str | None:
|
|
578
|
+
return self._cache_error
|
|
579
|
+
|
|
580
|
+
def _maybe_create_prefix_cache(self) -> AgenticVisionCacheInfo | None:
|
|
581
|
+
if self._cache_info is not None:
|
|
582
|
+
return self._cache_info
|
|
583
|
+
if not self._enable_explicit_context_cache:
|
|
584
|
+
return None
|
|
585
|
+
if self._expected_page_calls < 2:
|
|
586
|
+
return None
|
|
587
|
+
|
|
588
|
+
estimated_tokens = estimate_text_tokens(SYSTEM_PROMPT_AGENTIC_VISION + USER_PROMPT_AGENTIC_VISION_PREFIX)
|
|
589
|
+
if estimated_tokens < self._min_cacheable_tokens:
|
|
590
|
+
logger.info(
|
|
591
|
+
"Agentic Vision prompt estimate (%s) is below min_cacheable_tokens (%s); "
|
|
592
|
+
"attempting cache creation anyway because Gemini cache tokenization can exceed the heuristic.",
|
|
593
|
+
estimated_tokens,
|
|
594
|
+
self._min_cacheable_tokens,
|
|
595
|
+
)
|
|
596
|
+
|
|
597
|
+
display_name = (
|
|
598
|
+
"llamacloud-bench-gemini-agentic-vision-prefix-"
|
|
599
|
+
+ hashlib.sha256(
|
|
600
|
+
f"{self._model}|{SYSTEM_PROMPT_AGENTIC_VISION}|{USER_PROMPT_AGENTIC_VISION_PREFIX}".encode()
|
|
601
|
+
).hexdigest()[:16]
|
|
602
|
+
)
|
|
603
|
+
|
|
604
|
+
try:
|
|
605
|
+
cache = self._client.caches.create(
|
|
606
|
+
model=self._model,
|
|
607
|
+
config=self._types.CreateCachedContentConfig(
|
|
608
|
+
display_name=display_name,
|
|
609
|
+
system_instruction=SYSTEM_PROMPT_AGENTIC_VISION,
|
|
610
|
+
contents=[
|
|
611
|
+
self._types.Content(
|
|
612
|
+
role="user",
|
|
613
|
+
parts=[self._types.Part.from_text(text=USER_PROMPT_AGENTIC_VISION_PREFIX)],
|
|
614
|
+
)
|
|
615
|
+
],
|
|
616
|
+
tools=[self._types.Tool(code_execution=self._types.ToolCodeExecution())],
|
|
617
|
+
ttl=timedelta(seconds=self._context_cache_ttl_seconds),
|
|
618
|
+
),
|
|
619
|
+
)
|
|
620
|
+
except Exception as exc:
|
|
621
|
+
logger.warning("Failed to create Gemini context cache for Agentic Vision: %s", exc)
|
|
622
|
+
self._cache_error = str(exc)
|
|
623
|
+
return None
|
|
624
|
+
|
|
625
|
+
token_count = int(getattr(getattr(cache, "usage_metadata", None), "total_token_count", 0) or 0)
|
|
626
|
+
ttl_hours = self._context_cache_ttl_seconds / 3600.0
|
|
627
|
+
storage_cost_usd = (
|
|
628
|
+
token_count * self._cache_storage_cost_per_million_token_hour * ttl_hours / 1_000_000
|
|
629
|
+
if token_count > 0
|
|
630
|
+
else 0.0
|
|
631
|
+
)
|
|
632
|
+
self._cache_info = AgenticVisionCacheInfo(
|
|
633
|
+
name=str(getattr(cache, "name", "")),
|
|
634
|
+
display_name=display_name,
|
|
635
|
+
token_count=token_count,
|
|
636
|
+
ttl_seconds=self._context_cache_ttl_seconds,
|
|
637
|
+
storage_cost_usd=storage_cost_usd,
|
|
638
|
+
created=True,
|
|
639
|
+
)
|
|
640
|
+
return self._cache_info
|
|
641
|
+
|
|
642
|
+
def _build_generation_config(self, cache_name: str | None) -> Any:
|
|
643
|
+
config = self._types.GenerateContentConfig(
|
|
644
|
+
temperature=0,
|
|
645
|
+
max_output_tokens=self._max_output_tokens,
|
|
646
|
+
tools=[self._types.Tool(code_execution=self._types.ToolCodeExecution())],
|
|
647
|
+
)
|
|
648
|
+
if cache_name:
|
|
649
|
+
config.cached_content = cache_name
|
|
650
|
+
else:
|
|
651
|
+
config.system_instruction = SYSTEM_PROMPT_AGENTIC_VISION
|
|
652
|
+
if self._thinking_level is not None:
|
|
653
|
+
config.thinking_config = self._types.ThinkingConfig(
|
|
654
|
+
thinking_level=self._thinking_level,
|
|
655
|
+
include_thoughts=True,
|
|
656
|
+
)
|
|
657
|
+
return config
|
|
658
|
+
|
|
659
|
+
def _build_contents(
|
|
660
|
+
self,
|
|
661
|
+
*,
|
|
662
|
+
image_bytes: bytes,
|
|
663
|
+
image_mime_type: str,
|
|
664
|
+
page_width: int,
|
|
665
|
+
page_height: int,
|
|
666
|
+
use_cached_prefix: bool,
|
|
667
|
+
retry_instruction: str | None = None,
|
|
668
|
+
) -> list[Any]:
|
|
669
|
+
parts = []
|
|
670
|
+
if not use_cached_prefix:
|
|
671
|
+
parts.append(self._types.Part.from_text(text=USER_PROMPT_AGENTIC_VISION_PREFIX))
|
|
672
|
+
parts.append(self._types.Part.from_text(text=build_page_prompt_suffix(page_width, page_height)))
|
|
673
|
+
if retry_instruction:
|
|
674
|
+
parts.append(self._types.Part.from_text(text=retry_instruction))
|
|
675
|
+
parts.append(self._types.Part.from_bytes(data=image_bytes, mime_type=image_mime_type))
|
|
676
|
+
return [self._types.Content(role="user", parts=parts)]
|
|
677
|
+
|
|
678
|
+
def parse_page(
|
|
679
|
+
self,
|
|
680
|
+
*,
|
|
681
|
+
page_index: int,
|
|
682
|
+
image: Image.Image,
|
|
683
|
+
image_bytes: bytes,
|
|
684
|
+
image_mime_type: str,
|
|
685
|
+
max_attempts: int = 3,
|
|
686
|
+
) -> AgenticVisionPageResult:
|
|
687
|
+
"""Run one Agentic Vision page parse with retry on malformed final wrapped output."""
|
|
688
|
+
cache_info = self._maybe_create_prefix_cache()
|
|
689
|
+
use_cached_prefix = cache_info is not None
|
|
690
|
+
cache_name = cache_info.name if cache_info is not None else None
|
|
691
|
+
|
|
692
|
+
api_calls: list[dict[str, Any]] = []
|
|
693
|
+
last_error = "No attempts executed"
|
|
694
|
+
retry_instruction: str | None = None
|
|
695
|
+
width, height = image.size
|
|
696
|
+
|
|
697
|
+
for attempt in range(1, max_attempts + 1):
|
|
698
|
+
contents = self._build_contents(
|
|
699
|
+
image_bytes=image_bytes,
|
|
700
|
+
image_mime_type=image_mime_type,
|
|
701
|
+
page_width=width,
|
|
702
|
+
page_height=height,
|
|
703
|
+
use_cached_prefix=use_cached_prefix,
|
|
704
|
+
retry_instruction=retry_instruction,
|
|
705
|
+
)
|
|
706
|
+
request_summary = {
|
|
707
|
+
"system_instruction": SYSTEM_PROMPT_AGENTIC_VISION if not use_cached_prefix else None,
|
|
708
|
+
"user_prompt_prefix": None if use_cached_prefix else USER_PROMPT_AGENTIC_VISION_PREFIX,
|
|
709
|
+
"page_prompt_suffix": build_page_prompt_suffix(width, height),
|
|
710
|
+
"retry_instruction": retry_instruction,
|
|
711
|
+
"used_cached_content": bool(cache_name),
|
|
712
|
+
"cache_name": cache_name,
|
|
713
|
+
"contents": [
|
|
714
|
+
{
|
|
715
|
+
"role": getattr(content, "role", None),
|
|
716
|
+
"parts": [summarize_part_for_request(part) for part in getattr(content, "parts", []) or []],
|
|
717
|
+
}
|
|
718
|
+
for content in contents
|
|
719
|
+
],
|
|
720
|
+
}
|
|
721
|
+
|
|
722
|
+
try:
|
|
723
|
+
response = self._client.models.generate_content(
|
|
724
|
+
model=self._model,
|
|
725
|
+
contents=contents,
|
|
726
|
+
config=self._build_generation_config(cache_name),
|
|
727
|
+
)
|
|
728
|
+
except Exception as exc:
|
|
729
|
+
raise classify_gemini_api_exception(exc) from exc
|
|
730
|
+
|
|
731
|
+
usage = extract_usage_from_response(response)
|
|
732
|
+
response_parts = extract_serialized_response_parts(response)
|
|
733
|
+
thought_summaries = extract_thought_summaries(response)
|
|
734
|
+
thought_signatures = extract_thought_signatures(response)
|
|
735
|
+
generated_code = extract_generated_code(response)
|
|
736
|
+
execution_results = extract_code_execution_results(response)
|
|
737
|
+
final_text = extract_final_text(response)
|
|
738
|
+
|
|
739
|
+
try:
|
|
740
|
+
parsed = parse_page_response(response)
|
|
741
|
+
except Exception as exc:
|
|
742
|
+
last_error = str(exc)
|
|
743
|
+
parsed = None
|
|
744
|
+
|
|
745
|
+
call_record = {
|
|
746
|
+
"page_index": page_index,
|
|
747
|
+
"attempt": attempt,
|
|
748
|
+
"request": request_summary,
|
|
749
|
+
"response": safe_model_dump(response),
|
|
750
|
+
"response_parts": response_parts,
|
|
751
|
+
"usage": usage,
|
|
752
|
+
"final_text": final_text,
|
|
753
|
+
"cost_usd": 0.0,
|
|
754
|
+
}
|
|
755
|
+
api_calls.append(call_record)
|
|
756
|
+
|
|
757
|
+
if parsed is not None:
|
|
758
|
+
markdown = items_to_markdown(parsed.items)
|
|
759
|
+
return AgenticVisionPageResult(
|
|
760
|
+
page_index=page_index,
|
|
761
|
+
width=width,
|
|
762
|
+
height=height,
|
|
763
|
+
image_mime_type=image_mime_type,
|
|
764
|
+
items=parsed.items,
|
|
765
|
+
markdown=markdown,
|
|
766
|
+
raw_content=parsed.raw_content,
|
|
767
|
+
thought_summaries=thought_summaries,
|
|
768
|
+
thought_signatures=thought_signatures,
|
|
769
|
+
generated_code=generated_code,
|
|
770
|
+
code_execution_results=execution_results,
|
|
771
|
+
api_calls=api_calls,
|
|
772
|
+
)
|
|
773
|
+
|
|
774
|
+
retry_instruction = build_retry_instruction(response, last_error, attempt=attempt)
|
|
775
|
+
|
|
776
|
+
raise ProviderPermanentError(
|
|
777
|
+
f"Failed to obtain valid Agentic Vision wrapped layout output after {max_attempts} attempts: {last_error}",
|
|
778
|
+
debug_payload={
|
|
779
|
+
"mode": "parse_with_layout_agentic_vision",
|
|
780
|
+
"page_index": page_index,
|
|
781
|
+
"page_width": width,
|
|
782
|
+
"page_height": height,
|
|
783
|
+
"image_mime_type": image_mime_type,
|
|
784
|
+
"api_calls": api_calls,
|
|
785
|
+
"last_error": last_error,
|
|
786
|
+
},
|
|
787
|
+
)
|
|
788
|
+
|
|
789
|
+
|
|
790
|
+
def extract_usage_from_response(response: Any) -> dict[str, int]:
|
|
791
|
+
"""Extract all usage buckets relevant to Agentic Vision accounting."""
|
|
792
|
+
meta = getattr(response, "usage_metadata", None)
|
|
793
|
+
if meta is None:
|
|
794
|
+
return {
|
|
795
|
+
"input_tokens": 0,
|
|
796
|
+
"tool_use_prompt_tokens": 0,
|
|
797
|
+
"cached_content_tokens": 0,
|
|
798
|
+
"output_tokens": 0,
|
|
799
|
+
"thinking_tokens": 0,
|
|
800
|
+
"total_tokens": 0,
|
|
801
|
+
}
|
|
802
|
+
return {
|
|
803
|
+
"input_tokens": int(getattr(meta, "prompt_token_count", 0) or 0),
|
|
804
|
+
"tool_use_prompt_tokens": int(getattr(meta, "tool_use_prompt_token_count", 0) or 0),
|
|
805
|
+
"cached_content_tokens": int(getattr(meta, "cached_content_token_count", 0) or 0),
|
|
806
|
+
"output_tokens": int(getattr(meta, "candidates_token_count", 0) or 0),
|
|
807
|
+
"thinking_tokens": int(getattr(meta, "thoughts_token_count", 0) or 0),
|
|
808
|
+
"total_tokens": int(getattr(meta, "total_token_count", 0) or 0),
|
|
809
|
+
}
|
|
810
|
+
|
|
811
|
+
|
|
812
|
+
def classify_gemini_api_exception(exc: Exception) -> Exception:
|
|
813
|
+
"""Classify raw SDK exceptions into retryable provider errors when possible."""
|
|
814
|
+
error_str = str(exc).lower()
|
|
815
|
+
if any(keyword in error_str for keyword in TRANSIENT_ERROR_KEYWORDS):
|
|
816
|
+
return ProviderTransientError(f"Transient error calling Gemini API: {exc}")
|
|
817
|
+
if any(keyword in error_str for keyword in RATE_LIMIT_ERROR_KEYWORDS):
|
|
818
|
+
return ProviderTransientError(f"Rate limited: {exc}")
|
|
819
|
+
return ProviderPermanentError(f"Error calling Gemini API: {exc}")
|