langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,393 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import asdict, dataclass
|
|
6
|
+
|
|
7
|
+
from openpyxl.utils import get_column_letter, range_boundaries
|
|
8
|
+
|
|
9
|
+
from langparse.workbooks.modeling import RegionChoice
|
|
10
|
+
from langparse.workbooks.modeling.types import REGION_RULE_VERSION
|
|
11
|
+
from langparse.workbooks.types import CandidateRegion, SheetSnapshot, stable_id
|
|
12
|
+
|
|
13
|
+
PAGE_RE = re.compile(r"第\s*\d+\s*页\s*共\s*\d+\s*页")
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
@dataclass(frozen=True)
|
|
17
|
+
class RegionFeatures:
|
|
18
|
+
row_count: int
|
|
19
|
+
column_count: int
|
|
20
|
+
occupied_count: int
|
|
21
|
+
density: float
|
|
22
|
+
text_ratio: float
|
|
23
|
+
numeric_ratio: float
|
|
24
|
+
formula_ratio: float
|
|
25
|
+
nonempty_by_row: tuple[int, ...]
|
|
26
|
+
nonempty_by_column: tuple[int, ...]
|
|
27
|
+
positive_ordinal_rows: int
|
|
28
|
+
label_value_pairs: int
|
|
29
|
+
label_value_coverage: float
|
|
30
|
+
numeric_grid_rows: int
|
|
31
|
+
numeric_grid_columns: int
|
|
32
|
+
merged_title_rows: int
|
|
33
|
+
long_text_rows: int
|
|
34
|
+
has_page_sequence: bool
|
|
35
|
+
has_stable_table_schema: bool
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@dataclass(frozen=True)
|
|
39
|
+
class BlockClassification:
|
|
40
|
+
kind: str
|
|
41
|
+
confidence: float
|
|
42
|
+
reason_codes: list[str]
|
|
43
|
+
features: RegionFeatures
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class RegionAssessment:
|
|
48
|
+
deterministic: BlockClassification
|
|
49
|
+
choices: tuple[RegionChoice, ...]
|
|
50
|
+
ambiguous: bool
|
|
51
|
+
ambiguity_codes: tuple[str, ...]
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def extract_region_features(
|
|
55
|
+
sheet: SheetSnapshot,
|
|
56
|
+
candidate: CandidateRegion,
|
|
57
|
+
) -> RegionFeatures:
|
|
58
|
+
"""Compute deterministic, serializable signals for one candidate region."""
|
|
59
|
+
|
|
60
|
+
values, cells = _region_grid(sheet, candidate)
|
|
61
|
+
row_count = len(values)
|
|
62
|
+
column_count = len(values[0]) if values else 0
|
|
63
|
+
flat_values = [value for row in values for value in row if value != ""]
|
|
64
|
+
semantic_count = len(flat_values)
|
|
65
|
+
numeric_count = sum(_is_number(value) for value in flat_values)
|
|
66
|
+
text_count = semantic_count - numeric_count
|
|
67
|
+
formula_count = sum(
|
|
68
|
+
cell is not None and cell.merge_anchor is None and cell.formula is not None
|
|
69
|
+
for row in cells
|
|
70
|
+
for cell in row
|
|
71
|
+
)
|
|
72
|
+
nonempty_by_row = tuple(sum(value != "" for value in row) for row in values)
|
|
73
|
+
nonempty_by_column = tuple(
|
|
74
|
+
sum(values[row][column] != "" for row in range(row_count)) for column in range(column_count)
|
|
75
|
+
)
|
|
76
|
+
positive_ordinal_rows = sum(_is_row_ordinal(row[0]) for row in values[1:] if row)
|
|
77
|
+
label_value_pairs = sum(_label_value_pairs(row) for row in values)
|
|
78
|
+
numeric_grid_rows, numeric_grid_columns = _numeric_grid_shape(values)
|
|
79
|
+
merged_title_rows = sum(
|
|
80
|
+
sum(value != "" for value in value_row) == 1
|
|
81
|
+
and any(cell is not None and cell.colspan > 1 for cell in cell_row)
|
|
82
|
+
for value_row, cell_row in zip(values, cells, strict=True)
|
|
83
|
+
)
|
|
84
|
+
long_text_rows = sum(
|
|
85
|
+
len(nonempty) == 1 and len(nonempty[0]) >= 20
|
|
86
|
+
for row in values
|
|
87
|
+
for nonempty in [[value for value in row if value]]
|
|
88
|
+
)
|
|
89
|
+
has_page_sequence = any(PAGE_RE.search(value) for value in flat_values)
|
|
90
|
+
has_stable_table_schema = _has_stable_table_schema(
|
|
91
|
+
values,
|
|
92
|
+
numeric_grid_rows=numeric_grid_rows,
|
|
93
|
+
numeric_grid_columns=numeric_grid_columns,
|
|
94
|
+
)
|
|
95
|
+
area = row_count * column_count
|
|
96
|
+
return RegionFeatures(
|
|
97
|
+
row_count=row_count,
|
|
98
|
+
column_count=column_count,
|
|
99
|
+
occupied_count=len(candidate.cell_refs),
|
|
100
|
+
density=len(candidate.cell_refs) / area if area else 0.0,
|
|
101
|
+
text_ratio=text_count / semantic_count if semantic_count else 0.0,
|
|
102
|
+
numeric_ratio=numeric_count / semantic_count if semantic_count else 0.0,
|
|
103
|
+
formula_ratio=formula_count / semantic_count if semantic_count else 0.0,
|
|
104
|
+
nonempty_by_row=nonempty_by_row,
|
|
105
|
+
nonempty_by_column=nonempty_by_column,
|
|
106
|
+
positive_ordinal_rows=positive_ordinal_rows,
|
|
107
|
+
label_value_pairs=label_value_pairs,
|
|
108
|
+
label_value_coverage=label_value_pairs / row_count if row_count else 0.0,
|
|
109
|
+
numeric_grid_rows=numeric_grid_rows,
|
|
110
|
+
numeric_grid_columns=numeric_grid_columns,
|
|
111
|
+
merged_title_rows=merged_title_rows,
|
|
112
|
+
long_text_rows=long_text_rows,
|
|
113
|
+
has_page_sequence=has_page_sequence,
|
|
114
|
+
has_stable_table_schema=has_stable_table_schema,
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def classify_candidate_region(
|
|
119
|
+
sheet: SheetSnapshot,
|
|
120
|
+
candidate: CandidateRegion,
|
|
121
|
+
features: RegionFeatures | None = None,
|
|
122
|
+
) -> BlockClassification:
|
|
123
|
+
"""Classify a region conservatively with mutually exclusive rules."""
|
|
124
|
+
|
|
125
|
+
features = features if features is not None else extract_region_features(sheet, candidate)
|
|
126
|
+
values, _ = _region_grid(sheet, candidate)
|
|
127
|
+
return _classify_region(values, features, candidate.reason_codes)
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def assess_candidate_region(
|
|
131
|
+
sheet: SheetSnapshot,
|
|
132
|
+
candidate: CandidateRegion,
|
|
133
|
+
) -> RegionAssessment:
|
|
134
|
+
"""Assess a region and register only locally compatible alternative kinds."""
|
|
135
|
+
|
|
136
|
+
features = extract_region_features(sheet, candidate)
|
|
137
|
+
feature_digest = _structural_feature_digest(features)
|
|
138
|
+
values, _ = _region_grid(sheet, candidate)
|
|
139
|
+
deterministic = _classify_region(values, features, candidate.reason_codes)
|
|
140
|
+
choices = [
|
|
141
|
+
RegionChoice(
|
|
142
|
+
choice_id=_choice_id(
|
|
143
|
+
candidate,
|
|
144
|
+
feature_digest,
|
|
145
|
+
deterministic.kind,
|
|
146
|
+
deterministic.reason_codes[0],
|
|
147
|
+
),
|
|
148
|
+
kind=deterministic.kind,
|
|
149
|
+
local_score=deterministic.confidence,
|
|
150
|
+
reason_codes=tuple(deterministic.reason_codes),
|
|
151
|
+
)
|
|
152
|
+
]
|
|
153
|
+
if deterministic.kind != "unclassified":
|
|
154
|
+
return RegionAssessment(
|
|
155
|
+
deterministic=deterministic,
|
|
156
|
+
choices=tuple(choices),
|
|
157
|
+
ambiguous=False,
|
|
158
|
+
ambiguity_codes=(),
|
|
159
|
+
)
|
|
160
|
+
|
|
161
|
+
seen_kinds = {deterministic.kind}
|
|
162
|
+
for kind, score, reason_code in _weak_choice_kinds(features):
|
|
163
|
+
if kind in seen_kinds:
|
|
164
|
+
continue
|
|
165
|
+
choices.append(
|
|
166
|
+
RegionChoice(
|
|
167
|
+
choice_id=_choice_id(candidate, feature_digest, kind, reason_code),
|
|
168
|
+
kind=kind,
|
|
169
|
+
local_score=score,
|
|
170
|
+
reason_codes=(reason_code,),
|
|
171
|
+
)
|
|
172
|
+
)
|
|
173
|
+
seen_kinds.add(kind)
|
|
174
|
+
|
|
175
|
+
ambiguous = len(choices) >= 2
|
|
176
|
+
return RegionAssessment(
|
|
177
|
+
deterministic=deterministic,
|
|
178
|
+
choices=tuple(choices),
|
|
179
|
+
ambiguous=ambiguous,
|
|
180
|
+
ambiguity_codes=("unclassified_with_compatible_choices",) if ambiguous else (),
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _classify_region(
|
|
185
|
+
values: list[list[str]],
|
|
186
|
+
features: RegionFeatures,
|
|
187
|
+
region_reason_codes: list[str] | None = None,
|
|
188
|
+
) -> BlockClassification:
|
|
189
|
+
"""Apply the existing deterministic winner rules to precomputed facts."""
|
|
190
|
+
|
|
191
|
+
if "native_table_anchor" in (region_reason_codes or []):
|
|
192
|
+
return BlockClassification(
|
|
193
|
+
"logical_table",
|
|
194
|
+
0.98,
|
|
195
|
+
["native_table_anchor"],
|
|
196
|
+
features,
|
|
197
|
+
)
|
|
198
|
+
text_reason = _text_reason(features)
|
|
199
|
+
if text_reason:
|
|
200
|
+
return BlockClassification("text", 0.9, [text_reason], features)
|
|
201
|
+
if _is_form(features):
|
|
202
|
+
return BlockClassification("form", 0.9, ["stable_label_value_pairs"], features)
|
|
203
|
+
if _is_matrix(values, features):
|
|
204
|
+
return BlockClassification("matrix", 0.95, ["numeric_matrix_with_axes"], features)
|
|
205
|
+
if _is_logical_table(features):
|
|
206
|
+
reason = (
|
|
207
|
+
"consistent_print_fragments"
|
|
208
|
+
if features.has_page_sequence
|
|
209
|
+
else "stable_header_data_schema"
|
|
210
|
+
)
|
|
211
|
+
return BlockClassification("logical_table", 0.9, [reason], features)
|
|
212
|
+
return BlockClassification(
|
|
213
|
+
"unclassified",
|
|
214
|
+
0.5,
|
|
215
|
+
["insufficient_semantic_evidence"],
|
|
216
|
+
features,
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _weak_choice_kinds(features: RegionFeatures) -> list[tuple[str, float, str]]:
|
|
221
|
+
choices = []
|
|
222
|
+
if (
|
|
223
|
+
features.row_count >= 2
|
|
224
|
+
and features.column_count >= 2
|
|
225
|
+
and max(features.nonempty_by_row, default=0) >= 2
|
|
226
|
+
):
|
|
227
|
+
choices.append(("logical_table", 0.4, "weak_row_column_structure"))
|
|
228
|
+
if features.column_count >= 2 and features.label_value_pairs >= 1:
|
|
229
|
+
choices.append(("form", 0.4, "weak_label_value_pairs"))
|
|
230
|
+
if features.numeric_grid_rows >= 1 and features.numeric_grid_columns >= 1:
|
|
231
|
+
choices.append(("matrix", 0.4, "weak_numeric_axes"))
|
|
232
|
+
if features.occupied_count >= 2 and features.text_ratio >= 0.6:
|
|
233
|
+
choices.append(("text", 0.4, "weak_text_region"))
|
|
234
|
+
return choices
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def _structural_feature_digest(features: RegionFeatures) -> str:
|
|
238
|
+
payload = json.dumps(
|
|
239
|
+
asdict(features),
|
|
240
|
+
ensure_ascii=True,
|
|
241
|
+
sort_keys=True,
|
|
242
|
+
separators=(",", ":"),
|
|
243
|
+
)
|
|
244
|
+
return stable_id("region_features", payload)
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def _choice_id(
|
|
248
|
+
candidate: CandidateRegion,
|
|
249
|
+
feature_digest: str,
|
|
250
|
+
kind: str,
|
|
251
|
+
reason_code: str,
|
|
252
|
+
) -> str:
|
|
253
|
+
return stable_id(
|
|
254
|
+
"region_choice",
|
|
255
|
+
REGION_RULE_VERSION,
|
|
256
|
+
candidate.source_ref.key,
|
|
257
|
+
feature_digest,
|
|
258
|
+
kind,
|
|
259
|
+
reason_code,
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def _region_grid(sheet: SheetSnapshot, candidate: CandidateRegion):
|
|
264
|
+
min_col, min_row, max_col, max_row = range_boundaries(candidate.source_ref.range)
|
|
265
|
+
values = []
|
|
266
|
+
cells = []
|
|
267
|
+
for row_number in range(min_row, max_row + 1):
|
|
268
|
+
value_row = []
|
|
269
|
+
cell_row = []
|
|
270
|
+
for column_number in range(min_col, max_col + 1):
|
|
271
|
+
coordinate = f"{get_column_letter(column_number)}{row_number}"
|
|
272
|
+
cell = sheet.cells.get(coordinate)
|
|
273
|
+
cell_row.append(cell)
|
|
274
|
+
value_row.append(
|
|
275
|
+
"" if cell is None or cell.merge_anchor is not None else cell.display_value.strip()
|
|
276
|
+
)
|
|
277
|
+
values.append(value_row)
|
|
278
|
+
cells.append(cell_row)
|
|
279
|
+
return values, cells
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
def _is_number(value: str) -> bool:
|
|
283
|
+
try:
|
|
284
|
+
float(value.replace(",", ""))
|
|
285
|
+
except (TypeError, ValueError):
|
|
286
|
+
return False
|
|
287
|
+
return True
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _is_row_ordinal(value: str) -> bool:
|
|
291
|
+
# Printed schedules use both flat and hierarchical item numbers.
|
|
292
|
+
return bool(re.fullmatch(r"[1-9]\d*(?:\.\d+)*", value.strip()))
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _label_value_pairs(row: list[str]) -> int:
|
|
296
|
+
nonempty = [(index, value) for index, value in enumerate(row) if value]
|
|
297
|
+
if len(nonempty) % 2:
|
|
298
|
+
return 0
|
|
299
|
+
pairs = 0
|
|
300
|
+
for offset in range(0, len(nonempty) - 1, 2):
|
|
301
|
+
(label_index, label), (value_index, _) = nonempty[offset : offset + 2]
|
|
302
|
+
if value_index == label_index + 1 and not _is_number(label):
|
|
303
|
+
pairs += 1
|
|
304
|
+
return pairs
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _numeric_grid_shape(values: list[list[str]]) -> tuple[int, int]:
|
|
308
|
+
if len(values) < 2 or len(values[0]) < 2:
|
|
309
|
+
return 0, 0
|
|
310
|
+
interior = [row[1:] for row in values[1:]]
|
|
311
|
+
numeric_rows = sum(
|
|
312
|
+
row and all(value and _is_number(value) for value in row) for row in interior
|
|
313
|
+
)
|
|
314
|
+
numeric_columns = sum(
|
|
315
|
+
all(
|
|
316
|
+
interior[row][column] and _is_number(interior[row][column])
|
|
317
|
+
for row in range(len(interior))
|
|
318
|
+
)
|
|
319
|
+
for column in range(len(interior[0]))
|
|
320
|
+
)
|
|
321
|
+
return numeric_rows, numeric_columns
|
|
322
|
+
|
|
323
|
+
|
|
324
|
+
def _has_stable_table_schema(
|
|
325
|
+
values: list[list[str]],
|
|
326
|
+
*,
|
|
327
|
+
numeric_grid_rows: int,
|
|
328
|
+
numeric_grid_columns: int,
|
|
329
|
+
) -> bool:
|
|
330
|
+
if len(values) < 2 or len(values[0]) < 2:
|
|
331
|
+
return False
|
|
332
|
+
if numeric_grid_rows >= 2 and numeric_grid_columns >= 2:
|
|
333
|
+
return False
|
|
334
|
+
header = values[0]
|
|
335
|
+
if not all(value and not _is_number(value) for value in header):
|
|
336
|
+
return False
|
|
337
|
+
widths = [sum(value != "" for value in row) for row in values[1:]]
|
|
338
|
+
# Optional values and vertically merged group labels leave holes in valid
|
|
339
|
+
# records. A complete textual header plus consistently multi-field rows is
|
|
340
|
+
# sufficient; sparse covers/forms still fail the header requirement.
|
|
341
|
+
return bool(widths) and all(width >= 2 for width in widths)
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def _text_reason(features: RegionFeatures) -> str | None:
|
|
345
|
+
if (
|
|
346
|
+
features.row_count >= 2
|
|
347
|
+
and features.column_count == 1
|
|
348
|
+
and features.text_ratio == 1.0
|
|
349
|
+
and features.positive_ordinal_rows == 0
|
|
350
|
+
):
|
|
351
|
+
return "single_column_text"
|
|
352
|
+
if (
|
|
353
|
+
features.row_count >= 2
|
|
354
|
+
and features.column_count >= 2
|
|
355
|
+
and features.text_ratio >= 0.8
|
|
356
|
+
and features.numeric_grid_rows < 2
|
|
357
|
+
and features.positive_ordinal_rows == 0
|
|
358
|
+
and features.label_value_pairs < 2
|
|
359
|
+
and not features.has_stable_table_schema
|
|
360
|
+
and (features.merged_title_rows >= 1 or features.long_text_rows >= 1)
|
|
361
|
+
):
|
|
362
|
+
return "presentation_text_region"
|
|
363
|
+
return None
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def _is_form(features: RegionFeatures) -> bool:
|
|
367
|
+
return (
|
|
368
|
+
features.label_value_pairs >= 2
|
|
369
|
+
and features.label_value_coverage >= 0.5
|
|
370
|
+
and not features.has_stable_table_schema
|
|
371
|
+
and features.numeric_grid_columns < 2
|
|
372
|
+
and features.positive_ordinal_rows == 0
|
|
373
|
+
)
|
|
374
|
+
|
|
375
|
+
|
|
376
|
+
def _is_matrix(values: list[list[str]], features: RegionFeatures) -> bool:
|
|
377
|
+
if features.numeric_grid_rows < 2 or features.numeric_grid_columns < 2:
|
|
378
|
+
return False
|
|
379
|
+
top_axis = values and all(value and not _is_number(value) for value in values[0][1:])
|
|
380
|
+
left_axis = all(row[0] and not _is_number(row[0]) for row in values[1:])
|
|
381
|
+
return bool(top_axis and left_axis)
|
|
382
|
+
|
|
383
|
+
|
|
384
|
+
def _is_logical_table(features: RegionFeatures) -> bool:
|
|
385
|
+
return bool(
|
|
386
|
+
features.has_page_sequence
|
|
387
|
+
or features.has_stable_table_schema
|
|
388
|
+
or (
|
|
389
|
+
features.row_count >= 2
|
|
390
|
+
and features.column_count >= 2
|
|
391
|
+
and features.positive_ordinal_rows >= 1
|
|
392
|
+
)
|
|
393
|
+
)
|