langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,393 @@
1
+ from __future__ import annotations
2
+
3
+ import json
4
+ import re
5
+ from dataclasses import asdict, dataclass
6
+
7
+ from openpyxl.utils import get_column_letter, range_boundaries
8
+
9
+ from langparse.workbooks.modeling import RegionChoice
10
+ from langparse.workbooks.modeling.types import REGION_RULE_VERSION
11
+ from langparse.workbooks.types import CandidateRegion, SheetSnapshot, stable_id
12
+
13
+ PAGE_RE = re.compile(r"第\s*\d+\s*页\s*共\s*\d+\s*页")
14
+
15
+
16
+ @dataclass(frozen=True)
17
+ class RegionFeatures:
18
+ row_count: int
19
+ column_count: int
20
+ occupied_count: int
21
+ density: float
22
+ text_ratio: float
23
+ numeric_ratio: float
24
+ formula_ratio: float
25
+ nonempty_by_row: tuple[int, ...]
26
+ nonempty_by_column: tuple[int, ...]
27
+ positive_ordinal_rows: int
28
+ label_value_pairs: int
29
+ label_value_coverage: float
30
+ numeric_grid_rows: int
31
+ numeric_grid_columns: int
32
+ merged_title_rows: int
33
+ long_text_rows: int
34
+ has_page_sequence: bool
35
+ has_stable_table_schema: bool
36
+
37
+
38
+ @dataclass(frozen=True)
39
+ class BlockClassification:
40
+ kind: str
41
+ confidence: float
42
+ reason_codes: list[str]
43
+ features: RegionFeatures
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class RegionAssessment:
48
+ deterministic: BlockClassification
49
+ choices: tuple[RegionChoice, ...]
50
+ ambiguous: bool
51
+ ambiguity_codes: tuple[str, ...]
52
+
53
+
54
+ def extract_region_features(
55
+ sheet: SheetSnapshot,
56
+ candidate: CandidateRegion,
57
+ ) -> RegionFeatures:
58
+ """Compute deterministic, serializable signals for one candidate region."""
59
+
60
+ values, cells = _region_grid(sheet, candidate)
61
+ row_count = len(values)
62
+ column_count = len(values[0]) if values else 0
63
+ flat_values = [value for row in values for value in row if value != ""]
64
+ semantic_count = len(flat_values)
65
+ numeric_count = sum(_is_number(value) for value in flat_values)
66
+ text_count = semantic_count - numeric_count
67
+ formula_count = sum(
68
+ cell is not None and cell.merge_anchor is None and cell.formula is not None
69
+ for row in cells
70
+ for cell in row
71
+ )
72
+ nonempty_by_row = tuple(sum(value != "" for value in row) for row in values)
73
+ nonempty_by_column = tuple(
74
+ sum(values[row][column] != "" for row in range(row_count)) for column in range(column_count)
75
+ )
76
+ positive_ordinal_rows = sum(_is_row_ordinal(row[0]) for row in values[1:] if row)
77
+ label_value_pairs = sum(_label_value_pairs(row) for row in values)
78
+ numeric_grid_rows, numeric_grid_columns = _numeric_grid_shape(values)
79
+ merged_title_rows = sum(
80
+ sum(value != "" for value in value_row) == 1
81
+ and any(cell is not None and cell.colspan > 1 for cell in cell_row)
82
+ for value_row, cell_row in zip(values, cells, strict=True)
83
+ )
84
+ long_text_rows = sum(
85
+ len(nonempty) == 1 and len(nonempty[0]) >= 20
86
+ for row in values
87
+ for nonempty in [[value for value in row if value]]
88
+ )
89
+ has_page_sequence = any(PAGE_RE.search(value) for value in flat_values)
90
+ has_stable_table_schema = _has_stable_table_schema(
91
+ values,
92
+ numeric_grid_rows=numeric_grid_rows,
93
+ numeric_grid_columns=numeric_grid_columns,
94
+ )
95
+ area = row_count * column_count
96
+ return RegionFeatures(
97
+ row_count=row_count,
98
+ column_count=column_count,
99
+ occupied_count=len(candidate.cell_refs),
100
+ density=len(candidate.cell_refs) / area if area else 0.0,
101
+ text_ratio=text_count / semantic_count if semantic_count else 0.0,
102
+ numeric_ratio=numeric_count / semantic_count if semantic_count else 0.0,
103
+ formula_ratio=formula_count / semantic_count if semantic_count else 0.0,
104
+ nonempty_by_row=nonempty_by_row,
105
+ nonempty_by_column=nonempty_by_column,
106
+ positive_ordinal_rows=positive_ordinal_rows,
107
+ label_value_pairs=label_value_pairs,
108
+ label_value_coverage=label_value_pairs / row_count if row_count else 0.0,
109
+ numeric_grid_rows=numeric_grid_rows,
110
+ numeric_grid_columns=numeric_grid_columns,
111
+ merged_title_rows=merged_title_rows,
112
+ long_text_rows=long_text_rows,
113
+ has_page_sequence=has_page_sequence,
114
+ has_stable_table_schema=has_stable_table_schema,
115
+ )
116
+
117
+
118
+ def classify_candidate_region(
119
+ sheet: SheetSnapshot,
120
+ candidate: CandidateRegion,
121
+ features: RegionFeatures | None = None,
122
+ ) -> BlockClassification:
123
+ """Classify a region conservatively with mutually exclusive rules."""
124
+
125
+ features = features if features is not None else extract_region_features(sheet, candidate)
126
+ values, _ = _region_grid(sheet, candidate)
127
+ return _classify_region(values, features, candidate.reason_codes)
128
+
129
+
130
+ def assess_candidate_region(
131
+ sheet: SheetSnapshot,
132
+ candidate: CandidateRegion,
133
+ ) -> RegionAssessment:
134
+ """Assess a region and register only locally compatible alternative kinds."""
135
+
136
+ features = extract_region_features(sheet, candidate)
137
+ feature_digest = _structural_feature_digest(features)
138
+ values, _ = _region_grid(sheet, candidate)
139
+ deterministic = _classify_region(values, features, candidate.reason_codes)
140
+ choices = [
141
+ RegionChoice(
142
+ choice_id=_choice_id(
143
+ candidate,
144
+ feature_digest,
145
+ deterministic.kind,
146
+ deterministic.reason_codes[0],
147
+ ),
148
+ kind=deterministic.kind,
149
+ local_score=deterministic.confidence,
150
+ reason_codes=tuple(deterministic.reason_codes),
151
+ )
152
+ ]
153
+ if deterministic.kind != "unclassified":
154
+ return RegionAssessment(
155
+ deterministic=deterministic,
156
+ choices=tuple(choices),
157
+ ambiguous=False,
158
+ ambiguity_codes=(),
159
+ )
160
+
161
+ seen_kinds = {deterministic.kind}
162
+ for kind, score, reason_code in _weak_choice_kinds(features):
163
+ if kind in seen_kinds:
164
+ continue
165
+ choices.append(
166
+ RegionChoice(
167
+ choice_id=_choice_id(candidate, feature_digest, kind, reason_code),
168
+ kind=kind,
169
+ local_score=score,
170
+ reason_codes=(reason_code,),
171
+ )
172
+ )
173
+ seen_kinds.add(kind)
174
+
175
+ ambiguous = len(choices) >= 2
176
+ return RegionAssessment(
177
+ deterministic=deterministic,
178
+ choices=tuple(choices),
179
+ ambiguous=ambiguous,
180
+ ambiguity_codes=("unclassified_with_compatible_choices",) if ambiguous else (),
181
+ )
182
+
183
+
184
+ def _classify_region(
185
+ values: list[list[str]],
186
+ features: RegionFeatures,
187
+ region_reason_codes: list[str] | None = None,
188
+ ) -> BlockClassification:
189
+ """Apply the existing deterministic winner rules to precomputed facts."""
190
+
191
+ if "native_table_anchor" in (region_reason_codes or []):
192
+ return BlockClassification(
193
+ "logical_table",
194
+ 0.98,
195
+ ["native_table_anchor"],
196
+ features,
197
+ )
198
+ text_reason = _text_reason(features)
199
+ if text_reason:
200
+ return BlockClassification("text", 0.9, [text_reason], features)
201
+ if _is_form(features):
202
+ return BlockClassification("form", 0.9, ["stable_label_value_pairs"], features)
203
+ if _is_matrix(values, features):
204
+ return BlockClassification("matrix", 0.95, ["numeric_matrix_with_axes"], features)
205
+ if _is_logical_table(features):
206
+ reason = (
207
+ "consistent_print_fragments"
208
+ if features.has_page_sequence
209
+ else "stable_header_data_schema"
210
+ )
211
+ return BlockClassification("logical_table", 0.9, [reason], features)
212
+ return BlockClassification(
213
+ "unclassified",
214
+ 0.5,
215
+ ["insufficient_semantic_evidence"],
216
+ features,
217
+ )
218
+
219
+
220
+ def _weak_choice_kinds(features: RegionFeatures) -> list[tuple[str, float, str]]:
221
+ choices = []
222
+ if (
223
+ features.row_count >= 2
224
+ and features.column_count >= 2
225
+ and max(features.nonempty_by_row, default=0) >= 2
226
+ ):
227
+ choices.append(("logical_table", 0.4, "weak_row_column_structure"))
228
+ if features.column_count >= 2 and features.label_value_pairs >= 1:
229
+ choices.append(("form", 0.4, "weak_label_value_pairs"))
230
+ if features.numeric_grid_rows >= 1 and features.numeric_grid_columns >= 1:
231
+ choices.append(("matrix", 0.4, "weak_numeric_axes"))
232
+ if features.occupied_count >= 2 and features.text_ratio >= 0.6:
233
+ choices.append(("text", 0.4, "weak_text_region"))
234
+ return choices
235
+
236
+
237
+ def _structural_feature_digest(features: RegionFeatures) -> str:
238
+ payload = json.dumps(
239
+ asdict(features),
240
+ ensure_ascii=True,
241
+ sort_keys=True,
242
+ separators=(",", ":"),
243
+ )
244
+ return stable_id("region_features", payload)
245
+
246
+
247
+ def _choice_id(
248
+ candidate: CandidateRegion,
249
+ feature_digest: str,
250
+ kind: str,
251
+ reason_code: str,
252
+ ) -> str:
253
+ return stable_id(
254
+ "region_choice",
255
+ REGION_RULE_VERSION,
256
+ candidate.source_ref.key,
257
+ feature_digest,
258
+ kind,
259
+ reason_code,
260
+ )
261
+
262
+
263
+ def _region_grid(sheet: SheetSnapshot, candidate: CandidateRegion):
264
+ min_col, min_row, max_col, max_row = range_boundaries(candidate.source_ref.range)
265
+ values = []
266
+ cells = []
267
+ for row_number in range(min_row, max_row + 1):
268
+ value_row = []
269
+ cell_row = []
270
+ for column_number in range(min_col, max_col + 1):
271
+ coordinate = f"{get_column_letter(column_number)}{row_number}"
272
+ cell = sheet.cells.get(coordinate)
273
+ cell_row.append(cell)
274
+ value_row.append(
275
+ "" if cell is None or cell.merge_anchor is not None else cell.display_value.strip()
276
+ )
277
+ values.append(value_row)
278
+ cells.append(cell_row)
279
+ return values, cells
280
+
281
+
282
+ def _is_number(value: str) -> bool:
283
+ try:
284
+ float(value.replace(",", ""))
285
+ except (TypeError, ValueError):
286
+ return False
287
+ return True
288
+
289
+
290
+ def _is_row_ordinal(value: str) -> bool:
291
+ # Printed schedules use both flat and hierarchical item numbers.
292
+ return bool(re.fullmatch(r"[1-9]\d*(?:\.\d+)*", value.strip()))
293
+
294
+
295
+ def _label_value_pairs(row: list[str]) -> int:
296
+ nonempty = [(index, value) for index, value in enumerate(row) if value]
297
+ if len(nonempty) % 2:
298
+ return 0
299
+ pairs = 0
300
+ for offset in range(0, len(nonempty) - 1, 2):
301
+ (label_index, label), (value_index, _) = nonempty[offset : offset + 2]
302
+ if value_index == label_index + 1 and not _is_number(label):
303
+ pairs += 1
304
+ return pairs
305
+
306
+
307
+ def _numeric_grid_shape(values: list[list[str]]) -> tuple[int, int]:
308
+ if len(values) < 2 or len(values[0]) < 2:
309
+ return 0, 0
310
+ interior = [row[1:] for row in values[1:]]
311
+ numeric_rows = sum(
312
+ row and all(value and _is_number(value) for value in row) for row in interior
313
+ )
314
+ numeric_columns = sum(
315
+ all(
316
+ interior[row][column] and _is_number(interior[row][column])
317
+ for row in range(len(interior))
318
+ )
319
+ for column in range(len(interior[0]))
320
+ )
321
+ return numeric_rows, numeric_columns
322
+
323
+
324
+ def _has_stable_table_schema(
325
+ values: list[list[str]],
326
+ *,
327
+ numeric_grid_rows: int,
328
+ numeric_grid_columns: int,
329
+ ) -> bool:
330
+ if len(values) < 2 or len(values[0]) < 2:
331
+ return False
332
+ if numeric_grid_rows >= 2 and numeric_grid_columns >= 2:
333
+ return False
334
+ header = values[0]
335
+ if not all(value and not _is_number(value) for value in header):
336
+ return False
337
+ widths = [sum(value != "" for value in row) for row in values[1:]]
338
+ # Optional values and vertically merged group labels leave holes in valid
339
+ # records. A complete textual header plus consistently multi-field rows is
340
+ # sufficient; sparse covers/forms still fail the header requirement.
341
+ return bool(widths) and all(width >= 2 for width in widths)
342
+
343
+
344
+ def _text_reason(features: RegionFeatures) -> str | None:
345
+ if (
346
+ features.row_count >= 2
347
+ and features.column_count == 1
348
+ and features.text_ratio == 1.0
349
+ and features.positive_ordinal_rows == 0
350
+ ):
351
+ return "single_column_text"
352
+ if (
353
+ features.row_count >= 2
354
+ and features.column_count >= 2
355
+ and features.text_ratio >= 0.8
356
+ and features.numeric_grid_rows < 2
357
+ and features.positive_ordinal_rows == 0
358
+ and features.label_value_pairs < 2
359
+ and not features.has_stable_table_schema
360
+ and (features.merged_title_rows >= 1 or features.long_text_rows >= 1)
361
+ ):
362
+ return "presentation_text_region"
363
+ return None
364
+
365
+
366
+ def _is_form(features: RegionFeatures) -> bool:
367
+ return (
368
+ features.label_value_pairs >= 2
369
+ and features.label_value_coverage >= 0.5
370
+ and not features.has_stable_table_schema
371
+ and features.numeric_grid_columns < 2
372
+ and features.positive_ordinal_rows == 0
373
+ )
374
+
375
+
376
+ def _is_matrix(values: list[list[str]], features: RegionFeatures) -> bool:
377
+ if features.numeric_grid_rows < 2 or features.numeric_grid_columns < 2:
378
+ return False
379
+ top_axis = values and all(value and not _is_number(value) for value in values[0][1:])
380
+ left_axis = all(row[0] and not _is_number(row[0]) for row in values[1:])
381
+ return bool(top_axis and left_axis)
382
+
383
+
384
+ def _is_logical_table(features: RegionFeatures) -> bool:
385
+ return bool(
386
+ features.has_page_sequence
387
+ or features.has_stable_table_schema
388
+ or (
389
+ features.row_count >= 2
390
+ and features.column_count >= 2
391
+ and features.positive_ordinal_rows >= 1
392
+ )
393
+ )