langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,577 @@
1
+ """Deterministic evidence scoring for cross-Sheet table continuations."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ import unicodedata
7
+ from copy import deepcopy
8
+ from dataclasses import dataclass
9
+ from statistics import median
10
+
11
+ from langparse.types import StructuredData
12
+ from langparse.workbooks.types import (
13
+ HeaderColumn,
14
+ LogicalTable,
15
+ SheetIR,
16
+ SheetSnapshot,
17
+ TableContinuation,
18
+ TableSection,
19
+ WorkbookIR,
20
+ WorkbookSnapshot,
21
+ stable_id,
22
+ )
23
+
24
+ _PAGE_MARKER_RE = re.compile(r"第\s*\d+\s*页\s*共\s*\d+\s*页")
25
+ _CONTINUATION_SUFFIX_RE = re.compile(r"\s*(?:续表|续|continued)\s*$")
26
+ _PARENTHESIZED_CONTINUATION_SUFFIX_RE = re.compile(r"\s*\(\s*(?:续表|续|continued)\s*\)\s*$")
27
+ _SHEET_NUMBER_RE = re.compile(r"^(.*?)(\d+)$")
28
+ _PRESENTATION_ROLES = {
29
+ "title",
30
+ "context",
31
+ "header",
32
+ "repeated_title",
33
+ "repeated_context",
34
+ "repeated_header",
35
+ }
36
+ _REVIEW_THRESHOLD = 0.60
37
+ _AUTO_LINK_THRESHOLD = 0.85
38
+ _MIN_SCORE_LEAD = 0.10
39
+ _REPEATED_PRESENTATION_ROLES = {
40
+ "title": "repeated_title",
41
+ "context": "repeated_context",
42
+ "header": "repeated_header",
43
+ }
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class ContinuationCandidate:
48
+ left_sheet: str
49
+ right_sheet: str
50
+ left_table_id: str
51
+ right_table_id: str
52
+ confidence: float
53
+ reason_codes: tuple[str, ...] = ()
54
+ terminal_reason_codes: tuple[str, ...] = ()
55
+
56
+
57
+ def link_table_continuations(
58
+ snapshot: WorkbookSnapshot,
59
+ workbook_ir: WorkbookIR,
60
+ ) -> tuple[list[TableContinuation], list[StructuredData]]:
61
+ """Link unambiguous adjacent-Sheet table continuations and aggregate their views."""
62
+
63
+ table_order, tables_by_id = _workbook_tables(workbook_ir)
64
+ accepted_edges: list[ContinuationCandidate] = []
65
+ diagnostics: list[StructuredData] = []
66
+
67
+ for left_snapshot, left_ir, right_snapshot, right_ir in _adjacent_sheet_pairs(
68
+ snapshot, workbook_ir
69
+ ):
70
+ candidates = [
71
+ candidate
72
+ for left_table in _logical_tables(left_ir)
73
+ for right_table in _logical_tables(right_ir)
74
+ for candidate in [
75
+ score_continuation(left_snapshot, left_table, right_snapshot, right_table)
76
+ ]
77
+ if candidate is not None
78
+ ]
79
+ eligible = [
80
+ candidate
81
+ for candidate in candidates
82
+ if not candidate.terminal_reason_codes and candidate.confidence >= _REVIEW_THRESHOLD
83
+ ]
84
+ accepted = {
85
+ _candidate_key(candidate)
86
+ for candidate in eligible
87
+ if _is_mutual_unique_best(candidate, eligible)
88
+ and candidate.confidence >= _AUTO_LINK_THRESHOLD
89
+ }
90
+
91
+ for candidate in candidates:
92
+ extra_reason_codes: list[str] = []
93
+ if candidate.terminal_reason_codes:
94
+ status = "rejected"
95
+ extra_reason_codes.extend(candidate.terminal_reason_codes)
96
+ elif candidate.confidence < _REVIEW_THRESHOLD:
97
+ continue
98
+ elif _candidate_key(candidate) in accepted:
99
+ status = "accepted"
100
+ accepted_edges.append(candidate)
101
+ else:
102
+ status = "ambiguous"
103
+ if candidate.confidence < _AUTO_LINK_THRESHOLD:
104
+ extra_reason_codes.append("below_auto_accept_threshold")
105
+ if _has_close_competitor(candidate, eligible):
106
+ extra_reason_codes.append("competing_continuation_candidates")
107
+ elif candidate.confidence >= _AUTO_LINK_THRESHOLD:
108
+ extra_reason_codes.append("not_mutual_unique_best")
109
+ diagnostics.append(_candidate_diagnostic(candidate, status, extra_reason_codes))
110
+
111
+ groups = []
112
+ pending_assignments: list[tuple[LogicalTable, str, str]] = []
113
+ for member_table_ids, chain_edges in _continuation_chains(accepted_edges, table_order):
114
+ member_tables = [tables_by_id[table_id] for table_id in member_table_ids]
115
+ continuation_id = stable_id("continuation", snapshot.source, *member_table_ids)
116
+ reason_codes = _deduplicate_reason_codes(chain_edges)
117
+ aggregate = _aggregate_table(
118
+ continuation_id,
119
+ member_tables,
120
+ chain_edges,
121
+ reason_codes,
122
+ )
123
+ pending_assignments.extend(
124
+ (
125
+ member_table,
126
+ continuation_id,
127
+ _continuation_role(index, len(member_tables)),
128
+ )
129
+ for index, member_table in enumerate(member_tables)
130
+ )
131
+ groups.append(
132
+ TableContinuation(
133
+ continuation_id=continuation_id,
134
+ logical_table=aggregate,
135
+ member_table_ids=list(member_table_ids),
136
+ source_refs=deepcopy(aggregate.source_refs),
137
+ confidence=aggregate.confidence,
138
+ reason_codes=reason_codes,
139
+ )
140
+ )
141
+ for member_table, continuation_id, continuation_role in pending_assignments:
142
+ member_table.continuation_id = continuation_id
143
+ member_table.continuation_role = continuation_role
144
+ return groups, diagnostics
145
+
146
+
147
+ def _adjacent_sheet_pairs(
148
+ snapshot: WorkbookSnapshot,
149
+ workbook_ir: WorkbookIR,
150
+ ) -> list[tuple[SheetSnapshot, SheetIR, SheetSnapshot, SheetIR]]:
151
+ snapshot_by_index = {sheet.index: sheet for sheet in snapshot.sheets}
152
+ ir_by_index = {sheet.index: sheet for sheet in workbook_ir.sheets}
153
+ shared_indexes = sorted(set(snapshot_by_index) & set(ir_by_index))
154
+ return [
155
+ (
156
+ snapshot_by_index[index],
157
+ ir_by_index[index],
158
+ snapshot_by_index[index + 1],
159
+ ir_by_index[index + 1],
160
+ )
161
+ for index in shared_indexes
162
+ if index + 1 in snapshot_by_index and index + 1 in ir_by_index
163
+ ]
164
+
165
+
166
+ def _logical_tables(sheet_ir: SheetIR) -> list[LogicalTable]:
167
+ return [
168
+ block.logical_table
169
+ for block in sheet_ir.blocks
170
+ if block.kind == "logical_table" and block.logical_table is not None
171
+ ]
172
+
173
+
174
+ def _workbook_tables(workbook_ir: WorkbookIR) -> tuple[dict[str, int], dict[str, LogicalTable]]:
175
+ table_order = {}
176
+ tables_by_id = {}
177
+ for order, table in enumerate(
178
+ table
179
+ for sheet in sorted(workbook_ir.sheets, key=lambda sheet: sheet.index)
180
+ for table in _logical_tables(sheet)
181
+ ):
182
+ table_order[table.table_id] = order
183
+ tables_by_id[table.table_id] = table
184
+ return table_order, tables_by_id
185
+
186
+
187
+ def _is_mutual_unique_best(
188
+ candidate: ContinuationCandidate,
189
+ candidates: list[ContinuationCandidate],
190
+ ) -> bool:
191
+ left_options = [
192
+ option for option in candidates if option.left_table_id == candidate.left_table_id
193
+ ]
194
+ right_options = [
195
+ option for option in candidates if option.right_table_id == candidate.right_table_id
196
+ ]
197
+ return _has_score_lead(candidate, left_options) and _has_score_lead(candidate, right_options)
198
+
199
+
200
+ def _has_score_lead(
201
+ candidate: ContinuationCandidate,
202
+ alternatives: list[ContinuationCandidate],
203
+ ) -> bool:
204
+ return all(
205
+ alternative is candidate
206
+ or round(candidate.confidence - alternative.confidence, 4) >= _MIN_SCORE_LEAD
207
+ for alternative in alternatives
208
+ )
209
+
210
+
211
+ def _has_close_competitor(
212
+ candidate: ContinuationCandidate,
213
+ candidates: list[ContinuationCandidate],
214
+ ) -> bool:
215
+ return any(
216
+ alternative is not candidate
217
+ and (
218
+ alternative.left_table_id == candidate.left_table_id
219
+ or alternative.right_table_id == candidate.right_table_id
220
+ )
221
+ and round(abs(candidate.confidence - alternative.confidence), 4) < _MIN_SCORE_LEAD
222
+ for alternative in candidates
223
+ )
224
+
225
+
226
+ def _candidate_key(candidate: ContinuationCandidate) -> tuple[str, str]:
227
+ return candidate.left_table_id, candidate.right_table_id
228
+
229
+
230
+ def _candidate_diagnostic(
231
+ candidate: ContinuationCandidate,
232
+ status: str,
233
+ extra_reason_codes: list[str],
234
+ ) -> StructuredData:
235
+ return {
236
+ "left_table_id": candidate.left_table_id,
237
+ "right_table_id": candidate.right_table_id,
238
+ "left_sheet": candidate.left_sheet,
239
+ "right_sheet": candidate.right_sheet,
240
+ "confidence": candidate.confidence,
241
+ "status": status,
242
+ "reason_codes": [*candidate.reason_codes, *extra_reason_codes],
243
+ }
244
+
245
+
246
+ def _continuation_chains(
247
+ accepted_edges: list[ContinuationCandidate],
248
+ table_order: dict[str, int],
249
+ ) -> list[tuple[list[str], list[ContinuationCandidate]]]:
250
+ successors = {edge.left_table_id: edge for edge in accepted_edges}
251
+ predecessor_ids = {edge.right_table_id for edge in accepted_edges}
252
+ heads = sorted(
253
+ (table_id for table_id in successors if table_id not in predecessor_ids),
254
+ key=table_order.__getitem__,
255
+ )
256
+ chains = []
257
+ for head in heads:
258
+ member_table_ids = [head]
259
+ chain_edges = []
260
+ current_id = head
261
+ while current_id in successors:
262
+ edge = successors[current_id]
263
+ chain_edges.append(edge)
264
+ current_id = edge.right_table_id
265
+ member_table_ids.append(current_id)
266
+ if len(member_table_ids) >= 2:
267
+ chains.append((member_table_ids, chain_edges))
268
+ return chains
269
+
270
+
271
+ def _aggregate_table(
272
+ continuation_id: str,
273
+ member_tables: list[LogicalTable],
274
+ chain_edges: list[ContinuationCandidate],
275
+ reason_codes: list[str],
276
+ ) -> LogicalTable:
277
+ copied_members = [deepcopy(table) for table in member_tables]
278
+ aggregate = deepcopy(member_tables[0])
279
+ aggregate.table_id = stable_id("table", continuation_id, "aggregate")
280
+ aggregate.continuation_id = None
281
+ aggregate.continuation_role = None
282
+ aggregate.columns = deepcopy(copied_members[0].columns)
283
+ aggregate.rows = []
284
+ aggregate.fragments = []
285
+ aggregate.sections = [
286
+ section for copied_member in copied_members for section in copied_member.sections
287
+ ]
288
+ aggregate.source_refs = [
289
+ source_ref for copied_member in copied_members for source_ref in copied_member.source_refs
290
+ ]
291
+ aggregate.confidence = min(
292
+ *(table.confidence for table in member_tables),
293
+ *(edge.confidence for edge in chain_edges),
294
+ )
295
+ aggregate.diagnostics = [
296
+ {
297
+ "reason_code": "cross_sheet_continuation",
298
+ "continuation_id": continuation_id,
299
+ "member_table_ids": [table.table_id for table in member_tables],
300
+ "reason_codes": list(reason_codes),
301
+ }
302
+ ]
303
+
304
+ for member_index, copied_member in enumerate(copied_members):
305
+ if member_index:
306
+ for aggregate_column, member_column in zip(
307
+ aggregate.columns, copied_member.columns, strict=True
308
+ ):
309
+ aggregate_column.source_refs.extend(deepcopy(member_column.source_refs))
310
+ aggregate.fragments.extend(copied_member.fragments)
311
+
312
+ _append_aggregate_rows(aggregate, copied_members)
313
+ return aggregate
314
+
315
+
316
+ def _append_aggregate_rows(
317
+ aggregate: LogicalTable,
318
+ copied_members: list[LogicalTable],
319
+ ) -> None:
320
+ active_path: list[str] = []
321
+ active_section: TableSection | None = None
322
+ for member_index, copied_member in enumerate(copied_members):
323
+ member_sections_by_ref = {
324
+ section.source_ref.key: section for section in copied_member.sections
325
+ }
326
+ for row in copied_member.rows:
327
+ if member_index and row.role in _REPEATED_PRESENTATION_ROLES:
328
+ row.role = _REPEATED_PRESENTATION_ROLES[row.role]
329
+ if row.role == "section_header":
330
+ active_path = list(row.section_path)
331
+ active_section = member_sections_by_ref.get(row.source_ref.key)
332
+ if active_section is None:
333
+ active_section = _section_for_path(active_path, copied_member.sections)
334
+ if not active_path and active_section is not None:
335
+ active_path = [active_section.title]
336
+ elif row.section_path:
337
+ active_path = list(row.section_path)
338
+ member_section = _section_for_path(active_path, copied_member.sections)
339
+ if member_section is not None:
340
+ active_section = member_section
341
+ elif member_index and row.role == "data" and active_path:
342
+ row.section_path = list(active_path)
343
+ if active_section is not None and row.row_id not in active_section.row_ids:
344
+ active_section.row_ids.append(row.row_id)
345
+ aggregate.rows.append(row)
346
+
347
+
348
+ def _section_for_path(path: list[str], sections: list[TableSection]) -> TableSection | None:
349
+ if not path:
350
+ return None
351
+ return next(
352
+ (
353
+ section
354
+ for section in reversed(sections)
355
+ if [*section.parent_path, section.title] == path
356
+ ),
357
+ None,
358
+ )
359
+
360
+
361
+ def _continuation_role(index: int, member_count: int) -> str:
362
+ if index == 0:
363
+ return "head"
364
+ if index == member_count - 1:
365
+ return "tail"
366
+ return "member"
367
+
368
+
369
+ def _deduplicate_reason_codes(edges: list[ContinuationCandidate]) -> list[str]:
370
+ return list(dict.fromkeys(reason for edge in edges for reason in edge.reason_codes))
371
+
372
+
373
+ def score_continuation(
374
+ left_sheet: SheetSnapshot,
375
+ left_table: LogicalTable,
376
+ right_sheet: SheetSnapshot,
377
+ right_table: LogicalTable,
378
+ ) -> ContinuationCandidate | None:
379
+ """Score the explainable evidence that two table fragments continue each other."""
380
+
381
+ if header_fingerprint(left_table) != header_fingerprint(right_table):
382
+ return None
383
+
384
+ left_title = _normalize_title(left_table.title)
385
+ right_title = _normalize_title(right_table.title)
386
+ terminal = _terminal_reason_codes(left_table, left_title, right_table, right_title)
387
+
388
+ score = 0.35
389
+ reasons = ["header_fingerprint_match"]
390
+ if _has_valid_page_sequence(left_table, right_table):
391
+ score += 0.35
392
+ reasons.append("print_page_sequence")
393
+ if left_title and left_title == right_title:
394
+ score += 0.25
395
+ reasons.append("title_match")
396
+ if _has_sequential_sheet_names(left_sheet.name, right_sheet.name):
397
+ score += 0.25
398
+ reasons.append("sheet_name_sequence")
399
+ if _has_compatible_widths(left_sheet, left_table, right_sheet, right_table):
400
+ score += 0.15
401
+ reasons.append("column_width_compatibility")
402
+ if _has_compatible_units(left_table, right_table):
403
+ score += 0.10
404
+ reasons.append("unit_compatibility")
405
+
406
+ return ContinuationCandidate(
407
+ left_sheet=left_sheet.name,
408
+ right_sheet=right_sheet.name,
409
+ left_table_id=left_table.table_id,
410
+ right_table_id=right_table.table_id,
411
+ confidence=round(min(score, 1.0), 4),
412
+ reason_codes=tuple(reasons),
413
+ terminal_reason_codes=tuple(terminal),
414
+ )
415
+
416
+
417
+ def header_fingerprint(table: LogicalTable) -> tuple[tuple[str, ...], ...]:
418
+ """Return the positional, normalized schema required for a continuation."""
419
+
420
+ fingerprint = []
421
+ for index, column in enumerate(table.columns):
422
+ path = tuple(_normalize_text(part) for part in column.path if _normalize_text(part))
423
+ fingerprint.append(path or (f"<empty:{index}>",))
424
+ return tuple(fingerprint)
425
+
426
+
427
+ def _normalize_text(value: str) -> str:
428
+ return " ".join(unicodedata.normalize("NFKC", value).split()).casefold()
429
+
430
+
431
+ def _normalize_title(value: str) -> str:
432
+ normalized = _normalize_text(value)
433
+ normalized = _PAGE_MARKER_RE.sub("", normalized).strip()
434
+ normalized = _PARENTHESIZED_CONTINUATION_SUFFIX_RE.sub("", normalized).strip()
435
+ return _CONTINUATION_SUFFIX_RE.sub("", normalized).strip()
436
+
437
+
438
+ def _terminal_reason_codes(
439
+ left_table: LogicalTable,
440
+ left_title: str,
441
+ right_table: LogicalTable,
442
+ right_title: str,
443
+ ) -> list[str]:
444
+ terminal = []
445
+ if _ends_with_total(left_table):
446
+ terminal.append("terminal_total")
447
+ if left_title and right_title and left_title != right_title:
448
+ terminal.append("title_mismatch")
449
+ if _page_metadata_conflicts(left_table, right_table):
450
+ terminal.append("page_sequence_conflict")
451
+ return terminal
452
+
453
+
454
+ def _ends_with_total(table: LogicalTable) -> bool:
455
+ for row in reversed(table.rows):
456
+ if row.role not in _PRESENTATION_ROLES:
457
+ return row.role == "total"
458
+ return False
459
+
460
+
461
+ def _page_metadata(
462
+ left_table: LogicalTable, right_table: LogicalTable
463
+ ) -> tuple[int, int, int] | None:
464
+ left_fragment = next(
465
+ (
466
+ fragment
467
+ for fragment in reversed(left_table.fragments)
468
+ if fragment.page_number is not None and fragment.total_pages is not None
469
+ ),
470
+ None,
471
+ )
472
+ right_fragment = next(
473
+ (
474
+ fragment
475
+ for fragment in right_table.fragments
476
+ if fragment.page_number is not None and fragment.total_pages is not None
477
+ ),
478
+ None,
479
+ )
480
+ if left_fragment is None or right_fragment is None:
481
+ return None
482
+ if left_fragment.page_number is None or left_fragment.total_pages is None:
483
+ return None
484
+ if right_fragment.page_number is None or right_fragment.total_pages is None:
485
+ return None
486
+ if left_fragment.total_pages != right_fragment.total_pages:
487
+ return left_fragment.page_number, right_fragment.page_number, -1
488
+ return left_fragment.page_number, right_fragment.page_number, left_fragment.total_pages
489
+
490
+
491
+ def _has_valid_page_sequence(left_table: LogicalTable, right_table: LogicalTable) -> bool:
492
+ metadata = _page_metadata(left_table, right_table)
493
+ return (
494
+ metadata is not None
495
+ and 1 <= metadata[0] < metadata[2]
496
+ and metadata[1] == metadata[0] + 1 <= metadata[2]
497
+ )
498
+
499
+
500
+ def _page_metadata_conflicts(left_table: LogicalTable, right_table: LogicalTable) -> bool:
501
+ metadata = _page_metadata(left_table, right_table)
502
+ return metadata is not None and not (
503
+ 1 <= metadata[0] < metadata[2] and metadata[1] == metadata[0] + 1 <= metadata[2]
504
+ )
505
+
506
+
507
+ def _has_sequential_sheet_names(left_name: str, right_name: str) -> bool:
508
+ left = _normalize_text(left_name)
509
+ right = _normalize_text(right_name)
510
+ if right in {f"{left}续", f"{left}续表", f"{left}continued"}:
511
+ return True
512
+ left_match = _SHEET_NUMBER_RE.fullmatch(left)
513
+ right_match = _SHEET_NUMBER_RE.fullmatch(right)
514
+ return bool(
515
+ left_match
516
+ and right_match
517
+ and left_match.group(1)
518
+ and left_match.group(1) == right_match.group(1)
519
+ and int(right_match.group(2)) == int(left_match.group(2)) + 1
520
+ )
521
+
522
+
523
+ def _has_compatible_widths(
524
+ left_sheet: SheetSnapshot,
525
+ left_table: LogicalTable,
526
+ right_sheet: SheetSnapshot,
527
+ right_table: LogicalTable,
528
+ ) -> bool:
529
+ paired_columns = list(zip(left_table.columns, right_table.columns, strict=True))
530
+ if not paired_columns:
531
+ return False
532
+ differences = [
533
+ _relative_width_difference(
534
+ left_sheet.column_widths[left.coordinate], right_sheet.column_widths[right.coordinate]
535
+ )
536
+ for left, right in paired_columns
537
+ if left.coordinate in left_sheet.column_widths
538
+ and right.coordinate in right_sheet.column_widths
539
+ ]
540
+ return len(differences) * 2 >= len(paired_columns) and median(differences) <= 0.15
541
+
542
+
543
+ def _relative_width_difference(left_width: float, right_width: float) -> float:
544
+ denominator = max(abs(left_width), abs(right_width))
545
+ if denominator == 0:
546
+ return 0.0
547
+ return abs(left_width - right_width) / denominator
548
+
549
+
550
+ def _has_compatible_units(left_table: LogicalTable, right_table: LogicalTable) -> bool:
551
+ paired_columns = list(zip(left_table.columns, right_table.columns, strict=True))
552
+ if any(left.unit or right.unit for left, right in paired_columns):
553
+ return any(
554
+ left.unit and right.unit and _normalize_text(left.unit) == _normalize_text(right.unit)
555
+ for left, right in paired_columns
556
+ )
557
+ return any(
558
+ _unit_values(left_table, index) & _unit_values(right_table, index)
559
+ for index, (left, right) in enumerate(paired_columns)
560
+ if _is_unit_column(left) and _is_unit_column(right)
561
+ )
562
+
563
+
564
+ def _is_unit_column(column: HeaderColumn) -> bool:
565
+ return any(
566
+ "单位" in _normalize_text(part) or "unit" in _normalize_text(part) for part in column.path
567
+ )
568
+
569
+
570
+ def _unit_values(table: LogicalTable, column_index: int) -> set[str]:
571
+ return {
572
+ normalized
573
+ for row in table.rows
574
+ if row.role == "data" and column_index < len(row.values)
575
+ for normalized in [_normalize_text(str(row.values[column_index]))]
576
+ if normalized
577
+ }
@@ -0,0 +1,45 @@
1
+ from __future__ import annotations
2
+
3
+ from .evaluator import (
4
+ CaseEvaluationDetail,
5
+ CaseObservation,
6
+ CaseTruth,
7
+ RegionKind,
8
+ WorkbookEvaluationMetrics,
9
+ classify_case_evaluation,
10
+ evaluate_workbook_ambiguity,
11
+ )
12
+ from .schema import (
13
+ GoldenCase,
14
+ GoldenSample,
15
+ GoldenSetDriftError,
16
+ GoldenSetManifest,
17
+ InvalidGoldenSetError,
18
+ WorkbookEvaluationError,
19
+ compute_choices_digest,
20
+ compute_evaluation_id,
21
+ compute_sample_evaluation_id,
22
+ load_golden_set_manifest,
23
+ validate_output_dir_isolation,
24
+ )
25
+
26
+ __all__ = [
27
+ "CaseEvaluationDetail",
28
+ "CaseObservation",
29
+ "CaseTruth",
30
+ "GoldenCase",
31
+ "GoldenSample",
32
+ "GoldenSetDriftError",
33
+ "GoldenSetManifest",
34
+ "InvalidGoldenSetError",
35
+ "RegionKind",
36
+ "WorkbookEvaluationError",
37
+ "WorkbookEvaluationMetrics",
38
+ "classify_case_evaluation",
39
+ "compute_choices_digest",
40
+ "compute_evaluation_id",
41
+ "compute_sample_evaluation_id",
42
+ "evaluate_workbook_ambiguity",
43
+ "load_golden_set_manifest",
44
+ "validate_output_dir_isolation",
45
+ ]