langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,577 @@
|
|
|
1
|
+
"""Deterministic evidence scoring for cross-Sheet table continuations."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
import unicodedata
|
|
7
|
+
from copy import deepcopy
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from statistics import median
|
|
10
|
+
|
|
11
|
+
from langparse.types import StructuredData
|
|
12
|
+
from langparse.workbooks.types import (
|
|
13
|
+
HeaderColumn,
|
|
14
|
+
LogicalTable,
|
|
15
|
+
SheetIR,
|
|
16
|
+
SheetSnapshot,
|
|
17
|
+
TableContinuation,
|
|
18
|
+
TableSection,
|
|
19
|
+
WorkbookIR,
|
|
20
|
+
WorkbookSnapshot,
|
|
21
|
+
stable_id,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
_PAGE_MARKER_RE = re.compile(r"第\s*\d+\s*页\s*共\s*\d+\s*页")
|
|
25
|
+
_CONTINUATION_SUFFIX_RE = re.compile(r"\s*(?:续表|续|continued)\s*$")
|
|
26
|
+
_PARENTHESIZED_CONTINUATION_SUFFIX_RE = re.compile(r"\s*\(\s*(?:续表|续|continued)\s*\)\s*$")
|
|
27
|
+
_SHEET_NUMBER_RE = re.compile(r"^(.*?)(\d+)$")
|
|
28
|
+
_PRESENTATION_ROLES = {
|
|
29
|
+
"title",
|
|
30
|
+
"context",
|
|
31
|
+
"header",
|
|
32
|
+
"repeated_title",
|
|
33
|
+
"repeated_context",
|
|
34
|
+
"repeated_header",
|
|
35
|
+
}
|
|
36
|
+
_REVIEW_THRESHOLD = 0.60
|
|
37
|
+
_AUTO_LINK_THRESHOLD = 0.85
|
|
38
|
+
_MIN_SCORE_LEAD = 0.10
|
|
39
|
+
_REPEATED_PRESENTATION_ROLES = {
|
|
40
|
+
"title": "repeated_title",
|
|
41
|
+
"context": "repeated_context",
|
|
42
|
+
"header": "repeated_header",
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class ContinuationCandidate:
|
|
48
|
+
left_sheet: str
|
|
49
|
+
right_sheet: str
|
|
50
|
+
left_table_id: str
|
|
51
|
+
right_table_id: str
|
|
52
|
+
confidence: float
|
|
53
|
+
reason_codes: tuple[str, ...] = ()
|
|
54
|
+
terminal_reason_codes: tuple[str, ...] = ()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def link_table_continuations(
|
|
58
|
+
snapshot: WorkbookSnapshot,
|
|
59
|
+
workbook_ir: WorkbookIR,
|
|
60
|
+
) -> tuple[list[TableContinuation], list[StructuredData]]:
|
|
61
|
+
"""Link unambiguous adjacent-Sheet table continuations and aggregate their views."""
|
|
62
|
+
|
|
63
|
+
table_order, tables_by_id = _workbook_tables(workbook_ir)
|
|
64
|
+
accepted_edges: list[ContinuationCandidate] = []
|
|
65
|
+
diagnostics: list[StructuredData] = []
|
|
66
|
+
|
|
67
|
+
for left_snapshot, left_ir, right_snapshot, right_ir in _adjacent_sheet_pairs(
|
|
68
|
+
snapshot, workbook_ir
|
|
69
|
+
):
|
|
70
|
+
candidates = [
|
|
71
|
+
candidate
|
|
72
|
+
for left_table in _logical_tables(left_ir)
|
|
73
|
+
for right_table in _logical_tables(right_ir)
|
|
74
|
+
for candidate in [
|
|
75
|
+
score_continuation(left_snapshot, left_table, right_snapshot, right_table)
|
|
76
|
+
]
|
|
77
|
+
if candidate is not None
|
|
78
|
+
]
|
|
79
|
+
eligible = [
|
|
80
|
+
candidate
|
|
81
|
+
for candidate in candidates
|
|
82
|
+
if not candidate.terminal_reason_codes and candidate.confidence >= _REVIEW_THRESHOLD
|
|
83
|
+
]
|
|
84
|
+
accepted = {
|
|
85
|
+
_candidate_key(candidate)
|
|
86
|
+
for candidate in eligible
|
|
87
|
+
if _is_mutual_unique_best(candidate, eligible)
|
|
88
|
+
and candidate.confidence >= _AUTO_LINK_THRESHOLD
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
for candidate in candidates:
|
|
92
|
+
extra_reason_codes: list[str] = []
|
|
93
|
+
if candidate.terminal_reason_codes:
|
|
94
|
+
status = "rejected"
|
|
95
|
+
extra_reason_codes.extend(candidate.terminal_reason_codes)
|
|
96
|
+
elif candidate.confidence < _REVIEW_THRESHOLD:
|
|
97
|
+
continue
|
|
98
|
+
elif _candidate_key(candidate) in accepted:
|
|
99
|
+
status = "accepted"
|
|
100
|
+
accepted_edges.append(candidate)
|
|
101
|
+
else:
|
|
102
|
+
status = "ambiguous"
|
|
103
|
+
if candidate.confidence < _AUTO_LINK_THRESHOLD:
|
|
104
|
+
extra_reason_codes.append("below_auto_accept_threshold")
|
|
105
|
+
if _has_close_competitor(candidate, eligible):
|
|
106
|
+
extra_reason_codes.append("competing_continuation_candidates")
|
|
107
|
+
elif candidate.confidence >= _AUTO_LINK_THRESHOLD:
|
|
108
|
+
extra_reason_codes.append("not_mutual_unique_best")
|
|
109
|
+
diagnostics.append(_candidate_diagnostic(candidate, status, extra_reason_codes))
|
|
110
|
+
|
|
111
|
+
groups = []
|
|
112
|
+
pending_assignments: list[tuple[LogicalTable, str, str]] = []
|
|
113
|
+
for member_table_ids, chain_edges in _continuation_chains(accepted_edges, table_order):
|
|
114
|
+
member_tables = [tables_by_id[table_id] for table_id in member_table_ids]
|
|
115
|
+
continuation_id = stable_id("continuation", snapshot.source, *member_table_ids)
|
|
116
|
+
reason_codes = _deduplicate_reason_codes(chain_edges)
|
|
117
|
+
aggregate = _aggregate_table(
|
|
118
|
+
continuation_id,
|
|
119
|
+
member_tables,
|
|
120
|
+
chain_edges,
|
|
121
|
+
reason_codes,
|
|
122
|
+
)
|
|
123
|
+
pending_assignments.extend(
|
|
124
|
+
(
|
|
125
|
+
member_table,
|
|
126
|
+
continuation_id,
|
|
127
|
+
_continuation_role(index, len(member_tables)),
|
|
128
|
+
)
|
|
129
|
+
for index, member_table in enumerate(member_tables)
|
|
130
|
+
)
|
|
131
|
+
groups.append(
|
|
132
|
+
TableContinuation(
|
|
133
|
+
continuation_id=continuation_id,
|
|
134
|
+
logical_table=aggregate,
|
|
135
|
+
member_table_ids=list(member_table_ids),
|
|
136
|
+
source_refs=deepcopy(aggregate.source_refs),
|
|
137
|
+
confidence=aggregate.confidence,
|
|
138
|
+
reason_codes=reason_codes,
|
|
139
|
+
)
|
|
140
|
+
)
|
|
141
|
+
for member_table, continuation_id, continuation_role in pending_assignments:
|
|
142
|
+
member_table.continuation_id = continuation_id
|
|
143
|
+
member_table.continuation_role = continuation_role
|
|
144
|
+
return groups, diagnostics
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _adjacent_sheet_pairs(
|
|
148
|
+
snapshot: WorkbookSnapshot,
|
|
149
|
+
workbook_ir: WorkbookIR,
|
|
150
|
+
) -> list[tuple[SheetSnapshot, SheetIR, SheetSnapshot, SheetIR]]:
|
|
151
|
+
snapshot_by_index = {sheet.index: sheet for sheet in snapshot.sheets}
|
|
152
|
+
ir_by_index = {sheet.index: sheet for sheet in workbook_ir.sheets}
|
|
153
|
+
shared_indexes = sorted(set(snapshot_by_index) & set(ir_by_index))
|
|
154
|
+
return [
|
|
155
|
+
(
|
|
156
|
+
snapshot_by_index[index],
|
|
157
|
+
ir_by_index[index],
|
|
158
|
+
snapshot_by_index[index + 1],
|
|
159
|
+
ir_by_index[index + 1],
|
|
160
|
+
)
|
|
161
|
+
for index in shared_indexes
|
|
162
|
+
if index + 1 in snapshot_by_index and index + 1 in ir_by_index
|
|
163
|
+
]
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _logical_tables(sheet_ir: SheetIR) -> list[LogicalTable]:
|
|
167
|
+
return [
|
|
168
|
+
block.logical_table
|
|
169
|
+
for block in sheet_ir.blocks
|
|
170
|
+
if block.kind == "logical_table" and block.logical_table is not None
|
|
171
|
+
]
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _workbook_tables(workbook_ir: WorkbookIR) -> tuple[dict[str, int], dict[str, LogicalTable]]:
|
|
175
|
+
table_order = {}
|
|
176
|
+
tables_by_id = {}
|
|
177
|
+
for order, table in enumerate(
|
|
178
|
+
table
|
|
179
|
+
for sheet in sorted(workbook_ir.sheets, key=lambda sheet: sheet.index)
|
|
180
|
+
for table in _logical_tables(sheet)
|
|
181
|
+
):
|
|
182
|
+
table_order[table.table_id] = order
|
|
183
|
+
tables_by_id[table.table_id] = table
|
|
184
|
+
return table_order, tables_by_id
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
def _is_mutual_unique_best(
|
|
188
|
+
candidate: ContinuationCandidate,
|
|
189
|
+
candidates: list[ContinuationCandidate],
|
|
190
|
+
) -> bool:
|
|
191
|
+
left_options = [
|
|
192
|
+
option for option in candidates if option.left_table_id == candidate.left_table_id
|
|
193
|
+
]
|
|
194
|
+
right_options = [
|
|
195
|
+
option for option in candidates if option.right_table_id == candidate.right_table_id
|
|
196
|
+
]
|
|
197
|
+
return _has_score_lead(candidate, left_options) and _has_score_lead(candidate, right_options)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _has_score_lead(
|
|
201
|
+
candidate: ContinuationCandidate,
|
|
202
|
+
alternatives: list[ContinuationCandidate],
|
|
203
|
+
) -> bool:
|
|
204
|
+
return all(
|
|
205
|
+
alternative is candidate
|
|
206
|
+
or round(candidate.confidence - alternative.confidence, 4) >= _MIN_SCORE_LEAD
|
|
207
|
+
for alternative in alternatives
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _has_close_competitor(
|
|
212
|
+
candidate: ContinuationCandidate,
|
|
213
|
+
candidates: list[ContinuationCandidate],
|
|
214
|
+
) -> bool:
|
|
215
|
+
return any(
|
|
216
|
+
alternative is not candidate
|
|
217
|
+
and (
|
|
218
|
+
alternative.left_table_id == candidate.left_table_id
|
|
219
|
+
or alternative.right_table_id == candidate.right_table_id
|
|
220
|
+
)
|
|
221
|
+
and round(abs(candidate.confidence - alternative.confidence), 4) < _MIN_SCORE_LEAD
|
|
222
|
+
for alternative in candidates
|
|
223
|
+
)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _candidate_key(candidate: ContinuationCandidate) -> tuple[str, str]:
|
|
227
|
+
return candidate.left_table_id, candidate.right_table_id
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def _candidate_diagnostic(
|
|
231
|
+
candidate: ContinuationCandidate,
|
|
232
|
+
status: str,
|
|
233
|
+
extra_reason_codes: list[str],
|
|
234
|
+
) -> StructuredData:
|
|
235
|
+
return {
|
|
236
|
+
"left_table_id": candidate.left_table_id,
|
|
237
|
+
"right_table_id": candidate.right_table_id,
|
|
238
|
+
"left_sheet": candidate.left_sheet,
|
|
239
|
+
"right_sheet": candidate.right_sheet,
|
|
240
|
+
"confidence": candidate.confidence,
|
|
241
|
+
"status": status,
|
|
242
|
+
"reason_codes": [*candidate.reason_codes, *extra_reason_codes],
|
|
243
|
+
}
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _continuation_chains(
|
|
247
|
+
accepted_edges: list[ContinuationCandidate],
|
|
248
|
+
table_order: dict[str, int],
|
|
249
|
+
) -> list[tuple[list[str], list[ContinuationCandidate]]]:
|
|
250
|
+
successors = {edge.left_table_id: edge for edge in accepted_edges}
|
|
251
|
+
predecessor_ids = {edge.right_table_id for edge in accepted_edges}
|
|
252
|
+
heads = sorted(
|
|
253
|
+
(table_id for table_id in successors if table_id not in predecessor_ids),
|
|
254
|
+
key=table_order.__getitem__,
|
|
255
|
+
)
|
|
256
|
+
chains = []
|
|
257
|
+
for head in heads:
|
|
258
|
+
member_table_ids = [head]
|
|
259
|
+
chain_edges = []
|
|
260
|
+
current_id = head
|
|
261
|
+
while current_id in successors:
|
|
262
|
+
edge = successors[current_id]
|
|
263
|
+
chain_edges.append(edge)
|
|
264
|
+
current_id = edge.right_table_id
|
|
265
|
+
member_table_ids.append(current_id)
|
|
266
|
+
if len(member_table_ids) >= 2:
|
|
267
|
+
chains.append((member_table_ids, chain_edges))
|
|
268
|
+
return chains
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def _aggregate_table(
|
|
272
|
+
continuation_id: str,
|
|
273
|
+
member_tables: list[LogicalTable],
|
|
274
|
+
chain_edges: list[ContinuationCandidate],
|
|
275
|
+
reason_codes: list[str],
|
|
276
|
+
) -> LogicalTable:
|
|
277
|
+
copied_members = [deepcopy(table) for table in member_tables]
|
|
278
|
+
aggregate = deepcopy(member_tables[0])
|
|
279
|
+
aggregate.table_id = stable_id("table", continuation_id, "aggregate")
|
|
280
|
+
aggregate.continuation_id = None
|
|
281
|
+
aggregate.continuation_role = None
|
|
282
|
+
aggregate.columns = deepcopy(copied_members[0].columns)
|
|
283
|
+
aggregate.rows = []
|
|
284
|
+
aggregate.fragments = []
|
|
285
|
+
aggregate.sections = [
|
|
286
|
+
section for copied_member in copied_members for section in copied_member.sections
|
|
287
|
+
]
|
|
288
|
+
aggregate.source_refs = [
|
|
289
|
+
source_ref for copied_member in copied_members for source_ref in copied_member.source_refs
|
|
290
|
+
]
|
|
291
|
+
aggregate.confidence = min(
|
|
292
|
+
*(table.confidence for table in member_tables),
|
|
293
|
+
*(edge.confidence for edge in chain_edges),
|
|
294
|
+
)
|
|
295
|
+
aggregate.diagnostics = [
|
|
296
|
+
{
|
|
297
|
+
"reason_code": "cross_sheet_continuation",
|
|
298
|
+
"continuation_id": continuation_id,
|
|
299
|
+
"member_table_ids": [table.table_id for table in member_tables],
|
|
300
|
+
"reason_codes": list(reason_codes),
|
|
301
|
+
}
|
|
302
|
+
]
|
|
303
|
+
|
|
304
|
+
for member_index, copied_member in enumerate(copied_members):
|
|
305
|
+
if member_index:
|
|
306
|
+
for aggregate_column, member_column in zip(
|
|
307
|
+
aggregate.columns, copied_member.columns, strict=True
|
|
308
|
+
):
|
|
309
|
+
aggregate_column.source_refs.extend(deepcopy(member_column.source_refs))
|
|
310
|
+
aggregate.fragments.extend(copied_member.fragments)
|
|
311
|
+
|
|
312
|
+
_append_aggregate_rows(aggregate, copied_members)
|
|
313
|
+
return aggregate
|
|
314
|
+
|
|
315
|
+
|
|
316
|
+
def _append_aggregate_rows(
|
|
317
|
+
aggregate: LogicalTable,
|
|
318
|
+
copied_members: list[LogicalTable],
|
|
319
|
+
) -> None:
|
|
320
|
+
active_path: list[str] = []
|
|
321
|
+
active_section: TableSection | None = None
|
|
322
|
+
for member_index, copied_member in enumerate(copied_members):
|
|
323
|
+
member_sections_by_ref = {
|
|
324
|
+
section.source_ref.key: section for section in copied_member.sections
|
|
325
|
+
}
|
|
326
|
+
for row in copied_member.rows:
|
|
327
|
+
if member_index and row.role in _REPEATED_PRESENTATION_ROLES:
|
|
328
|
+
row.role = _REPEATED_PRESENTATION_ROLES[row.role]
|
|
329
|
+
if row.role == "section_header":
|
|
330
|
+
active_path = list(row.section_path)
|
|
331
|
+
active_section = member_sections_by_ref.get(row.source_ref.key)
|
|
332
|
+
if active_section is None:
|
|
333
|
+
active_section = _section_for_path(active_path, copied_member.sections)
|
|
334
|
+
if not active_path and active_section is not None:
|
|
335
|
+
active_path = [active_section.title]
|
|
336
|
+
elif row.section_path:
|
|
337
|
+
active_path = list(row.section_path)
|
|
338
|
+
member_section = _section_for_path(active_path, copied_member.sections)
|
|
339
|
+
if member_section is not None:
|
|
340
|
+
active_section = member_section
|
|
341
|
+
elif member_index and row.role == "data" and active_path:
|
|
342
|
+
row.section_path = list(active_path)
|
|
343
|
+
if active_section is not None and row.row_id not in active_section.row_ids:
|
|
344
|
+
active_section.row_ids.append(row.row_id)
|
|
345
|
+
aggregate.rows.append(row)
|
|
346
|
+
|
|
347
|
+
|
|
348
|
+
def _section_for_path(path: list[str], sections: list[TableSection]) -> TableSection | None:
|
|
349
|
+
if not path:
|
|
350
|
+
return None
|
|
351
|
+
return next(
|
|
352
|
+
(
|
|
353
|
+
section
|
|
354
|
+
for section in reversed(sections)
|
|
355
|
+
if [*section.parent_path, section.title] == path
|
|
356
|
+
),
|
|
357
|
+
None,
|
|
358
|
+
)
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _continuation_role(index: int, member_count: int) -> str:
|
|
362
|
+
if index == 0:
|
|
363
|
+
return "head"
|
|
364
|
+
if index == member_count - 1:
|
|
365
|
+
return "tail"
|
|
366
|
+
return "member"
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def _deduplicate_reason_codes(edges: list[ContinuationCandidate]) -> list[str]:
|
|
370
|
+
return list(dict.fromkeys(reason for edge in edges for reason in edge.reason_codes))
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def score_continuation(
|
|
374
|
+
left_sheet: SheetSnapshot,
|
|
375
|
+
left_table: LogicalTable,
|
|
376
|
+
right_sheet: SheetSnapshot,
|
|
377
|
+
right_table: LogicalTable,
|
|
378
|
+
) -> ContinuationCandidate | None:
|
|
379
|
+
"""Score the explainable evidence that two table fragments continue each other."""
|
|
380
|
+
|
|
381
|
+
if header_fingerprint(left_table) != header_fingerprint(right_table):
|
|
382
|
+
return None
|
|
383
|
+
|
|
384
|
+
left_title = _normalize_title(left_table.title)
|
|
385
|
+
right_title = _normalize_title(right_table.title)
|
|
386
|
+
terminal = _terminal_reason_codes(left_table, left_title, right_table, right_title)
|
|
387
|
+
|
|
388
|
+
score = 0.35
|
|
389
|
+
reasons = ["header_fingerprint_match"]
|
|
390
|
+
if _has_valid_page_sequence(left_table, right_table):
|
|
391
|
+
score += 0.35
|
|
392
|
+
reasons.append("print_page_sequence")
|
|
393
|
+
if left_title and left_title == right_title:
|
|
394
|
+
score += 0.25
|
|
395
|
+
reasons.append("title_match")
|
|
396
|
+
if _has_sequential_sheet_names(left_sheet.name, right_sheet.name):
|
|
397
|
+
score += 0.25
|
|
398
|
+
reasons.append("sheet_name_sequence")
|
|
399
|
+
if _has_compatible_widths(left_sheet, left_table, right_sheet, right_table):
|
|
400
|
+
score += 0.15
|
|
401
|
+
reasons.append("column_width_compatibility")
|
|
402
|
+
if _has_compatible_units(left_table, right_table):
|
|
403
|
+
score += 0.10
|
|
404
|
+
reasons.append("unit_compatibility")
|
|
405
|
+
|
|
406
|
+
return ContinuationCandidate(
|
|
407
|
+
left_sheet=left_sheet.name,
|
|
408
|
+
right_sheet=right_sheet.name,
|
|
409
|
+
left_table_id=left_table.table_id,
|
|
410
|
+
right_table_id=right_table.table_id,
|
|
411
|
+
confidence=round(min(score, 1.0), 4),
|
|
412
|
+
reason_codes=tuple(reasons),
|
|
413
|
+
terminal_reason_codes=tuple(terminal),
|
|
414
|
+
)
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def header_fingerprint(table: LogicalTable) -> tuple[tuple[str, ...], ...]:
|
|
418
|
+
"""Return the positional, normalized schema required for a continuation."""
|
|
419
|
+
|
|
420
|
+
fingerprint = []
|
|
421
|
+
for index, column in enumerate(table.columns):
|
|
422
|
+
path = tuple(_normalize_text(part) for part in column.path if _normalize_text(part))
|
|
423
|
+
fingerprint.append(path or (f"<empty:{index}>",))
|
|
424
|
+
return tuple(fingerprint)
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def _normalize_text(value: str) -> str:
|
|
428
|
+
return " ".join(unicodedata.normalize("NFKC", value).split()).casefold()
|
|
429
|
+
|
|
430
|
+
|
|
431
|
+
def _normalize_title(value: str) -> str:
|
|
432
|
+
normalized = _normalize_text(value)
|
|
433
|
+
normalized = _PAGE_MARKER_RE.sub("", normalized).strip()
|
|
434
|
+
normalized = _PARENTHESIZED_CONTINUATION_SUFFIX_RE.sub("", normalized).strip()
|
|
435
|
+
return _CONTINUATION_SUFFIX_RE.sub("", normalized).strip()
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
def _terminal_reason_codes(
|
|
439
|
+
left_table: LogicalTable,
|
|
440
|
+
left_title: str,
|
|
441
|
+
right_table: LogicalTable,
|
|
442
|
+
right_title: str,
|
|
443
|
+
) -> list[str]:
|
|
444
|
+
terminal = []
|
|
445
|
+
if _ends_with_total(left_table):
|
|
446
|
+
terminal.append("terminal_total")
|
|
447
|
+
if left_title and right_title and left_title != right_title:
|
|
448
|
+
terminal.append("title_mismatch")
|
|
449
|
+
if _page_metadata_conflicts(left_table, right_table):
|
|
450
|
+
terminal.append("page_sequence_conflict")
|
|
451
|
+
return terminal
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _ends_with_total(table: LogicalTable) -> bool:
|
|
455
|
+
for row in reversed(table.rows):
|
|
456
|
+
if row.role not in _PRESENTATION_ROLES:
|
|
457
|
+
return row.role == "total"
|
|
458
|
+
return False
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def _page_metadata(
|
|
462
|
+
left_table: LogicalTable, right_table: LogicalTable
|
|
463
|
+
) -> tuple[int, int, int] | None:
|
|
464
|
+
left_fragment = next(
|
|
465
|
+
(
|
|
466
|
+
fragment
|
|
467
|
+
for fragment in reversed(left_table.fragments)
|
|
468
|
+
if fragment.page_number is not None and fragment.total_pages is not None
|
|
469
|
+
),
|
|
470
|
+
None,
|
|
471
|
+
)
|
|
472
|
+
right_fragment = next(
|
|
473
|
+
(
|
|
474
|
+
fragment
|
|
475
|
+
for fragment in right_table.fragments
|
|
476
|
+
if fragment.page_number is not None and fragment.total_pages is not None
|
|
477
|
+
),
|
|
478
|
+
None,
|
|
479
|
+
)
|
|
480
|
+
if left_fragment is None or right_fragment is None:
|
|
481
|
+
return None
|
|
482
|
+
if left_fragment.page_number is None or left_fragment.total_pages is None:
|
|
483
|
+
return None
|
|
484
|
+
if right_fragment.page_number is None or right_fragment.total_pages is None:
|
|
485
|
+
return None
|
|
486
|
+
if left_fragment.total_pages != right_fragment.total_pages:
|
|
487
|
+
return left_fragment.page_number, right_fragment.page_number, -1
|
|
488
|
+
return left_fragment.page_number, right_fragment.page_number, left_fragment.total_pages
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def _has_valid_page_sequence(left_table: LogicalTable, right_table: LogicalTable) -> bool:
|
|
492
|
+
metadata = _page_metadata(left_table, right_table)
|
|
493
|
+
return (
|
|
494
|
+
metadata is not None
|
|
495
|
+
and 1 <= metadata[0] < metadata[2]
|
|
496
|
+
and metadata[1] == metadata[0] + 1 <= metadata[2]
|
|
497
|
+
)
|
|
498
|
+
|
|
499
|
+
|
|
500
|
+
def _page_metadata_conflicts(left_table: LogicalTable, right_table: LogicalTable) -> bool:
|
|
501
|
+
metadata = _page_metadata(left_table, right_table)
|
|
502
|
+
return metadata is not None and not (
|
|
503
|
+
1 <= metadata[0] < metadata[2] and metadata[1] == metadata[0] + 1 <= metadata[2]
|
|
504
|
+
)
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def _has_sequential_sheet_names(left_name: str, right_name: str) -> bool:
|
|
508
|
+
left = _normalize_text(left_name)
|
|
509
|
+
right = _normalize_text(right_name)
|
|
510
|
+
if right in {f"{left}续", f"{left}续表", f"{left}continued"}:
|
|
511
|
+
return True
|
|
512
|
+
left_match = _SHEET_NUMBER_RE.fullmatch(left)
|
|
513
|
+
right_match = _SHEET_NUMBER_RE.fullmatch(right)
|
|
514
|
+
return bool(
|
|
515
|
+
left_match
|
|
516
|
+
and right_match
|
|
517
|
+
and left_match.group(1)
|
|
518
|
+
and left_match.group(1) == right_match.group(1)
|
|
519
|
+
and int(right_match.group(2)) == int(left_match.group(2)) + 1
|
|
520
|
+
)
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def _has_compatible_widths(
|
|
524
|
+
left_sheet: SheetSnapshot,
|
|
525
|
+
left_table: LogicalTable,
|
|
526
|
+
right_sheet: SheetSnapshot,
|
|
527
|
+
right_table: LogicalTable,
|
|
528
|
+
) -> bool:
|
|
529
|
+
paired_columns = list(zip(left_table.columns, right_table.columns, strict=True))
|
|
530
|
+
if not paired_columns:
|
|
531
|
+
return False
|
|
532
|
+
differences = [
|
|
533
|
+
_relative_width_difference(
|
|
534
|
+
left_sheet.column_widths[left.coordinate], right_sheet.column_widths[right.coordinate]
|
|
535
|
+
)
|
|
536
|
+
for left, right in paired_columns
|
|
537
|
+
if left.coordinate in left_sheet.column_widths
|
|
538
|
+
and right.coordinate in right_sheet.column_widths
|
|
539
|
+
]
|
|
540
|
+
return len(differences) * 2 >= len(paired_columns) and median(differences) <= 0.15
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
def _relative_width_difference(left_width: float, right_width: float) -> float:
|
|
544
|
+
denominator = max(abs(left_width), abs(right_width))
|
|
545
|
+
if denominator == 0:
|
|
546
|
+
return 0.0
|
|
547
|
+
return abs(left_width - right_width) / denominator
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
def _has_compatible_units(left_table: LogicalTable, right_table: LogicalTable) -> bool:
|
|
551
|
+
paired_columns = list(zip(left_table.columns, right_table.columns, strict=True))
|
|
552
|
+
if any(left.unit or right.unit for left, right in paired_columns):
|
|
553
|
+
return any(
|
|
554
|
+
left.unit and right.unit and _normalize_text(left.unit) == _normalize_text(right.unit)
|
|
555
|
+
for left, right in paired_columns
|
|
556
|
+
)
|
|
557
|
+
return any(
|
|
558
|
+
_unit_values(left_table, index) & _unit_values(right_table, index)
|
|
559
|
+
for index, (left, right) in enumerate(paired_columns)
|
|
560
|
+
if _is_unit_column(left) and _is_unit_column(right)
|
|
561
|
+
)
|
|
562
|
+
|
|
563
|
+
|
|
564
|
+
def _is_unit_column(column: HeaderColumn) -> bool:
|
|
565
|
+
return any(
|
|
566
|
+
"单位" in _normalize_text(part) or "unit" in _normalize_text(part) for part in column.path
|
|
567
|
+
)
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
def _unit_values(table: LogicalTable, column_index: int) -> set[str]:
|
|
571
|
+
return {
|
|
572
|
+
normalized
|
|
573
|
+
for row in table.rows
|
|
574
|
+
if row.role == "data" and column_index < len(row.values)
|
|
575
|
+
for normalized in [_normalize_text(str(row.values[column_index]))]
|
|
576
|
+
if normalized
|
|
577
|
+
}
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from .evaluator import (
|
|
4
|
+
CaseEvaluationDetail,
|
|
5
|
+
CaseObservation,
|
|
6
|
+
CaseTruth,
|
|
7
|
+
RegionKind,
|
|
8
|
+
WorkbookEvaluationMetrics,
|
|
9
|
+
classify_case_evaluation,
|
|
10
|
+
evaluate_workbook_ambiguity,
|
|
11
|
+
)
|
|
12
|
+
from .schema import (
|
|
13
|
+
GoldenCase,
|
|
14
|
+
GoldenSample,
|
|
15
|
+
GoldenSetDriftError,
|
|
16
|
+
GoldenSetManifest,
|
|
17
|
+
InvalidGoldenSetError,
|
|
18
|
+
WorkbookEvaluationError,
|
|
19
|
+
compute_choices_digest,
|
|
20
|
+
compute_evaluation_id,
|
|
21
|
+
compute_sample_evaluation_id,
|
|
22
|
+
load_golden_set_manifest,
|
|
23
|
+
validate_output_dir_isolation,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
__all__ = [
|
|
27
|
+
"CaseEvaluationDetail",
|
|
28
|
+
"CaseObservation",
|
|
29
|
+
"CaseTruth",
|
|
30
|
+
"GoldenCase",
|
|
31
|
+
"GoldenSample",
|
|
32
|
+
"GoldenSetDriftError",
|
|
33
|
+
"GoldenSetManifest",
|
|
34
|
+
"InvalidGoldenSetError",
|
|
35
|
+
"RegionKind",
|
|
36
|
+
"WorkbookEvaluationError",
|
|
37
|
+
"WorkbookEvaluationMetrics",
|
|
38
|
+
"classify_case_evaluation",
|
|
39
|
+
"compute_choices_digest",
|
|
40
|
+
"compute_evaluation_id",
|
|
41
|
+
"compute_sample_evaluation_id",
|
|
42
|
+
"evaluate_workbook_ambiguity",
|
|
43
|
+
"load_golden_set_manifest",
|
|
44
|
+
"validate_output_dir_isolation",
|
|
45
|
+
]
|