langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,222 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from openpyxl.utils import get_column_letter, range_boundaries
|
|
4
|
+
|
|
5
|
+
from langparse.types import ParsedElement, ParsedPageResult
|
|
6
|
+
from langparse.workbooks.types import (
|
|
7
|
+
FormBlock,
|
|
8
|
+
LogicalRow,
|
|
9
|
+
LogicalTable,
|
|
10
|
+
MatrixBlock,
|
|
11
|
+
SheetIR,
|
|
12
|
+
SheetSnapshot,
|
|
13
|
+
TextBlock,
|
|
14
|
+
WorkbookBlock,
|
|
15
|
+
WorkbookIR,
|
|
16
|
+
WorkbookSnapshot,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def render_workbook_markdown(snapshot: WorkbookSnapshot, ir: WorkbookIR) -> str:
|
|
21
|
+
"""Render semantic logical tables while retaining source coordinates."""
|
|
22
|
+
|
|
23
|
+
ir_by_index = {sheet.index: sheet for sheet in ir.sheets}
|
|
24
|
+
sections = [
|
|
25
|
+
_render_sheet_markdown(sheet, ir_by_index.get(sheet.index)) for sheet in snapshot.sheets
|
|
26
|
+
]
|
|
27
|
+
return "\n\n".join(sections)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def compatibility_pages(
|
|
31
|
+
snapshot: WorkbookSnapshot,
|
|
32
|
+
ir: WorkbookIR,
|
|
33
|
+
) -> list[ParsedPageResult]:
|
|
34
|
+
"""Provide stable one-sheet compatibility parts for existing consumers."""
|
|
35
|
+
|
|
36
|
+
ir_by_index = {sheet.index: sheet for sheet in ir.sheets}
|
|
37
|
+
pages: list[ParsedPageResult] = []
|
|
38
|
+
for sheet in snapshot.sheets:
|
|
39
|
+
sheet_ir = ir_by_index.get(sheet.index)
|
|
40
|
+
source_range = _source_range(sheet, sheet_ir)
|
|
41
|
+
markdown = _render_sheet_markdown(sheet, sheet_ir)
|
|
42
|
+
metadata = {
|
|
43
|
+
"part_kind": "sheet",
|
|
44
|
+
"sheet_name": sheet.name,
|
|
45
|
+
"source_range": source_range,
|
|
46
|
+
}
|
|
47
|
+
tables = []
|
|
48
|
+
elements = []
|
|
49
|
+
if source_range is not None:
|
|
50
|
+
columns, row_numbers, data_rows = _sheet_grid(sheet, source_range)
|
|
51
|
+
table = {
|
|
52
|
+
"rows": [columns, *data_rows],
|
|
53
|
+
"columns": columns,
|
|
54
|
+
"row_numbers": row_numbers,
|
|
55
|
+
"sheet_name": sheet.name,
|
|
56
|
+
"source_range": source_range,
|
|
57
|
+
}
|
|
58
|
+
tables.append(table)
|
|
59
|
+
elements.append(
|
|
60
|
+
ParsedElement(
|
|
61
|
+
kind="table",
|
|
62
|
+
text=_render_grid(columns, data_rows),
|
|
63
|
+
metadata={"source_range": source_range},
|
|
64
|
+
)
|
|
65
|
+
)
|
|
66
|
+
pages.append(
|
|
67
|
+
ParsedPageResult(
|
|
68
|
+
page_number=sheet.index + 1,
|
|
69
|
+
markdown_content=markdown,
|
|
70
|
+
plain_text=markdown,
|
|
71
|
+
elements=elements,
|
|
72
|
+
tables=tables,
|
|
73
|
+
metadata=metadata,
|
|
74
|
+
)
|
|
75
|
+
)
|
|
76
|
+
return pages
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _render_sheet_markdown(sheet: SheetSnapshot, sheet_ir: SheetIR | None) -> str:
|
|
80
|
+
heading = f"## Sheet: {sheet.name}"
|
|
81
|
+
source_range = _source_range(sheet, sheet_ir)
|
|
82
|
+
if sheet_ir is not None and sheet_ir.blocks:
|
|
83
|
+
rendered = [_render_block(sheet, block) for block in sheet_ir.blocks]
|
|
84
|
+
return "\n\n".join([heading, *rendered])
|
|
85
|
+
if source_range is None:
|
|
86
|
+
return f"{heading}\n\n_Empty sheet._"
|
|
87
|
+
columns, _, data_rows = _sheet_grid(sheet, source_range)
|
|
88
|
+
source_comment = f"<!-- source_range: {sheet.name}!{source_range} -->"
|
|
89
|
+
return f"{heading}\n\n{source_comment}\n\n{_render_grid(columns, data_rows)}"
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _render_block(sheet: SheetSnapshot, block: WorkbookBlock) -> str:
|
|
93
|
+
if block.kind in {"chart", "image"}:
|
|
94
|
+
from langparse.workbooks.objects import render_object
|
|
95
|
+
|
|
96
|
+
return render_object(block)
|
|
97
|
+
if block.logical_table is not None:
|
|
98
|
+
return _render_logical_table(block.logical_table)
|
|
99
|
+
if block.form is not None:
|
|
100
|
+
return _render_form(block.form)
|
|
101
|
+
if block.matrix is not None:
|
|
102
|
+
return _render_matrix(block.matrix)
|
|
103
|
+
if block.text is not None:
|
|
104
|
+
return _render_text(block.text)
|
|
105
|
+
source_range = block.source_refs[0].range
|
|
106
|
+
columns, _, rows = _sheet_grid(sheet, source_range)
|
|
107
|
+
source_comment = f"<!-- source_range: {sheet.name}!{source_range} -->"
|
|
108
|
+
return f"{source_comment}\n\n{_render_grid(columns, rows)}"
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _render_logical_table(table: LogicalTable) -> str:
|
|
112
|
+
source_ranges = ", ".join(source_ref.key for source_ref in table.source_refs)
|
|
113
|
+
parts = [f"<!-- source_ranges: {source_ranges} -->"]
|
|
114
|
+
if table.continuation_id is not None:
|
|
115
|
+
parts.append(
|
|
116
|
+
f"<!-- continuation_id: {table.continuation_id}; role: {table.continuation_role} -->"
|
|
117
|
+
)
|
|
118
|
+
if table.title:
|
|
119
|
+
parts.append(f"### Table: {table.title}")
|
|
120
|
+
if table.context:
|
|
121
|
+
parts.append("\n".join(f"> {line}" for line in table.context))
|
|
122
|
+
|
|
123
|
+
columns = [" / ".join(column.path) or column.coordinate for column in table.columns]
|
|
124
|
+
eligible_rows = [row for row in table.rows if row.role in {"data", "total", "unknown"}]
|
|
125
|
+
groups = _group_rows_by_section(eligible_rows)
|
|
126
|
+
for section_path, rows in groups:
|
|
127
|
+
if section_path:
|
|
128
|
+
parts.append(f"#### Section: {' / '.join(section_path)}")
|
|
129
|
+
parts.append(_render_grid(columns, [row.values for row in rows]))
|
|
130
|
+
if not groups:
|
|
131
|
+
parts.append("_No semantic data rows._")
|
|
132
|
+
return "\n\n".join(parts)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _render_form(form: FormBlock) -> str:
|
|
136
|
+
source_ranges = ", ".join(source_ref.key for source_ref in form.source_refs)
|
|
137
|
+
parts = [f"<!-- source_ranges: {source_ranges} -->"]
|
|
138
|
+
if form.title:
|
|
139
|
+
parts.append(f"### Form: {form.title}")
|
|
140
|
+
if form.fields:
|
|
141
|
+
parts.append(
|
|
142
|
+
_render_grid(["Field", "Value"], [[field.label, field.value] for field in form.fields])
|
|
143
|
+
)
|
|
144
|
+
parts.extend(line.text for line in form.free_text)
|
|
145
|
+
return "\n\n".join(parts)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _render_matrix(matrix: MatrixBlock) -> str:
|
|
149
|
+
source_ranges = ", ".join(source_ref.key for source_ref in matrix.source_refs)
|
|
150
|
+
parts = [f"<!-- source_ranges: {source_ranges} -->"]
|
|
151
|
+
if matrix.title:
|
|
152
|
+
parts.append(f"### Matrix: {matrix.title}")
|
|
153
|
+
columns = ["", *[header.value for header in matrix.column_headers]]
|
|
154
|
+
rows = [
|
|
155
|
+
[header.value, *values]
|
|
156
|
+
for header, values in zip(matrix.row_headers, matrix.values, strict=True)
|
|
157
|
+
]
|
|
158
|
+
parts.append(_render_grid(columns, rows))
|
|
159
|
+
return "\n\n".join(parts)
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def _render_text(text: TextBlock) -> str:
|
|
163
|
+
source_ranges = ", ".join(source_ref.key for source_ref in text.source_refs)
|
|
164
|
+
return "\n\n".join(
|
|
165
|
+
[f"<!-- source_ranges: {source_ranges} -->", *[line.text for line in text.lines]]
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _group_rows_by_section(rows: list[LogicalRow]) -> list[tuple[list[str], list[LogicalRow]]]:
|
|
170
|
+
groups: list[tuple[list[str], list[LogicalRow]]] = []
|
|
171
|
+
for row in rows:
|
|
172
|
+
if not groups or groups[-1][0] != row.section_path:
|
|
173
|
+
groups.append((list(row.section_path), []))
|
|
174
|
+
groups[-1][1].append(row)
|
|
175
|
+
return groups
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _source_range(sheet: SheetSnapshot, sheet_ir: SheetIR | None) -> str | None:
|
|
179
|
+
if sheet.used_range is not None:
|
|
180
|
+
return sheet.used_range
|
|
181
|
+
if sheet_ir is not None:
|
|
182
|
+
for block in sheet_ir.blocks:
|
|
183
|
+
if block.kind not in {"chart", "image"} and block.source_refs:
|
|
184
|
+
return block.source_refs[0].range
|
|
185
|
+
return None
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _sheet_grid(
|
|
189
|
+
sheet: SheetSnapshot,
|
|
190
|
+
source_range: str,
|
|
191
|
+
) -> tuple[list[str], list[int], list[list[str]]]:
|
|
192
|
+
min_col, min_row, max_col, max_row = range_boundaries(source_range)
|
|
193
|
+
columns = [get_column_letter(column) for column in range(min_col, max_col + 1)]
|
|
194
|
+
row_numbers = list(range(min_row, max_row + 1))
|
|
195
|
+
rows: list[list[str]] = []
|
|
196
|
+
for row_number in row_numbers:
|
|
197
|
+
row: list[str] = []
|
|
198
|
+
for column in columns:
|
|
199
|
+
cell = sheet.cells.get(f"{column}{row_number}")
|
|
200
|
+
if cell is None or cell.merge_anchor is not None:
|
|
201
|
+
row.append("")
|
|
202
|
+
else:
|
|
203
|
+
row.append(cell.display_value)
|
|
204
|
+
rows.append(row)
|
|
205
|
+
return columns, row_numbers, rows
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _render_grid(columns: list[str], rows: list[list[str]]) -> str:
|
|
209
|
+
header = "| " + " | ".join(_escape_markdown(value) for value in columns) + " |"
|
|
210
|
+
separator = "| " + " | ".join("---" for _ in columns) + " |"
|
|
211
|
+
body = ["| " + " | ".join(_escape_markdown(value) for value in row) + " |" for row in rows]
|
|
212
|
+
return "\n".join([header, separator, *body])
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def _escape_markdown(value: str) -> str:
|
|
216
|
+
return (
|
|
217
|
+
str(value)
|
|
218
|
+
.replace("\r\n", "\n")
|
|
219
|
+
.replace("\r", "\n")
|
|
220
|
+
.replace("|", r"\|")
|
|
221
|
+
.replace("\n", "<br>")
|
|
222
|
+
)
|
|
@@ -0,0 +1,477 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
from openpyxl.utils import get_column_letter, range_boundaries
|
|
6
|
+
|
|
7
|
+
from langparse.workbooks.labels import is_section_label, is_total_label
|
|
8
|
+
from langparse.workbooks.types import (
|
|
9
|
+
CandidateRegion,
|
|
10
|
+
HeaderColumn,
|
|
11
|
+
LogicalRow,
|
|
12
|
+
LogicalTable,
|
|
13
|
+
SheetSnapshot,
|
|
14
|
+
SourceRef,
|
|
15
|
+
TableFragment,
|
|
16
|
+
TableSection,
|
|
17
|
+
stable_id,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
PAGE_RE = re.compile(r"第\s*(\d+)\s*页\s*共\s*(\d+)\s*页")
|
|
21
|
+
ROW_NUMBER_RE = re.compile(r"[1-9]\d*(?:\.\d+)*")
|
|
22
|
+
CONTEXT_LABEL_RE = re.compile(
|
|
23
|
+
r"^(?:工程名称|单位工程名称|项目名称|编制单位|编制日期|建设单位|单位|币种)\s*[::]\s*\S"
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def interpret_logical_table(
|
|
28
|
+
sheet: SheetSnapshot,
|
|
29
|
+
candidate: CandidateRegion,
|
|
30
|
+
) -> LogicalTable:
|
|
31
|
+
"""Interpret deterministic print fragments and their shared header."""
|
|
32
|
+
|
|
33
|
+
min_col, min_row, max_col, max_row = range_boundaries(candidate.source_ref.range)
|
|
34
|
+
page_markers = _page_markers(sheet, min_row, max_row, min_col, max_col)
|
|
35
|
+
fragments = _build_fragments(
|
|
36
|
+
sheet,
|
|
37
|
+
candidate,
|
|
38
|
+
page_markers,
|
|
39
|
+
min_col,
|
|
40
|
+
min_row,
|
|
41
|
+
max_col,
|
|
42
|
+
max_row,
|
|
43
|
+
)
|
|
44
|
+
header_rows = fragments[0].header_row_numbers if fragments else []
|
|
45
|
+
columns = _build_header_columns(
|
|
46
|
+
sheet,
|
|
47
|
+
candidate.source_ref,
|
|
48
|
+
header_rows,
|
|
49
|
+
min_col,
|
|
50
|
+
max_col,
|
|
51
|
+
)
|
|
52
|
+
title_rows = fragments[0].title_row_numbers if fragments else []
|
|
53
|
+
title = _first_text(sheet, title_rows[0], min_col, max_col) if title_rows else ""
|
|
54
|
+
context = [
|
|
55
|
+
text
|
|
56
|
+
for row_number in fragments[0].context_row_numbers
|
|
57
|
+
for text in [_row_text(sheet, row_number, min_col, max_col)]
|
|
58
|
+
if text
|
|
59
|
+
]
|
|
60
|
+
rows, sections = _build_rows_and_sections(
|
|
61
|
+
sheet,
|
|
62
|
+
candidate,
|
|
63
|
+
fragments,
|
|
64
|
+
min_col,
|
|
65
|
+
min_row,
|
|
66
|
+
max_col,
|
|
67
|
+
max_row,
|
|
68
|
+
)
|
|
69
|
+
return LogicalTable(
|
|
70
|
+
table_id=stable_id("table", candidate.source_ref.key),
|
|
71
|
+
title=title,
|
|
72
|
+
context=context,
|
|
73
|
+
columns=columns,
|
|
74
|
+
rows=rows,
|
|
75
|
+
fragments=fragments,
|
|
76
|
+
sections=sections,
|
|
77
|
+
source_refs=[candidate.source_ref],
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def _build_rows_and_sections(
|
|
82
|
+
sheet: SheetSnapshot,
|
|
83
|
+
candidate: CandidateRegion,
|
|
84
|
+
fragments: list[TableFragment],
|
|
85
|
+
min_col: int,
|
|
86
|
+
min_row: int,
|
|
87
|
+
max_col: int,
|
|
88
|
+
max_row: int,
|
|
89
|
+
) -> tuple[list[LogicalRow], list[TableSection]]:
|
|
90
|
+
roles: dict[int, str] = {}
|
|
91
|
+
for fragment_index, fragment in enumerate(fragments):
|
|
92
|
+
for row_number in fragment.title_row_numbers:
|
|
93
|
+
roles[row_number] = "title" if fragment_index == 0 else "repeated_title"
|
|
94
|
+
for row_number in fragment.context_row_numbers:
|
|
95
|
+
roles[row_number] = "context" if fragment_index == 0 else "repeated_context"
|
|
96
|
+
for row_number in fragment.header_row_numbers:
|
|
97
|
+
roles[row_number] = "header" if fragment_index == 0 else "repeated_header"
|
|
98
|
+
|
|
99
|
+
header_rows = fragments[0].header_row_numbers if fragments else []
|
|
100
|
+
ordinal_column = any(
|
|
101
|
+
_cell_text(sheet, row, min_col).strip() in {"序号", "编号"} for row in header_rows
|
|
102
|
+
)
|
|
103
|
+
logical_rows: list[LogicalRow] = []
|
|
104
|
+
sections: list[TableSection] = []
|
|
105
|
+
current_section: TableSection | None = None
|
|
106
|
+
for row_number in range(min_row, max_row + 1):
|
|
107
|
+
values = [
|
|
108
|
+
_raw_cell_text(sheet, row_number, column) for column in range(min_col, max_col + 1)
|
|
109
|
+
]
|
|
110
|
+
first_cell = sheet.cells.get(f"{get_column_letter(min_col)}{row_number}")
|
|
111
|
+
role = roles.get(row_number) or _classify_content_row(
|
|
112
|
+
values,
|
|
113
|
+
merged_section=bool(first_cell and first_cell.colspan > 1),
|
|
114
|
+
ordinal_column=ordinal_column,
|
|
115
|
+
)
|
|
116
|
+
row_range = (
|
|
117
|
+
f"{get_column_letter(min_col)}{row_number}:{get_column_letter(max_col)}{row_number}"
|
|
118
|
+
)
|
|
119
|
+
source_ref = SourceRef(sheet_name=sheet.name, range=row_range)
|
|
120
|
+
row_id = stable_id("row", candidate.source_ref.key, str(row_number))
|
|
121
|
+
if role == "section_header":
|
|
122
|
+
title = _section_title(values)
|
|
123
|
+
current_section = TableSection(
|
|
124
|
+
section_id=stable_id("section", candidate.source_ref.key, str(row_number), title),
|
|
125
|
+
title=title,
|
|
126
|
+
source_ref=source_ref,
|
|
127
|
+
)
|
|
128
|
+
sections.append(current_section)
|
|
129
|
+
section_path = [current_section.title] if current_section is not None else []
|
|
130
|
+
source_cells = [
|
|
131
|
+
f"{get_column_letter(column)}{row_number}"
|
|
132
|
+
for column in range(min_col, max_col + 1)
|
|
133
|
+
if f"{get_column_letter(column)}{row_number}" in sheet.cells
|
|
134
|
+
]
|
|
135
|
+
logical_row = LogicalRow(
|
|
136
|
+
row_id=row_id,
|
|
137
|
+
source_ref=source_ref,
|
|
138
|
+
role=role,
|
|
139
|
+
values=values,
|
|
140
|
+
source_cells=source_cells,
|
|
141
|
+
section_path=section_path,
|
|
142
|
+
metadata={"row_number": row_number},
|
|
143
|
+
)
|
|
144
|
+
logical_rows.append(logical_row)
|
|
145
|
+
if current_section is not None and role in {"data", "total"}:
|
|
146
|
+
current_section.row_ids.append(row_id)
|
|
147
|
+
return logical_rows, sections
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _classify_content_row(
|
|
151
|
+
values: list[str], *, merged_section: bool = False, ordinal_column: bool = False
|
|
152
|
+
) -> str:
|
|
153
|
+
nonempty = [value.strip() for value in values if value.strip()]
|
|
154
|
+
first = values[0].strip() if values else ""
|
|
155
|
+
# A total is identified by its leading label, never by a word embedded in
|
|
156
|
+
# arbitrary description columns. A numbered record remains a record.
|
|
157
|
+
if nonempty and is_total_label(nonempty[0]):
|
|
158
|
+
return "total"
|
|
159
|
+
if merged_section and is_section_label(first):
|
|
160
|
+
return "section_header"
|
|
161
|
+
# Printed budget sections use empty/zero code columns followed by a title
|
|
162
|
+
# and amounts. A blank merged product cell plus an item/unit is a detail.
|
|
163
|
+
if ordinal_column and first in {"", "0"}:
|
|
164
|
+
if _section_title(values) and any(
|
|
165
|
+
_is_number(value) and float(value) != 0 for value in values[3:]
|
|
166
|
+
):
|
|
167
|
+
return "section_header"
|
|
168
|
+
if ROW_NUMBER_RE.fullmatch(first):
|
|
169
|
+
return "data"
|
|
170
|
+
if ordinal_column:
|
|
171
|
+
# Preserve non-record annotations such as “其中” in numbered schedules.
|
|
172
|
+
return "data" if len(values) > 1 and any(c.isdigit() for c in values[1]) else "unknown"
|
|
173
|
+
if len(nonempty) >= 2:
|
|
174
|
+
return "data"
|
|
175
|
+
return "unknown"
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _section_title(values: list[str]) -> str:
|
|
179
|
+
if values and is_section_label(values[0]):
|
|
180
|
+
return values[0].strip()
|
|
181
|
+
for value in values[1:6]:
|
|
182
|
+
text = value.strip()
|
|
183
|
+
if text and not _is_number(text):
|
|
184
|
+
return text
|
|
185
|
+
return ""
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _is_number(value: str) -> bool:
|
|
189
|
+
try:
|
|
190
|
+
float(value)
|
|
191
|
+
except (TypeError, ValueError):
|
|
192
|
+
return False
|
|
193
|
+
return True
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _page_markers(
|
|
197
|
+
sheet: SheetSnapshot,
|
|
198
|
+
min_row: int,
|
|
199
|
+
max_row: int,
|
|
200
|
+
min_col: int,
|
|
201
|
+
max_col: int,
|
|
202
|
+
) -> list[tuple[int, int, int]]:
|
|
203
|
+
markers = []
|
|
204
|
+
for row_number in range(min_row, max_row + 1):
|
|
205
|
+
match = PAGE_RE.search(_row_text(sheet, row_number, min_col, max_col))
|
|
206
|
+
if match:
|
|
207
|
+
markers.append((row_number, int(match.group(1)), int(match.group(2))))
|
|
208
|
+
return markers
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _build_fragments(
|
|
212
|
+
sheet: SheetSnapshot,
|
|
213
|
+
candidate: CandidateRegion,
|
|
214
|
+
page_markers: list[tuple[int, int, int]],
|
|
215
|
+
min_col: int,
|
|
216
|
+
min_row: int,
|
|
217
|
+
max_col: int,
|
|
218
|
+
max_row: int,
|
|
219
|
+
) -> list[TableFragment]:
|
|
220
|
+
if len(page_markers) == 1:
|
|
221
|
+
context_row, page_number, total_pages = page_markers[0]
|
|
222
|
+
title_rows, context_rows = _print_preamble(sheet, min_row, context_row, min_col, max_col)
|
|
223
|
+
header_rows = _header_rows_after(
|
|
224
|
+
sheet, context_row, context_row + 1, max_row, min_col, max_col
|
|
225
|
+
)
|
|
226
|
+
return [
|
|
227
|
+
TableFragment(
|
|
228
|
+
fragment_id=stable_id("fragment", candidate.source_ref.key, str(page_number)),
|
|
229
|
+
source_ref=candidate.source_ref,
|
|
230
|
+
page_number=page_number,
|
|
231
|
+
total_pages=total_pages,
|
|
232
|
+
title_row_numbers=title_rows,
|
|
233
|
+
context_row_numbers=context_rows,
|
|
234
|
+
header_row_numbers=header_rows,
|
|
235
|
+
diagnostics=[{"reason_code": "single_print_fragment"}],
|
|
236
|
+
)
|
|
237
|
+
]
|
|
238
|
+
|
|
239
|
+
if not _valid_page_sequence(page_markers):
|
|
240
|
+
title_rows, context_rows = (
|
|
241
|
+
([], [])
|
|
242
|
+
if "native_table_anchor" in candidate.reason_codes
|
|
243
|
+
else _unpaged_preamble(sheet, min_row, max_row, min_col, max_col)
|
|
244
|
+
)
|
|
245
|
+
header_start = context_rows[-1] + 1 if context_rows else min_row
|
|
246
|
+
header_rows = _header_rows_after(
|
|
247
|
+
sheet, header_start - 1, header_start, max_row, min_col, max_col
|
|
248
|
+
)
|
|
249
|
+
return [
|
|
250
|
+
TableFragment(
|
|
251
|
+
fragment_id=stable_id("fragment", candidate.source_ref.key),
|
|
252
|
+
source_ref=candidate.source_ref,
|
|
253
|
+
title_row_numbers=title_rows,
|
|
254
|
+
context_row_numbers=context_rows,
|
|
255
|
+
header_row_numbers=header_rows,
|
|
256
|
+
diagnostics=[{"reason_code": "no_consistent_print_sequence"}],
|
|
257
|
+
)
|
|
258
|
+
]
|
|
259
|
+
|
|
260
|
+
starts = [max(min_row, row_number - 1) for row_number, _, _ in page_markers]
|
|
261
|
+
fragments: list[TableFragment] = []
|
|
262
|
+
header_fingerprints: list[tuple[tuple[str, ...], ...]] = []
|
|
263
|
+
for index, ((context_row, page_number, total_pages), start_row) in enumerate(
|
|
264
|
+
zip(page_markers, starts, strict=True)
|
|
265
|
+
):
|
|
266
|
+
end_row = starts[index + 1] - 1 if index + 1 < len(starts) else max_row
|
|
267
|
+
header_rows = _header_rows_after(
|
|
268
|
+
sheet,
|
|
269
|
+
context_row,
|
|
270
|
+
start_row,
|
|
271
|
+
end_row,
|
|
272
|
+
min_col,
|
|
273
|
+
max_col,
|
|
274
|
+
)
|
|
275
|
+
header_fingerprints.append(
|
|
276
|
+
tuple(
|
|
277
|
+
tuple(
|
|
278
|
+
_cell_text(sheet, row_number, column) for column in range(min_col, max_col + 1)
|
|
279
|
+
)
|
|
280
|
+
for row_number in header_rows
|
|
281
|
+
)
|
|
282
|
+
)
|
|
283
|
+
source_range = (
|
|
284
|
+
f"{get_column_letter(min_col)}{start_row}:{get_column_letter(max_col)}{end_row}"
|
|
285
|
+
)
|
|
286
|
+
fragments.append(
|
|
287
|
+
TableFragment(
|
|
288
|
+
fragment_id=stable_id("fragment", candidate.source_ref.key, str(page_number)),
|
|
289
|
+
source_ref=SourceRef(sheet_name=sheet.name, range=source_range),
|
|
290
|
+
page_number=page_number,
|
|
291
|
+
total_pages=total_pages,
|
|
292
|
+
title_row_numbers=[start_row] if start_row < context_row else [],
|
|
293
|
+
context_row_numbers=[context_row],
|
|
294
|
+
header_row_numbers=header_rows,
|
|
295
|
+
)
|
|
296
|
+
)
|
|
297
|
+
|
|
298
|
+
if len(set(header_fingerprints)) != 1:
|
|
299
|
+
diagnostic = {"reason_code": "header_fingerprint_mismatch"}
|
|
300
|
+
for fragment in fragments:
|
|
301
|
+
fragment.confidence = 0.5
|
|
302
|
+
fragment.diagnostics.append(diagnostic)
|
|
303
|
+
return fragments
|
|
304
|
+
|
|
305
|
+
|
|
306
|
+
def _valid_page_sequence(markers: list[tuple[int, int, int]]) -> bool:
|
|
307
|
+
if len(markers) < 2:
|
|
308
|
+
return False
|
|
309
|
+
pages = [page for _, page, _ in markers]
|
|
310
|
+
totals = {total for _, _, total in markers}
|
|
311
|
+
return len(totals) == 1 and pages == list(range(pages[0], pages[0] + len(pages)))
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
def _is_context_row(sheet: SheetSnapshot, row: int, min_col: int, max_col: int) -> bool:
|
|
315
|
+
values = [
|
|
316
|
+
text.strip()
|
|
317
|
+
for column in range(min_col, max_col + 1)
|
|
318
|
+
if (text := _raw_cell_text(sheet, row, column)).strip()
|
|
319
|
+
]
|
|
320
|
+
return bool(values) and all(
|
|
321
|
+
CONTEXT_LABEL_RE.match(value) or PAGE_RE.search(value) for value in values
|
|
322
|
+
)
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def _print_preamble(
|
|
326
|
+
sheet: SheetSnapshot, start: int, marker_row: int, min_col: int, max_col: int
|
|
327
|
+
) -> tuple[list[int], list[int]]:
|
|
328
|
+
# A page marker anchors the preamble; keep preceding labelled context as well.
|
|
329
|
+
contexts = [marker_row]
|
|
330
|
+
for row in range(marker_row - 1, start - 1, -1):
|
|
331
|
+
if not _is_context_row(sheet, row, min_col, max_col):
|
|
332
|
+
break
|
|
333
|
+
contexts.insert(0, row)
|
|
334
|
+
titles = [start] if start < contexts[0] else []
|
|
335
|
+
return titles, contexts
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def _unpaged_preamble(
|
|
339
|
+
sheet: SheetSnapshot, start: int, end: int, min_col: int, max_col: int
|
|
340
|
+
) -> tuple[list[int], list[int]]:
|
|
341
|
+
# A sparse title alone is ambiguous with a merged parent header. Require
|
|
342
|
+
# explicit labelled context immediately afterwards before assigning ownership.
|
|
343
|
+
first_values = [
|
|
344
|
+
_raw_cell_text(sheet, start, column).strip() for column in range(min_col, max_col + 1)
|
|
345
|
+
]
|
|
346
|
+
title = (
|
|
347
|
+
not _is_context_row(sheet, start, min_col, max_col)
|
|
348
|
+
and sum(bool(value) for value in first_values) == 1
|
|
349
|
+
)
|
|
350
|
+
context_start = start + 1 if title else start
|
|
351
|
+
contexts = []
|
|
352
|
+
for row in range(context_start, end + 1):
|
|
353
|
+
if not _is_context_row(sheet, row, min_col, max_col):
|
|
354
|
+
break
|
|
355
|
+
contexts.append(row)
|
|
356
|
+
if not contexts or contexts[-1] == end:
|
|
357
|
+
return [], []
|
|
358
|
+
return [start] if title else [], contexts
|
|
359
|
+
|
|
360
|
+
|
|
361
|
+
def _header_rows_after(
|
|
362
|
+
sheet: SheetSnapshot,
|
|
363
|
+
context_row: int,
|
|
364
|
+
start_row: int,
|
|
365
|
+
end_row: int,
|
|
366
|
+
min_col: int,
|
|
367
|
+
max_col: int,
|
|
368
|
+
) -> list[int]:
|
|
369
|
+
rows = []
|
|
370
|
+
for row_number in range(max(start_row, context_row + 1), end_row + 1):
|
|
371
|
+
if _looks_like_data_or_section(sheet, row_number, min_col, max_col):
|
|
372
|
+
break
|
|
373
|
+
if rows and _flat_header(sheet, rows[-1], min_col, max_col):
|
|
374
|
+
# A complete flat schema also supports text-only records. Do not
|
|
375
|
+
# consume the entire body while waiting for a numeric identifier.
|
|
376
|
+
break
|
|
377
|
+
if _row_text(sheet, row_number, min_col, max_col):
|
|
378
|
+
rows.append(row_number)
|
|
379
|
+
return rows
|
|
380
|
+
|
|
381
|
+
|
|
382
|
+
def _flat_header(sheet: SheetSnapshot, row: int, min_col: int, max_col: int) -> bool:
|
|
383
|
+
cells = [
|
|
384
|
+
sheet.cells.get(f"{get_column_letter(col)}{row}") for col in range(min_col, max_col + 1)
|
|
385
|
+
]
|
|
386
|
+
return len(cells) >= 2 and all(
|
|
387
|
+
cell is not None
|
|
388
|
+
and cell.display_value.strip()
|
|
389
|
+
and not cell.merge_anchor
|
|
390
|
+
and cell.colspan == 1
|
|
391
|
+
and cell.rowspan == 1
|
|
392
|
+
for cell in cells
|
|
393
|
+
)
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _looks_like_data_or_section(
|
|
397
|
+
sheet: SheetSnapshot,
|
|
398
|
+
row_number: int,
|
|
399
|
+
min_col: int,
|
|
400
|
+
max_col: int,
|
|
401
|
+
) -> bool:
|
|
402
|
+
first = _cell_text(sheet, row_number, min_col).strip()
|
|
403
|
+
if re.fullmatch(r"\d+", first) or ROW_NUMBER_RE.fullmatch(first):
|
|
404
|
+
return True
|
|
405
|
+
values = [_raw_cell_text(sheet, row_number, column) for column in range(min_col, max_col + 1)]
|
|
406
|
+
if any(_is_number(value) for value in values) and any(
|
|
407
|
+
value.strip() and not _is_number(value) for value in values
|
|
408
|
+
):
|
|
409
|
+
return True
|
|
410
|
+
first_cell = sheet.cells.get(f"{get_column_letter(min_col)}{row_number}")
|
|
411
|
+
return is_total_label(first) or bool(
|
|
412
|
+
first_cell and first_cell.colspan > 1 and is_section_label(first)
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def _build_header_columns(
|
|
417
|
+
sheet: SheetSnapshot,
|
|
418
|
+
candidate_ref: SourceRef,
|
|
419
|
+
header_rows: list[int],
|
|
420
|
+
min_col: int,
|
|
421
|
+
max_col: int,
|
|
422
|
+
) -> list[HeaderColumn]:
|
|
423
|
+
columns = []
|
|
424
|
+
for column in range(min_col, max_col + 1):
|
|
425
|
+
coordinate = get_column_letter(column)
|
|
426
|
+
path = []
|
|
427
|
+
source_refs = []
|
|
428
|
+
for row_number in header_rows:
|
|
429
|
+
text = _cell_text(sheet, row_number, column).strip()
|
|
430
|
+
if text and (not path or path[-1] != text):
|
|
431
|
+
path.append(text)
|
|
432
|
+
source_refs.append(SourceRef(sheet_name=sheet.name, range=f"{coordinate}{row_number}"))
|
|
433
|
+
columns.append(
|
|
434
|
+
HeaderColumn(
|
|
435
|
+
column_id=stable_id("column", candidate_ref.key, coordinate),
|
|
436
|
+
coordinate=coordinate,
|
|
437
|
+
path=path,
|
|
438
|
+
source_refs=source_refs,
|
|
439
|
+
)
|
|
440
|
+
)
|
|
441
|
+
return columns
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _cell_text(sheet: SheetSnapshot, row_number: int, column: int) -> str:
|
|
445
|
+
coordinate = f"{get_column_letter(column)}{row_number}"
|
|
446
|
+
cell = sheet.cells.get(coordinate)
|
|
447
|
+
if cell is None:
|
|
448
|
+
return ""
|
|
449
|
+
if cell.merge_anchor:
|
|
450
|
+
anchor = sheet.cells.get(cell.merge_anchor)
|
|
451
|
+
return anchor.display_value if anchor is not None else ""
|
|
452
|
+
return cell.display_value
|
|
453
|
+
|
|
454
|
+
|
|
455
|
+
def _raw_cell_text(sheet: SheetSnapshot, row_number: int, column: int) -> str:
|
|
456
|
+
coordinate = f"{get_column_letter(column)}{row_number}"
|
|
457
|
+
cell = sheet.cells.get(coordinate)
|
|
458
|
+
if cell is None or cell.merge_anchor is not None:
|
|
459
|
+
return ""
|
|
460
|
+
return cell.display_value
|
|
461
|
+
|
|
462
|
+
|
|
463
|
+
def _row_text(sheet: SheetSnapshot, row_number: int, min_col: int, max_col: int) -> str:
|
|
464
|
+
return " | ".join(
|
|
465
|
+
text
|
|
466
|
+
for column in range(min_col, max_col + 1)
|
|
467
|
+
for text in [_cell_text(sheet, row_number, column).strip()]
|
|
468
|
+
if text
|
|
469
|
+
)
|
|
470
|
+
|
|
471
|
+
|
|
472
|
+
def _first_text(sheet: SheetSnapshot, row_number: int, min_col: int, max_col: int) -> str:
|
|
473
|
+
for column in range(min_col, max_col + 1):
|
|
474
|
+
text = _cell_text(sheet, row_number, column).strip()
|
|
475
|
+
if text:
|
|
476
|
+
return text
|
|
477
|
+
return ""
|