langparse 0.1.0rc1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +35 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +0 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/workbook.py +900 -0
- langparse/cli.py +257 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +202 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +140 -0
- langparse/engines/pdf/mineru.py +235 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +127 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +52 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +190 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +5 -0
- langparse/services/batch_service.py +253 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +468 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/types.py +94 -0
- langparse/workbooks/__init__.py +97 -0
- langparse/workbooks/adapters.py +309 -0
- langparse/workbooks/assembly.py +928 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/classification.py +381 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/regions.py +77 -0
- langparse/workbooks/rendering.py +216 -0
- langparse/workbooks/tables.py +375 -0
- langparse/workbooks/types.py +239 -0
- langparse-0.1.0rc1.dist-info/METADATA +720 -0
- langparse-0.1.0rc1.dist-info/RECORD +85 -0
- langparse-0.1.0rc1.dist-info/WHEEL +5 -0
- langparse-0.1.0rc1.dist-info/entry_points.txt +2 -0
- langparse-0.1.0rc1.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0rc1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,900 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Callable
|
|
4
|
+
|
|
5
|
+
from openpyxl.utils import get_column_letter, range_boundaries
|
|
6
|
+
|
|
7
|
+
from langparse.chunkers.profiles import (
|
|
8
|
+
WorkbookChunkPolicy,
|
|
9
|
+
WorkbookChunkProfile,
|
|
10
|
+
resolve_workbook_chunk_policy,
|
|
11
|
+
)
|
|
12
|
+
from langparse.core.rendering import document_metadata
|
|
13
|
+
from langparse.types import Chunk, ParsedDocumentResult
|
|
14
|
+
from langparse.workbooks.types import (
|
|
15
|
+
FormBlock,
|
|
16
|
+
FormField,
|
|
17
|
+
LogicalRow,
|
|
18
|
+
LogicalTable,
|
|
19
|
+
MatrixBlock,
|
|
20
|
+
MatrixHeader,
|
|
21
|
+
TableContinuation,
|
|
22
|
+
TextBlock,
|
|
23
|
+
TextLine,
|
|
24
|
+
WorkbookBlock,
|
|
25
|
+
WorkbookIR,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class WorkbookStructuralChunker:
|
|
30
|
+
"""Pack complete raw-grid rows directly from workbook compatibility facts."""
|
|
31
|
+
|
|
32
|
+
def __init__(
|
|
33
|
+
self,
|
|
34
|
+
max_chunk_size: int | None = None,
|
|
35
|
+
length_function: Callable[[str], int] = len,
|
|
36
|
+
*,
|
|
37
|
+
profile: str | WorkbookChunkProfile | None = None,
|
|
38
|
+
):
|
|
39
|
+
self.policy: WorkbookChunkPolicy = resolve_workbook_chunk_policy(profile)
|
|
40
|
+
resolved_size = (
|
|
41
|
+
self.policy.default_max_chunk_size if max_chunk_size is None else max_chunk_size
|
|
42
|
+
)
|
|
43
|
+
if resolved_size <= 0:
|
|
44
|
+
raise ValueError("max_chunk_size must be positive")
|
|
45
|
+
self.max_chunk_size = resolved_size
|
|
46
|
+
self.length_function = length_function
|
|
47
|
+
|
|
48
|
+
def chunk(self, parsed: ParsedDocumentResult) -> list[Chunk]:
|
|
49
|
+
if not isinstance(parsed.structure, WorkbookIR):
|
|
50
|
+
raise TypeError("WorkbookStructuralChunker requires WorkbookIR structure")
|
|
51
|
+
|
|
52
|
+
ir_sheets = {sheet.index: sheet for sheet in parsed.structure.sheets}
|
|
53
|
+
chunks: list[Chunk] = []
|
|
54
|
+
for page in parsed.pages:
|
|
55
|
+
sheet_name = page.metadata.get("sheet_name")
|
|
56
|
+
sheet_ir = ir_sheets.get(page.page_number - 1)
|
|
57
|
+
if sheet_ir is not None and sheet_ir.blocks:
|
|
58
|
+
for block in sheet_ir.blocks:
|
|
59
|
+
chunks.extend(
|
|
60
|
+
self._chunk_block(
|
|
61
|
+
parsed,
|
|
62
|
+
str(sheet_name),
|
|
63
|
+
page.page_number,
|
|
64
|
+
block,
|
|
65
|
+
len(chunks),
|
|
66
|
+
)
|
|
67
|
+
)
|
|
68
|
+
continue
|
|
69
|
+
confidence = (
|
|
70
|
+
sheet_ir.blocks[0].confidence if sheet_ir is not None and sheet_ir.blocks else 1.0
|
|
71
|
+
)
|
|
72
|
+
for table in page.tables:
|
|
73
|
+
rows = table.get("rows", [])
|
|
74
|
+
if not rows:
|
|
75
|
+
continue
|
|
76
|
+
columns = [str(value) for value in table.get("columns") or rows[0]]
|
|
77
|
+
data_rows = [[str(value) for value in row] for row in rows[1:]]
|
|
78
|
+
row_numbers = list(table.get("row_numbers", []))
|
|
79
|
+
if len(row_numbers) != len(data_rows):
|
|
80
|
+
row_numbers = _row_numbers(table.get("source_range"), len(data_rows))
|
|
81
|
+
chunks.extend(
|
|
82
|
+
self._pack_table(
|
|
83
|
+
parsed=parsed,
|
|
84
|
+
sheet_name=str(sheet_name),
|
|
85
|
+
sheet_ordinal=page.page_number,
|
|
86
|
+
columns=columns,
|
|
87
|
+
rows=data_rows,
|
|
88
|
+
row_numbers=row_numbers,
|
|
89
|
+
confidence=confidence,
|
|
90
|
+
chunk_index_offset=len(chunks),
|
|
91
|
+
)
|
|
92
|
+
)
|
|
93
|
+
self._finalize_chunks(parsed, chunks)
|
|
94
|
+
self._validate_chunks(parsed, chunks)
|
|
95
|
+
return chunks
|
|
96
|
+
|
|
97
|
+
def _finalize_chunks(self, parsed: ParsedDocumentResult, chunks: list[Chunk]) -> None:
|
|
98
|
+
workbook_ir = parsed.structure
|
|
99
|
+
assert isinstance(workbook_ir, WorkbookIR)
|
|
100
|
+
for index, chunk in enumerate(chunks):
|
|
101
|
+
chunk.metadata["chunk_index"] = index
|
|
102
|
+
chunk.metadata["chunk_profile"] = self.policy.name.value
|
|
103
|
+
chunk.metadata["chunk_profile_version"] = self.policy.version
|
|
104
|
+
ordinal = int(chunk.metadata["sheet_ordinal"])
|
|
105
|
+
sheet_ir = workbook_ir.sheets[ordinal - 1]
|
|
106
|
+
snapshot = workbook_ir.snapshot
|
|
107
|
+
sheet_snapshot = snapshot.sheets[ordinal - 1] if snapshot is not None else None
|
|
108
|
+
hidden_rows = set(sheet_snapshot.hidden_rows) if sheet_snapshot is not None else set()
|
|
109
|
+
referenced_rows = set(chunk.metadata.get("row_numbers", []))
|
|
110
|
+
if not referenced_rows:
|
|
111
|
+
referenced_rows = _row_numbers_from_source_ranges(chunk.metadata["source_ranges"])
|
|
112
|
+
chunk.metadata["sheet_visibility"] = (
|
|
113
|
+
sheet_snapshot.visibility if sheet_snapshot is not None else sheet_ir.visibility
|
|
114
|
+
)
|
|
115
|
+
chunk.metadata["hidden_row_numbers"] = sorted(referenced_rows & hidden_rows)
|
|
116
|
+
|
|
117
|
+
def _validate_chunks(self, parsed: ParsedDocumentResult, chunks: list[Chunk]) -> None:
|
|
118
|
+
workbook_ir = parsed.structure
|
|
119
|
+
assert isinstance(workbook_ir, WorkbookIR)
|
|
120
|
+
expected_row_ids = [
|
|
121
|
+
row.row_id
|
|
122
|
+
for sheet in workbook_ir.sheets
|
|
123
|
+
for block in sheet.blocks
|
|
124
|
+
if block.logical_table is not None
|
|
125
|
+
for row in block.logical_table.rows
|
|
126
|
+
if row.role in {"data", "total"}
|
|
127
|
+
]
|
|
128
|
+
actual_row_ids = [
|
|
129
|
+
row_id
|
|
130
|
+
for chunk in chunks
|
|
131
|
+
if chunk.metadata["chunk_type"] == "table_rows"
|
|
132
|
+
for row_id in chunk.metadata["row_ids"]
|
|
133
|
+
]
|
|
134
|
+
if len(actual_row_ids) != len(set(actual_row_ids)) or set(actual_row_ids) != set(
|
|
135
|
+
expected_row_ids
|
|
136
|
+
):
|
|
137
|
+
raise ValueError("Workbook chunk row conservation failed")
|
|
138
|
+
if [chunk.metadata["chunk_index"] for chunk in chunks] != list(range(len(chunks))):
|
|
139
|
+
raise ValueError("Workbook chunk indexes are not contiguous")
|
|
140
|
+
|
|
141
|
+
for chunk in chunks:
|
|
142
|
+
source_ranges = chunk.metadata["source_ranges"]
|
|
143
|
+
for source_range in source_ranges:
|
|
144
|
+
_source_range_is_valid(workbook_ir.snapshot, source_range)
|
|
145
|
+
if chunk.metadata["chunk_type"] != "table_rows":
|
|
146
|
+
continue
|
|
147
|
+
payload = chunk.structured_payload
|
|
148
|
+
if len(chunk.metadata["row_ids"]) != len(payload["rows"]):
|
|
149
|
+
raise ValueError("Workbook table chunk row payload mismatch")
|
|
150
|
+
if self.policy.analysis_records:
|
|
151
|
+
records = payload["records"]
|
|
152
|
+
if len(chunk.metadata["row_ids"]) != len(records):
|
|
153
|
+
raise ValueError("Workbook table chunk analysis record mismatch")
|
|
154
|
+
record_ranges = list(
|
|
155
|
+
dict.fromkeys(
|
|
156
|
+
source_ref for record in records for source_ref in record["source_refs"]
|
|
157
|
+
)
|
|
158
|
+
)
|
|
159
|
+
if source_ranges != record_ranges:
|
|
160
|
+
raise ValueError("Workbook table chunk source ranges mismatch")
|
|
161
|
+
|
|
162
|
+
def _chunk_block(
|
|
163
|
+
self,
|
|
164
|
+
parsed: ParsedDocumentResult,
|
|
165
|
+
sheet_name: str,
|
|
166
|
+
sheet_ordinal: int,
|
|
167
|
+
block: WorkbookBlock,
|
|
168
|
+
chunk_index_offset: int,
|
|
169
|
+
) -> list[Chunk]:
|
|
170
|
+
if block.logical_table is not None:
|
|
171
|
+
return self._chunk_logical_table(
|
|
172
|
+
parsed,
|
|
173
|
+
sheet_name,
|
|
174
|
+
sheet_ordinal,
|
|
175
|
+
block.logical_table,
|
|
176
|
+
chunk_index_offset,
|
|
177
|
+
)
|
|
178
|
+
if block.form is not None:
|
|
179
|
+
return self._chunk_form(
|
|
180
|
+
parsed, sheet_name, sheet_ordinal, block.form, chunk_index_offset
|
|
181
|
+
)
|
|
182
|
+
if block.matrix is not None:
|
|
183
|
+
return self._chunk_matrix(
|
|
184
|
+
parsed, sheet_name, sheet_ordinal, block.matrix, chunk_index_offset
|
|
185
|
+
)
|
|
186
|
+
if block.text is not None:
|
|
187
|
+
return self._chunk_text(
|
|
188
|
+
parsed, sheet_name, sheet_ordinal, block.text, chunk_index_offset
|
|
189
|
+
)
|
|
190
|
+
source_range = block.source_refs[0].range
|
|
191
|
+
columns, rows, row_numbers = _raw_block_grid(
|
|
192
|
+
parsed.structure,
|
|
193
|
+
sheet_ordinal,
|
|
194
|
+
source_range,
|
|
195
|
+
)
|
|
196
|
+
return self._pack_table(
|
|
197
|
+
parsed=parsed,
|
|
198
|
+
sheet_name=sheet_name,
|
|
199
|
+
sheet_ordinal=sheet_ordinal,
|
|
200
|
+
columns=columns,
|
|
201
|
+
rows=rows,
|
|
202
|
+
row_numbers=row_numbers,
|
|
203
|
+
confidence=block.confidence,
|
|
204
|
+
chunk_index_offset=chunk_index_offset,
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
def _chunk_logical_table(
|
|
208
|
+
self,
|
|
209
|
+
parsed: ParsedDocumentResult,
|
|
210
|
+
sheet_name: str,
|
|
211
|
+
sheet_ordinal: int,
|
|
212
|
+
table: LogicalTable,
|
|
213
|
+
chunk_index_offset: int,
|
|
214
|
+
) -> list[Chunk]:
|
|
215
|
+
columns = [" / ".join(column.path) or column.coordinate for column in table.columns]
|
|
216
|
+
continuation = _continuation_for_table(parsed.structure, table)
|
|
217
|
+
eligible = [row for row in table.rows if row.role in {"data", "total"}]
|
|
218
|
+
grouped: list[tuple[list[str], list[LogicalRow]]] = []
|
|
219
|
+
for row in eligible:
|
|
220
|
+
if not grouped or grouped[-1][0] != row.section_path:
|
|
221
|
+
grouped.append((list(row.section_path), []))
|
|
222
|
+
grouped[-1][1].append(row)
|
|
223
|
+
|
|
224
|
+
chunks: list[Chunk] = []
|
|
225
|
+
for section_path, rows in grouped:
|
|
226
|
+
pending = []
|
|
227
|
+
for row in rows:
|
|
228
|
+
candidate = [*pending, row]
|
|
229
|
+
content = _render_logical_chunk(table.title, section_path, columns, candidate)
|
|
230
|
+
if pending and self.length_function(content) > self.max_chunk_size:
|
|
231
|
+
chunks.append(
|
|
232
|
+
_logical_chunk(
|
|
233
|
+
parsed,
|
|
234
|
+
table,
|
|
235
|
+
continuation,
|
|
236
|
+
sheet_name,
|
|
237
|
+
sheet_ordinal,
|
|
238
|
+
section_path,
|
|
239
|
+
columns,
|
|
240
|
+
pending,
|
|
241
|
+
chunk_index_offset + len(chunks),
|
|
242
|
+
self.max_chunk_size,
|
|
243
|
+
self.length_function,
|
|
244
|
+
self.policy,
|
|
245
|
+
)
|
|
246
|
+
)
|
|
247
|
+
pending = []
|
|
248
|
+
pending.append(row)
|
|
249
|
+
if pending:
|
|
250
|
+
chunks.append(
|
|
251
|
+
_logical_chunk(
|
|
252
|
+
parsed,
|
|
253
|
+
table,
|
|
254
|
+
continuation,
|
|
255
|
+
sheet_name,
|
|
256
|
+
sheet_ordinal,
|
|
257
|
+
section_path,
|
|
258
|
+
columns,
|
|
259
|
+
pending,
|
|
260
|
+
chunk_index_offset + len(chunks),
|
|
261
|
+
self.max_chunk_size,
|
|
262
|
+
self.length_function,
|
|
263
|
+
self.policy,
|
|
264
|
+
)
|
|
265
|
+
)
|
|
266
|
+
return chunks
|
|
267
|
+
|
|
268
|
+
def _chunk_form(
|
|
269
|
+
self,
|
|
270
|
+
parsed: ParsedDocumentResult,
|
|
271
|
+
sheet_name: str,
|
|
272
|
+
sheet_ordinal: int,
|
|
273
|
+
form: FormBlock,
|
|
274
|
+
chunk_index_offset: int,
|
|
275
|
+
) -> list[Chunk]:
|
|
276
|
+
chunks: list[Chunk] = []
|
|
277
|
+
pending: list[FormField] = []
|
|
278
|
+
|
|
279
|
+
def emit(*, oversized: bool = False) -> None:
|
|
280
|
+
if not pending:
|
|
281
|
+
return
|
|
282
|
+
chunks.append(
|
|
283
|
+
_form_chunk(
|
|
284
|
+
parsed,
|
|
285
|
+
form,
|
|
286
|
+
sheet_name,
|
|
287
|
+
sheet_ordinal,
|
|
288
|
+
pending,
|
|
289
|
+
[],
|
|
290
|
+
chunk_index_offset + len(chunks),
|
|
291
|
+
oversized,
|
|
292
|
+
analysis_records=self.policy.analysis_records,
|
|
293
|
+
)
|
|
294
|
+
)
|
|
295
|
+
pending.clear()
|
|
296
|
+
|
|
297
|
+
for field in form.fields:
|
|
298
|
+
candidate = [*pending, field]
|
|
299
|
+
content = _render_form_chunk(form.title, candidate, [])
|
|
300
|
+
if pending and self.length_function(content) > self.max_chunk_size:
|
|
301
|
+
emit()
|
|
302
|
+
candidate = [field]
|
|
303
|
+
content = _render_form_chunk(form.title, candidate, [])
|
|
304
|
+
pending.append(field)
|
|
305
|
+
if self.length_function(content) > self.max_chunk_size:
|
|
306
|
+
emit(oversized=True)
|
|
307
|
+
emit()
|
|
308
|
+
if form.free_text:
|
|
309
|
+
content = _render_form_chunk(form.title, [], form.free_text)
|
|
310
|
+
chunks.append(
|
|
311
|
+
_form_chunk(
|
|
312
|
+
parsed,
|
|
313
|
+
form,
|
|
314
|
+
sheet_name,
|
|
315
|
+
sheet_ordinal,
|
|
316
|
+
[],
|
|
317
|
+
form.free_text,
|
|
318
|
+
chunk_index_offset + len(chunks),
|
|
319
|
+
self.length_function(content) > self.max_chunk_size,
|
|
320
|
+
analysis_records=self.policy.analysis_records,
|
|
321
|
+
)
|
|
322
|
+
)
|
|
323
|
+
return chunks
|
|
324
|
+
|
|
325
|
+
def _chunk_matrix(
|
|
326
|
+
self,
|
|
327
|
+
parsed: ParsedDocumentResult,
|
|
328
|
+
sheet_name: str,
|
|
329
|
+
sheet_ordinal: int,
|
|
330
|
+
matrix: MatrixBlock,
|
|
331
|
+
chunk_index_offset: int,
|
|
332
|
+
) -> list[Chunk]:
|
|
333
|
+
chunks: list[Chunk] = []
|
|
334
|
+
pending: list[tuple[MatrixHeader, list[str], list]] = []
|
|
335
|
+
|
|
336
|
+
def emit(*, oversized: bool = False) -> None:
|
|
337
|
+
if not pending:
|
|
338
|
+
return
|
|
339
|
+
chunks.append(
|
|
340
|
+
_matrix_chunk(
|
|
341
|
+
parsed,
|
|
342
|
+
matrix,
|
|
343
|
+
sheet_name,
|
|
344
|
+
sheet_ordinal,
|
|
345
|
+
pending,
|
|
346
|
+
chunk_index_offset + len(chunks),
|
|
347
|
+
oversized,
|
|
348
|
+
analysis_records=self.policy.analysis_records,
|
|
349
|
+
)
|
|
350
|
+
)
|
|
351
|
+
pending.clear()
|
|
352
|
+
|
|
353
|
+
rows = zip(
|
|
354
|
+
matrix.row_headers,
|
|
355
|
+
matrix.values,
|
|
356
|
+
matrix.value_source_refs,
|
|
357
|
+
strict=True,
|
|
358
|
+
)
|
|
359
|
+
for header, values, refs in rows:
|
|
360
|
+
candidate = [*pending, (header, values, refs)]
|
|
361
|
+
content = _render_matrix_chunk(matrix, candidate)
|
|
362
|
+
if pending and self.length_function(content) > self.max_chunk_size:
|
|
363
|
+
emit()
|
|
364
|
+
candidate = [(header, values, refs)]
|
|
365
|
+
content = _render_matrix_chunk(matrix, candidate)
|
|
366
|
+
pending.append((header, values, refs))
|
|
367
|
+
if self.length_function(content) > self.max_chunk_size:
|
|
368
|
+
emit(oversized=True)
|
|
369
|
+
emit()
|
|
370
|
+
return chunks
|
|
371
|
+
|
|
372
|
+
def _chunk_text(
|
|
373
|
+
self,
|
|
374
|
+
parsed: ParsedDocumentResult,
|
|
375
|
+
sheet_name: str,
|
|
376
|
+
sheet_ordinal: int,
|
|
377
|
+
text: TextBlock,
|
|
378
|
+
chunk_index_offset: int,
|
|
379
|
+
) -> list[Chunk]:
|
|
380
|
+
chunks: list[Chunk] = []
|
|
381
|
+
pending: list[TextLine] = []
|
|
382
|
+
|
|
383
|
+
def emit(*, oversized: bool = False) -> None:
|
|
384
|
+
if not pending:
|
|
385
|
+
return
|
|
386
|
+
chunks.append(
|
|
387
|
+
_text_chunk(
|
|
388
|
+
parsed,
|
|
389
|
+
text,
|
|
390
|
+
sheet_name,
|
|
391
|
+
sheet_ordinal,
|
|
392
|
+
pending,
|
|
393
|
+
chunk_index_offset + len(chunks),
|
|
394
|
+
oversized,
|
|
395
|
+
analysis_records=self.policy.analysis_records,
|
|
396
|
+
)
|
|
397
|
+
)
|
|
398
|
+
pending.clear()
|
|
399
|
+
|
|
400
|
+
for line in text.lines:
|
|
401
|
+
candidate = [*pending, line]
|
|
402
|
+
content = "\n".join(item.text for item in candidate)
|
|
403
|
+
if pending and self.length_function(content) > self.max_chunk_size:
|
|
404
|
+
emit()
|
|
405
|
+
candidate = [line]
|
|
406
|
+
content = line.text
|
|
407
|
+
pending.append(line)
|
|
408
|
+
if self.length_function(content) > self.max_chunk_size:
|
|
409
|
+
emit(oversized=True)
|
|
410
|
+
emit()
|
|
411
|
+
return chunks
|
|
412
|
+
|
|
413
|
+
def _pack_table(
|
|
414
|
+
self,
|
|
415
|
+
*,
|
|
416
|
+
parsed: ParsedDocumentResult,
|
|
417
|
+
sheet_name: str,
|
|
418
|
+
sheet_ordinal: int,
|
|
419
|
+
columns: list[str],
|
|
420
|
+
rows: list[list[str]],
|
|
421
|
+
row_numbers: list[int],
|
|
422
|
+
confidence: float,
|
|
423
|
+
chunk_index_offset: int,
|
|
424
|
+
) -> list[Chunk]:
|
|
425
|
+
packed: list[Chunk] = []
|
|
426
|
+
pending_rows: list[list[str]] = []
|
|
427
|
+
pending_numbers: list[int] = []
|
|
428
|
+
|
|
429
|
+
def emit(*, oversized: bool = False) -> None:
|
|
430
|
+
if not pending_rows:
|
|
431
|
+
return
|
|
432
|
+
source_range = _source_range(sheet_name, columns, pending_numbers)
|
|
433
|
+
content = _render_chunk(sheet_name, source_range, columns, pending_rows)
|
|
434
|
+
payload = {
|
|
435
|
+
"columns": list(columns),
|
|
436
|
+
"rows": [list(row) for row in pending_rows],
|
|
437
|
+
}
|
|
438
|
+
if self.policy.analysis_records:
|
|
439
|
+
payload["column_schema"] = [
|
|
440
|
+
{
|
|
441
|
+
"column_index": index,
|
|
442
|
+
"coordinate": column,
|
|
443
|
+
"header_path": [],
|
|
444
|
+
}
|
|
445
|
+
for index, column in enumerate(columns)
|
|
446
|
+
]
|
|
447
|
+
payload["records"] = [
|
|
448
|
+
{
|
|
449
|
+
"row_number": row_number,
|
|
450
|
+
"role": "raw",
|
|
451
|
+
"section_path": [],
|
|
452
|
+
"values": list(row),
|
|
453
|
+
"source_refs": [_source_range(sheet_name, columns, [row_number])],
|
|
454
|
+
}
|
|
455
|
+
for row, row_number in zip(pending_rows, pending_numbers, strict=True)
|
|
456
|
+
]
|
|
457
|
+
metadata = document_metadata(parsed)
|
|
458
|
+
metadata.update(
|
|
459
|
+
{
|
|
460
|
+
"chunk_type": "raw_grid_rows",
|
|
461
|
+
"chunk_index": chunk_index_offset + len(packed),
|
|
462
|
+
"sheet_name": sheet_name,
|
|
463
|
+
"sheet_ordinal": sheet_ordinal,
|
|
464
|
+
"source_ranges": [source_range],
|
|
465
|
+
"row_numbers": list(pending_numbers),
|
|
466
|
+
"confidence": confidence,
|
|
467
|
+
"warnings": list(parsed.diagnostics.warnings)
|
|
468
|
+
if parsed.diagnostics is not None
|
|
469
|
+
else [],
|
|
470
|
+
}
|
|
471
|
+
)
|
|
472
|
+
if oversized:
|
|
473
|
+
metadata["oversized"] = True
|
|
474
|
+
packed.append(
|
|
475
|
+
Chunk(
|
|
476
|
+
content=content,
|
|
477
|
+
metadata=metadata,
|
|
478
|
+
structured_payload=payload,
|
|
479
|
+
)
|
|
480
|
+
)
|
|
481
|
+
pending_rows.clear()
|
|
482
|
+
pending_numbers.clear()
|
|
483
|
+
|
|
484
|
+
for row, row_number in zip(rows, row_numbers, strict=True):
|
|
485
|
+
candidate_rows = [*pending_rows, row]
|
|
486
|
+
candidate_numbers = [*pending_numbers, row_number]
|
|
487
|
+
candidate_range = _source_range(sheet_name, columns, candidate_numbers)
|
|
488
|
+
candidate = _render_chunk(sheet_name, candidate_range, columns, candidate_rows)
|
|
489
|
+
if pending_rows and self.length_function(candidate) > self.max_chunk_size:
|
|
490
|
+
emit()
|
|
491
|
+
candidate_rows = [row]
|
|
492
|
+
candidate_numbers = [row_number]
|
|
493
|
+
candidate_range = _source_range(sheet_name, columns, candidate_numbers)
|
|
494
|
+
candidate = _render_chunk(sheet_name, candidate_range, columns, candidate_rows)
|
|
495
|
+
|
|
496
|
+
pending_rows.append(row)
|
|
497
|
+
pending_numbers.append(row_number)
|
|
498
|
+
if self.length_function(candidate) > self.max_chunk_size:
|
|
499
|
+
emit(oversized=True)
|
|
500
|
+
|
|
501
|
+
emit()
|
|
502
|
+
return packed
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
def _row_numbers(source_range: str | None, count: int) -> list[int]:
|
|
506
|
+
if not source_range:
|
|
507
|
+
return list(range(1, count + 1))
|
|
508
|
+
_, min_row, _, _ = range_boundaries(source_range)
|
|
509
|
+
return list(range(min_row, min_row + count))
|
|
510
|
+
|
|
511
|
+
|
|
512
|
+
def _row_numbers_from_source_ranges(source_ranges: list[str]) -> set[int]:
|
|
513
|
+
row_numbers = set()
|
|
514
|
+
for source_range in source_ranges:
|
|
515
|
+
_, cell_range = source_range.rsplit("!", 1)
|
|
516
|
+
_, min_row, _, max_row = range_boundaries(cell_range)
|
|
517
|
+
row_numbers.update(range(min_row, max_row + 1))
|
|
518
|
+
return row_numbers
|
|
519
|
+
|
|
520
|
+
|
|
521
|
+
def _source_range_is_valid(snapshot, source_ref: str) -> None:
|
|
522
|
+
if snapshot is None:
|
|
523
|
+
raise ValueError("WorkbookIR snapshot is required for source-range validation")
|
|
524
|
+
sheet_name, cell_range = source_ref.rsplit("!", 1)
|
|
525
|
+
sheet = next((item for item in snapshot.sheets if item.name == sheet_name), None)
|
|
526
|
+
if sheet is None:
|
|
527
|
+
raise ValueError(f"Workbook source range references unknown sheet: {sheet_name}")
|
|
528
|
+
if sheet.used_range is None:
|
|
529
|
+
raise ValueError(f"Workbook sheet used_range is required: {sheet_name}")
|
|
530
|
+
try:
|
|
531
|
+
min_col, min_row, max_col, max_row = range_boundaries(cell_range)
|
|
532
|
+
used_min_col, used_min_row, used_max_col, used_max_row = range_boundaries(sheet.used_range)
|
|
533
|
+
except ValueError as exc:
|
|
534
|
+
raise ValueError(f"Workbook source range is invalid: {source_ref}") from exc
|
|
535
|
+
if not (
|
|
536
|
+
used_min_col <= min_col <= max_col <= used_max_col
|
|
537
|
+
and used_min_row <= min_row <= max_row <= used_max_row
|
|
538
|
+
):
|
|
539
|
+
raise ValueError(f"Workbook source range is outside sheet used_range: {source_ref}")
|
|
540
|
+
|
|
541
|
+
|
|
542
|
+
def _source_range(sheet_name: str, columns: list[str], row_numbers: list[int]) -> str:
|
|
543
|
+
first_row = min(row_numbers)
|
|
544
|
+
last_row = max(row_numbers)
|
|
545
|
+
return f"{sheet_name}!{columns[0]}{first_row}:{columns[-1]}{last_row}"
|
|
546
|
+
|
|
547
|
+
|
|
548
|
+
def _render_chunk(
|
|
549
|
+
sheet_name: str,
|
|
550
|
+
source_range: str,
|
|
551
|
+
columns: list[str],
|
|
552
|
+
rows: list[list[str]],
|
|
553
|
+
) -> str:
|
|
554
|
+
heading = f"## Sheet: {sheet_name}"
|
|
555
|
+
source_comment = f"<!-- source_range: {source_range} -->"
|
|
556
|
+
header = _markdown_row(columns)
|
|
557
|
+
separator = "| " + " | ".join("---" for _ in columns) + " |"
|
|
558
|
+
body = [_markdown_row(row) for row in rows]
|
|
559
|
+
return "\n\n".join((heading, source_comment, "\n".join((header, separator, *body))))
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def _markdown_row(row: list[str]) -> str:
|
|
563
|
+
escaped = [
|
|
564
|
+
str(value)
|
|
565
|
+
.replace("\r\n", "\n")
|
|
566
|
+
.replace("\r", "\n")
|
|
567
|
+
.replace("|", r"\|")
|
|
568
|
+
.replace("\n", "<br>")
|
|
569
|
+
for value in row
|
|
570
|
+
]
|
|
571
|
+
return "| " + " | ".join(escaped) + " |"
|
|
572
|
+
|
|
573
|
+
|
|
574
|
+
def _render_form_chunk(
|
|
575
|
+
title: str,
|
|
576
|
+
fields: list[FormField],
|
|
577
|
+
lines: list[TextLine],
|
|
578
|
+
) -> str:
|
|
579
|
+
parts = [f"### Form: {title}"] if title else []
|
|
580
|
+
if fields:
|
|
581
|
+
parts.append(
|
|
582
|
+
"\n".join(
|
|
583
|
+
[
|
|
584
|
+
_markdown_row(["Field", "Value"]),
|
|
585
|
+
"| --- | --- |",
|
|
586
|
+
*[_markdown_row([field.label, field.value]) for field in fields],
|
|
587
|
+
]
|
|
588
|
+
)
|
|
589
|
+
)
|
|
590
|
+
parts.extend(line.text for line in lines)
|
|
591
|
+
return "\n\n".join(parts)
|
|
592
|
+
|
|
593
|
+
|
|
594
|
+
def _form_chunk(
|
|
595
|
+
parsed: ParsedDocumentResult,
|
|
596
|
+
form: FormBlock,
|
|
597
|
+
sheet_name: str,
|
|
598
|
+
sheet_ordinal: int,
|
|
599
|
+
fields: list[FormField],
|
|
600
|
+
lines: list[TextLine],
|
|
601
|
+
chunk_index: int,
|
|
602
|
+
oversized: bool,
|
|
603
|
+
*,
|
|
604
|
+
analysis_records: bool,
|
|
605
|
+
) -> Chunk:
|
|
606
|
+
metadata = document_metadata(parsed)
|
|
607
|
+
source_ranges = [
|
|
608
|
+
ref.key for field in fields for ref in [*field.label_source_refs, *field.value_source_refs]
|
|
609
|
+
]
|
|
610
|
+
source_ranges.extend(ref.key for line in lines for ref in line.source_refs)
|
|
611
|
+
metadata.update(
|
|
612
|
+
{
|
|
613
|
+
"chunk_type": "form_fields",
|
|
614
|
+
"chunk_index": chunk_index,
|
|
615
|
+
"sheet_name": sheet_name,
|
|
616
|
+
"sheet_ordinal": sheet_ordinal,
|
|
617
|
+
"form_id": form.form_id,
|
|
618
|
+
"field_ids": [field.field_id for field in fields],
|
|
619
|
+
"source_ranges": source_ranges,
|
|
620
|
+
"confidence": min([form.confidence, *[field.confidence for field in fields]]),
|
|
621
|
+
"warnings": list(parsed.diagnostics.warnings) if parsed.diagnostics is not None else [],
|
|
622
|
+
}
|
|
623
|
+
)
|
|
624
|
+
if oversized:
|
|
625
|
+
metadata["oversized"] = True
|
|
626
|
+
payload = {
|
|
627
|
+
"fields": [[field.label, field.value] for field in fields],
|
|
628
|
+
"free_text": [line.text for line in lines],
|
|
629
|
+
}
|
|
630
|
+
if analysis_records:
|
|
631
|
+
payload["records"] = [
|
|
632
|
+
{
|
|
633
|
+
"record_type": "field",
|
|
634
|
+
"field_id": field.field_id,
|
|
635
|
+
"label": field.label,
|
|
636
|
+
"value": field.value,
|
|
637
|
+
"label_source_refs": [ref.key for ref in field.label_source_refs],
|
|
638
|
+
"value_source_refs": [ref.key for ref in field.value_source_refs],
|
|
639
|
+
}
|
|
640
|
+
for field in fields
|
|
641
|
+
]
|
|
642
|
+
payload["records"].extend(
|
|
643
|
+
{
|
|
644
|
+
"record_type": "text",
|
|
645
|
+
"text": line.text,
|
|
646
|
+
"source_refs": [ref.key for ref in line.source_refs],
|
|
647
|
+
}
|
|
648
|
+
for line in lines
|
|
649
|
+
)
|
|
650
|
+
return Chunk(
|
|
651
|
+
content=_render_form_chunk(form.title, fields, lines),
|
|
652
|
+
metadata=metadata,
|
|
653
|
+
structured_payload=payload,
|
|
654
|
+
)
|
|
655
|
+
|
|
656
|
+
|
|
657
|
+
def _render_matrix_chunk(matrix: MatrixBlock, rows: list[tuple]) -> str:
|
|
658
|
+
parts = [f"### Matrix: {matrix.title}"] if matrix.title else []
|
|
659
|
+
columns = ["", *[header.value for header in matrix.column_headers]]
|
|
660
|
+
table_lines = [
|
|
661
|
+
_markdown_row(columns),
|
|
662
|
+
"| " + " | ".join("---" for _ in columns) + " |",
|
|
663
|
+
*[_markdown_row([header.value, *values]) for header, values, _ in rows],
|
|
664
|
+
]
|
|
665
|
+
parts.append("\n".join(table_lines))
|
|
666
|
+
return "\n\n".join(parts)
|
|
667
|
+
|
|
668
|
+
|
|
669
|
+
def _matrix_chunk(
|
|
670
|
+
parsed: ParsedDocumentResult,
|
|
671
|
+
matrix: MatrixBlock,
|
|
672
|
+
sheet_name: str,
|
|
673
|
+
sheet_ordinal: int,
|
|
674
|
+
rows: list[tuple],
|
|
675
|
+
chunk_index: int,
|
|
676
|
+
oversized: bool,
|
|
677
|
+
*,
|
|
678
|
+
analysis_records: bool,
|
|
679
|
+
) -> Chunk:
|
|
680
|
+
metadata = document_metadata(parsed)
|
|
681
|
+
source_ranges = []
|
|
682
|
+
for header, _, refs in rows:
|
|
683
|
+
source_ranges.extend(ref.key for ref in header.source_refs)
|
|
684
|
+
source_ranges.extend(ref.key for ref in refs if ref is not None)
|
|
685
|
+
metadata.update(
|
|
686
|
+
{
|
|
687
|
+
"chunk_type": "matrix_rows",
|
|
688
|
+
"chunk_index": chunk_index,
|
|
689
|
+
"sheet_name": sheet_name,
|
|
690
|
+
"sheet_ordinal": sheet_ordinal,
|
|
691
|
+
"matrix_id": matrix.matrix_id,
|
|
692
|
+
"row_headers": [header.value for header, _, _ in rows],
|
|
693
|
+
"source_ranges": source_ranges,
|
|
694
|
+
"confidence": matrix.confidence,
|
|
695
|
+
"warnings": list(parsed.diagnostics.warnings) if parsed.diagnostics is not None else [],
|
|
696
|
+
}
|
|
697
|
+
)
|
|
698
|
+
if oversized:
|
|
699
|
+
metadata["oversized"] = True
|
|
700
|
+
payload = {
|
|
701
|
+
"column_headers": [header.value for header in matrix.column_headers],
|
|
702
|
+
"row_headers": [header.value for header, _, _ in rows],
|
|
703
|
+
"values": [list(values) for _, values, _ in rows],
|
|
704
|
+
}
|
|
705
|
+
if analysis_records:
|
|
706
|
+
payload["records"] = [
|
|
707
|
+
{
|
|
708
|
+
"row_header": header.value,
|
|
709
|
+
"row_header_source_refs": [ref.key for ref in header.source_refs],
|
|
710
|
+
"values": list(values),
|
|
711
|
+
"value_source_refs": [ref.key if ref is not None else None for ref in refs],
|
|
712
|
+
}
|
|
713
|
+
for header, values, refs in rows
|
|
714
|
+
]
|
|
715
|
+
return Chunk(
|
|
716
|
+
content=_render_matrix_chunk(matrix, rows),
|
|
717
|
+
metadata=metadata,
|
|
718
|
+
structured_payload=payload,
|
|
719
|
+
)
|
|
720
|
+
|
|
721
|
+
|
|
722
|
+
def _text_chunk(
|
|
723
|
+
parsed: ParsedDocumentResult,
|
|
724
|
+
text: TextBlock,
|
|
725
|
+
sheet_name: str,
|
|
726
|
+
sheet_ordinal: int,
|
|
727
|
+
lines: list[TextLine],
|
|
728
|
+
chunk_index: int,
|
|
729
|
+
oversized: bool,
|
|
730
|
+
*,
|
|
731
|
+
analysis_records: bool,
|
|
732
|
+
) -> Chunk:
|
|
733
|
+
metadata = document_metadata(parsed)
|
|
734
|
+
metadata.update(
|
|
735
|
+
{
|
|
736
|
+
"chunk_type": "text_block",
|
|
737
|
+
"chunk_index": chunk_index,
|
|
738
|
+
"sheet_name": sheet_name,
|
|
739
|
+
"sheet_ordinal": sheet_ordinal,
|
|
740
|
+
"text_id": text.text_id,
|
|
741
|
+
"source_ranges": [ref.key for line in lines for ref in line.source_refs],
|
|
742
|
+
"confidence": text.confidence,
|
|
743
|
+
"warnings": list(parsed.diagnostics.warnings) if parsed.diagnostics is not None else [],
|
|
744
|
+
}
|
|
745
|
+
)
|
|
746
|
+
if oversized:
|
|
747
|
+
metadata["oversized"] = True
|
|
748
|
+
payload = {"lines": [line.text for line in lines]}
|
|
749
|
+
if analysis_records:
|
|
750
|
+
payload["records"] = [
|
|
751
|
+
{"text": line.text, "source_refs": [ref.key for ref in line.source_refs]}
|
|
752
|
+
for line in lines
|
|
753
|
+
]
|
|
754
|
+
return Chunk(
|
|
755
|
+
content="\n".join(line.text for line in lines),
|
|
756
|
+
metadata=metadata,
|
|
757
|
+
structured_payload=payload,
|
|
758
|
+
)
|
|
759
|
+
|
|
760
|
+
|
|
761
|
+
def _raw_block_grid(
|
|
762
|
+
workbook_ir: WorkbookIR,
|
|
763
|
+
sheet_ordinal: int,
|
|
764
|
+
source_range: str,
|
|
765
|
+
) -> tuple[list[str], list[list[str]], list[int]]:
|
|
766
|
+
if workbook_ir.snapshot is None:
|
|
767
|
+
raise ValueError("WorkbookIR snapshot is required for raw block chunks")
|
|
768
|
+
sheet = workbook_ir.snapshot.sheets[sheet_ordinal - 1]
|
|
769
|
+
min_col, min_row, max_col, max_row = range_boundaries(source_range)
|
|
770
|
+
columns = [get_column_letter(column) for column in range(min_col, max_col + 1)]
|
|
771
|
+
row_numbers = list(range(min_row, max_row + 1))
|
|
772
|
+
rows = []
|
|
773
|
+
for row_number in row_numbers:
|
|
774
|
+
row = []
|
|
775
|
+
for column in range(min_col, max_col + 1):
|
|
776
|
+
coordinate = f"{get_column_letter(column)}{row_number}"
|
|
777
|
+
cell = sheet.cells.get(coordinate)
|
|
778
|
+
row.append("" if cell is None or cell.merge_anchor is not None else cell.display_value)
|
|
779
|
+
rows.append(row)
|
|
780
|
+
return columns, rows, row_numbers
|
|
781
|
+
|
|
782
|
+
|
|
783
|
+
def _render_logical_chunk(
|
|
784
|
+
title: str,
|
|
785
|
+
section_path: list[str],
|
|
786
|
+
columns: list[str],
|
|
787
|
+
rows: list[LogicalRow],
|
|
788
|
+
) -> str:
|
|
789
|
+
headings = [f"### Table: {title}"] if title else []
|
|
790
|
+
if section_path:
|
|
791
|
+
headings.append(f"#### Section: {' / '.join(section_path)}")
|
|
792
|
+
table_lines = [
|
|
793
|
+
_markdown_row(columns),
|
|
794
|
+
"| " + " | ".join("---" for _ in columns) + " |",
|
|
795
|
+
*[_markdown_row(row.values) for row in rows],
|
|
796
|
+
]
|
|
797
|
+
return "\n\n".join([*headings, "\n".join(table_lines)])
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def _logical_chunk(
|
|
801
|
+
parsed: ParsedDocumentResult,
|
|
802
|
+
table: LogicalTable,
|
|
803
|
+
continuation: TableContinuation | None,
|
|
804
|
+
sheet_name: str,
|
|
805
|
+
sheet_ordinal: int,
|
|
806
|
+
section_path: list[str],
|
|
807
|
+
columns: list[str],
|
|
808
|
+
rows: list[LogicalRow],
|
|
809
|
+
chunk_index: int,
|
|
810
|
+
max_chunk_size: int,
|
|
811
|
+
length_function: Callable[[str], int],
|
|
812
|
+
policy: WorkbookChunkPolicy,
|
|
813
|
+
) -> Chunk:
|
|
814
|
+
content = _render_logical_chunk(table.title, section_path, columns, rows)
|
|
815
|
+
metadata = document_metadata(parsed)
|
|
816
|
+
metadata.update(
|
|
817
|
+
{
|
|
818
|
+
"chunk_type": "table_rows",
|
|
819
|
+
"chunk_index": chunk_index,
|
|
820
|
+
"sheet_name": sheet_name,
|
|
821
|
+
"sheet_ordinal": sheet_ordinal,
|
|
822
|
+
"table_id": table.table_id,
|
|
823
|
+
"section_path": list(section_path),
|
|
824
|
+
"header_paths": [list(column.path) for column in table.columns],
|
|
825
|
+
"row_ids": [row.row_id for row in rows],
|
|
826
|
+
"row_numbers": [row.metadata["row_number"] for row in rows],
|
|
827
|
+
"source_ranges": [row.source_ref.key for row in rows],
|
|
828
|
+
"fragment_ranges": _fragment_ranges_for_rows(table, rows),
|
|
829
|
+
"confidence": min([table.confidence, *[row.confidence for row in rows]]),
|
|
830
|
+
"warnings": list(parsed.diagnostics.warnings) if parsed.diagnostics is not None else [],
|
|
831
|
+
}
|
|
832
|
+
)
|
|
833
|
+
if continuation is not None:
|
|
834
|
+
metadata.update(
|
|
835
|
+
{
|
|
836
|
+
"continuation_id": continuation.continuation_id,
|
|
837
|
+
"continuation_role": table.continuation_role,
|
|
838
|
+
"continuation_member_table_ids": list(continuation.member_table_ids),
|
|
839
|
+
"continuation_source_ranges": [ref.key for ref in continuation.source_refs],
|
|
840
|
+
}
|
|
841
|
+
)
|
|
842
|
+
if length_function(content) > max_chunk_size:
|
|
843
|
+
metadata["oversized"] = True
|
|
844
|
+
payload = {
|
|
845
|
+
"columns": columns,
|
|
846
|
+
"rows": [list(row.values) for row in rows],
|
|
847
|
+
"roles": [row.role for row in rows],
|
|
848
|
+
}
|
|
849
|
+
if policy.analysis_records:
|
|
850
|
+
payload["column_schema"] = [
|
|
851
|
+
{
|
|
852
|
+
"column_index": index,
|
|
853
|
+
"coordinate": column.coordinate,
|
|
854
|
+
"header_path": list(column.path),
|
|
855
|
+
}
|
|
856
|
+
for index, column in enumerate(table.columns)
|
|
857
|
+
]
|
|
858
|
+
payload["records"] = [
|
|
859
|
+
{
|
|
860
|
+
"row_id": row.row_id,
|
|
861
|
+
"row_number": int(row.metadata["row_number"]),
|
|
862
|
+
"role": row.role,
|
|
863
|
+
"section_path": list(row.section_path),
|
|
864
|
+
"values": list(row.values),
|
|
865
|
+
"source_refs": [row.source_ref.key],
|
|
866
|
+
}
|
|
867
|
+
for row in rows
|
|
868
|
+
]
|
|
869
|
+
return Chunk(
|
|
870
|
+
content=content,
|
|
871
|
+
metadata=metadata,
|
|
872
|
+
structured_payload=payload,
|
|
873
|
+
)
|
|
874
|
+
|
|
875
|
+
|
|
876
|
+
def _continuation_for_table(
|
|
877
|
+
workbook_ir: WorkbookIR,
|
|
878
|
+
table: LogicalTable,
|
|
879
|
+
) -> TableContinuation | None:
|
|
880
|
+
if table.continuation_id is None:
|
|
881
|
+
return None
|
|
882
|
+
return next(
|
|
883
|
+
(
|
|
884
|
+
group
|
|
885
|
+
for group in workbook_ir.table_continuations
|
|
886
|
+
if group.continuation_id == table.continuation_id
|
|
887
|
+
and table.table_id in group.member_table_ids
|
|
888
|
+
),
|
|
889
|
+
None,
|
|
890
|
+
)
|
|
891
|
+
|
|
892
|
+
|
|
893
|
+
def _fragment_ranges_for_rows(table: LogicalTable, rows: list[LogicalRow]) -> list[str]:
|
|
894
|
+
row_numbers = {int(row.metadata["row_number"]) for row in rows}
|
|
895
|
+
ranges = []
|
|
896
|
+
for fragment in table.fragments:
|
|
897
|
+
_, min_row, _, max_row = range_boundaries(fragment.source_ref.range)
|
|
898
|
+
if any(min_row <= row_number <= max_row for row_number in row_numbers):
|
|
899
|
+
ranges.append(fragment.source_ref.key)
|
|
900
|
+
return ranges
|