langparse 0.1.0rc1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (85) hide show
  1. langparse/__init__.py +35 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +0 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/semantic.py +242 -0
  7. langparse/chunkers/workbook.py +900 -0
  8. langparse/cli.py +257 -0
  9. langparse/config.py +169 -0
  10. langparse/core/__init__.py +0 -0
  11. langparse/core/chunker.py +16 -0
  12. langparse/core/engine.py +37 -0
  13. langparse/core/parser.py +35 -0
  14. langparse/core/rendering.py +49 -0
  15. langparse/engines/__init__.py +1 -0
  16. langparse/engines/pdf/__init__.py +1 -0
  17. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  18. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  19. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  20. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  21. langparse/engines/pdf/deepdoc/operators.py +684 -0
  22. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  23. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  24. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  25. langparse/engines/pdf/deepdoc/rendering.py +202 -0
  26. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  27. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  28. langparse/engines/pdf/deepdoc/utils.py +36 -0
  29. langparse/engines/pdf/deepdoc_engine.py +140 -0
  30. langparse/engines/pdf/mineru.py +235 -0
  31. langparse/engines/pdf/mineru_client.py +318 -0
  32. langparse/engines/pdf/mineru_service.py +225 -0
  33. langparse/engines/pdf/ocr.py +101 -0
  34. langparse/engines/pdf/other.py +20 -0
  35. langparse/engines/pdf/simple.py +127 -0
  36. langparse/engines/pdf/vision_llm.py +27 -0
  37. langparse/errors.py +52 -0
  38. langparse/logging.py +27 -0
  39. langparse/metrics.py +129 -0
  40. langparse/parsers/__init__.py +0 -0
  41. langparse/parsers/docx_parser.py +114 -0
  42. langparse/parsers/excel_parser.py +190 -0
  43. langparse/parsers/markdown_parser.py +34 -0
  44. langparse/parsers/pdf_parser.py +31 -0
  45. langparse/parsers/registry.py +48 -0
  46. langparse/parsers/sniff.py +72 -0
  47. langparse/py.typed +0 -0
  48. langparse/services/__init__.py +5 -0
  49. langparse/services/batch_service.py +253 -0
  50. langparse/services/benchmark_service.py +202 -0
  51. langparse/services/fidelity.py +154 -0
  52. langparse/services/output_paths.py +86 -0
  53. langparse/services/parse_service.py +468 -0
  54. langparse/services/quality.py +65 -0
  55. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  56. langparse/types.py +94 -0
  57. langparse/workbooks/__init__.py +97 -0
  58. langparse/workbooks/adapters.py +309 -0
  59. langparse/workbooks/assembly.py +928 -0
  60. langparse/workbooks/blocks.py +209 -0
  61. langparse/workbooks/classification.py +381 -0
  62. langparse/workbooks/continuation.py +577 -0
  63. langparse/workbooks/evaluation/__init__.py +45 -0
  64. langparse/workbooks/evaluation/evaluator.py +381 -0
  65. langparse/workbooks/evaluation/schema.py +419 -0
  66. langparse/workbooks/modeling/__init__.py +52 -0
  67. langparse/workbooks/modeling/cache.py +20 -0
  68. langparse/workbooks/modeling/config.py +87 -0
  69. langparse/workbooks/modeling/contract.py +628 -0
  70. langparse/workbooks/modeling/disambiguation.py +800 -0
  71. langparse/workbooks/modeling/openai_adapter.py +192 -0
  72. langparse/workbooks/modeling/policy.py +79 -0
  73. langparse/workbooks/modeling/ports.py +44 -0
  74. langparse/workbooks/modeling/pricing.py +17 -0
  75. langparse/workbooks/modeling/types.py +251 -0
  76. langparse/workbooks/regions.py +77 -0
  77. langparse/workbooks/rendering.py +216 -0
  78. langparse/workbooks/tables.py +375 -0
  79. langparse/workbooks/types.py +239 -0
  80. langparse-0.1.0rc1.dist-info/METADATA +720 -0
  81. langparse-0.1.0rc1.dist-info/RECORD +85 -0
  82. langparse-0.1.0rc1.dist-info/WHEEL +5 -0
  83. langparse-0.1.0rc1.dist-info/entry_points.txt +2 -0
  84. langparse-0.1.0rc1.dist-info/licenses/LICENSE +192 -0
  85. langparse-0.1.0rc1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,900 @@
1
+ from __future__ import annotations
2
+
3
+ from collections.abc import Callable
4
+
5
+ from openpyxl.utils import get_column_letter, range_boundaries
6
+
7
+ from langparse.chunkers.profiles import (
8
+ WorkbookChunkPolicy,
9
+ WorkbookChunkProfile,
10
+ resolve_workbook_chunk_policy,
11
+ )
12
+ from langparse.core.rendering import document_metadata
13
+ from langparse.types import Chunk, ParsedDocumentResult
14
+ from langparse.workbooks.types import (
15
+ FormBlock,
16
+ FormField,
17
+ LogicalRow,
18
+ LogicalTable,
19
+ MatrixBlock,
20
+ MatrixHeader,
21
+ TableContinuation,
22
+ TextBlock,
23
+ TextLine,
24
+ WorkbookBlock,
25
+ WorkbookIR,
26
+ )
27
+
28
+
29
+ class WorkbookStructuralChunker:
30
+ """Pack complete raw-grid rows directly from workbook compatibility facts."""
31
+
32
+ def __init__(
33
+ self,
34
+ max_chunk_size: int | None = None,
35
+ length_function: Callable[[str], int] = len,
36
+ *,
37
+ profile: str | WorkbookChunkProfile | None = None,
38
+ ):
39
+ self.policy: WorkbookChunkPolicy = resolve_workbook_chunk_policy(profile)
40
+ resolved_size = (
41
+ self.policy.default_max_chunk_size if max_chunk_size is None else max_chunk_size
42
+ )
43
+ if resolved_size <= 0:
44
+ raise ValueError("max_chunk_size must be positive")
45
+ self.max_chunk_size = resolved_size
46
+ self.length_function = length_function
47
+
48
+ def chunk(self, parsed: ParsedDocumentResult) -> list[Chunk]:
49
+ if not isinstance(parsed.structure, WorkbookIR):
50
+ raise TypeError("WorkbookStructuralChunker requires WorkbookIR structure")
51
+
52
+ ir_sheets = {sheet.index: sheet for sheet in parsed.structure.sheets}
53
+ chunks: list[Chunk] = []
54
+ for page in parsed.pages:
55
+ sheet_name = page.metadata.get("sheet_name")
56
+ sheet_ir = ir_sheets.get(page.page_number - 1)
57
+ if sheet_ir is not None and sheet_ir.blocks:
58
+ for block in sheet_ir.blocks:
59
+ chunks.extend(
60
+ self._chunk_block(
61
+ parsed,
62
+ str(sheet_name),
63
+ page.page_number,
64
+ block,
65
+ len(chunks),
66
+ )
67
+ )
68
+ continue
69
+ confidence = (
70
+ sheet_ir.blocks[0].confidence if sheet_ir is not None and sheet_ir.blocks else 1.0
71
+ )
72
+ for table in page.tables:
73
+ rows = table.get("rows", [])
74
+ if not rows:
75
+ continue
76
+ columns = [str(value) for value in table.get("columns") or rows[0]]
77
+ data_rows = [[str(value) for value in row] for row in rows[1:]]
78
+ row_numbers = list(table.get("row_numbers", []))
79
+ if len(row_numbers) != len(data_rows):
80
+ row_numbers = _row_numbers(table.get("source_range"), len(data_rows))
81
+ chunks.extend(
82
+ self._pack_table(
83
+ parsed=parsed,
84
+ sheet_name=str(sheet_name),
85
+ sheet_ordinal=page.page_number,
86
+ columns=columns,
87
+ rows=data_rows,
88
+ row_numbers=row_numbers,
89
+ confidence=confidence,
90
+ chunk_index_offset=len(chunks),
91
+ )
92
+ )
93
+ self._finalize_chunks(parsed, chunks)
94
+ self._validate_chunks(parsed, chunks)
95
+ return chunks
96
+
97
+ def _finalize_chunks(self, parsed: ParsedDocumentResult, chunks: list[Chunk]) -> None:
98
+ workbook_ir = parsed.structure
99
+ assert isinstance(workbook_ir, WorkbookIR)
100
+ for index, chunk in enumerate(chunks):
101
+ chunk.metadata["chunk_index"] = index
102
+ chunk.metadata["chunk_profile"] = self.policy.name.value
103
+ chunk.metadata["chunk_profile_version"] = self.policy.version
104
+ ordinal = int(chunk.metadata["sheet_ordinal"])
105
+ sheet_ir = workbook_ir.sheets[ordinal - 1]
106
+ snapshot = workbook_ir.snapshot
107
+ sheet_snapshot = snapshot.sheets[ordinal - 1] if snapshot is not None else None
108
+ hidden_rows = set(sheet_snapshot.hidden_rows) if sheet_snapshot is not None else set()
109
+ referenced_rows = set(chunk.metadata.get("row_numbers", []))
110
+ if not referenced_rows:
111
+ referenced_rows = _row_numbers_from_source_ranges(chunk.metadata["source_ranges"])
112
+ chunk.metadata["sheet_visibility"] = (
113
+ sheet_snapshot.visibility if sheet_snapshot is not None else sheet_ir.visibility
114
+ )
115
+ chunk.metadata["hidden_row_numbers"] = sorted(referenced_rows & hidden_rows)
116
+
117
+ def _validate_chunks(self, parsed: ParsedDocumentResult, chunks: list[Chunk]) -> None:
118
+ workbook_ir = parsed.structure
119
+ assert isinstance(workbook_ir, WorkbookIR)
120
+ expected_row_ids = [
121
+ row.row_id
122
+ for sheet in workbook_ir.sheets
123
+ for block in sheet.blocks
124
+ if block.logical_table is not None
125
+ for row in block.logical_table.rows
126
+ if row.role in {"data", "total"}
127
+ ]
128
+ actual_row_ids = [
129
+ row_id
130
+ for chunk in chunks
131
+ if chunk.metadata["chunk_type"] == "table_rows"
132
+ for row_id in chunk.metadata["row_ids"]
133
+ ]
134
+ if len(actual_row_ids) != len(set(actual_row_ids)) or set(actual_row_ids) != set(
135
+ expected_row_ids
136
+ ):
137
+ raise ValueError("Workbook chunk row conservation failed")
138
+ if [chunk.metadata["chunk_index"] for chunk in chunks] != list(range(len(chunks))):
139
+ raise ValueError("Workbook chunk indexes are not contiguous")
140
+
141
+ for chunk in chunks:
142
+ source_ranges = chunk.metadata["source_ranges"]
143
+ for source_range in source_ranges:
144
+ _source_range_is_valid(workbook_ir.snapshot, source_range)
145
+ if chunk.metadata["chunk_type"] != "table_rows":
146
+ continue
147
+ payload = chunk.structured_payload
148
+ if len(chunk.metadata["row_ids"]) != len(payload["rows"]):
149
+ raise ValueError("Workbook table chunk row payload mismatch")
150
+ if self.policy.analysis_records:
151
+ records = payload["records"]
152
+ if len(chunk.metadata["row_ids"]) != len(records):
153
+ raise ValueError("Workbook table chunk analysis record mismatch")
154
+ record_ranges = list(
155
+ dict.fromkeys(
156
+ source_ref for record in records for source_ref in record["source_refs"]
157
+ )
158
+ )
159
+ if source_ranges != record_ranges:
160
+ raise ValueError("Workbook table chunk source ranges mismatch")
161
+
162
+ def _chunk_block(
163
+ self,
164
+ parsed: ParsedDocumentResult,
165
+ sheet_name: str,
166
+ sheet_ordinal: int,
167
+ block: WorkbookBlock,
168
+ chunk_index_offset: int,
169
+ ) -> list[Chunk]:
170
+ if block.logical_table is not None:
171
+ return self._chunk_logical_table(
172
+ parsed,
173
+ sheet_name,
174
+ sheet_ordinal,
175
+ block.logical_table,
176
+ chunk_index_offset,
177
+ )
178
+ if block.form is not None:
179
+ return self._chunk_form(
180
+ parsed, sheet_name, sheet_ordinal, block.form, chunk_index_offset
181
+ )
182
+ if block.matrix is not None:
183
+ return self._chunk_matrix(
184
+ parsed, sheet_name, sheet_ordinal, block.matrix, chunk_index_offset
185
+ )
186
+ if block.text is not None:
187
+ return self._chunk_text(
188
+ parsed, sheet_name, sheet_ordinal, block.text, chunk_index_offset
189
+ )
190
+ source_range = block.source_refs[0].range
191
+ columns, rows, row_numbers = _raw_block_grid(
192
+ parsed.structure,
193
+ sheet_ordinal,
194
+ source_range,
195
+ )
196
+ return self._pack_table(
197
+ parsed=parsed,
198
+ sheet_name=sheet_name,
199
+ sheet_ordinal=sheet_ordinal,
200
+ columns=columns,
201
+ rows=rows,
202
+ row_numbers=row_numbers,
203
+ confidence=block.confidence,
204
+ chunk_index_offset=chunk_index_offset,
205
+ )
206
+
207
+ def _chunk_logical_table(
208
+ self,
209
+ parsed: ParsedDocumentResult,
210
+ sheet_name: str,
211
+ sheet_ordinal: int,
212
+ table: LogicalTable,
213
+ chunk_index_offset: int,
214
+ ) -> list[Chunk]:
215
+ columns = [" / ".join(column.path) or column.coordinate for column in table.columns]
216
+ continuation = _continuation_for_table(parsed.structure, table)
217
+ eligible = [row for row in table.rows if row.role in {"data", "total"}]
218
+ grouped: list[tuple[list[str], list[LogicalRow]]] = []
219
+ for row in eligible:
220
+ if not grouped or grouped[-1][0] != row.section_path:
221
+ grouped.append((list(row.section_path), []))
222
+ grouped[-1][1].append(row)
223
+
224
+ chunks: list[Chunk] = []
225
+ for section_path, rows in grouped:
226
+ pending = []
227
+ for row in rows:
228
+ candidate = [*pending, row]
229
+ content = _render_logical_chunk(table.title, section_path, columns, candidate)
230
+ if pending and self.length_function(content) > self.max_chunk_size:
231
+ chunks.append(
232
+ _logical_chunk(
233
+ parsed,
234
+ table,
235
+ continuation,
236
+ sheet_name,
237
+ sheet_ordinal,
238
+ section_path,
239
+ columns,
240
+ pending,
241
+ chunk_index_offset + len(chunks),
242
+ self.max_chunk_size,
243
+ self.length_function,
244
+ self.policy,
245
+ )
246
+ )
247
+ pending = []
248
+ pending.append(row)
249
+ if pending:
250
+ chunks.append(
251
+ _logical_chunk(
252
+ parsed,
253
+ table,
254
+ continuation,
255
+ sheet_name,
256
+ sheet_ordinal,
257
+ section_path,
258
+ columns,
259
+ pending,
260
+ chunk_index_offset + len(chunks),
261
+ self.max_chunk_size,
262
+ self.length_function,
263
+ self.policy,
264
+ )
265
+ )
266
+ return chunks
267
+
268
+ def _chunk_form(
269
+ self,
270
+ parsed: ParsedDocumentResult,
271
+ sheet_name: str,
272
+ sheet_ordinal: int,
273
+ form: FormBlock,
274
+ chunk_index_offset: int,
275
+ ) -> list[Chunk]:
276
+ chunks: list[Chunk] = []
277
+ pending: list[FormField] = []
278
+
279
+ def emit(*, oversized: bool = False) -> None:
280
+ if not pending:
281
+ return
282
+ chunks.append(
283
+ _form_chunk(
284
+ parsed,
285
+ form,
286
+ sheet_name,
287
+ sheet_ordinal,
288
+ pending,
289
+ [],
290
+ chunk_index_offset + len(chunks),
291
+ oversized,
292
+ analysis_records=self.policy.analysis_records,
293
+ )
294
+ )
295
+ pending.clear()
296
+
297
+ for field in form.fields:
298
+ candidate = [*pending, field]
299
+ content = _render_form_chunk(form.title, candidate, [])
300
+ if pending and self.length_function(content) > self.max_chunk_size:
301
+ emit()
302
+ candidate = [field]
303
+ content = _render_form_chunk(form.title, candidate, [])
304
+ pending.append(field)
305
+ if self.length_function(content) > self.max_chunk_size:
306
+ emit(oversized=True)
307
+ emit()
308
+ if form.free_text:
309
+ content = _render_form_chunk(form.title, [], form.free_text)
310
+ chunks.append(
311
+ _form_chunk(
312
+ parsed,
313
+ form,
314
+ sheet_name,
315
+ sheet_ordinal,
316
+ [],
317
+ form.free_text,
318
+ chunk_index_offset + len(chunks),
319
+ self.length_function(content) > self.max_chunk_size,
320
+ analysis_records=self.policy.analysis_records,
321
+ )
322
+ )
323
+ return chunks
324
+
325
+ def _chunk_matrix(
326
+ self,
327
+ parsed: ParsedDocumentResult,
328
+ sheet_name: str,
329
+ sheet_ordinal: int,
330
+ matrix: MatrixBlock,
331
+ chunk_index_offset: int,
332
+ ) -> list[Chunk]:
333
+ chunks: list[Chunk] = []
334
+ pending: list[tuple[MatrixHeader, list[str], list]] = []
335
+
336
+ def emit(*, oversized: bool = False) -> None:
337
+ if not pending:
338
+ return
339
+ chunks.append(
340
+ _matrix_chunk(
341
+ parsed,
342
+ matrix,
343
+ sheet_name,
344
+ sheet_ordinal,
345
+ pending,
346
+ chunk_index_offset + len(chunks),
347
+ oversized,
348
+ analysis_records=self.policy.analysis_records,
349
+ )
350
+ )
351
+ pending.clear()
352
+
353
+ rows = zip(
354
+ matrix.row_headers,
355
+ matrix.values,
356
+ matrix.value_source_refs,
357
+ strict=True,
358
+ )
359
+ for header, values, refs in rows:
360
+ candidate = [*pending, (header, values, refs)]
361
+ content = _render_matrix_chunk(matrix, candidate)
362
+ if pending and self.length_function(content) > self.max_chunk_size:
363
+ emit()
364
+ candidate = [(header, values, refs)]
365
+ content = _render_matrix_chunk(matrix, candidate)
366
+ pending.append((header, values, refs))
367
+ if self.length_function(content) > self.max_chunk_size:
368
+ emit(oversized=True)
369
+ emit()
370
+ return chunks
371
+
372
+ def _chunk_text(
373
+ self,
374
+ parsed: ParsedDocumentResult,
375
+ sheet_name: str,
376
+ sheet_ordinal: int,
377
+ text: TextBlock,
378
+ chunk_index_offset: int,
379
+ ) -> list[Chunk]:
380
+ chunks: list[Chunk] = []
381
+ pending: list[TextLine] = []
382
+
383
+ def emit(*, oversized: bool = False) -> None:
384
+ if not pending:
385
+ return
386
+ chunks.append(
387
+ _text_chunk(
388
+ parsed,
389
+ text,
390
+ sheet_name,
391
+ sheet_ordinal,
392
+ pending,
393
+ chunk_index_offset + len(chunks),
394
+ oversized,
395
+ analysis_records=self.policy.analysis_records,
396
+ )
397
+ )
398
+ pending.clear()
399
+
400
+ for line in text.lines:
401
+ candidate = [*pending, line]
402
+ content = "\n".join(item.text for item in candidate)
403
+ if pending and self.length_function(content) > self.max_chunk_size:
404
+ emit()
405
+ candidate = [line]
406
+ content = line.text
407
+ pending.append(line)
408
+ if self.length_function(content) > self.max_chunk_size:
409
+ emit(oversized=True)
410
+ emit()
411
+ return chunks
412
+
413
+ def _pack_table(
414
+ self,
415
+ *,
416
+ parsed: ParsedDocumentResult,
417
+ sheet_name: str,
418
+ sheet_ordinal: int,
419
+ columns: list[str],
420
+ rows: list[list[str]],
421
+ row_numbers: list[int],
422
+ confidence: float,
423
+ chunk_index_offset: int,
424
+ ) -> list[Chunk]:
425
+ packed: list[Chunk] = []
426
+ pending_rows: list[list[str]] = []
427
+ pending_numbers: list[int] = []
428
+
429
+ def emit(*, oversized: bool = False) -> None:
430
+ if not pending_rows:
431
+ return
432
+ source_range = _source_range(sheet_name, columns, pending_numbers)
433
+ content = _render_chunk(sheet_name, source_range, columns, pending_rows)
434
+ payload = {
435
+ "columns": list(columns),
436
+ "rows": [list(row) for row in pending_rows],
437
+ }
438
+ if self.policy.analysis_records:
439
+ payload["column_schema"] = [
440
+ {
441
+ "column_index": index,
442
+ "coordinate": column,
443
+ "header_path": [],
444
+ }
445
+ for index, column in enumerate(columns)
446
+ ]
447
+ payload["records"] = [
448
+ {
449
+ "row_number": row_number,
450
+ "role": "raw",
451
+ "section_path": [],
452
+ "values": list(row),
453
+ "source_refs": [_source_range(sheet_name, columns, [row_number])],
454
+ }
455
+ for row, row_number in zip(pending_rows, pending_numbers, strict=True)
456
+ ]
457
+ metadata = document_metadata(parsed)
458
+ metadata.update(
459
+ {
460
+ "chunk_type": "raw_grid_rows",
461
+ "chunk_index": chunk_index_offset + len(packed),
462
+ "sheet_name": sheet_name,
463
+ "sheet_ordinal": sheet_ordinal,
464
+ "source_ranges": [source_range],
465
+ "row_numbers": list(pending_numbers),
466
+ "confidence": confidence,
467
+ "warnings": list(parsed.diagnostics.warnings)
468
+ if parsed.diagnostics is not None
469
+ else [],
470
+ }
471
+ )
472
+ if oversized:
473
+ metadata["oversized"] = True
474
+ packed.append(
475
+ Chunk(
476
+ content=content,
477
+ metadata=metadata,
478
+ structured_payload=payload,
479
+ )
480
+ )
481
+ pending_rows.clear()
482
+ pending_numbers.clear()
483
+
484
+ for row, row_number in zip(rows, row_numbers, strict=True):
485
+ candidate_rows = [*pending_rows, row]
486
+ candidate_numbers = [*pending_numbers, row_number]
487
+ candidate_range = _source_range(sheet_name, columns, candidate_numbers)
488
+ candidate = _render_chunk(sheet_name, candidate_range, columns, candidate_rows)
489
+ if pending_rows and self.length_function(candidate) > self.max_chunk_size:
490
+ emit()
491
+ candidate_rows = [row]
492
+ candidate_numbers = [row_number]
493
+ candidate_range = _source_range(sheet_name, columns, candidate_numbers)
494
+ candidate = _render_chunk(sheet_name, candidate_range, columns, candidate_rows)
495
+
496
+ pending_rows.append(row)
497
+ pending_numbers.append(row_number)
498
+ if self.length_function(candidate) > self.max_chunk_size:
499
+ emit(oversized=True)
500
+
501
+ emit()
502
+ return packed
503
+
504
+
505
+ def _row_numbers(source_range: str | None, count: int) -> list[int]:
506
+ if not source_range:
507
+ return list(range(1, count + 1))
508
+ _, min_row, _, _ = range_boundaries(source_range)
509
+ return list(range(min_row, min_row + count))
510
+
511
+
512
+ def _row_numbers_from_source_ranges(source_ranges: list[str]) -> set[int]:
513
+ row_numbers = set()
514
+ for source_range in source_ranges:
515
+ _, cell_range = source_range.rsplit("!", 1)
516
+ _, min_row, _, max_row = range_boundaries(cell_range)
517
+ row_numbers.update(range(min_row, max_row + 1))
518
+ return row_numbers
519
+
520
+
521
+ def _source_range_is_valid(snapshot, source_ref: str) -> None:
522
+ if snapshot is None:
523
+ raise ValueError("WorkbookIR snapshot is required for source-range validation")
524
+ sheet_name, cell_range = source_ref.rsplit("!", 1)
525
+ sheet = next((item for item in snapshot.sheets if item.name == sheet_name), None)
526
+ if sheet is None:
527
+ raise ValueError(f"Workbook source range references unknown sheet: {sheet_name}")
528
+ if sheet.used_range is None:
529
+ raise ValueError(f"Workbook sheet used_range is required: {sheet_name}")
530
+ try:
531
+ min_col, min_row, max_col, max_row = range_boundaries(cell_range)
532
+ used_min_col, used_min_row, used_max_col, used_max_row = range_boundaries(sheet.used_range)
533
+ except ValueError as exc:
534
+ raise ValueError(f"Workbook source range is invalid: {source_ref}") from exc
535
+ if not (
536
+ used_min_col <= min_col <= max_col <= used_max_col
537
+ and used_min_row <= min_row <= max_row <= used_max_row
538
+ ):
539
+ raise ValueError(f"Workbook source range is outside sheet used_range: {source_ref}")
540
+
541
+
542
+ def _source_range(sheet_name: str, columns: list[str], row_numbers: list[int]) -> str:
543
+ first_row = min(row_numbers)
544
+ last_row = max(row_numbers)
545
+ return f"{sheet_name}!{columns[0]}{first_row}:{columns[-1]}{last_row}"
546
+
547
+
548
+ def _render_chunk(
549
+ sheet_name: str,
550
+ source_range: str,
551
+ columns: list[str],
552
+ rows: list[list[str]],
553
+ ) -> str:
554
+ heading = f"## Sheet: {sheet_name}"
555
+ source_comment = f"<!-- source_range: {source_range} -->"
556
+ header = _markdown_row(columns)
557
+ separator = "| " + " | ".join("---" for _ in columns) + " |"
558
+ body = [_markdown_row(row) for row in rows]
559
+ return "\n\n".join((heading, source_comment, "\n".join((header, separator, *body))))
560
+
561
+
562
+ def _markdown_row(row: list[str]) -> str:
563
+ escaped = [
564
+ str(value)
565
+ .replace("\r\n", "\n")
566
+ .replace("\r", "\n")
567
+ .replace("|", r"\|")
568
+ .replace("\n", "<br>")
569
+ for value in row
570
+ ]
571
+ return "| " + " | ".join(escaped) + " |"
572
+
573
+
574
+ def _render_form_chunk(
575
+ title: str,
576
+ fields: list[FormField],
577
+ lines: list[TextLine],
578
+ ) -> str:
579
+ parts = [f"### Form: {title}"] if title else []
580
+ if fields:
581
+ parts.append(
582
+ "\n".join(
583
+ [
584
+ _markdown_row(["Field", "Value"]),
585
+ "| --- | --- |",
586
+ *[_markdown_row([field.label, field.value]) for field in fields],
587
+ ]
588
+ )
589
+ )
590
+ parts.extend(line.text for line in lines)
591
+ return "\n\n".join(parts)
592
+
593
+
594
+ def _form_chunk(
595
+ parsed: ParsedDocumentResult,
596
+ form: FormBlock,
597
+ sheet_name: str,
598
+ sheet_ordinal: int,
599
+ fields: list[FormField],
600
+ lines: list[TextLine],
601
+ chunk_index: int,
602
+ oversized: bool,
603
+ *,
604
+ analysis_records: bool,
605
+ ) -> Chunk:
606
+ metadata = document_metadata(parsed)
607
+ source_ranges = [
608
+ ref.key for field in fields for ref in [*field.label_source_refs, *field.value_source_refs]
609
+ ]
610
+ source_ranges.extend(ref.key for line in lines for ref in line.source_refs)
611
+ metadata.update(
612
+ {
613
+ "chunk_type": "form_fields",
614
+ "chunk_index": chunk_index,
615
+ "sheet_name": sheet_name,
616
+ "sheet_ordinal": sheet_ordinal,
617
+ "form_id": form.form_id,
618
+ "field_ids": [field.field_id for field in fields],
619
+ "source_ranges": source_ranges,
620
+ "confidence": min([form.confidence, *[field.confidence for field in fields]]),
621
+ "warnings": list(parsed.diagnostics.warnings) if parsed.diagnostics is not None else [],
622
+ }
623
+ )
624
+ if oversized:
625
+ metadata["oversized"] = True
626
+ payload = {
627
+ "fields": [[field.label, field.value] for field in fields],
628
+ "free_text": [line.text for line in lines],
629
+ }
630
+ if analysis_records:
631
+ payload["records"] = [
632
+ {
633
+ "record_type": "field",
634
+ "field_id": field.field_id,
635
+ "label": field.label,
636
+ "value": field.value,
637
+ "label_source_refs": [ref.key for ref in field.label_source_refs],
638
+ "value_source_refs": [ref.key for ref in field.value_source_refs],
639
+ }
640
+ for field in fields
641
+ ]
642
+ payload["records"].extend(
643
+ {
644
+ "record_type": "text",
645
+ "text": line.text,
646
+ "source_refs": [ref.key for ref in line.source_refs],
647
+ }
648
+ for line in lines
649
+ )
650
+ return Chunk(
651
+ content=_render_form_chunk(form.title, fields, lines),
652
+ metadata=metadata,
653
+ structured_payload=payload,
654
+ )
655
+
656
+
657
+ def _render_matrix_chunk(matrix: MatrixBlock, rows: list[tuple]) -> str:
658
+ parts = [f"### Matrix: {matrix.title}"] if matrix.title else []
659
+ columns = ["", *[header.value for header in matrix.column_headers]]
660
+ table_lines = [
661
+ _markdown_row(columns),
662
+ "| " + " | ".join("---" for _ in columns) + " |",
663
+ *[_markdown_row([header.value, *values]) for header, values, _ in rows],
664
+ ]
665
+ parts.append("\n".join(table_lines))
666
+ return "\n\n".join(parts)
667
+
668
+
669
+ def _matrix_chunk(
670
+ parsed: ParsedDocumentResult,
671
+ matrix: MatrixBlock,
672
+ sheet_name: str,
673
+ sheet_ordinal: int,
674
+ rows: list[tuple],
675
+ chunk_index: int,
676
+ oversized: bool,
677
+ *,
678
+ analysis_records: bool,
679
+ ) -> Chunk:
680
+ metadata = document_metadata(parsed)
681
+ source_ranges = []
682
+ for header, _, refs in rows:
683
+ source_ranges.extend(ref.key for ref in header.source_refs)
684
+ source_ranges.extend(ref.key for ref in refs if ref is not None)
685
+ metadata.update(
686
+ {
687
+ "chunk_type": "matrix_rows",
688
+ "chunk_index": chunk_index,
689
+ "sheet_name": sheet_name,
690
+ "sheet_ordinal": sheet_ordinal,
691
+ "matrix_id": matrix.matrix_id,
692
+ "row_headers": [header.value for header, _, _ in rows],
693
+ "source_ranges": source_ranges,
694
+ "confidence": matrix.confidence,
695
+ "warnings": list(parsed.diagnostics.warnings) if parsed.diagnostics is not None else [],
696
+ }
697
+ )
698
+ if oversized:
699
+ metadata["oversized"] = True
700
+ payload = {
701
+ "column_headers": [header.value for header in matrix.column_headers],
702
+ "row_headers": [header.value for header, _, _ in rows],
703
+ "values": [list(values) for _, values, _ in rows],
704
+ }
705
+ if analysis_records:
706
+ payload["records"] = [
707
+ {
708
+ "row_header": header.value,
709
+ "row_header_source_refs": [ref.key for ref in header.source_refs],
710
+ "values": list(values),
711
+ "value_source_refs": [ref.key if ref is not None else None for ref in refs],
712
+ }
713
+ for header, values, refs in rows
714
+ ]
715
+ return Chunk(
716
+ content=_render_matrix_chunk(matrix, rows),
717
+ metadata=metadata,
718
+ structured_payload=payload,
719
+ )
720
+
721
+
722
+ def _text_chunk(
723
+ parsed: ParsedDocumentResult,
724
+ text: TextBlock,
725
+ sheet_name: str,
726
+ sheet_ordinal: int,
727
+ lines: list[TextLine],
728
+ chunk_index: int,
729
+ oversized: bool,
730
+ *,
731
+ analysis_records: bool,
732
+ ) -> Chunk:
733
+ metadata = document_metadata(parsed)
734
+ metadata.update(
735
+ {
736
+ "chunk_type": "text_block",
737
+ "chunk_index": chunk_index,
738
+ "sheet_name": sheet_name,
739
+ "sheet_ordinal": sheet_ordinal,
740
+ "text_id": text.text_id,
741
+ "source_ranges": [ref.key for line in lines for ref in line.source_refs],
742
+ "confidence": text.confidence,
743
+ "warnings": list(parsed.diagnostics.warnings) if parsed.diagnostics is not None else [],
744
+ }
745
+ )
746
+ if oversized:
747
+ metadata["oversized"] = True
748
+ payload = {"lines": [line.text for line in lines]}
749
+ if analysis_records:
750
+ payload["records"] = [
751
+ {"text": line.text, "source_refs": [ref.key for ref in line.source_refs]}
752
+ for line in lines
753
+ ]
754
+ return Chunk(
755
+ content="\n".join(line.text for line in lines),
756
+ metadata=metadata,
757
+ structured_payload=payload,
758
+ )
759
+
760
+
761
+ def _raw_block_grid(
762
+ workbook_ir: WorkbookIR,
763
+ sheet_ordinal: int,
764
+ source_range: str,
765
+ ) -> tuple[list[str], list[list[str]], list[int]]:
766
+ if workbook_ir.snapshot is None:
767
+ raise ValueError("WorkbookIR snapshot is required for raw block chunks")
768
+ sheet = workbook_ir.snapshot.sheets[sheet_ordinal - 1]
769
+ min_col, min_row, max_col, max_row = range_boundaries(source_range)
770
+ columns = [get_column_letter(column) for column in range(min_col, max_col + 1)]
771
+ row_numbers = list(range(min_row, max_row + 1))
772
+ rows = []
773
+ for row_number in row_numbers:
774
+ row = []
775
+ for column in range(min_col, max_col + 1):
776
+ coordinate = f"{get_column_letter(column)}{row_number}"
777
+ cell = sheet.cells.get(coordinate)
778
+ row.append("" if cell is None or cell.merge_anchor is not None else cell.display_value)
779
+ rows.append(row)
780
+ return columns, rows, row_numbers
781
+
782
+
783
+ def _render_logical_chunk(
784
+ title: str,
785
+ section_path: list[str],
786
+ columns: list[str],
787
+ rows: list[LogicalRow],
788
+ ) -> str:
789
+ headings = [f"### Table: {title}"] if title else []
790
+ if section_path:
791
+ headings.append(f"#### Section: {' / '.join(section_path)}")
792
+ table_lines = [
793
+ _markdown_row(columns),
794
+ "| " + " | ".join("---" for _ in columns) + " |",
795
+ *[_markdown_row(row.values) for row in rows],
796
+ ]
797
+ return "\n\n".join([*headings, "\n".join(table_lines)])
798
+
799
+
800
+ def _logical_chunk(
801
+ parsed: ParsedDocumentResult,
802
+ table: LogicalTable,
803
+ continuation: TableContinuation | None,
804
+ sheet_name: str,
805
+ sheet_ordinal: int,
806
+ section_path: list[str],
807
+ columns: list[str],
808
+ rows: list[LogicalRow],
809
+ chunk_index: int,
810
+ max_chunk_size: int,
811
+ length_function: Callable[[str], int],
812
+ policy: WorkbookChunkPolicy,
813
+ ) -> Chunk:
814
+ content = _render_logical_chunk(table.title, section_path, columns, rows)
815
+ metadata = document_metadata(parsed)
816
+ metadata.update(
817
+ {
818
+ "chunk_type": "table_rows",
819
+ "chunk_index": chunk_index,
820
+ "sheet_name": sheet_name,
821
+ "sheet_ordinal": sheet_ordinal,
822
+ "table_id": table.table_id,
823
+ "section_path": list(section_path),
824
+ "header_paths": [list(column.path) for column in table.columns],
825
+ "row_ids": [row.row_id for row in rows],
826
+ "row_numbers": [row.metadata["row_number"] for row in rows],
827
+ "source_ranges": [row.source_ref.key for row in rows],
828
+ "fragment_ranges": _fragment_ranges_for_rows(table, rows),
829
+ "confidence": min([table.confidence, *[row.confidence for row in rows]]),
830
+ "warnings": list(parsed.diagnostics.warnings) if parsed.diagnostics is not None else [],
831
+ }
832
+ )
833
+ if continuation is not None:
834
+ metadata.update(
835
+ {
836
+ "continuation_id": continuation.continuation_id,
837
+ "continuation_role": table.continuation_role,
838
+ "continuation_member_table_ids": list(continuation.member_table_ids),
839
+ "continuation_source_ranges": [ref.key for ref in continuation.source_refs],
840
+ }
841
+ )
842
+ if length_function(content) > max_chunk_size:
843
+ metadata["oversized"] = True
844
+ payload = {
845
+ "columns": columns,
846
+ "rows": [list(row.values) for row in rows],
847
+ "roles": [row.role for row in rows],
848
+ }
849
+ if policy.analysis_records:
850
+ payload["column_schema"] = [
851
+ {
852
+ "column_index": index,
853
+ "coordinate": column.coordinate,
854
+ "header_path": list(column.path),
855
+ }
856
+ for index, column in enumerate(table.columns)
857
+ ]
858
+ payload["records"] = [
859
+ {
860
+ "row_id": row.row_id,
861
+ "row_number": int(row.metadata["row_number"]),
862
+ "role": row.role,
863
+ "section_path": list(row.section_path),
864
+ "values": list(row.values),
865
+ "source_refs": [row.source_ref.key],
866
+ }
867
+ for row in rows
868
+ ]
869
+ return Chunk(
870
+ content=content,
871
+ metadata=metadata,
872
+ structured_payload=payload,
873
+ )
874
+
875
+
876
+ def _continuation_for_table(
877
+ workbook_ir: WorkbookIR,
878
+ table: LogicalTable,
879
+ ) -> TableContinuation | None:
880
+ if table.continuation_id is None:
881
+ return None
882
+ return next(
883
+ (
884
+ group
885
+ for group in workbook_ir.table_continuations
886
+ if group.continuation_id == table.continuation_id
887
+ and table.table_id in group.member_table_ids
888
+ ),
889
+ None,
890
+ )
891
+
892
+
893
+ def _fragment_ranges_for_rows(table: LogicalTable, rows: list[LogicalRow]) -> list[str]:
894
+ row_numbers = {int(row.metadata["row_number"]) for row in rows}
895
+ ranges = []
896
+ for fragment in table.fragments:
897
+ _, min_row, _, max_row = range_boundaries(fragment.source_ref.range)
898
+ if any(min_row <= row_number <= max_row for row_number in row_numbers):
899
+ ranges.append(fragment.source_ref.key)
900
+ return ranges