langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,993 @@
1
+ from __future__ import annotations
2
+
3
+ from collections import Counter
4
+ from copy import deepcopy
5
+ from dataclasses import asdict, dataclass, fields, replace
6
+ from typing import Any
7
+
8
+ from openpyxl.utils import get_column_letter
9
+ from openpyxl.utils.cell import coordinate_to_tuple, range_boundaries
10
+
11
+ from langparse.types import ParseDiagnostics
12
+ from langparse.workbooks.blocks import (
13
+ interpret_form_block,
14
+ interpret_matrix_block,
15
+ interpret_text_block,
16
+ )
17
+ from langparse.workbooks.classification import (
18
+ BlockClassification,
19
+ RegionAssessment,
20
+ assess_candidate_region,
21
+ classify_candidate_region,
22
+ )
23
+ from langparse.workbooks.continuation import link_table_continuations
24
+ from langparse.workbooks.modeling import (
25
+ InvalidRegionAmbiguityCaseError,
26
+ ModelCallAudit,
27
+ RegionAmbiguityCase,
28
+ RequiredWorkbookDisambiguationError,
29
+ WorkbookDisambiguation,
30
+ WorkbookModelMode,
31
+ )
32
+ from langparse.workbooks.modeling.contract import (
33
+ _candidate_envelope_has_formula,
34
+ build_region_case,
35
+ )
36
+ from langparse.workbooks.modeling.disambiguation import _audit_payload, _safe_error_type
37
+ from langparse.workbooks.modeling.types import (
38
+ REGION_PRIVACY_VERSION,
39
+ REGION_PROMPT_VERSION,
40
+ REGION_RULE_VERSION,
41
+ REGION_SCHEMA_VERSION,
42
+ REGION_VALIDATOR_VERSION,
43
+ )
44
+ from langparse.workbooks.regions import detect_candidate_regions
45
+ from langparse.workbooks.tables import interpret_logical_table
46
+ from langparse.workbooks.types import (
47
+ CandidateRegion,
48
+ CellSnapshot,
49
+ LogicalTable,
50
+ SheetIR,
51
+ SheetSnapshot,
52
+ SourceRef,
53
+ WorkbookBlock,
54
+ WorkbookIR,
55
+ WorkbookSnapshot,
56
+ stable_id,
57
+ )
58
+
59
+
60
+ @dataclass(frozen=True)
61
+ class _RegionDraft:
62
+ sheet_index: int
63
+ sheet: SheetSnapshot
64
+ candidate: CandidateRegion
65
+ assessment: RegionAssessment
66
+ case: RegionAmbiguityCase | None
67
+ case_id: str | None = None
68
+ unavailable_audit: ModelCallAudit | None = None
69
+
70
+
71
+ @dataclass(frozen=True)
72
+ class _MaterializedRegion:
73
+ draft: _RegionDraft
74
+ block: WorkbookBlock
75
+ deterministic_block: WorkbookBlock
76
+ audit: ModelCallAudit | None = None
77
+ selection_attempted: bool = False
78
+ model_selected: bool = False
79
+ materialization_failed: bool = False
80
+
81
+
82
+ def assemble_workbook(
83
+ snapshot: WorkbookSnapshot,
84
+ *,
85
+ disambiguation: WorkbookDisambiguation | None = None,
86
+ ) -> tuple[WorkbookIR, ParseDiagnostics]:
87
+ """Classify and interpret candidate regions with local raw-grid fallback."""
88
+
89
+ configured = WorkbookDisambiguation.off() if disambiguation is None else disambiguation
90
+ if configured.mode is WorkbookModelMode.OFF:
91
+ return _assemble_deterministic(snapshot)
92
+
93
+ drafts = _region_drafts(snapshot, configured)
94
+ try:
95
+ resolutions_by_case_id = _resolve_region_cases(drafts, configured)
96
+ except RequiredWorkbookDisambiguationError as error:
97
+ _raise_required_with_deterministic_fallback(snapshot, drafts, error)
98
+ materialized = [
99
+ _materialize_region(snapshot.source, draft, resolutions_by_case_id) for draft in drafts
100
+ ]
101
+ workbook_ir = _workbook_from_materialized(snapshot, materialized, rollback_selected=False)
102
+ diagnostics, tentative_validation_codes = _finalize_workbook(snapshot, workbook_ir)
103
+
104
+ attempted_regions = [region for region in materialized if region.selection_attempted]
105
+ selected_regions = [region for region in attempted_regions if region.model_selected]
106
+ reverted_case_ids: set[str] = set()
107
+ rollback_validation_codes: tuple[str, ...] = ()
108
+ materialization_rollback = any(region.materialization_failed for region in attempted_regions)
109
+ if materialization_rollback:
110
+ reverted_case_ids = {
111
+ region.draft.case.case_id
112
+ for region in attempted_regions
113
+ if region.draft.case is not None
114
+ }
115
+ workbook_ir = _workbook_from_materialized(snapshot, materialized, rollback_selected=True)
116
+ diagnostics, rollback_validation_codes = _finalize_workbook(snapshot, workbook_ir)
117
+ elif tentative_validation_codes and selected_regions:
118
+ reverted_case_ids = {
119
+ region.draft.case.case_id
120
+ for region in selected_regions
121
+ if region.draft.case is not None
122
+ }
123
+ workbook_ir = _workbook_from_materialized(snapshot, materialized, rollback_selected=True)
124
+ diagnostics, rollback_validation_codes = _finalize_workbook(snapshot, workbook_ir)
125
+
126
+ unresolved_case_ids: list[str] = []
127
+ finalized_audits: list[ModelCallAudit] = []
128
+ for region in materialized:
129
+ if region.audit is None or region.draft.case_id is None:
130
+ continue
131
+ case_id = region.draft.case_id
132
+ audit = region.audit
133
+ if region.draft.case is None:
134
+ if configured.mode is WorkbookModelMode.REQUIRED:
135
+ unresolved_case_ids.append(case_id)
136
+ elif case_id in reverted_case_ids:
137
+ if materialization_rollback:
138
+ audit = replace(
139
+ audit,
140
+ outcome="materialization_error",
141
+ validation_codes=_stable_codes(
142
+ *audit.validation_codes,
143
+ "materialization_error",
144
+ *tentative_validation_codes,
145
+ *rollback_validation_codes,
146
+ ),
147
+ reason_codes=("deterministic_fallback",),
148
+ error_type=audit.error_type if region.materialization_failed else None,
149
+ )
150
+ else:
151
+ audit = replace(
152
+ audit,
153
+ outcome="validation_error",
154
+ validation_codes=_stable_codes(
155
+ *audit.validation_codes,
156
+ *tentative_validation_codes,
157
+ *rollback_validation_codes,
158
+ ),
159
+ reason_codes=("deterministic_fallback",),
160
+ error_type=None,
161
+ )
162
+ if configured.mode is WorkbookModelMode.REQUIRED:
163
+ unresolved_case_ids.append(case_id)
164
+ elif region.model_selected:
165
+ audit = replace(
166
+ audit,
167
+ outcome="accepted",
168
+ reason_codes=("model_selected_choice",),
169
+ error_type=None,
170
+ )
171
+ finalized_audits.append(audit)
172
+
173
+ diagnostics.model_calls = [_audit_payload(audit) for audit in finalized_audits]
174
+ if unresolved_case_ids:
175
+ diagnostics.status = "failed"
176
+ raise RequiredWorkbookDisambiguationError(
177
+ tuple(unresolved_case_ids),
178
+ diagnostics,
179
+ )
180
+ return workbook_ir, diagnostics
181
+
182
+
183
+ def _assemble_deterministic(
184
+ snapshot: WorkbookSnapshot,
185
+ ) -> tuple[WorkbookIR, ParseDiagnostics]:
186
+ """Preserve the pre-model semantic assembly path byte-for-byte in behavior."""
187
+
188
+ workbook_ir, diagnostics = assemble_baseline(snapshot)
189
+ block_counts: Counter[str] = Counter()
190
+ ambiguous_regions = []
191
+ region_diagnostics = []
192
+ for sheet, sheet_ir in zip(snapshot.sheets, workbook_ir.sheets, strict=True):
193
+ semantic_blocks: list[WorkbookBlock] = []
194
+ for candidate in detect_candidate_regions(sheet):
195
+ region_diagnostics.append(_candidate_region_diagnostic(sheet.name, candidate))
196
+ try:
197
+ classification = classify_candidate_region(sheet, candidate)
198
+ block = _block_for_candidate(
199
+ snapshot.source,
200
+ sheet,
201
+ candidate,
202
+ classification,
203
+ )
204
+ except Exception as exc:
205
+ block = _unclassified_block(
206
+ snapshot.source,
207
+ candidate,
208
+ confidence=0.0,
209
+ reason_codes=["semantic_block_fallback"],
210
+ extra_diagnostic={"error_type": type(exc).__name__},
211
+ )
212
+ if block.kind == "unclassified":
213
+ reason_codes = [
214
+ diagnostic["reason_code"]
215
+ for diagnostic in block.diagnostics
216
+ if "reason_code" in diagnostic
217
+ ]
218
+ ambiguous_regions.append(
219
+ {
220
+ "sheet_name": sheet.name,
221
+ "range": candidate.source_ref.range,
222
+ "candidate_kind": "unclassified",
223
+ "confidence": block.confidence,
224
+ "reason_codes": reason_codes,
225
+ }
226
+ )
227
+ semantic_blocks.append(block)
228
+ block_counts[block.kind] += 1
229
+ sheet_ir.blocks = semantic_blocks
230
+
231
+ diagnostics.block_count_by_kind = dict(sorted(block_counts.items()))
232
+ diagnostics.ambiguous_regions = ambiguous_regions
233
+ diagnostics.region_diagnostics = region_diagnostics
234
+ try:
235
+ groups, candidates = link_table_continuations(snapshot, workbook_ir)
236
+ except Exception as exc:
237
+ diagnostics.warnings.append(f"cross_sheet_continuation_fallback:{type(exc).__name__}")
238
+ else:
239
+ workbook_ir.table_continuations = groups
240
+ diagnostics.continuation_candidates = candidates
241
+ ambiguous_count = sum(item["status"] == "ambiguous" for item in candidates)
242
+ if ambiguous_count:
243
+ diagnostics.warnings.append(
244
+ f"Workbook contains {ambiguous_count} ambiguous continuation candidates"
245
+ )
246
+ _update_coverage(snapshot, workbook_ir, diagnostics)
247
+ validity_ratio, invalid_refs = validate_workbook_source_refs(snapshot, workbook_ir)
248
+ diagnostics.source_ref_validity_ratio = validity_ratio
249
+ if invalid_refs:
250
+ diagnostics.status = "partial"
251
+ diagnostics.warnings.append(
252
+ f"Workbook IR contains {len(invalid_refs)} invalid source refs: {invalid_refs[:10]}"
253
+ )
254
+ return workbook_ir, diagnostics
255
+
256
+
257
+ def _region_drafts(
258
+ snapshot: WorkbookSnapshot,
259
+ configured: WorkbookDisambiguation,
260
+ ) -> list[_RegionDraft]:
261
+ drafts = []
262
+ for sheet_index, sheet in enumerate(snapshot.sheets):
263
+ for candidate in detect_candidate_regions(sheet):
264
+ assessment = assess_candidate_region(sheet, candidate)
265
+ case = None
266
+ case_id = None
267
+ unavailable_audit = None
268
+ if configured.mode is not WorkbookModelMode.OFF and assessment.ambiguous:
269
+ case_id = _local_region_case_id(candidate, assessment)
270
+ unavailable_outcome = _unavailable_case_outcome(sheet, candidate)
271
+ if unavailable_outcome is not None:
272
+ unavailable_audit = _local_unavailable_audit(
273
+ case_id,
274
+ candidate,
275
+ configured,
276
+ rule_confidence=assessment.deterministic.confidence,
277
+ outcome=unavailable_outcome,
278
+ )
279
+ else:
280
+ try:
281
+ case = build_region_case(sheet, candidate, assessment)
282
+ except InvalidRegionAmbiguityCaseError as error:
283
+ unavailable_audit = _local_unavailable_audit(
284
+ case_id,
285
+ candidate,
286
+ configured,
287
+ rule_confidence=assessment.deterministic.confidence,
288
+ outcome="case_unavailable",
289
+ error=error,
290
+ )
291
+ else:
292
+ case_id = case.case_id
293
+ drafts.append(
294
+ _RegionDraft(
295
+ sheet_index=sheet_index,
296
+ sheet=sheet,
297
+ candidate=candidate,
298
+ assessment=assessment,
299
+ case=case,
300
+ case_id=case_id,
301
+ unavailable_audit=unavailable_audit,
302
+ )
303
+ )
304
+ return drafts
305
+
306
+
307
+ def _local_region_case_id(
308
+ candidate: CandidateRegion,
309
+ assessment: RegionAssessment,
310
+ ) -> str:
311
+ return stable_id(
312
+ "region_case_unavailable",
313
+ REGION_RULE_VERSION,
314
+ candidate.source_ref.key,
315
+ *(choice.choice_id for choice in assessment.choices),
316
+ )
317
+
318
+
319
+ def _unavailable_case_outcome(
320
+ sheet: SheetSnapshot,
321
+ candidate: CandidateRegion,
322
+ ) -> str | None:
323
+ if sheet.visibility != "visible":
324
+ return "hidden_content"
325
+ min_column, min_row, max_column, max_row = range_boundaries(candidate.source_ref.range)
326
+ if any(min_row <= row <= max_row for row in sheet.hidden_rows):
327
+ return "hidden_content"
328
+ hidden_columns = {get_column_letter(column) for column in range(min_column, max_column + 1)}
329
+ if hidden_columns.intersection(sheet.hidden_columns):
330
+ return "hidden_content"
331
+ for coordinate, cell in sheet.cells.items():
332
+ row, column = coordinate_to_tuple(coordinate)
333
+ if min_row <= row <= max_row and min_column <= column <= max_column and cell.hidden:
334
+ return "hidden_content"
335
+ if _candidate_envelope_has_formula(sheet, candidate):
336
+ return "formula_content"
337
+ return None
338
+
339
+
340
+ def _local_unavailable_audit(
341
+ case_id: str,
342
+ candidate: CandidateRegion,
343
+ configured: WorkbookDisambiguation,
344
+ *,
345
+ rule_confidence: float,
346
+ outcome: str,
347
+ error: Exception | None = None,
348
+ ) -> ModelCallAudit:
349
+ return ModelCallAudit(
350
+ case_id=case_id,
351
+ source_range=candidate.source_ref.range,
352
+ mode=configured.mode.value,
353
+ schema_version=REGION_SCHEMA_VERSION,
354
+ prompt_version=REGION_PROMPT_VERSION,
355
+ rule_version=REGION_RULE_VERSION,
356
+ validator_version=REGION_VALIDATOR_VERSION,
357
+ privacy_version=REGION_PRIVACY_VERSION,
358
+ rule_confidence=rule_confidence,
359
+ provider=None,
360
+ model=None,
361
+ model_revision=None,
362
+ request_checksum=None,
363
+ response_checksum=None,
364
+ cache_status="not_checked",
365
+ attempts=0,
366
+ elapsed_ms=0,
367
+ request_bytes=0,
368
+ response_bytes=0,
369
+ outcome=outcome,
370
+ selected_choice_id=None,
371
+ reported_confidence=None,
372
+ validation_codes=(outcome,),
373
+ reason_codes=("deterministic_fallback",),
374
+ error_type=_safe_error_type(error),
375
+ )
376
+
377
+
378
+ def _resolve_region_cases(
379
+ drafts: list[_RegionDraft],
380
+ configured: WorkbookDisambiguation,
381
+ ):
382
+ cases = [draft.case for draft in drafts if draft.case is not None]
383
+ if not cases:
384
+ return {}
385
+ runtime = configured._runtime
386
+ if runtime is None:
387
+ raise RuntimeError("enabled workbook disambiguation runtime is unavailable")
388
+ resolutions = runtime.resolve(cases, configured)
389
+ return {resolution.case_id: resolution for resolution in resolutions.resolutions}
390
+
391
+
392
+ def _raise_required_with_deterministic_fallback(
393
+ snapshot: WorkbookSnapshot,
394
+ drafts: list[_RegionDraft],
395
+ error: RequiredWorkbookDisambiguationError,
396
+ ) -> None:
397
+ materialized = [_deterministic_materialized_region(snapshot.source, draft) for draft in drafts]
398
+ workbook_ir = _workbook_from_materialized(snapshot, materialized, rollback_selected=False)
399
+ diagnostics, _ = _finalize_workbook(snapshot, workbook_ir)
400
+ diagnostics.model_calls = _ordered_error_audits(drafts, error.diagnostics.model_calls)
401
+ diagnostics.status = "failed"
402
+ raise RequiredWorkbookDisambiguationError(
403
+ _ordered_unresolved_case_ids(drafts, error.case_ids),
404
+ diagnostics,
405
+ ) from None
406
+
407
+
408
+ def _ordered_error_audits(
409
+ drafts: list[_RegionDraft],
410
+ error_audits: list[dict[str, object]],
411
+ ) -> list[dict[str, object]]:
412
+ audits_by_case_id = {
413
+ audit["case_id"]: _audit_field_payload(audit)
414
+ for audit in error_audits
415
+ if isinstance(audit.get("case_id"), str)
416
+ }
417
+ audits = []
418
+ for draft in drafts:
419
+ if draft.unavailable_audit is not None:
420
+ audits.append(_audit_payload(draft.unavailable_audit))
421
+ elif draft.case_id is not None and draft.case_id in audits_by_case_id:
422
+ audits.append(audits_by_case_id[draft.case_id])
423
+ return audits
424
+
425
+
426
+ def _audit_field_payload(audit: dict[str, object]) -> dict[str, object]:
427
+ return {field.name: audit[field.name] for field in fields(ModelCallAudit)}
428
+
429
+
430
+ def _ordered_unresolved_case_ids(
431
+ drafts: list[_RegionDraft],
432
+ disambiguator_case_ids: tuple[str, ...],
433
+ ) -> tuple[str, ...]:
434
+ unresolved_case_ids = set(disambiguator_case_ids)
435
+ unresolved_case_ids.update(
436
+ draft.case_id
437
+ for draft in drafts
438
+ if draft.case_id is not None and draft.unavailable_audit is not None
439
+ )
440
+ ordered = [
441
+ draft.case_id
442
+ for draft in drafts
443
+ if draft.case_id is not None and draft.case_id in unresolved_case_ids
444
+ ]
445
+ ordered.extend(case_id for case_id in disambiguator_case_ids if case_id not in ordered)
446
+ return tuple(ordered)
447
+
448
+
449
+ def _deterministic_materialized_region(
450
+ snapshot_source: str,
451
+ draft: _RegionDraft,
452
+ ) -> _MaterializedRegion:
453
+ deterministic_block = _materialize_deterministic(snapshot_source, draft)
454
+ return _MaterializedRegion(
455
+ draft=draft,
456
+ block=deterministic_block,
457
+ deterministic_block=deterministic_block,
458
+ audit=draft.unavailable_audit,
459
+ )
460
+
461
+
462
+ def _materialize_region(snapshot_source, draft: _RegionDraft, resolutions_by_case_id):
463
+ deterministic_region = _deterministic_materialized_region(snapshot_source, draft)
464
+ deterministic_block = deterministic_region.deterministic_block
465
+ if draft.case is None:
466
+ return deterministic_region
467
+
468
+ resolution = resolutions_by_case_id[draft.case.case_id]
469
+ if resolution.status == "local_fallback":
470
+ return _MaterializedRegion(
471
+ draft=draft,
472
+ block=deterministic_block,
473
+ deterministic_block=deterministic_block,
474
+ audit=_normalize_fallback_audit(resolution.audit),
475
+ )
476
+
477
+ choice = next(
478
+ choice for choice in draft.case.choices if choice.choice_id == resolution.choice_id
479
+ )
480
+ classification = BlockClassification(
481
+ kind=choice.kind,
482
+ confidence=choice.local_score,
483
+ reason_codes=[*choice.reason_codes, "model_selected_choice"],
484
+ features=draft.assessment.deterministic.features,
485
+ )
486
+ assert resolution.audit is not None
487
+ try:
488
+ block = _block_for_candidate(
489
+ snapshot_source,
490
+ draft.sheet,
491
+ draft.candidate,
492
+ classification,
493
+ )
494
+ except Exception as exc:
495
+ audit = replace(
496
+ resolution.audit,
497
+ outcome="materialization_error",
498
+ validation_codes=_stable_codes(
499
+ *resolution.audit.validation_codes,
500
+ "materialization_error",
501
+ ),
502
+ reason_codes=("semantic_block_fallback",),
503
+ error_type=_safe_error_type(exc),
504
+ )
505
+ return _MaterializedRegion(
506
+ draft=draft,
507
+ block=deterministic_block,
508
+ deterministic_block=deterministic_block,
509
+ audit=audit,
510
+ selection_attempted=True,
511
+ materialization_failed=True,
512
+ )
513
+ return _MaterializedRegion(
514
+ draft=draft,
515
+ block=block,
516
+ deterministic_block=deterministic_block,
517
+ audit=resolution.audit,
518
+ selection_attempted=True,
519
+ model_selected=True,
520
+ )
521
+
522
+
523
+ def _materialize_deterministic(snapshot_source, draft: _RegionDraft) -> WorkbookBlock:
524
+ try:
525
+ return _block_for_candidate(
526
+ snapshot_source,
527
+ draft.sheet,
528
+ draft.candidate,
529
+ draft.assessment.deterministic,
530
+ )
531
+ except Exception as exc:
532
+ return _unclassified_block(
533
+ snapshot_source,
534
+ draft.candidate,
535
+ confidence=0.0,
536
+ reason_codes=["semantic_block_fallback"],
537
+ extra_diagnostic={"error_type": type(exc).__name__},
538
+ )
539
+
540
+
541
+ def _normalize_fallback_audit(audit: ModelCallAudit | None) -> ModelCallAudit | None:
542
+ if audit is None:
543
+ return None
544
+ outcome = (
545
+ "provider_error"
546
+ if audit.outcome in {"adapter_error", "deadline_exceeded", "timeout"}
547
+ else audit.outcome
548
+ )
549
+ return replace(
550
+ audit,
551
+ outcome=outcome,
552
+ reason_codes=("deterministic_fallback",),
553
+ )
554
+
555
+
556
+ def _workbook_from_materialized(
557
+ snapshot: WorkbookSnapshot,
558
+ materialized: list[_MaterializedRegion],
559
+ *,
560
+ rollback_selected: bool,
561
+ ) -> WorkbookIR:
562
+ workbook_ir, _ = assemble_baseline(snapshot)
563
+ blocks_by_sheet: dict[int, list[WorkbookBlock]] = {
564
+ index: [] for index in range(len(snapshot.sheets))
565
+ }
566
+ for region in materialized:
567
+ block = (
568
+ region.deterministic_block
569
+ if rollback_selected and region.selection_attempted
570
+ else region.block
571
+ )
572
+ blocks_by_sheet[region.draft.sheet_index].append(deepcopy(block))
573
+ for sheet_index, sheet_ir in enumerate(workbook_ir.sheets):
574
+ sheet_ir.blocks = blocks_by_sheet[sheet_index]
575
+ return workbook_ir
576
+
577
+
578
+ def _finalize_workbook(
579
+ snapshot: WorkbookSnapshot,
580
+ workbook_ir: WorkbookIR,
581
+ ) -> tuple[ParseDiagnostics, tuple[str, ...]]:
582
+ baseline, diagnostics = assemble_baseline(snapshot)
583
+ workbook_ir.lineage = baseline.lineage
584
+ block_counts: Counter[str] = Counter()
585
+ ambiguous_regions = []
586
+ region_diagnostics = []
587
+ for sheet_ir in workbook_ir.sheets:
588
+ for block in sheet_ir.blocks:
589
+ region_diagnostic = _block_region_diagnostic(sheet_ir.name, block)
590
+ if region_diagnostic is not None:
591
+ region_diagnostics.append(region_diagnostic)
592
+ if block.kind == "unclassified":
593
+ reason_codes = [
594
+ diagnostic["reason_code"]
595
+ for diagnostic in block.diagnostics
596
+ if "reason_code" in diagnostic
597
+ ]
598
+ ambiguous_regions.append(
599
+ {
600
+ "sheet_name": sheet_ir.name,
601
+ "range": block.source_refs[0].range,
602
+ "candidate_kind": "unclassified",
603
+ "confidence": block.confidence,
604
+ "reason_codes": reason_codes,
605
+ }
606
+ )
607
+ block_counts[block.kind] += 1
608
+
609
+ diagnostics.block_count_by_kind = dict(sorted(block_counts.items()))
610
+ diagnostics.ambiguous_regions = ambiguous_regions
611
+ diagnostics.region_diagnostics = region_diagnostics
612
+ continuation_failed = False
613
+ try:
614
+ groups, candidates = link_table_continuations(snapshot, workbook_ir)
615
+ except Exception as exc:
616
+ continuation_failed = True
617
+ diagnostics.warnings.append(f"cross_sheet_continuation_fallback:{type(exc).__name__}")
618
+ else:
619
+ workbook_ir.table_continuations = groups
620
+ diagnostics.continuation_candidates = candidates
621
+ ambiguous_count = sum(item["status"] == "ambiguous" for item in candidates)
622
+ if ambiguous_count:
623
+ diagnostics.warnings.append(
624
+ f"Workbook contains {ambiguous_count} ambiguous continuation candidates"
625
+ )
626
+ _update_coverage(snapshot, workbook_ir, diagnostics)
627
+ row_conservation_passed = _row_conservation_passed(workbook_ir)
628
+ if not row_conservation_passed:
629
+ diagnostics.status = "partial"
630
+ diagnostics.warnings.append("Workbook IR failed logical row conservation")
631
+ validity_ratio, invalid_refs = validate_workbook_source_refs(snapshot, workbook_ir)
632
+ diagnostics.source_ref_validity_ratio = validity_ratio
633
+ if invalid_refs:
634
+ diagnostics.status = "partial"
635
+ diagnostics.warnings.append(
636
+ f"Workbook IR contains {len(invalid_refs)} invalid source refs: {invalid_refs[:10]}"
637
+ )
638
+ validation_codes = []
639
+ if diagnostics.coverage_ratio != 1.0:
640
+ validation_codes.append("invalid_coverage")
641
+ if not diagnostics.reconstruction_passed:
642
+ validation_codes.append("reconstruction_failed")
643
+ if not row_conservation_passed:
644
+ validation_codes.append("row_conservation_failed")
645
+ if diagnostics.source_ref_validity_ratio != 1.0 or invalid_refs:
646
+ validation_codes.append("invalid_source_refs")
647
+ if continuation_failed:
648
+ validation_codes.append("continuation_error")
649
+ return diagnostics, tuple(validation_codes)
650
+
651
+
652
+ def _row_conservation_passed(workbook_ir: WorkbookIR) -> bool:
653
+ for sheet_ir in workbook_ir.sheets:
654
+ for block in sheet_ir.blocks:
655
+ if block.logical_table is None or not block.source_refs:
656
+ continue
657
+ min_col, min_row, max_col, max_row = range_boundaries(block.source_refs[0].range)
658
+ expected = [
659
+ (sheet_ir.name, min_col, row_number, max_col, row_number)
660
+ for row_number in range(min_row, max_row + 1)
661
+ ]
662
+ actual = []
663
+ for row in block.logical_table.rows:
664
+ row_min_col, row_min, row_max_col, row_max = range_boundaries(row.source_ref.range)
665
+ actual.append(
666
+ (
667
+ row.source_ref.sheet_name,
668
+ row_min_col,
669
+ row_min,
670
+ row_max_col,
671
+ row_max,
672
+ )
673
+ )
674
+ if actual != expected:
675
+ return False
676
+ return True
677
+
678
+
679
+ def _stable_codes(*codes: str) -> tuple[str, ...]:
680
+ return tuple(dict.fromkeys(codes))
681
+
682
+
683
+ def _candidate_region_diagnostic(
684
+ sheet_name: str,
685
+ candidate: CandidateRegion,
686
+ ) -> dict[str, Any]:
687
+ return {
688
+ "sheet_name": sheet_name,
689
+ "range": candidate.source_ref.range,
690
+ "reason_codes": list(candidate.reason_codes),
691
+ "confidence": candidate.confidence,
692
+ "conflicts": deepcopy(candidate.diagnostics),
693
+ }
694
+
695
+
696
+ def _block_region_diagnostic(
697
+ sheet_name: str,
698
+ block: WorkbookBlock,
699
+ ) -> dict[str, Any] | None:
700
+ reason_codes = block.metadata.get("region_reason_codes")
701
+ confidence = block.metadata.get("region_confidence")
702
+ if reason_codes is None or confidence is None or not block.source_refs:
703
+ return None
704
+ return {
705
+ "sheet_name": sheet_name,
706
+ "range": block.source_refs[0].range,
707
+ "reason_codes": list(reason_codes),
708
+ "confidence": confidence,
709
+ "conflicts": deepcopy(block.metadata.get("region_diagnostics", [])),
710
+ }
711
+
712
+
713
+ def _block_for_candidate(snapshot_source, sheet, candidate, classification):
714
+ common = {
715
+ "block_id": stable_id(
716
+ "block", snapshot_source, candidate.source_ref.key, classification.kind
717
+ ),
718
+ "kind": classification.kind,
719
+ "source_refs": [candidate.source_ref],
720
+ "cell_refs": candidate.cell_refs,
721
+ "confidence": classification.confidence,
722
+ "metadata": {
723
+ "view": "raw_grid"
724
+ if classification.kind == "unclassified"
725
+ else f"semantic_{classification.kind}",
726
+ "cell_count": len(candidate.cell_refs),
727
+ "features": asdict(classification.features),
728
+ "reason_codes": list(classification.reason_codes),
729
+ "region_reason_codes": list(candidate.reason_codes),
730
+ "region_confidence": candidate.confidence,
731
+ "region_diagnostics": deepcopy(candidate.diagnostics),
732
+ },
733
+ "diagnostics": [
734
+ {"reason_code": reason_code} for reason_code in classification.reason_codes
735
+ ],
736
+ }
737
+ if classification.kind == "logical_table":
738
+ table = interpret_logical_table(sheet, candidate)
739
+ common["confidence"] = min(classification.confidence, table.confidence)
740
+ return WorkbookBlock(**common, logical_table=table)
741
+ if classification.kind == "form":
742
+ return WorkbookBlock(
743
+ **common,
744
+ form=interpret_form_block(sheet, candidate, classification),
745
+ )
746
+ if classification.kind == "matrix":
747
+ return WorkbookBlock(
748
+ **common,
749
+ matrix=interpret_matrix_block(sheet, candidate, classification),
750
+ )
751
+ if classification.kind == "text":
752
+ return WorkbookBlock(
753
+ **common,
754
+ text=interpret_text_block(sheet, candidate, classification),
755
+ )
756
+ return WorkbookBlock(**common)
757
+
758
+
759
+ def _unclassified_block(
760
+ snapshot_source,
761
+ candidate,
762
+ *,
763
+ confidence,
764
+ reason_codes,
765
+ extra_diagnostic=None,
766
+ ):
767
+ diagnostics = [{"reason_code": reason_code} for reason_code in reason_codes]
768
+ if extra_diagnostic:
769
+ diagnostics[0].update(extra_diagnostic)
770
+ return WorkbookBlock(
771
+ block_id=stable_id("block", snapshot_source, candidate.source_ref.key, "unclassified"),
772
+ kind="unclassified",
773
+ source_refs=[candidate.source_ref],
774
+ cell_refs=candidate.cell_refs,
775
+ confidence=confidence,
776
+ metadata={
777
+ "view": "raw_grid",
778
+ "cell_count": len(candidate.cell_refs),
779
+ "reason_codes": list(reason_codes),
780
+ "region_reason_codes": list(candidate.reason_codes),
781
+ "region_confidence": candidate.confidence,
782
+ "region_diagnostics": deepcopy(candidate.diagnostics),
783
+ },
784
+ diagnostics=diagnostics,
785
+ )
786
+
787
+
788
+ def validate_workbook_source_refs(
789
+ snapshot: WorkbookSnapshot,
790
+ workbook_ir: WorkbookIR,
791
+ ) -> tuple[float, list[str]]:
792
+ """Return the ratio of derived refs bounded by an existing source Sheet."""
793
+
794
+ sheet_bounds = {
795
+ sheet.name: range_boundaries(sheet.used_range or _range_for_coordinates(list(sheet.cells)))
796
+ for sheet in snapshot.sheets
797
+ if sheet.used_range or sheet.cells
798
+ }
799
+ refs = []
800
+ drawing_refs = []
801
+ for sheet_ir in workbook_ir.sheets:
802
+ for block in sheet_ir.blocks:
803
+ if block.kind in {"chart", "image"}:
804
+ drawing_refs.extend(block.source_refs)
805
+ continue
806
+ refs.extend(block.source_refs)
807
+ if block.logical_table is not None:
808
+ refs.extend(_logical_table_source_refs(block.logical_table))
809
+ if block.form is not None:
810
+ refs.extend(block.form.source_refs)
811
+ refs.extend(ref for field in block.form.fields for ref in field.label_source_refs)
812
+ refs.extend(ref for field in block.form.fields for ref in field.value_source_refs)
813
+ refs.extend(ref for line in block.form.free_text for ref in line.source_refs)
814
+ if block.matrix is not None:
815
+ refs.extend(block.matrix.source_refs)
816
+ refs.extend(
817
+ ref for header in block.matrix.row_headers for ref in header.source_refs
818
+ )
819
+ refs.extend(
820
+ ref for header in block.matrix.column_headers for ref in header.source_refs
821
+ )
822
+ refs.extend(
823
+ ref for row in block.matrix.value_source_refs for ref in row if ref is not None
824
+ )
825
+ if block.text is not None:
826
+ refs.extend(block.text.source_refs)
827
+ refs.extend(ref for line in block.text.lines for ref in line.source_refs)
828
+
829
+ for continuation in workbook_ir.table_continuations:
830
+ refs.extend(_logical_table_source_refs(continuation.logical_table))
831
+ refs.extend(continuation.source_refs)
832
+
833
+ invalid = sorted({ref.key for ref in refs if not _valid_source_ref(ref, sheet_bounds)})
834
+ invalid_count = sum(not _valid_source_ref(ref, sheet_bounds) for ref in refs)
835
+ if drawing_refs:
836
+ from langparse.workbooks.objects import validate_object_source
837
+
838
+ for ref in drawing_refs:
839
+ try:
840
+ validate_object_source(snapshot, ref.key)
841
+ except ValueError:
842
+ invalid_count += 1
843
+ invalid.append(ref.key)
844
+ count = len(refs) + len(drawing_refs)
845
+ ratio = (count - invalid_count) / count if count else 1.0
846
+ invalid = sorted(set(invalid))
847
+ return ratio, invalid
848
+
849
+
850
+ def _logical_table_source_refs(table: LogicalTable) -> list[SourceRef]:
851
+ return [
852
+ *table.source_refs,
853
+ *(ref for column in table.columns for ref in column.source_refs),
854
+ *(row.source_ref for row in table.rows),
855
+ *(fragment.source_ref for fragment in table.fragments),
856
+ *(section.source_ref for section in table.sections),
857
+ ]
858
+
859
+
860
+ def _valid_source_ref(ref: SourceRef, sheet_bounds) -> bool:
861
+ bounds = sheet_bounds.get(ref.sheet_name)
862
+ if bounds is None:
863
+ return False
864
+ min_col, min_row, max_col, max_row = range_boundaries(ref.range)
865
+ sheet_min_col, sheet_min_row, sheet_max_col, sheet_max_row = bounds
866
+ return (
867
+ sheet_min_col <= min_col <= max_col <= sheet_max_col
868
+ and sheet_min_row <= min_row <= max_row <= sheet_max_row
869
+ )
870
+
871
+
872
+ def _update_coverage(snapshot, workbook_ir, diagnostics):
873
+ source_refs = {
874
+ f"{sheet.name}!{coordinate}"
875
+ for sheet in snapshot.sheets
876
+ for coordinate, cell in sheet.cells.items()
877
+ if _is_assignable_cell(cell)
878
+ }
879
+ assigned_refs = {
880
+ f"{sheet_ir.name}!{coordinate}"
881
+ for sheet_ir in workbook_ir.sheets
882
+ for block in sheet_ir.blocks
883
+ for coordinate in block.cell_refs
884
+ }
885
+ diagnostics.coverage_ratio = (
886
+ len(source_refs & assigned_refs) / len(source_refs) if source_refs else 1.0
887
+ )
888
+ diagnostics.reconstruction_passed = source_refs == assigned_refs
889
+ if not diagnostics.reconstruction_passed:
890
+ diagnostics.status = "partial"
891
+
892
+
893
+ def assemble_baseline(snapshot: WorkbookSnapshot) -> tuple[WorkbookIR, ParseDiagnostics]:
894
+ """Create a lossless raw-grid IR before any semantic table interpretation."""
895
+
896
+ sheet_irs: list[SheetIR] = []
897
+ source_cell_keys: set[str] = set()
898
+ assigned_cell_keys: set[str] = set()
899
+ block_counts: Counter[str] = Counter()
900
+
901
+ for sheet in snapshot.sheets:
902
+ cell_refs = sorted(
903
+ (coordinate for coordinate, cell in sheet.cells.items() if _is_assignable_cell(cell)),
904
+ key=coordinate_to_tuple,
905
+ )
906
+ blocks: list[WorkbookBlock] = []
907
+ if cell_refs:
908
+ source_range = sheet.used_range or _range_for_coordinates(cell_refs)
909
+ source_ref = SourceRef(sheet_name=sheet.name, range=source_range)
910
+ block = WorkbookBlock(
911
+ block_id=stable_id("block", snapshot.source, source_ref.key, "unclassified"),
912
+ kind="unclassified",
913
+ source_refs=[source_ref],
914
+ cell_refs=cell_refs,
915
+ metadata={"view": "raw_grid", "cell_count": len(cell_refs)},
916
+ )
917
+ blocks.append(block)
918
+ block_counts[block.kind] += 1
919
+
920
+ sheet_irs.append(
921
+ SheetIR(
922
+ sheet_id=stable_id("sheet", snapshot.source, str(sheet.index), sheet.name),
923
+ name=sheet.name,
924
+ index=sheet.index,
925
+ blocks=blocks,
926
+ visibility=sheet.visibility,
927
+ metadata={
928
+ "used_range": sheet.used_range,
929
+ "print_area": sheet.print_area,
930
+ "merged_ranges": sheet.merged_ranges,
931
+ "object_count": len(sheet.objects),
932
+ },
933
+ )
934
+ )
935
+
936
+ qualified_refs = {f"{sheet.name}!{coordinate}" for coordinate in cell_refs}
937
+ source_cell_keys.update(qualified_refs)
938
+ assigned_cell_keys.update(qualified_refs if blocks else set())
939
+
940
+ reconstruction_passed = assigned_cell_keys == source_cell_keys
941
+ coverage_ratio = (
942
+ len(assigned_cell_keys & source_cell_keys) / len(source_cell_keys)
943
+ if source_cell_keys
944
+ else 1.0
945
+ )
946
+ warnings = list(snapshot.metadata.get("warnings", []))
947
+ if not reconstruction_passed:
948
+ missing = sorted(source_cell_keys - assigned_cell_keys)
949
+ warnings.append(f"Workbook IR omitted {len(missing)} source cells: {missing[:10]}")
950
+
951
+ diagnostics = ParseDiagnostics(
952
+ status="success" if reconstruction_passed and coverage_ratio == 1.0 else "partial",
953
+ coverage_ratio=coverage_ratio,
954
+ reconstruction_passed=reconstruction_passed,
955
+ block_count_by_kind=dict(sorted(block_counts.items())),
956
+ unsupported_features=list(snapshot.metadata.get("unsupported_features", [])),
957
+ warnings=warnings,
958
+ )
959
+ workbook_ir = WorkbookIR(
960
+ kind="workbook",
961
+ workbook_id=stable_id("workbook", snapshot.source, snapshot.filename),
962
+ source=snapshot.source,
963
+ sheets=sheet_irs,
964
+ filename=snapshot.filename,
965
+ snapshot=snapshot,
966
+ metadata={"snapshot": snapshot.metadata},
967
+ )
968
+ from langparse.workbooks.lineage import build_lineage
969
+
970
+ workbook_ir.lineage = build_lineage(snapshot)
971
+ diagnostics.reference_diagnostics = [asdict(item) for item in workbook_ir.lineage.diagnostics]
972
+ return workbook_ir, diagnostics
973
+
974
+
975
+ def _is_assignable_cell(cell: CellSnapshot) -> bool:
976
+ return any(
977
+ (
978
+ cell.raw_value is not None,
979
+ cell.formula is not None,
980
+ cell.comment is not None,
981
+ cell.hyperlink is not None,
982
+ cell.merge_anchor is not None,
983
+ )
984
+ )
985
+
986
+
987
+ def _range_for_coordinates(coordinates: list[str]) -> str:
988
+ positions = [coordinate_to_tuple(coordinate) for coordinate in coordinates]
989
+ rows = [row for row, _ in positions]
990
+ columns = [column for _, column in positions]
991
+ return (
992
+ f"{get_column_letter(min(columns))}{min(rows)}:{get_column_letter(max(columns))}{max(rows)}"
993
+ )