langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,993 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections import Counter
|
|
4
|
+
from copy import deepcopy
|
|
5
|
+
from dataclasses import asdict, dataclass, fields, replace
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from openpyxl.utils import get_column_letter
|
|
9
|
+
from openpyxl.utils.cell import coordinate_to_tuple, range_boundaries
|
|
10
|
+
|
|
11
|
+
from langparse.types import ParseDiagnostics
|
|
12
|
+
from langparse.workbooks.blocks import (
|
|
13
|
+
interpret_form_block,
|
|
14
|
+
interpret_matrix_block,
|
|
15
|
+
interpret_text_block,
|
|
16
|
+
)
|
|
17
|
+
from langparse.workbooks.classification import (
|
|
18
|
+
BlockClassification,
|
|
19
|
+
RegionAssessment,
|
|
20
|
+
assess_candidate_region,
|
|
21
|
+
classify_candidate_region,
|
|
22
|
+
)
|
|
23
|
+
from langparse.workbooks.continuation import link_table_continuations
|
|
24
|
+
from langparse.workbooks.modeling import (
|
|
25
|
+
InvalidRegionAmbiguityCaseError,
|
|
26
|
+
ModelCallAudit,
|
|
27
|
+
RegionAmbiguityCase,
|
|
28
|
+
RequiredWorkbookDisambiguationError,
|
|
29
|
+
WorkbookDisambiguation,
|
|
30
|
+
WorkbookModelMode,
|
|
31
|
+
)
|
|
32
|
+
from langparse.workbooks.modeling.contract import (
|
|
33
|
+
_candidate_envelope_has_formula,
|
|
34
|
+
build_region_case,
|
|
35
|
+
)
|
|
36
|
+
from langparse.workbooks.modeling.disambiguation import _audit_payload, _safe_error_type
|
|
37
|
+
from langparse.workbooks.modeling.types import (
|
|
38
|
+
REGION_PRIVACY_VERSION,
|
|
39
|
+
REGION_PROMPT_VERSION,
|
|
40
|
+
REGION_RULE_VERSION,
|
|
41
|
+
REGION_SCHEMA_VERSION,
|
|
42
|
+
REGION_VALIDATOR_VERSION,
|
|
43
|
+
)
|
|
44
|
+
from langparse.workbooks.regions import detect_candidate_regions
|
|
45
|
+
from langparse.workbooks.tables import interpret_logical_table
|
|
46
|
+
from langparse.workbooks.types import (
|
|
47
|
+
CandidateRegion,
|
|
48
|
+
CellSnapshot,
|
|
49
|
+
LogicalTable,
|
|
50
|
+
SheetIR,
|
|
51
|
+
SheetSnapshot,
|
|
52
|
+
SourceRef,
|
|
53
|
+
WorkbookBlock,
|
|
54
|
+
WorkbookIR,
|
|
55
|
+
WorkbookSnapshot,
|
|
56
|
+
stable_id,
|
|
57
|
+
)
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass(frozen=True)
|
|
61
|
+
class _RegionDraft:
|
|
62
|
+
sheet_index: int
|
|
63
|
+
sheet: SheetSnapshot
|
|
64
|
+
candidate: CandidateRegion
|
|
65
|
+
assessment: RegionAssessment
|
|
66
|
+
case: RegionAmbiguityCase | None
|
|
67
|
+
case_id: str | None = None
|
|
68
|
+
unavailable_audit: ModelCallAudit | None = None
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
@dataclass(frozen=True)
|
|
72
|
+
class _MaterializedRegion:
|
|
73
|
+
draft: _RegionDraft
|
|
74
|
+
block: WorkbookBlock
|
|
75
|
+
deterministic_block: WorkbookBlock
|
|
76
|
+
audit: ModelCallAudit | None = None
|
|
77
|
+
selection_attempted: bool = False
|
|
78
|
+
model_selected: bool = False
|
|
79
|
+
materialization_failed: bool = False
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def assemble_workbook(
|
|
83
|
+
snapshot: WorkbookSnapshot,
|
|
84
|
+
*,
|
|
85
|
+
disambiguation: WorkbookDisambiguation | None = None,
|
|
86
|
+
) -> tuple[WorkbookIR, ParseDiagnostics]:
|
|
87
|
+
"""Classify and interpret candidate regions with local raw-grid fallback."""
|
|
88
|
+
|
|
89
|
+
configured = WorkbookDisambiguation.off() if disambiguation is None else disambiguation
|
|
90
|
+
if configured.mode is WorkbookModelMode.OFF:
|
|
91
|
+
return _assemble_deterministic(snapshot)
|
|
92
|
+
|
|
93
|
+
drafts = _region_drafts(snapshot, configured)
|
|
94
|
+
try:
|
|
95
|
+
resolutions_by_case_id = _resolve_region_cases(drafts, configured)
|
|
96
|
+
except RequiredWorkbookDisambiguationError as error:
|
|
97
|
+
_raise_required_with_deterministic_fallback(snapshot, drafts, error)
|
|
98
|
+
materialized = [
|
|
99
|
+
_materialize_region(snapshot.source, draft, resolutions_by_case_id) for draft in drafts
|
|
100
|
+
]
|
|
101
|
+
workbook_ir = _workbook_from_materialized(snapshot, materialized, rollback_selected=False)
|
|
102
|
+
diagnostics, tentative_validation_codes = _finalize_workbook(snapshot, workbook_ir)
|
|
103
|
+
|
|
104
|
+
attempted_regions = [region for region in materialized if region.selection_attempted]
|
|
105
|
+
selected_regions = [region for region in attempted_regions if region.model_selected]
|
|
106
|
+
reverted_case_ids: set[str] = set()
|
|
107
|
+
rollback_validation_codes: tuple[str, ...] = ()
|
|
108
|
+
materialization_rollback = any(region.materialization_failed for region in attempted_regions)
|
|
109
|
+
if materialization_rollback:
|
|
110
|
+
reverted_case_ids = {
|
|
111
|
+
region.draft.case.case_id
|
|
112
|
+
for region in attempted_regions
|
|
113
|
+
if region.draft.case is not None
|
|
114
|
+
}
|
|
115
|
+
workbook_ir = _workbook_from_materialized(snapshot, materialized, rollback_selected=True)
|
|
116
|
+
diagnostics, rollback_validation_codes = _finalize_workbook(snapshot, workbook_ir)
|
|
117
|
+
elif tentative_validation_codes and selected_regions:
|
|
118
|
+
reverted_case_ids = {
|
|
119
|
+
region.draft.case.case_id
|
|
120
|
+
for region in selected_regions
|
|
121
|
+
if region.draft.case is not None
|
|
122
|
+
}
|
|
123
|
+
workbook_ir = _workbook_from_materialized(snapshot, materialized, rollback_selected=True)
|
|
124
|
+
diagnostics, rollback_validation_codes = _finalize_workbook(snapshot, workbook_ir)
|
|
125
|
+
|
|
126
|
+
unresolved_case_ids: list[str] = []
|
|
127
|
+
finalized_audits: list[ModelCallAudit] = []
|
|
128
|
+
for region in materialized:
|
|
129
|
+
if region.audit is None or region.draft.case_id is None:
|
|
130
|
+
continue
|
|
131
|
+
case_id = region.draft.case_id
|
|
132
|
+
audit = region.audit
|
|
133
|
+
if region.draft.case is None:
|
|
134
|
+
if configured.mode is WorkbookModelMode.REQUIRED:
|
|
135
|
+
unresolved_case_ids.append(case_id)
|
|
136
|
+
elif case_id in reverted_case_ids:
|
|
137
|
+
if materialization_rollback:
|
|
138
|
+
audit = replace(
|
|
139
|
+
audit,
|
|
140
|
+
outcome="materialization_error",
|
|
141
|
+
validation_codes=_stable_codes(
|
|
142
|
+
*audit.validation_codes,
|
|
143
|
+
"materialization_error",
|
|
144
|
+
*tentative_validation_codes,
|
|
145
|
+
*rollback_validation_codes,
|
|
146
|
+
),
|
|
147
|
+
reason_codes=("deterministic_fallback",),
|
|
148
|
+
error_type=audit.error_type if region.materialization_failed else None,
|
|
149
|
+
)
|
|
150
|
+
else:
|
|
151
|
+
audit = replace(
|
|
152
|
+
audit,
|
|
153
|
+
outcome="validation_error",
|
|
154
|
+
validation_codes=_stable_codes(
|
|
155
|
+
*audit.validation_codes,
|
|
156
|
+
*tentative_validation_codes,
|
|
157
|
+
*rollback_validation_codes,
|
|
158
|
+
),
|
|
159
|
+
reason_codes=("deterministic_fallback",),
|
|
160
|
+
error_type=None,
|
|
161
|
+
)
|
|
162
|
+
if configured.mode is WorkbookModelMode.REQUIRED:
|
|
163
|
+
unresolved_case_ids.append(case_id)
|
|
164
|
+
elif region.model_selected:
|
|
165
|
+
audit = replace(
|
|
166
|
+
audit,
|
|
167
|
+
outcome="accepted",
|
|
168
|
+
reason_codes=("model_selected_choice",),
|
|
169
|
+
error_type=None,
|
|
170
|
+
)
|
|
171
|
+
finalized_audits.append(audit)
|
|
172
|
+
|
|
173
|
+
diagnostics.model_calls = [_audit_payload(audit) for audit in finalized_audits]
|
|
174
|
+
if unresolved_case_ids:
|
|
175
|
+
diagnostics.status = "failed"
|
|
176
|
+
raise RequiredWorkbookDisambiguationError(
|
|
177
|
+
tuple(unresolved_case_ids),
|
|
178
|
+
diagnostics,
|
|
179
|
+
)
|
|
180
|
+
return workbook_ir, diagnostics
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _assemble_deterministic(
|
|
184
|
+
snapshot: WorkbookSnapshot,
|
|
185
|
+
) -> tuple[WorkbookIR, ParseDiagnostics]:
|
|
186
|
+
"""Preserve the pre-model semantic assembly path byte-for-byte in behavior."""
|
|
187
|
+
|
|
188
|
+
workbook_ir, diagnostics = assemble_baseline(snapshot)
|
|
189
|
+
block_counts: Counter[str] = Counter()
|
|
190
|
+
ambiguous_regions = []
|
|
191
|
+
region_diagnostics = []
|
|
192
|
+
for sheet, sheet_ir in zip(snapshot.sheets, workbook_ir.sheets, strict=True):
|
|
193
|
+
semantic_blocks: list[WorkbookBlock] = []
|
|
194
|
+
for candidate in detect_candidate_regions(sheet):
|
|
195
|
+
region_diagnostics.append(_candidate_region_diagnostic(sheet.name, candidate))
|
|
196
|
+
try:
|
|
197
|
+
classification = classify_candidate_region(sheet, candidate)
|
|
198
|
+
block = _block_for_candidate(
|
|
199
|
+
snapshot.source,
|
|
200
|
+
sheet,
|
|
201
|
+
candidate,
|
|
202
|
+
classification,
|
|
203
|
+
)
|
|
204
|
+
except Exception as exc:
|
|
205
|
+
block = _unclassified_block(
|
|
206
|
+
snapshot.source,
|
|
207
|
+
candidate,
|
|
208
|
+
confidence=0.0,
|
|
209
|
+
reason_codes=["semantic_block_fallback"],
|
|
210
|
+
extra_diagnostic={"error_type": type(exc).__name__},
|
|
211
|
+
)
|
|
212
|
+
if block.kind == "unclassified":
|
|
213
|
+
reason_codes = [
|
|
214
|
+
diagnostic["reason_code"]
|
|
215
|
+
for diagnostic in block.diagnostics
|
|
216
|
+
if "reason_code" in diagnostic
|
|
217
|
+
]
|
|
218
|
+
ambiguous_regions.append(
|
|
219
|
+
{
|
|
220
|
+
"sheet_name": sheet.name,
|
|
221
|
+
"range": candidate.source_ref.range,
|
|
222
|
+
"candidate_kind": "unclassified",
|
|
223
|
+
"confidence": block.confidence,
|
|
224
|
+
"reason_codes": reason_codes,
|
|
225
|
+
}
|
|
226
|
+
)
|
|
227
|
+
semantic_blocks.append(block)
|
|
228
|
+
block_counts[block.kind] += 1
|
|
229
|
+
sheet_ir.blocks = semantic_blocks
|
|
230
|
+
|
|
231
|
+
diagnostics.block_count_by_kind = dict(sorted(block_counts.items()))
|
|
232
|
+
diagnostics.ambiguous_regions = ambiguous_regions
|
|
233
|
+
diagnostics.region_diagnostics = region_diagnostics
|
|
234
|
+
try:
|
|
235
|
+
groups, candidates = link_table_continuations(snapshot, workbook_ir)
|
|
236
|
+
except Exception as exc:
|
|
237
|
+
diagnostics.warnings.append(f"cross_sheet_continuation_fallback:{type(exc).__name__}")
|
|
238
|
+
else:
|
|
239
|
+
workbook_ir.table_continuations = groups
|
|
240
|
+
diagnostics.continuation_candidates = candidates
|
|
241
|
+
ambiguous_count = sum(item["status"] == "ambiguous" for item in candidates)
|
|
242
|
+
if ambiguous_count:
|
|
243
|
+
diagnostics.warnings.append(
|
|
244
|
+
f"Workbook contains {ambiguous_count} ambiguous continuation candidates"
|
|
245
|
+
)
|
|
246
|
+
_update_coverage(snapshot, workbook_ir, diagnostics)
|
|
247
|
+
validity_ratio, invalid_refs = validate_workbook_source_refs(snapshot, workbook_ir)
|
|
248
|
+
diagnostics.source_ref_validity_ratio = validity_ratio
|
|
249
|
+
if invalid_refs:
|
|
250
|
+
diagnostics.status = "partial"
|
|
251
|
+
diagnostics.warnings.append(
|
|
252
|
+
f"Workbook IR contains {len(invalid_refs)} invalid source refs: {invalid_refs[:10]}"
|
|
253
|
+
)
|
|
254
|
+
return workbook_ir, diagnostics
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def _region_drafts(
|
|
258
|
+
snapshot: WorkbookSnapshot,
|
|
259
|
+
configured: WorkbookDisambiguation,
|
|
260
|
+
) -> list[_RegionDraft]:
|
|
261
|
+
drafts = []
|
|
262
|
+
for sheet_index, sheet in enumerate(snapshot.sheets):
|
|
263
|
+
for candidate in detect_candidate_regions(sheet):
|
|
264
|
+
assessment = assess_candidate_region(sheet, candidate)
|
|
265
|
+
case = None
|
|
266
|
+
case_id = None
|
|
267
|
+
unavailable_audit = None
|
|
268
|
+
if configured.mode is not WorkbookModelMode.OFF and assessment.ambiguous:
|
|
269
|
+
case_id = _local_region_case_id(candidate, assessment)
|
|
270
|
+
unavailable_outcome = _unavailable_case_outcome(sheet, candidate)
|
|
271
|
+
if unavailable_outcome is not None:
|
|
272
|
+
unavailable_audit = _local_unavailable_audit(
|
|
273
|
+
case_id,
|
|
274
|
+
candidate,
|
|
275
|
+
configured,
|
|
276
|
+
rule_confidence=assessment.deterministic.confidence,
|
|
277
|
+
outcome=unavailable_outcome,
|
|
278
|
+
)
|
|
279
|
+
else:
|
|
280
|
+
try:
|
|
281
|
+
case = build_region_case(sheet, candidate, assessment)
|
|
282
|
+
except InvalidRegionAmbiguityCaseError as error:
|
|
283
|
+
unavailable_audit = _local_unavailable_audit(
|
|
284
|
+
case_id,
|
|
285
|
+
candidate,
|
|
286
|
+
configured,
|
|
287
|
+
rule_confidence=assessment.deterministic.confidence,
|
|
288
|
+
outcome="case_unavailable",
|
|
289
|
+
error=error,
|
|
290
|
+
)
|
|
291
|
+
else:
|
|
292
|
+
case_id = case.case_id
|
|
293
|
+
drafts.append(
|
|
294
|
+
_RegionDraft(
|
|
295
|
+
sheet_index=sheet_index,
|
|
296
|
+
sheet=sheet,
|
|
297
|
+
candidate=candidate,
|
|
298
|
+
assessment=assessment,
|
|
299
|
+
case=case,
|
|
300
|
+
case_id=case_id,
|
|
301
|
+
unavailable_audit=unavailable_audit,
|
|
302
|
+
)
|
|
303
|
+
)
|
|
304
|
+
return drafts
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _local_region_case_id(
|
|
308
|
+
candidate: CandidateRegion,
|
|
309
|
+
assessment: RegionAssessment,
|
|
310
|
+
) -> str:
|
|
311
|
+
return stable_id(
|
|
312
|
+
"region_case_unavailable",
|
|
313
|
+
REGION_RULE_VERSION,
|
|
314
|
+
candidate.source_ref.key,
|
|
315
|
+
*(choice.choice_id for choice in assessment.choices),
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
|
|
319
|
+
def _unavailable_case_outcome(
|
|
320
|
+
sheet: SheetSnapshot,
|
|
321
|
+
candidate: CandidateRegion,
|
|
322
|
+
) -> str | None:
|
|
323
|
+
if sheet.visibility != "visible":
|
|
324
|
+
return "hidden_content"
|
|
325
|
+
min_column, min_row, max_column, max_row = range_boundaries(candidate.source_ref.range)
|
|
326
|
+
if any(min_row <= row <= max_row for row in sheet.hidden_rows):
|
|
327
|
+
return "hidden_content"
|
|
328
|
+
hidden_columns = {get_column_letter(column) for column in range(min_column, max_column + 1)}
|
|
329
|
+
if hidden_columns.intersection(sheet.hidden_columns):
|
|
330
|
+
return "hidden_content"
|
|
331
|
+
for coordinate, cell in sheet.cells.items():
|
|
332
|
+
row, column = coordinate_to_tuple(coordinate)
|
|
333
|
+
if min_row <= row <= max_row and min_column <= column <= max_column and cell.hidden:
|
|
334
|
+
return "hidden_content"
|
|
335
|
+
if _candidate_envelope_has_formula(sheet, candidate):
|
|
336
|
+
return "formula_content"
|
|
337
|
+
return None
|
|
338
|
+
|
|
339
|
+
|
|
340
|
+
def _local_unavailable_audit(
|
|
341
|
+
case_id: str,
|
|
342
|
+
candidate: CandidateRegion,
|
|
343
|
+
configured: WorkbookDisambiguation,
|
|
344
|
+
*,
|
|
345
|
+
rule_confidence: float,
|
|
346
|
+
outcome: str,
|
|
347
|
+
error: Exception | None = None,
|
|
348
|
+
) -> ModelCallAudit:
|
|
349
|
+
return ModelCallAudit(
|
|
350
|
+
case_id=case_id,
|
|
351
|
+
source_range=candidate.source_ref.range,
|
|
352
|
+
mode=configured.mode.value,
|
|
353
|
+
schema_version=REGION_SCHEMA_VERSION,
|
|
354
|
+
prompt_version=REGION_PROMPT_VERSION,
|
|
355
|
+
rule_version=REGION_RULE_VERSION,
|
|
356
|
+
validator_version=REGION_VALIDATOR_VERSION,
|
|
357
|
+
privacy_version=REGION_PRIVACY_VERSION,
|
|
358
|
+
rule_confidence=rule_confidence,
|
|
359
|
+
provider=None,
|
|
360
|
+
model=None,
|
|
361
|
+
model_revision=None,
|
|
362
|
+
request_checksum=None,
|
|
363
|
+
response_checksum=None,
|
|
364
|
+
cache_status="not_checked",
|
|
365
|
+
attempts=0,
|
|
366
|
+
elapsed_ms=0,
|
|
367
|
+
request_bytes=0,
|
|
368
|
+
response_bytes=0,
|
|
369
|
+
outcome=outcome,
|
|
370
|
+
selected_choice_id=None,
|
|
371
|
+
reported_confidence=None,
|
|
372
|
+
validation_codes=(outcome,),
|
|
373
|
+
reason_codes=("deterministic_fallback",),
|
|
374
|
+
error_type=_safe_error_type(error),
|
|
375
|
+
)
|
|
376
|
+
|
|
377
|
+
|
|
378
|
+
def _resolve_region_cases(
|
|
379
|
+
drafts: list[_RegionDraft],
|
|
380
|
+
configured: WorkbookDisambiguation,
|
|
381
|
+
):
|
|
382
|
+
cases = [draft.case for draft in drafts if draft.case is not None]
|
|
383
|
+
if not cases:
|
|
384
|
+
return {}
|
|
385
|
+
runtime = configured._runtime
|
|
386
|
+
if runtime is None:
|
|
387
|
+
raise RuntimeError("enabled workbook disambiguation runtime is unavailable")
|
|
388
|
+
resolutions = runtime.resolve(cases, configured)
|
|
389
|
+
return {resolution.case_id: resolution for resolution in resolutions.resolutions}
|
|
390
|
+
|
|
391
|
+
|
|
392
|
+
def _raise_required_with_deterministic_fallback(
|
|
393
|
+
snapshot: WorkbookSnapshot,
|
|
394
|
+
drafts: list[_RegionDraft],
|
|
395
|
+
error: RequiredWorkbookDisambiguationError,
|
|
396
|
+
) -> None:
|
|
397
|
+
materialized = [_deterministic_materialized_region(snapshot.source, draft) for draft in drafts]
|
|
398
|
+
workbook_ir = _workbook_from_materialized(snapshot, materialized, rollback_selected=False)
|
|
399
|
+
diagnostics, _ = _finalize_workbook(snapshot, workbook_ir)
|
|
400
|
+
diagnostics.model_calls = _ordered_error_audits(drafts, error.diagnostics.model_calls)
|
|
401
|
+
diagnostics.status = "failed"
|
|
402
|
+
raise RequiredWorkbookDisambiguationError(
|
|
403
|
+
_ordered_unresolved_case_ids(drafts, error.case_ids),
|
|
404
|
+
diagnostics,
|
|
405
|
+
) from None
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def _ordered_error_audits(
|
|
409
|
+
drafts: list[_RegionDraft],
|
|
410
|
+
error_audits: list[dict[str, object]],
|
|
411
|
+
) -> list[dict[str, object]]:
|
|
412
|
+
audits_by_case_id = {
|
|
413
|
+
audit["case_id"]: _audit_field_payload(audit)
|
|
414
|
+
for audit in error_audits
|
|
415
|
+
if isinstance(audit.get("case_id"), str)
|
|
416
|
+
}
|
|
417
|
+
audits = []
|
|
418
|
+
for draft in drafts:
|
|
419
|
+
if draft.unavailable_audit is not None:
|
|
420
|
+
audits.append(_audit_payload(draft.unavailable_audit))
|
|
421
|
+
elif draft.case_id is not None and draft.case_id in audits_by_case_id:
|
|
422
|
+
audits.append(audits_by_case_id[draft.case_id])
|
|
423
|
+
return audits
|
|
424
|
+
|
|
425
|
+
|
|
426
|
+
def _audit_field_payload(audit: dict[str, object]) -> dict[str, object]:
|
|
427
|
+
return {field.name: audit[field.name] for field in fields(ModelCallAudit)}
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def _ordered_unresolved_case_ids(
|
|
431
|
+
drafts: list[_RegionDraft],
|
|
432
|
+
disambiguator_case_ids: tuple[str, ...],
|
|
433
|
+
) -> tuple[str, ...]:
|
|
434
|
+
unresolved_case_ids = set(disambiguator_case_ids)
|
|
435
|
+
unresolved_case_ids.update(
|
|
436
|
+
draft.case_id
|
|
437
|
+
for draft in drafts
|
|
438
|
+
if draft.case_id is not None and draft.unavailable_audit is not None
|
|
439
|
+
)
|
|
440
|
+
ordered = [
|
|
441
|
+
draft.case_id
|
|
442
|
+
for draft in drafts
|
|
443
|
+
if draft.case_id is not None and draft.case_id in unresolved_case_ids
|
|
444
|
+
]
|
|
445
|
+
ordered.extend(case_id for case_id in disambiguator_case_ids if case_id not in ordered)
|
|
446
|
+
return tuple(ordered)
|
|
447
|
+
|
|
448
|
+
|
|
449
|
+
def _deterministic_materialized_region(
|
|
450
|
+
snapshot_source: str,
|
|
451
|
+
draft: _RegionDraft,
|
|
452
|
+
) -> _MaterializedRegion:
|
|
453
|
+
deterministic_block = _materialize_deterministic(snapshot_source, draft)
|
|
454
|
+
return _MaterializedRegion(
|
|
455
|
+
draft=draft,
|
|
456
|
+
block=deterministic_block,
|
|
457
|
+
deterministic_block=deterministic_block,
|
|
458
|
+
audit=draft.unavailable_audit,
|
|
459
|
+
)
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _materialize_region(snapshot_source, draft: _RegionDraft, resolutions_by_case_id):
|
|
463
|
+
deterministic_region = _deterministic_materialized_region(snapshot_source, draft)
|
|
464
|
+
deterministic_block = deterministic_region.deterministic_block
|
|
465
|
+
if draft.case is None:
|
|
466
|
+
return deterministic_region
|
|
467
|
+
|
|
468
|
+
resolution = resolutions_by_case_id[draft.case.case_id]
|
|
469
|
+
if resolution.status == "local_fallback":
|
|
470
|
+
return _MaterializedRegion(
|
|
471
|
+
draft=draft,
|
|
472
|
+
block=deterministic_block,
|
|
473
|
+
deterministic_block=deterministic_block,
|
|
474
|
+
audit=_normalize_fallback_audit(resolution.audit),
|
|
475
|
+
)
|
|
476
|
+
|
|
477
|
+
choice = next(
|
|
478
|
+
choice for choice in draft.case.choices if choice.choice_id == resolution.choice_id
|
|
479
|
+
)
|
|
480
|
+
classification = BlockClassification(
|
|
481
|
+
kind=choice.kind,
|
|
482
|
+
confidence=choice.local_score,
|
|
483
|
+
reason_codes=[*choice.reason_codes, "model_selected_choice"],
|
|
484
|
+
features=draft.assessment.deterministic.features,
|
|
485
|
+
)
|
|
486
|
+
assert resolution.audit is not None
|
|
487
|
+
try:
|
|
488
|
+
block = _block_for_candidate(
|
|
489
|
+
snapshot_source,
|
|
490
|
+
draft.sheet,
|
|
491
|
+
draft.candidate,
|
|
492
|
+
classification,
|
|
493
|
+
)
|
|
494
|
+
except Exception as exc:
|
|
495
|
+
audit = replace(
|
|
496
|
+
resolution.audit,
|
|
497
|
+
outcome="materialization_error",
|
|
498
|
+
validation_codes=_stable_codes(
|
|
499
|
+
*resolution.audit.validation_codes,
|
|
500
|
+
"materialization_error",
|
|
501
|
+
),
|
|
502
|
+
reason_codes=("semantic_block_fallback",),
|
|
503
|
+
error_type=_safe_error_type(exc),
|
|
504
|
+
)
|
|
505
|
+
return _MaterializedRegion(
|
|
506
|
+
draft=draft,
|
|
507
|
+
block=deterministic_block,
|
|
508
|
+
deterministic_block=deterministic_block,
|
|
509
|
+
audit=audit,
|
|
510
|
+
selection_attempted=True,
|
|
511
|
+
materialization_failed=True,
|
|
512
|
+
)
|
|
513
|
+
return _MaterializedRegion(
|
|
514
|
+
draft=draft,
|
|
515
|
+
block=block,
|
|
516
|
+
deterministic_block=deterministic_block,
|
|
517
|
+
audit=resolution.audit,
|
|
518
|
+
selection_attempted=True,
|
|
519
|
+
model_selected=True,
|
|
520
|
+
)
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def _materialize_deterministic(snapshot_source, draft: _RegionDraft) -> WorkbookBlock:
|
|
524
|
+
try:
|
|
525
|
+
return _block_for_candidate(
|
|
526
|
+
snapshot_source,
|
|
527
|
+
draft.sheet,
|
|
528
|
+
draft.candidate,
|
|
529
|
+
draft.assessment.deterministic,
|
|
530
|
+
)
|
|
531
|
+
except Exception as exc:
|
|
532
|
+
return _unclassified_block(
|
|
533
|
+
snapshot_source,
|
|
534
|
+
draft.candidate,
|
|
535
|
+
confidence=0.0,
|
|
536
|
+
reason_codes=["semantic_block_fallback"],
|
|
537
|
+
extra_diagnostic={"error_type": type(exc).__name__},
|
|
538
|
+
)
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
def _normalize_fallback_audit(audit: ModelCallAudit | None) -> ModelCallAudit | None:
|
|
542
|
+
if audit is None:
|
|
543
|
+
return None
|
|
544
|
+
outcome = (
|
|
545
|
+
"provider_error"
|
|
546
|
+
if audit.outcome in {"adapter_error", "deadline_exceeded", "timeout"}
|
|
547
|
+
else audit.outcome
|
|
548
|
+
)
|
|
549
|
+
return replace(
|
|
550
|
+
audit,
|
|
551
|
+
outcome=outcome,
|
|
552
|
+
reason_codes=("deterministic_fallback",),
|
|
553
|
+
)
|
|
554
|
+
|
|
555
|
+
|
|
556
|
+
def _workbook_from_materialized(
|
|
557
|
+
snapshot: WorkbookSnapshot,
|
|
558
|
+
materialized: list[_MaterializedRegion],
|
|
559
|
+
*,
|
|
560
|
+
rollback_selected: bool,
|
|
561
|
+
) -> WorkbookIR:
|
|
562
|
+
workbook_ir, _ = assemble_baseline(snapshot)
|
|
563
|
+
blocks_by_sheet: dict[int, list[WorkbookBlock]] = {
|
|
564
|
+
index: [] for index in range(len(snapshot.sheets))
|
|
565
|
+
}
|
|
566
|
+
for region in materialized:
|
|
567
|
+
block = (
|
|
568
|
+
region.deterministic_block
|
|
569
|
+
if rollback_selected and region.selection_attempted
|
|
570
|
+
else region.block
|
|
571
|
+
)
|
|
572
|
+
blocks_by_sheet[region.draft.sheet_index].append(deepcopy(block))
|
|
573
|
+
for sheet_index, sheet_ir in enumerate(workbook_ir.sheets):
|
|
574
|
+
sheet_ir.blocks = blocks_by_sheet[sheet_index]
|
|
575
|
+
return workbook_ir
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def _finalize_workbook(
|
|
579
|
+
snapshot: WorkbookSnapshot,
|
|
580
|
+
workbook_ir: WorkbookIR,
|
|
581
|
+
) -> tuple[ParseDiagnostics, tuple[str, ...]]:
|
|
582
|
+
baseline, diagnostics = assemble_baseline(snapshot)
|
|
583
|
+
workbook_ir.lineage = baseline.lineage
|
|
584
|
+
block_counts: Counter[str] = Counter()
|
|
585
|
+
ambiguous_regions = []
|
|
586
|
+
region_diagnostics = []
|
|
587
|
+
for sheet_ir in workbook_ir.sheets:
|
|
588
|
+
for block in sheet_ir.blocks:
|
|
589
|
+
region_diagnostic = _block_region_diagnostic(sheet_ir.name, block)
|
|
590
|
+
if region_diagnostic is not None:
|
|
591
|
+
region_diagnostics.append(region_diagnostic)
|
|
592
|
+
if block.kind == "unclassified":
|
|
593
|
+
reason_codes = [
|
|
594
|
+
diagnostic["reason_code"]
|
|
595
|
+
for diagnostic in block.diagnostics
|
|
596
|
+
if "reason_code" in diagnostic
|
|
597
|
+
]
|
|
598
|
+
ambiguous_regions.append(
|
|
599
|
+
{
|
|
600
|
+
"sheet_name": sheet_ir.name,
|
|
601
|
+
"range": block.source_refs[0].range,
|
|
602
|
+
"candidate_kind": "unclassified",
|
|
603
|
+
"confidence": block.confidence,
|
|
604
|
+
"reason_codes": reason_codes,
|
|
605
|
+
}
|
|
606
|
+
)
|
|
607
|
+
block_counts[block.kind] += 1
|
|
608
|
+
|
|
609
|
+
diagnostics.block_count_by_kind = dict(sorted(block_counts.items()))
|
|
610
|
+
diagnostics.ambiguous_regions = ambiguous_regions
|
|
611
|
+
diagnostics.region_diagnostics = region_diagnostics
|
|
612
|
+
continuation_failed = False
|
|
613
|
+
try:
|
|
614
|
+
groups, candidates = link_table_continuations(snapshot, workbook_ir)
|
|
615
|
+
except Exception as exc:
|
|
616
|
+
continuation_failed = True
|
|
617
|
+
diagnostics.warnings.append(f"cross_sheet_continuation_fallback:{type(exc).__name__}")
|
|
618
|
+
else:
|
|
619
|
+
workbook_ir.table_continuations = groups
|
|
620
|
+
diagnostics.continuation_candidates = candidates
|
|
621
|
+
ambiguous_count = sum(item["status"] == "ambiguous" for item in candidates)
|
|
622
|
+
if ambiguous_count:
|
|
623
|
+
diagnostics.warnings.append(
|
|
624
|
+
f"Workbook contains {ambiguous_count} ambiguous continuation candidates"
|
|
625
|
+
)
|
|
626
|
+
_update_coverage(snapshot, workbook_ir, diagnostics)
|
|
627
|
+
row_conservation_passed = _row_conservation_passed(workbook_ir)
|
|
628
|
+
if not row_conservation_passed:
|
|
629
|
+
diagnostics.status = "partial"
|
|
630
|
+
diagnostics.warnings.append("Workbook IR failed logical row conservation")
|
|
631
|
+
validity_ratio, invalid_refs = validate_workbook_source_refs(snapshot, workbook_ir)
|
|
632
|
+
diagnostics.source_ref_validity_ratio = validity_ratio
|
|
633
|
+
if invalid_refs:
|
|
634
|
+
diagnostics.status = "partial"
|
|
635
|
+
diagnostics.warnings.append(
|
|
636
|
+
f"Workbook IR contains {len(invalid_refs)} invalid source refs: {invalid_refs[:10]}"
|
|
637
|
+
)
|
|
638
|
+
validation_codes = []
|
|
639
|
+
if diagnostics.coverage_ratio != 1.0:
|
|
640
|
+
validation_codes.append("invalid_coverage")
|
|
641
|
+
if not diagnostics.reconstruction_passed:
|
|
642
|
+
validation_codes.append("reconstruction_failed")
|
|
643
|
+
if not row_conservation_passed:
|
|
644
|
+
validation_codes.append("row_conservation_failed")
|
|
645
|
+
if diagnostics.source_ref_validity_ratio != 1.0 or invalid_refs:
|
|
646
|
+
validation_codes.append("invalid_source_refs")
|
|
647
|
+
if continuation_failed:
|
|
648
|
+
validation_codes.append("continuation_error")
|
|
649
|
+
return diagnostics, tuple(validation_codes)
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
def _row_conservation_passed(workbook_ir: WorkbookIR) -> bool:
|
|
653
|
+
for sheet_ir in workbook_ir.sheets:
|
|
654
|
+
for block in sheet_ir.blocks:
|
|
655
|
+
if block.logical_table is None or not block.source_refs:
|
|
656
|
+
continue
|
|
657
|
+
min_col, min_row, max_col, max_row = range_boundaries(block.source_refs[0].range)
|
|
658
|
+
expected = [
|
|
659
|
+
(sheet_ir.name, min_col, row_number, max_col, row_number)
|
|
660
|
+
for row_number in range(min_row, max_row + 1)
|
|
661
|
+
]
|
|
662
|
+
actual = []
|
|
663
|
+
for row in block.logical_table.rows:
|
|
664
|
+
row_min_col, row_min, row_max_col, row_max = range_boundaries(row.source_ref.range)
|
|
665
|
+
actual.append(
|
|
666
|
+
(
|
|
667
|
+
row.source_ref.sheet_name,
|
|
668
|
+
row_min_col,
|
|
669
|
+
row_min,
|
|
670
|
+
row_max_col,
|
|
671
|
+
row_max,
|
|
672
|
+
)
|
|
673
|
+
)
|
|
674
|
+
if actual != expected:
|
|
675
|
+
return False
|
|
676
|
+
return True
|
|
677
|
+
|
|
678
|
+
|
|
679
|
+
def _stable_codes(*codes: str) -> tuple[str, ...]:
|
|
680
|
+
return tuple(dict.fromkeys(codes))
|
|
681
|
+
|
|
682
|
+
|
|
683
|
+
def _candidate_region_diagnostic(
|
|
684
|
+
sheet_name: str,
|
|
685
|
+
candidate: CandidateRegion,
|
|
686
|
+
) -> dict[str, Any]:
|
|
687
|
+
return {
|
|
688
|
+
"sheet_name": sheet_name,
|
|
689
|
+
"range": candidate.source_ref.range,
|
|
690
|
+
"reason_codes": list(candidate.reason_codes),
|
|
691
|
+
"confidence": candidate.confidence,
|
|
692
|
+
"conflicts": deepcopy(candidate.diagnostics),
|
|
693
|
+
}
|
|
694
|
+
|
|
695
|
+
|
|
696
|
+
def _block_region_diagnostic(
|
|
697
|
+
sheet_name: str,
|
|
698
|
+
block: WorkbookBlock,
|
|
699
|
+
) -> dict[str, Any] | None:
|
|
700
|
+
reason_codes = block.metadata.get("region_reason_codes")
|
|
701
|
+
confidence = block.metadata.get("region_confidence")
|
|
702
|
+
if reason_codes is None or confidence is None or not block.source_refs:
|
|
703
|
+
return None
|
|
704
|
+
return {
|
|
705
|
+
"sheet_name": sheet_name,
|
|
706
|
+
"range": block.source_refs[0].range,
|
|
707
|
+
"reason_codes": list(reason_codes),
|
|
708
|
+
"confidence": confidence,
|
|
709
|
+
"conflicts": deepcopy(block.metadata.get("region_diagnostics", [])),
|
|
710
|
+
}
|
|
711
|
+
|
|
712
|
+
|
|
713
|
+
def _block_for_candidate(snapshot_source, sheet, candidate, classification):
|
|
714
|
+
common = {
|
|
715
|
+
"block_id": stable_id(
|
|
716
|
+
"block", snapshot_source, candidate.source_ref.key, classification.kind
|
|
717
|
+
),
|
|
718
|
+
"kind": classification.kind,
|
|
719
|
+
"source_refs": [candidate.source_ref],
|
|
720
|
+
"cell_refs": candidate.cell_refs,
|
|
721
|
+
"confidence": classification.confidence,
|
|
722
|
+
"metadata": {
|
|
723
|
+
"view": "raw_grid"
|
|
724
|
+
if classification.kind == "unclassified"
|
|
725
|
+
else f"semantic_{classification.kind}",
|
|
726
|
+
"cell_count": len(candidate.cell_refs),
|
|
727
|
+
"features": asdict(classification.features),
|
|
728
|
+
"reason_codes": list(classification.reason_codes),
|
|
729
|
+
"region_reason_codes": list(candidate.reason_codes),
|
|
730
|
+
"region_confidence": candidate.confidence,
|
|
731
|
+
"region_diagnostics": deepcopy(candidate.diagnostics),
|
|
732
|
+
},
|
|
733
|
+
"diagnostics": [
|
|
734
|
+
{"reason_code": reason_code} for reason_code in classification.reason_codes
|
|
735
|
+
],
|
|
736
|
+
}
|
|
737
|
+
if classification.kind == "logical_table":
|
|
738
|
+
table = interpret_logical_table(sheet, candidate)
|
|
739
|
+
common["confidence"] = min(classification.confidence, table.confidence)
|
|
740
|
+
return WorkbookBlock(**common, logical_table=table)
|
|
741
|
+
if classification.kind == "form":
|
|
742
|
+
return WorkbookBlock(
|
|
743
|
+
**common,
|
|
744
|
+
form=interpret_form_block(sheet, candidate, classification),
|
|
745
|
+
)
|
|
746
|
+
if classification.kind == "matrix":
|
|
747
|
+
return WorkbookBlock(
|
|
748
|
+
**common,
|
|
749
|
+
matrix=interpret_matrix_block(sheet, candidate, classification),
|
|
750
|
+
)
|
|
751
|
+
if classification.kind == "text":
|
|
752
|
+
return WorkbookBlock(
|
|
753
|
+
**common,
|
|
754
|
+
text=interpret_text_block(sheet, candidate, classification),
|
|
755
|
+
)
|
|
756
|
+
return WorkbookBlock(**common)
|
|
757
|
+
|
|
758
|
+
|
|
759
|
+
def _unclassified_block(
|
|
760
|
+
snapshot_source,
|
|
761
|
+
candidate,
|
|
762
|
+
*,
|
|
763
|
+
confidence,
|
|
764
|
+
reason_codes,
|
|
765
|
+
extra_diagnostic=None,
|
|
766
|
+
):
|
|
767
|
+
diagnostics = [{"reason_code": reason_code} for reason_code in reason_codes]
|
|
768
|
+
if extra_diagnostic:
|
|
769
|
+
diagnostics[0].update(extra_diagnostic)
|
|
770
|
+
return WorkbookBlock(
|
|
771
|
+
block_id=stable_id("block", snapshot_source, candidate.source_ref.key, "unclassified"),
|
|
772
|
+
kind="unclassified",
|
|
773
|
+
source_refs=[candidate.source_ref],
|
|
774
|
+
cell_refs=candidate.cell_refs,
|
|
775
|
+
confidence=confidence,
|
|
776
|
+
metadata={
|
|
777
|
+
"view": "raw_grid",
|
|
778
|
+
"cell_count": len(candidate.cell_refs),
|
|
779
|
+
"reason_codes": list(reason_codes),
|
|
780
|
+
"region_reason_codes": list(candidate.reason_codes),
|
|
781
|
+
"region_confidence": candidate.confidence,
|
|
782
|
+
"region_diagnostics": deepcopy(candidate.diagnostics),
|
|
783
|
+
},
|
|
784
|
+
diagnostics=diagnostics,
|
|
785
|
+
)
|
|
786
|
+
|
|
787
|
+
|
|
788
|
+
def validate_workbook_source_refs(
|
|
789
|
+
snapshot: WorkbookSnapshot,
|
|
790
|
+
workbook_ir: WorkbookIR,
|
|
791
|
+
) -> tuple[float, list[str]]:
|
|
792
|
+
"""Return the ratio of derived refs bounded by an existing source Sheet."""
|
|
793
|
+
|
|
794
|
+
sheet_bounds = {
|
|
795
|
+
sheet.name: range_boundaries(sheet.used_range or _range_for_coordinates(list(sheet.cells)))
|
|
796
|
+
for sheet in snapshot.sheets
|
|
797
|
+
if sheet.used_range or sheet.cells
|
|
798
|
+
}
|
|
799
|
+
refs = []
|
|
800
|
+
drawing_refs = []
|
|
801
|
+
for sheet_ir in workbook_ir.sheets:
|
|
802
|
+
for block in sheet_ir.blocks:
|
|
803
|
+
if block.kind in {"chart", "image"}:
|
|
804
|
+
drawing_refs.extend(block.source_refs)
|
|
805
|
+
continue
|
|
806
|
+
refs.extend(block.source_refs)
|
|
807
|
+
if block.logical_table is not None:
|
|
808
|
+
refs.extend(_logical_table_source_refs(block.logical_table))
|
|
809
|
+
if block.form is not None:
|
|
810
|
+
refs.extend(block.form.source_refs)
|
|
811
|
+
refs.extend(ref for field in block.form.fields for ref in field.label_source_refs)
|
|
812
|
+
refs.extend(ref for field in block.form.fields for ref in field.value_source_refs)
|
|
813
|
+
refs.extend(ref for line in block.form.free_text for ref in line.source_refs)
|
|
814
|
+
if block.matrix is not None:
|
|
815
|
+
refs.extend(block.matrix.source_refs)
|
|
816
|
+
refs.extend(
|
|
817
|
+
ref for header in block.matrix.row_headers for ref in header.source_refs
|
|
818
|
+
)
|
|
819
|
+
refs.extend(
|
|
820
|
+
ref for header in block.matrix.column_headers for ref in header.source_refs
|
|
821
|
+
)
|
|
822
|
+
refs.extend(
|
|
823
|
+
ref for row in block.matrix.value_source_refs for ref in row if ref is not None
|
|
824
|
+
)
|
|
825
|
+
if block.text is not None:
|
|
826
|
+
refs.extend(block.text.source_refs)
|
|
827
|
+
refs.extend(ref for line in block.text.lines for ref in line.source_refs)
|
|
828
|
+
|
|
829
|
+
for continuation in workbook_ir.table_continuations:
|
|
830
|
+
refs.extend(_logical_table_source_refs(continuation.logical_table))
|
|
831
|
+
refs.extend(continuation.source_refs)
|
|
832
|
+
|
|
833
|
+
invalid = sorted({ref.key for ref in refs if not _valid_source_ref(ref, sheet_bounds)})
|
|
834
|
+
invalid_count = sum(not _valid_source_ref(ref, sheet_bounds) for ref in refs)
|
|
835
|
+
if drawing_refs:
|
|
836
|
+
from langparse.workbooks.objects import validate_object_source
|
|
837
|
+
|
|
838
|
+
for ref in drawing_refs:
|
|
839
|
+
try:
|
|
840
|
+
validate_object_source(snapshot, ref.key)
|
|
841
|
+
except ValueError:
|
|
842
|
+
invalid_count += 1
|
|
843
|
+
invalid.append(ref.key)
|
|
844
|
+
count = len(refs) + len(drawing_refs)
|
|
845
|
+
ratio = (count - invalid_count) / count if count else 1.0
|
|
846
|
+
invalid = sorted(set(invalid))
|
|
847
|
+
return ratio, invalid
|
|
848
|
+
|
|
849
|
+
|
|
850
|
+
def _logical_table_source_refs(table: LogicalTable) -> list[SourceRef]:
|
|
851
|
+
return [
|
|
852
|
+
*table.source_refs,
|
|
853
|
+
*(ref for column in table.columns for ref in column.source_refs),
|
|
854
|
+
*(row.source_ref for row in table.rows),
|
|
855
|
+
*(fragment.source_ref for fragment in table.fragments),
|
|
856
|
+
*(section.source_ref for section in table.sections),
|
|
857
|
+
]
|
|
858
|
+
|
|
859
|
+
|
|
860
|
+
def _valid_source_ref(ref: SourceRef, sheet_bounds) -> bool:
|
|
861
|
+
bounds = sheet_bounds.get(ref.sheet_name)
|
|
862
|
+
if bounds is None:
|
|
863
|
+
return False
|
|
864
|
+
min_col, min_row, max_col, max_row = range_boundaries(ref.range)
|
|
865
|
+
sheet_min_col, sheet_min_row, sheet_max_col, sheet_max_row = bounds
|
|
866
|
+
return (
|
|
867
|
+
sheet_min_col <= min_col <= max_col <= sheet_max_col
|
|
868
|
+
and sheet_min_row <= min_row <= max_row <= sheet_max_row
|
|
869
|
+
)
|
|
870
|
+
|
|
871
|
+
|
|
872
|
+
def _update_coverage(snapshot, workbook_ir, diagnostics):
|
|
873
|
+
source_refs = {
|
|
874
|
+
f"{sheet.name}!{coordinate}"
|
|
875
|
+
for sheet in snapshot.sheets
|
|
876
|
+
for coordinate, cell in sheet.cells.items()
|
|
877
|
+
if _is_assignable_cell(cell)
|
|
878
|
+
}
|
|
879
|
+
assigned_refs = {
|
|
880
|
+
f"{sheet_ir.name}!{coordinate}"
|
|
881
|
+
for sheet_ir in workbook_ir.sheets
|
|
882
|
+
for block in sheet_ir.blocks
|
|
883
|
+
for coordinate in block.cell_refs
|
|
884
|
+
}
|
|
885
|
+
diagnostics.coverage_ratio = (
|
|
886
|
+
len(source_refs & assigned_refs) / len(source_refs) if source_refs else 1.0
|
|
887
|
+
)
|
|
888
|
+
diagnostics.reconstruction_passed = source_refs == assigned_refs
|
|
889
|
+
if not diagnostics.reconstruction_passed:
|
|
890
|
+
diagnostics.status = "partial"
|
|
891
|
+
|
|
892
|
+
|
|
893
|
+
def assemble_baseline(snapshot: WorkbookSnapshot) -> tuple[WorkbookIR, ParseDiagnostics]:
|
|
894
|
+
"""Create a lossless raw-grid IR before any semantic table interpretation."""
|
|
895
|
+
|
|
896
|
+
sheet_irs: list[SheetIR] = []
|
|
897
|
+
source_cell_keys: set[str] = set()
|
|
898
|
+
assigned_cell_keys: set[str] = set()
|
|
899
|
+
block_counts: Counter[str] = Counter()
|
|
900
|
+
|
|
901
|
+
for sheet in snapshot.sheets:
|
|
902
|
+
cell_refs = sorted(
|
|
903
|
+
(coordinate for coordinate, cell in sheet.cells.items() if _is_assignable_cell(cell)),
|
|
904
|
+
key=coordinate_to_tuple,
|
|
905
|
+
)
|
|
906
|
+
blocks: list[WorkbookBlock] = []
|
|
907
|
+
if cell_refs:
|
|
908
|
+
source_range = sheet.used_range or _range_for_coordinates(cell_refs)
|
|
909
|
+
source_ref = SourceRef(sheet_name=sheet.name, range=source_range)
|
|
910
|
+
block = WorkbookBlock(
|
|
911
|
+
block_id=stable_id("block", snapshot.source, source_ref.key, "unclassified"),
|
|
912
|
+
kind="unclassified",
|
|
913
|
+
source_refs=[source_ref],
|
|
914
|
+
cell_refs=cell_refs,
|
|
915
|
+
metadata={"view": "raw_grid", "cell_count": len(cell_refs)},
|
|
916
|
+
)
|
|
917
|
+
blocks.append(block)
|
|
918
|
+
block_counts[block.kind] += 1
|
|
919
|
+
|
|
920
|
+
sheet_irs.append(
|
|
921
|
+
SheetIR(
|
|
922
|
+
sheet_id=stable_id("sheet", snapshot.source, str(sheet.index), sheet.name),
|
|
923
|
+
name=sheet.name,
|
|
924
|
+
index=sheet.index,
|
|
925
|
+
blocks=blocks,
|
|
926
|
+
visibility=sheet.visibility,
|
|
927
|
+
metadata={
|
|
928
|
+
"used_range": sheet.used_range,
|
|
929
|
+
"print_area": sheet.print_area,
|
|
930
|
+
"merged_ranges": sheet.merged_ranges,
|
|
931
|
+
"object_count": len(sheet.objects),
|
|
932
|
+
},
|
|
933
|
+
)
|
|
934
|
+
)
|
|
935
|
+
|
|
936
|
+
qualified_refs = {f"{sheet.name}!{coordinate}" for coordinate in cell_refs}
|
|
937
|
+
source_cell_keys.update(qualified_refs)
|
|
938
|
+
assigned_cell_keys.update(qualified_refs if blocks else set())
|
|
939
|
+
|
|
940
|
+
reconstruction_passed = assigned_cell_keys == source_cell_keys
|
|
941
|
+
coverage_ratio = (
|
|
942
|
+
len(assigned_cell_keys & source_cell_keys) / len(source_cell_keys)
|
|
943
|
+
if source_cell_keys
|
|
944
|
+
else 1.0
|
|
945
|
+
)
|
|
946
|
+
warnings = list(snapshot.metadata.get("warnings", []))
|
|
947
|
+
if not reconstruction_passed:
|
|
948
|
+
missing = sorted(source_cell_keys - assigned_cell_keys)
|
|
949
|
+
warnings.append(f"Workbook IR omitted {len(missing)} source cells: {missing[:10]}")
|
|
950
|
+
|
|
951
|
+
diagnostics = ParseDiagnostics(
|
|
952
|
+
status="success" if reconstruction_passed and coverage_ratio == 1.0 else "partial",
|
|
953
|
+
coverage_ratio=coverage_ratio,
|
|
954
|
+
reconstruction_passed=reconstruction_passed,
|
|
955
|
+
block_count_by_kind=dict(sorted(block_counts.items())),
|
|
956
|
+
unsupported_features=list(snapshot.metadata.get("unsupported_features", [])),
|
|
957
|
+
warnings=warnings,
|
|
958
|
+
)
|
|
959
|
+
workbook_ir = WorkbookIR(
|
|
960
|
+
kind="workbook",
|
|
961
|
+
workbook_id=stable_id("workbook", snapshot.source, snapshot.filename),
|
|
962
|
+
source=snapshot.source,
|
|
963
|
+
sheets=sheet_irs,
|
|
964
|
+
filename=snapshot.filename,
|
|
965
|
+
snapshot=snapshot,
|
|
966
|
+
metadata={"snapshot": snapshot.metadata},
|
|
967
|
+
)
|
|
968
|
+
from langparse.workbooks.lineage import build_lineage
|
|
969
|
+
|
|
970
|
+
workbook_ir.lineage = build_lineage(snapshot)
|
|
971
|
+
diagnostics.reference_diagnostics = [asdict(item) for item in workbook_ir.lineage.diagnostics]
|
|
972
|
+
return workbook_ir, diagnostics
|
|
973
|
+
|
|
974
|
+
|
|
975
|
+
def _is_assignable_cell(cell: CellSnapshot) -> bool:
|
|
976
|
+
return any(
|
|
977
|
+
(
|
|
978
|
+
cell.raw_value is not None,
|
|
979
|
+
cell.formula is not None,
|
|
980
|
+
cell.comment is not None,
|
|
981
|
+
cell.hyperlink is not None,
|
|
982
|
+
cell.merge_anchor is not None,
|
|
983
|
+
)
|
|
984
|
+
)
|
|
985
|
+
|
|
986
|
+
|
|
987
|
+
def _range_for_coordinates(coordinates: list[str]) -> str:
|
|
988
|
+
positions = [coordinate_to_tuple(coordinate) for coordinate in coordinates]
|
|
989
|
+
rows = [row for row, _ in positions]
|
|
990
|
+
columns = [column for _, column in positions]
|
|
991
|
+
return (
|
|
992
|
+
f"{get_column_letter(min(columns))}{min(rows)}:{get_column_letter(max(columns))}{max(rows)}"
|
|
993
|
+
)
|