langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,229 @@
1
+ """Source-grounded drawing facts and semantic blocks; no image inference or I/O."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from dataclasses import dataclass, field
7
+ from typing import Any
8
+
9
+ from langparse.types import ParseDiagnostics
10
+ from langparse.workbooks.types import (
11
+ SourceRef,
12
+ WorkbookBlock,
13
+ WorkbookIR,
14
+ WorkbookSnapshot,
15
+ stable_id,
16
+ )
17
+
18
+
19
+ @dataclass
20
+ class ChartSeries:
21
+ title: str | None = None
22
+ title_reference: str | None = None
23
+ categories: str | None = None
24
+ values: str | None = None
25
+ x_values: str | None = None
26
+ y_values: str | None = None
27
+ source_refs: list[SourceRef] = field(default_factory=list)
28
+
29
+
30
+ @dataclass
31
+ class DrawingFact:
32
+ object_id: str
33
+ kind: str
34
+ anchor: str | None
35
+ title: str | None = None
36
+ chart_types: list[str] = field(default_factory=list)
37
+ series: list[ChartSeries] = field(default_factory=list)
38
+ axes: list[dict[str, Any]] = field(default_factory=list)
39
+ width: float | None = None
40
+ height: float | None = None
41
+ media_id: str | None = None
42
+ enhancement_status: str = "not_requested"
43
+ diagnostics: list[str] = field(default_factory=list)
44
+
45
+
46
+ def _text(value: Any) -> str | None:
47
+ if isinstance(value, str):
48
+ return value
49
+ # DrawingML rich text can split a title over multiple runs/paragraphs.
50
+ rich = getattr(getattr(value, "tx", None), "rich", None)
51
+ paragraphs = getattr(rich, "p", [])
52
+ lines = []
53
+ for paragraph in paragraphs:
54
+ runs = [getattr(run, "t", "") for run in getattr(paragraph, "r", [])]
55
+ lines.append("".join(runs))
56
+ return "\n".join(lines).strip() or None
57
+
58
+
59
+ def _formula(value: Any) -> str | None:
60
+ for field_name in ("numRef", "strRef", "multiLvlStrRef"):
61
+ reference = getattr(value, field_name, None)
62
+ if reference is not None:
63
+ return getattr(reference, "f", None)
64
+ return None
65
+
66
+
67
+ def _source(formula: str, diagnostics: list[str], sheet_names: dict[str, str]) -> SourceRef | None:
68
+ # External, union and dynamic expressions remain explicit raw facts.
69
+ match = re.fullmatch(
70
+ r"(?:'((?:[^']|'')+)'|([^'!]+))!(\$?[A-Z]+\$?[1-9][0-9]*(?::\$?[A-Z]+\$?[1-9][0-9]*)?)",
71
+ formula,
72
+ )
73
+ if match and "[" not in formula:
74
+ from langparse.workbooks.lineage import _bounds
75
+
76
+ name = (match[1] or match[2]).replace("''", "'")
77
+ cell_range = match[3].replace("$", "")
78
+ try:
79
+ _bounds(cell_range)
80
+ except ValueError:
81
+ pass
82
+ else:
83
+ if name.casefold() in sheet_names:
84
+ return SourceRef(sheet_names[name.casefold()], cell_range)
85
+ diagnostics.append(f"unresolved_chart_reference: {formula}")
86
+ return None
87
+
88
+
89
+ def extract_objects(sheet: Any) -> list[DrawingFact]:
90
+ from openpyxl.utils import get_column_letter
91
+
92
+ sheet_names = {name.casefold(): name for name in sheet.parent.sheetnames}
93
+ result = []
94
+ for kind, objects in (("chart", sheet._charts), ("image", sheet._images)):
95
+ for index, obj in enumerate(objects):
96
+ anchor = obj.anchor if isinstance(obj.anchor, str) else None
97
+ marker = getattr(obj.anchor, "_from", None)
98
+ if marker is not None:
99
+ anchor = f"{get_column_letter(marker.col + 1)}{marker.row + 1}"
100
+ fact = DrawingFact(stable_id("object", sheet.title, kind, str(index)), kind, anchor)
101
+ if anchor is None:
102
+ fact.diagnostics.append("unresolved_anchor")
103
+ if kind == "image":
104
+ fact.width, fact.height = obj.width, obj.height
105
+ fact.media_id = stable_id("media", sheet.title, str(index))
106
+ else:
107
+ fact.title = _text(obj.title)
108
+ for chart in getattr(obj, "_charts", [obj]):
109
+ fact.chart_types.append(type(chart).__name__)
110
+ for axis_name in ("x_axis", "y_axis", "z_axis"):
111
+ axis = getattr(chart, axis_name, None)
112
+ if axis is not None:
113
+ fact.axes.append(
114
+ {
115
+ "axis": axis_name,
116
+ "id": axis.axId,
117
+ "title": _text(axis.title),
118
+ "position": axis.axPos,
119
+ }
120
+ )
121
+ for series in chart.series:
122
+ tx = getattr(series, "tx", None)
123
+ item = ChartSeries(
124
+ title=getattr(tx, "v", None),
125
+ title_reference=_formula(tx),
126
+ categories=_formula(getattr(series, "cat", None)),
127
+ values=_formula(getattr(series, "val", None)),
128
+ x_values=_formula(getattr(series, "xVal", None)),
129
+ y_values=_formula(getattr(series, "yVal", None)),
130
+ )
131
+ for expression in (
132
+ item.title_reference,
133
+ item.categories,
134
+ item.values,
135
+ item.x_values,
136
+ item.y_values,
137
+ ):
138
+ if expression:
139
+ ref = _source(expression, fact.diagnostics, sheet_names)
140
+ if ref is not None:
141
+ item.source_refs.append(ref)
142
+ fact.series.append(item)
143
+ result.append(fact)
144
+ return result
145
+
146
+
147
+ def attach_objects(
148
+ snapshot: WorkbookSnapshot, ir: WorkbookIR, diagnostics: ParseDiagnostics
149
+ ) -> None:
150
+ """Add semantic objects after cell interpretation without claiming cell coverage."""
151
+ total = covered = 0
152
+ sheets = {sheet.name: sheet for sheet in ir.sheets}
153
+ for source in snapshot.sheets:
154
+ target = sheets[source.name]
155
+ for obj in source.objects:
156
+ total += 1
157
+ anchor = obj.get("anchor")
158
+ notes = list(obj.get("diagnostics", []))
159
+ try:
160
+ validate_object_source(snapshot, f"{source.name}!{anchor or ''}")
161
+ except (TypeError, ValueError):
162
+ notes.append("unresolved_anchor")
163
+ diagnostics.warnings.append(f"Object on {source.name}: unresolved_anchor")
164
+ diagnostics.status = "partial"
165
+ continue
166
+ refs = [SourceRef(source.name, anchor)]
167
+ for series in obj.get("series", []):
168
+ for raw in series.get("source_refs", []):
169
+ if raw["sheet_name"] in sheets:
170
+ ref = SourceRef(**raw)
171
+ if ref not in refs:
172
+ refs.append(ref)
173
+ else:
174
+ notes.append(f"missing_chart_sheet: {raw['sheet_name']}")
175
+ if notes:
176
+ diagnostics.status = "partial"
177
+ diagnostics.warnings.extend(f"Object on {source.name}: {note}" for note in notes)
178
+ target.blocks.append(
179
+ WorkbookBlock(
180
+ block_id=obj.get("object_id")
181
+ or stable_id("object", source.name, anchor, str(total)),
182
+ kind=obj["kind"],
183
+ source_refs=refs,
184
+ metadata={"anchor": anchor, "object": obj},
185
+ diagnostics=[{"code": note} for note in notes],
186
+ )
187
+ )
188
+ covered += 1
189
+ kind = obj["kind"]
190
+ diagnostics.block_count_by_kind[kind] = diagnostics.block_count_by_kind.get(kind, 0) + 1
191
+ diagnostics.object_coverage_ratio = covered / total if total else 1.0
192
+ if snapshot.metadata.get("drawing_read_errors"):
193
+ diagnostics.status = "partial"
194
+ diagnostics.object_coverage_ratio = None
195
+ diagnostics.unsupported_features.extend(snapshot.metadata["drawing_read_errors"])
196
+
197
+
198
+ def render_object(block: WorkbookBlock) -> str:
199
+ obj = block.metadata["object"]
200
+ source = ", ".join(ref.key for ref in block.source_refs)
201
+ title = obj.get("title") or obj["kind"].capitalize()
202
+ lines = [f"<!-- source_ranges: {source} -->", f"**{title}** ({obj['kind']})"]
203
+ for series in obj.get("series", []):
204
+ references = [
205
+ series[key]
206
+ for key in ("categories", "values", "x_values", "y_values")
207
+ if series.get(key)
208
+ ]
209
+ lines.append("- " + (series.get("title") or "Series") + ": " + ", ".join(references))
210
+ if obj["kind"] == "image":
211
+ lines.append(
212
+ f"Image {obj.get('media_id')}; {obj.get('width')} × {obj.get('height')}; OCR/VLM not requested."
213
+ )
214
+ return "\n\n".join(lines)
215
+
216
+
217
+ def validate_object_source(snapshot: WorkbookSnapshot | None, source: str) -> None:
218
+ """Drawing anchors can be outside used cells, but must be valid Excel coordinates."""
219
+ from openpyxl.utils.cell import range_boundaries
220
+
221
+ sheet_name, cell_range = source.rsplit("!", 1)
222
+ if snapshot is None or not any(sheet.name == sheet_name for sheet in snapshot.sheets):
223
+ raise ValueError(f"Object source references unknown sheet: {sheet_name}")
224
+ try:
225
+ left, top, right, bottom = range_boundaries(cell_range)
226
+ if not (1 <= left <= right <= 16384 and 1 <= top <= bottom <= 1048576):
227
+ raise ValueError("out of bounds")
228
+ except (ValueError, TypeError) as exc:
229
+ raise ValueError(f"Invalid object source range: {source}") from exc
@@ -0,0 +1,23 @@
1
+ """Whole-workbook quality evaluation primitives."""
2
+
3
+ from .evaluator import (
4
+ WORKBOOK_QUALITY_METRIC_SCHEMA_VERSION,
5
+ WorkbookQualityMetrics,
6
+ evaluate_workbook_result,
7
+ )
8
+ from .schema import (
9
+ WorkbookQualityGate,
10
+ WorkbookQualityManifest,
11
+ WorkbookQualityManifestError,
12
+ load_workbook_quality_manifest,
13
+ )
14
+
15
+ __all__ = [
16
+ "WorkbookQualityManifest",
17
+ "WorkbookQualityManifestError",
18
+ "WorkbookQualityMetrics",
19
+ "WorkbookQualityGate",
20
+ "WORKBOOK_QUALITY_METRIC_SCHEMA_VERSION",
21
+ "evaluate_workbook_result",
22
+ "load_workbook_quality_manifest",
23
+ ]
@@ -0,0 +1,53 @@
1
+ """Check exported workbook consumption contracts against the same frozen truth."""
2
+
3
+ from langparse.workbooks.bundle import WorkbookBundle
4
+
5
+
6
+ def evaluate_bundle_contract(expectation, parsed) -> dict[str, float]:
7
+ bundle = WorkbookBundle.from_result(parsed, max_rows=None)
8
+ payload = bundle.to_dict()
9
+ refs: set[str] = set()
10
+
11
+ def collect(value):
12
+ if isinstance(value, str):
13
+ refs.add(value)
14
+ elif isinstance(value, list):
15
+ for item in value:
16
+ collect(item)
17
+
18
+ def visit(value):
19
+ if isinstance(value, dict):
20
+ for key, item in value.items():
21
+ if key in {"source_ref", "dependent", "target"} and isinstance(item, str):
22
+ refs.add(item)
23
+ elif key.endswith("source_refs"):
24
+ collect(item)
25
+ else:
26
+ visit(item)
27
+ elif isinstance(value, list):
28
+ for item in value:
29
+ visit(item)
30
+
31
+ visit(payload)
32
+ expected_refs = set(expectation.required_source_refs)
33
+ expected_blocks = {
34
+ (sheet.name, block.kind, block.source_range)
35
+ for sheet in expectation.sheets
36
+ for block in sheet.blocks
37
+ }
38
+ actual_blocks = {
39
+ (block["sheet"], block["kind"], block["source_refs"][0].rsplit("!", 1)[1])
40
+ for block in payload["structures"]["blocks"]
41
+ if block["source_refs"]
42
+ }
43
+ encoded = bundle.to_json()
44
+ stable = WorkbookBundle.from_json(encoded).to_json() == encoded
45
+ return {
46
+ "bundle_source_ref_completeness": len(expected_refs & refs) / len(expected_refs)
47
+ if expected_refs
48
+ else 1.0,
49
+ "bundle_structure_recall": len(expected_blocks & actual_blocks) / len(expected_blocks)
50
+ if expected_blocks
51
+ else 1.0,
52
+ "bundle_roundtrip_stability": float(stable),
53
+ }
@@ -0,0 +1,266 @@
1
+ from __future__ import annotations
2
+
3
+ from collections import Counter
4
+ from collections.abc import Iterable
5
+ from dataclasses import dataclass, fields, is_dataclass
6
+ from typing import Any
7
+
8
+ from langparse.types import ParsedDocumentResult
9
+ from langparse.workbooks.quality.bundle import evaluate_bundle_contract
10
+ from langparse.workbooks.quality.facts import evaluate_fact_truth
11
+ from langparse.workbooks.quality.schema import WorkbookExpectation
12
+ from langparse.workbooks.types import SourceRef, WorkbookIR
13
+
14
+ WORKBOOK_QUALITY_METRIC_SCHEMA_VERSION = 2
15
+
16
+
17
+ @dataclass(frozen=True)
18
+ class WorkbookQualityMetrics:
19
+ block_precision: float
20
+ block_recall: float
21
+ header_path_accuracy: float | None
22
+ row_role_f1: float | None
23
+ form_field_exact_match: float | None
24
+ matrix_axis_accuracy: float | None
25
+ continuation_precision: float | None
26
+ continuation_recall: float | None
27
+ source_ref_completeness: float
28
+ source_ref_validity_ratio: float
29
+ cell_coverage_ratio: float
30
+ fallback_rate: float
31
+ object_fact_precision: float | None
32
+ object_fact_recall: float | None
33
+ object_semantic_recall: float | None
34
+ formula_accuracy: float | None = None
35
+ named_range_accuracy: float | None = None
36
+ excel_table_accuracy: float | None = None
37
+ visibility_accuracy: float | None = None
38
+ dependency_precision: float | None = None
39
+ dependency_recall: float | None = None
40
+ bundle_source_ref_completeness: float = 0.0
41
+ bundle_structure_recall: float = 0.0
42
+ bundle_roundtrip_stability: float = 0.0
43
+
44
+
45
+ def evaluate_workbook_result(
46
+ expectation: WorkbookExpectation,
47
+ parsed: ParsedDocumentResult,
48
+ ) -> WorkbookQualityMetrics:
49
+ if not isinstance(parsed.structure, WorkbookIR):
50
+ raise TypeError("Workbook quality evaluation requires WorkbookIR")
51
+ workbook = parsed.structure
52
+
53
+ expected_blocks = [
54
+ (sheet.name, block.source_range, block.kind)
55
+ for sheet in expectation.sheets
56
+ for block in sheet.blocks
57
+ ]
58
+ observed_blocks = [
59
+ (sheet.name, _block_range(block), block.kind)
60
+ for sheet in workbook.sheets
61
+ for block in sheet.blocks
62
+ if block.kind not in {"chart", "image"}
63
+ ]
64
+ block_precision, block_recall = _precision_recall(expected_blocks, observed_blocks)
65
+
66
+ expected_headers = [
67
+ (sheet.name, block.source_range, header.coordinate, header.path)
68
+ for sheet in expectation.sheets
69
+ for block in sheet.blocks
70
+ for header in block.headers
71
+ ]
72
+ observed_headers = [
73
+ (sheet.name, _block_range(block), column.coordinate, tuple(column.path))
74
+ for sheet in workbook.sheets
75
+ for block in sheet.blocks
76
+ if block.logical_table is not None
77
+ for column in block.logical_table.columns
78
+ ]
79
+
80
+ expected_rows = [
81
+ (sheet.name, row.source_range, row.role)
82
+ for sheet in expectation.sheets
83
+ for block in sheet.blocks
84
+ for row in block.rows
85
+ ]
86
+ observed_rows = [
87
+ (sheet.name, row.source_ref.range, row.role)
88
+ for sheet in workbook.sheets
89
+ for block in sheet.blocks
90
+ if block.logical_table is not None
91
+ for row in block.logical_table.rows
92
+ ]
93
+
94
+ expected_fields = [
95
+ (sheet.name, block.source_range, label, _stable_value(value))
96
+ for sheet in expectation.sheets
97
+ for block in sheet.blocks
98
+ for label, value in block.form_fields
99
+ ]
100
+ observed_fields = [
101
+ (sheet.name, _block_range(block), field.label, _stable_value(field.value))
102
+ for sheet in workbook.sheets
103
+ for block in sheet.blocks
104
+ if block.form is not None
105
+ for field in block.form.fields
106
+ ]
107
+
108
+ expected_axes = [
109
+ (sheet.name, block.source_range, axis, value)
110
+ for sheet in expectation.sheets
111
+ for block in sheet.blocks
112
+ for axis, values in (
113
+ ("row", block.matrix_axes.rows),
114
+ ("column", block.matrix_axes.columns),
115
+ )
116
+ for value in values
117
+ ]
118
+ observed_axes = [
119
+ (sheet.name, _block_range(block), axis, header.value)
120
+ for sheet in workbook.sheets
121
+ for block in sheet.blocks
122
+ if block.matrix is not None
123
+ for axis, headers in (
124
+ ("row", block.matrix.row_headers),
125
+ ("column", block.matrix.column_headers),
126
+ )
127
+ for header in headers
128
+ ]
129
+
130
+ expected_continuations = list(expectation.continuations)
131
+ observed_continuations = [
132
+ tuple(ref.key for ref in continuation.source_refs)
133
+ for continuation in workbook.table_continuations
134
+ ]
135
+ continuation_precision, continuation_recall = _optional_precision_recall(
136
+ expected_continuations,
137
+ observed_continuations,
138
+ )
139
+
140
+ available_refs = _collect_source_refs(workbook)
141
+ required_refs = set(expectation.required_source_refs)
142
+ source_ref_completeness = (
143
+ len(required_refs & available_refs) / len(required_refs) if required_refs else 1.0
144
+ )
145
+
146
+ expected_objects = [(item.sheet_name, item.kind, item.anchor) for item in expectation.objects]
147
+ observed_object_facts = [
148
+ (sheet.name, str(item.get("kind")), str(item.get("anchor")))
149
+ for sheet in (workbook.snapshot.sheets if workbook.snapshot is not None else [])
150
+ for item in sheet.objects
151
+ if item.get("kind") is not None and item.get("anchor") is not None
152
+ ]
153
+ observed_semantic_objects = [
154
+ (sheet.name, block.kind, str(block.metadata.get("anchor")))
155
+ for sheet in workbook.sheets
156
+ for block in sheet.blocks
157
+ if block.kind in {"chart", "image"} and block.metadata.get("anchor") is not None
158
+ ]
159
+ object_fact_precision, object_fact_recall = _optional_precision_recall(
160
+ expected_objects,
161
+ observed_object_facts,
162
+ )
163
+ _, object_semantic_recall = _optional_precision_recall(
164
+ expected_objects,
165
+ observed_semantic_objects,
166
+ )
167
+
168
+ block_count = sum(
169
+ block.kind not in {"chart", "image"} for sheet in workbook.sheets for block in sheet.blocks
170
+ )
171
+ fallback_count = sum(
172
+ block.kind == "unclassified" or not block.source_refs
173
+ for sheet in workbook.sheets
174
+ for block in sheet.blocks
175
+ )
176
+ diagnostics = parsed.diagnostics
177
+ return WorkbookQualityMetrics(
178
+ block_precision=block_precision,
179
+ block_recall=block_recall,
180
+ header_path_accuracy=_set_accuracy(expected_headers, observed_headers),
181
+ row_role_f1=_set_f1(expected_rows, observed_rows),
182
+ form_field_exact_match=_set_accuracy(expected_fields, observed_fields),
183
+ matrix_axis_accuracy=_set_accuracy(expected_axes, observed_axes),
184
+ continuation_precision=continuation_precision,
185
+ continuation_recall=continuation_recall,
186
+ source_ref_completeness=source_ref_completeness,
187
+ source_ref_validity_ratio=(
188
+ diagnostics.source_ref_validity_ratio if diagnostics is not None else 0.0
189
+ ),
190
+ cell_coverage_ratio=diagnostics.coverage_ratio if diagnostics is not None else 0.0,
191
+ fallback_rate=fallback_count / block_count if block_count else 0.0,
192
+ object_fact_precision=object_fact_precision,
193
+ object_fact_recall=object_fact_recall,
194
+ object_semantic_recall=object_semantic_recall,
195
+ **evaluate_fact_truth(expectation.facts, workbook),
196
+ **evaluate_bundle_contract(expectation, parsed),
197
+ )
198
+
199
+
200
+ def _precision_recall(expected: Iterable[Any], observed: Iterable[Any]) -> tuple[float, float]:
201
+ expected_counts = Counter(expected)
202
+ observed_counts = Counter(observed)
203
+ matched = sum((expected_counts & observed_counts).values())
204
+ expected_total = expected_counts.total()
205
+ observed_total = observed_counts.total()
206
+ precision = matched / observed_total if observed_total else float(not expected_total)
207
+ recall = matched / expected_total if expected_total else float(not observed_total)
208
+ return precision, recall
209
+
210
+
211
+ def _optional_precision_recall(
212
+ expected: Iterable[Any], observed: Iterable[Any]
213
+ ) -> tuple[float | None, float | None]:
214
+ expected = tuple(expected)
215
+ observed = tuple(observed)
216
+ if not expected and not observed:
217
+ return None, None
218
+ return _precision_recall(expected, observed)
219
+
220
+
221
+ def _set_accuracy(expected: Iterable[Any], observed: Iterable[Any]) -> float | None:
222
+ expected_counts = Counter(expected)
223
+ observed_counts = Counter(observed)
224
+ if not expected_counts and not observed_counts:
225
+ return None
226
+ return sum((expected_counts & observed_counts).values()) / max(
227
+ expected_counts.total(), observed_counts.total()
228
+ )
229
+
230
+
231
+ def _set_f1(expected: Iterable[Any], observed: Iterable[Any]) -> float | None:
232
+ expected = tuple(expected)
233
+ observed = tuple(observed)
234
+ if not expected and not observed:
235
+ return None
236
+ precision, recall = _precision_recall(expected, observed)
237
+ return 2 * precision * recall / (precision + recall) if precision + recall else 0.0
238
+
239
+
240
+ def _block_range(block: Any) -> str:
241
+ return block.source_refs[0].range if block.source_refs else "<missing-source-ref>"
242
+
243
+
244
+ def _stable_value(value: Any) -> str:
245
+ return repr(value)
246
+
247
+
248
+ def _collect_source_refs(value: Any) -> set[str]:
249
+ refs: set[str] = set()
250
+
251
+ def visit(item: Any) -> None:
252
+ if isinstance(item, SourceRef):
253
+ refs.add(item.key)
254
+ elif is_dataclass(item):
255
+ for item_field in fields(item):
256
+ if item_field.name != "snapshot":
257
+ visit(getattr(item, item_field.name))
258
+ elif isinstance(item, dict):
259
+ for nested in item.values():
260
+ visit(nested)
261
+ elif isinstance(item, (list, tuple, set)):
262
+ for nested in item:
263
+ visit(nested)
264
+
265
+ visit(value)
266
+ return refs