langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,229 @@
|
|
|
1
|
+
"""Source-grounded drawing facts and semantic blocks; no image inference or I/O."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from langparse.types import ParseDiagnostics
|
|
10
|
+
from langparse.workbooks.types import (
|
|
11
|
+
SourceRef,
|
|
12
|
+
WorkbookBlock,
|
|
13
|
+
WorkbookIR,
|
|
14
|
+
WorkbookSnapshot,
|
|
15
|
+
stable_id,
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class ChartSeries:
|
|
21
|
+
title: str | None = None
|
|
22
|
+
title_reference: str | None = None
|
|
23
|
+
categories: str | None = None
|
|
24
|
+
values: str | None = None
|
|
25
|
+
x_values: str | None = None
|
|
26
|
+
y_values: str | None = None
|
|
27
|
+
source_refs: list[SourceRef] = field(default_factory=list)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class DrawingFact:
|
|
32
|
+
object_id: str
|
|
33
|
+
kind: str
|
|
34
|
+
anchor: str | None
|
|
35
|
+
title: str | None = None
|
|
36
|
+
chart_types: list[str] = field(default_factory=list)
|
|
37
|
+
series: list[ChartSeries] = field(default_factory=list)
|
|
38
|
+
axes: list[dict[str, Any]] = field(default_factory=list)
|
|
39
|
+
width: float | None = None
|
|
40
|
+
height: float | None = None
|
|
41
|
+
media_id: str | None = None
|
|
42
|
+
enhancement_status: str = "not_requested"
|
|
43
|
+
diagnostics: list[str] = field(default_factory=list)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _text(value: Any) -> str | None:
|
|
47
|
+
if isinstance(value, str):
|
|
48
|
+
return value
|
|
49
|
+
# DrawingML rich text can split a title over multiple runs/paragraphs.
|
|
50
|
+
rich = getattr(getattr(value, "tx", None), "rich", None)
|
|
51
|
+
paragraphs = getattr(rich, "p", [])
|
|
52
|
+
lines = []
|
|
53
|
+
for paragraph in paragraphs:
|
|
54
|
+
runs = [getattr(run, "t", "") for run in getattr(paragraph, "r", [])]
|
|
55
|
+
lines.append("".join(runs))
|
|
56
|
+
return "\n".join(lines).strip() or None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _formula(value: Any) -> str | None:
|
|
60
|
+
for field_name in ("numRef", "strRef", "multiLvlStrRef"):
|
|
61
|
+
reference = getattr(value, field_name, None)
|
|
62
|
+
if reference is not None:
|
|
63
|
+
return getattr(reference, "f", None)
|
|
64
|
+
return None
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _source(formula: str, diagnostics: list[str], sheet_names: dict[str, str]) -> SourceRef | None:
|
|
68
|
+
# External, union and dynamic expressions remain explicit raw facts.
|
|
69
|
+
match = re.fullmatch(
|
|
70
|
+
r"(?:'((?:[^']|'')+)'|([^'!]+))!(\$?[A-Z]+\$?[1-9][0-9]*(?::\$?[A-Z]+\$?[1-9][0-9]*)?)",
|
|
71
|
+
formula,
|
|
72
|
+
)
|
|
73
|
+
if match and "[" not in formula:
|
|
74
|
+
from langparse.workbooks.lineage import _bounds
|
|
75
|
+
|
|
76
|
+
name = (match[1] or match[2]).replace("''", "'")
|
|
77
|
+
cell_range = match[3].replace("$", "")
|
|
78
|
+
try:
|
|
79
|
+
_bounds(cell_range)
|
|
80
|
+
except ValueError:
|
|
81
|
+
pass
|
|
82
|
+
else:
|
|
83
|
+
if name.casefold() in sheet_names:
|
|
84
|
+
return SourceRef(sheet_names[name.casefold()], cell_range)
|
|
85
|
+
diagnostics.append(f"unresolved_chart_reference: {formula}")
|
|
86
|
+
return None
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def extract_objects(sheet: Any) -> list[DrawingFact]:
|
|
90
|
+
from openpyxl.utils import get_column_letter
|
|
91
|
+
|
|
92
|
+
sheet_names = {name.casefold(): name for name in sheet.parent.sheetnames}
|
|
93
|
+
result = []
|
|
94
|
+
for kind, objects in (("chart", sheet._charts), ("image", sheet._images)):
|
|
95
|
+
for index, obj in enumerate(objects):
|
|
96
|
+
anchor = obj.anchor if isinstance(obj.anchor, str) else None
|
|
97
|
+
marker = getattr(obj.anchor, "_from", None)
|
|
98
|
+
if marker is not None:
|
|
99
|
+
anchor = f"{get_column_letter(marker.col + 1)}{marker.row + 1}"
|
|
100
|
+
fact = DrawingFact(stable_id("object", sheet.title, kind, str(index)), kind, anchor)
|
|
101
|
+
if anchor is None:
|
|
102
|
+
fact.diagnostics.append("unresolved_anchor")
|
|
103
|
+
if kind == "image":
|
|
104
|
+
fact.width, fact.height = obj.width, obj.height
|
|
105
|
+
fact.media_id = stable_id("media", sheet.title, str(index))
|
|
106
|
+
else:
|
|
107
|
+
fact.title = _text(obj.title)
|
|
108
|
+
for chart in getattr(obj, "_charts", [obj]):
|
|
109
|
+
fact.chart_types.append(type(chart).__name__)
|
|
110
|
+
for axis_name in ("x_axis", "y_axis", "z_axis"):
|
|
111
|
+
axis = getattr(chart, axis_name, None)
|
|
112
|
+
if axis is not None:
|
|
113
|
+
fact.axes.append(
|
|
114
|
+
{
|
|
115
|
+
"axis": axis_name,
|
|
116
|
+
"id": axis.axId,
|
|
117
|
+
"title": _text(axis.title),
|
|
118
|
+
"position": axis.axPos,
|
|
119
|
+
}
|
|
120
|
+
)
|
|
121
|
+
for series in chart.series:
|
|
122
|
+
tx = getattr(series, "tx", None)
|
|
123
|
+
item = ChartSeries(
|
|
124
|
+
title=getattr(tx, "v", None),
|
|
125
|
+
title_reference=_formula(tx),
|
|
126
|
+
categories=_formula(getattr(series, "cat", None)),
|
|
127
|
+
values=_formula(getattr(series, "val", None)),
|
|
128
|
+
x_values=_formula(getattr(series, "xVal", None)),
|
|
129
|
+
y_values=_formula(getattr(series, "yVal", None)),
|
|
130
|
+
)
|
|
131
|
+
for expression in (
|
|
132
|
+
item.title_reference,
|
|
133
|
+
item.categories,
|
|
134
|
+
item.values,
|
|
135
|
+
item.x_values,
|
|
136
|
+
item.y_values,
|
|
137
|
+
):
|
|
138
|
+
if expression:
|
|
139
|
+
ref = _source(expression, fact.diagnostics, sheet_names)
|
|
140
|
+
if ref is not None:
|
|
141
|
+
item.source_refs.append(ref)
|
|
142
|
+
fact.series.append(item)
|
|
143
|
+
result.append(fact)
|
|
144
|
+
return result
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def attach_objects(
|
|
148
|
+
snapshot: WorkbookSnapshot, ir: WorkbookIR, diagnostics: ParseDiagnostics
|
|
149
|
+
) -> None:
|
|
150
|
+
"""Add semantic objects after cell interpretation without claiming cell coverage."""
|
|
151
|
+
total = covered = 0
|
|
152
|
+
sheets = {sheet.name: sheet for sheet in ir.sheets}
|
|
153
|
+
for source in snapshot.sheets:
|
|
154
|
+
target = sheets[source.name]
|
|
155
|
+
for obj in source.objects:
|
|
156
|
+
total += 1
|
|
157
|
+
anchor = obj.get("anchor")
|
|
158
|
+
notes = list(obj.get("diagnostics", []))
|
|
159
|
+
try:
|
|
160
|
+
validate_object_source(snapshot, f"{source.name}!{anchor or ''}")
|
|
161
|
+
except (TypeError, ValueError):
|
|
162
|
+
notes.append("unresolved_anchor")
|
|
163
|
+
diagnostics.warnings.append(f"Object on {source.name}: unresolved_anchor")
|
|
164
|
+
diagnostics.status = "partial"
|
|
165
|
+
continue
|
|
166
|
+
refs = [SourceRef(source.name, anchor)]
|
|
167
|
+
for series in obj.get("series", []):
|
|
168
|
+
for raw in series.get("source_refs", []):
|
|
169
|
+
if raw["sheet_name"] in sheets:
|
|
170
|
+
ref = SourceRef(**raw)
|
|
171
|
+
if ref not in refs:
|
|
172
|
+
refs.append(ref)
|
|
173
|
+
else:
|
|
174
|
+
notes.append(f"missing_chart_sheet: {raw['sheet_name']}")
|
|
175
|
+
if notes:
|
|
176
|
+
diagnostics.status = "partial"
|
|
177
|
+
diagnostics.warnings.extend(f"Object on {source.name}: {note}" for note in notes)
|
|
178
|
+
target.blocks.append(
|
|
179
|
+
WorkbookBlock(
|
|
180
|
+
block_id=obj.get("object_id")
|
|
181
|
+
or stable_id("object", source.name, anchor, str(total)),
|
|
182
|
+
kind=obj["kind"],
|
|
183
|
+
source_refs=refs,
|
|
184
|
+
metadata={"anchor": anchor, "object": obj},
|
|
185
|
+
diagnostics=[{"code": note} for note in notes],
|
|
186
|
+
)
|
|
187
|
+
)
|
|
188
|
+
covered += 1
|
|
189
|
+
kind = obj["kind"]
|
|
190
|
+
diagnostics.block_count_by_kind[kind] = diagnostics.block_count_by_kind.get(kind, 0) + 1
|
|
191
|
+
diagnostics.object_coverage_ratio = covered / total if total else 1.0
|
|
192
|
+
if snapshot.metadata.get("drawing_read_errors"):
|
|
193
|
+
diagnostics.status = "partial"
|
|
194
|
+
diagnostics.object_coverage_ratio = None
|
|
195
|
+
diagnostics.unsupported_features.extend(snapshot.metadata["drawing_read_errors"])
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def render_object(block: WorkbookBlock) -> str:
|
|
199
|
+
obj = block.metadata["object"]
|
|
200
|
+
source = ", ".join(ref.key for ref in block.source_refs)
|
|
201
|
+
title = obj.get("title") or obj["kind"].capitalize()
|
|
202
|
+
lines = [f"<!-- source_ranges: {source} -->", f"**{title}** ({obj['kind']})"]
|
|
203
|
+
for series in obj.get("series", []):
|
|
204
|
+
references = [
|
|
205
|
+
series[key]
|
|
206
|
+
for key in ("categories", "values", "x_values", "y_values")
|
|
207
|
+
if series.get(key)
|
|
208
|
+
]
|
|
209
|
+
lines.append("- " + (series.get("title") or "Series") + ": " + ", ".join(references))
|
|
210
|
+
if obj["kind"] == "image":
|
|
211
|
+
lines.append(
|
|
212
|
+
f"Image {obj.get('media_id')}; {obj.get('width')} × {obj.get('height')}; OCR/VLM not requested."
|
|
213
|
+
)
|
|
214
|
+
return "\n\n".join(lines)
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def validate_object_source(snapshot: WorkbookSnapshot | None, source: str) -> None:
|
|
218
|
+
"""Drawing anchors can be outside used cells, but must be valid Excel coordinates."""
|
|
219
|
+
from openpyxl.utils.cell import range_boundaries
|
|
220
|
+
|
|
221
|
+
sheet_name, cell_range = source.rsplit("!", 1)
|
|
222
|
+
if snapshot is None or not any(sheet.name == sheet_name for sheet in snapshot.sheets):
|
|
223
|
+
raise ValueError(f"Object source references unknown sheet: {sheet_name}")
|
|
224
|
+
try:
|
|
225
|
+
left, top, right, bottom = range_boundaries(cell_range)
|
|
226
|
+
if not (1 <= left <= right <= 16384 and 1 <= top <= bottom <= 1048576):
|
|
227
|
+
raise ValueError("out of bounds")
|
|
228
|
+
except (ValueError, TypeError) as exc:
|
|
229
|
+
raise ValueError(f"Invalid object source range: {source}") from exc
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""Whole-workbook quality evaluation primitives."""
|
|
2
|
+
|
|
3
|
+
from .evaluator import (
|
|
4
|
+
WORKBOOK_QUALITY_METRIC_SCHEMA_VERSION,
|
|
5
|
+
WorkbookQualityMetrics,
|
|
6
|
+
evaluate_workbook_result,
|
|
7
|
+
)
|
|
8
|
+
from .schema import (
|
|
9
|
+
WorkbookQualityGate,
|
|
10
|
+
WorkbookQualityManifest,
|
|
11
|
+
WorkbookQualityManifestError,
|
|
12
|
+
load_workbook_quality_manifest,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
"WorkbookQualityManifest",
|
|
17
|
+
"WorkbookQualityManifestError",
|
|
18
|
+
"WorkbookQualityMetrics",
|
|
19
|
+
"WorkbookQualityGate",
|
|
20
|
+
"WORKBOOK_QUALITY_METRIC_SCHEMA_VERSION",
|
|
21
|
+
"evaluate_workbook_result",
|
|
22
|
+
"load_workbook_quality_manifest",
|
|
23
|
+
]
|
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
"""Check exported workbook consumption contracts against the same frozen truth."""
|
|
2
|
+
|
|
3
|
+
from langparse.workbooks.bundle import WorkbookBundle
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def evaluate_bundle_contract(expectation, parsed) -> dict[str, float]:
|
|
7
|
+
bundle = WorkbookBundle.from_result(parsed, max_rows=None)
|
|
8
|
+
payload = bundle.to_dict()
|
|
9
|
+
refs: set[str] = set()
|
|
10
|
+
|
|
11
|
+
def collect(value):
|
|
12
|
+
if isinstance(value, str):
|
|
13
|
+
refs.add(value)
|
|
14
|
+
elif isinstance(value, list):
|
|
15
|
+
for item in value:
|
|
16
|
+
collect(item)
|
|
17
|
+
|
|
18
|
+
def visit(value):
|
|
19
|
+
if isinstance(value, dict):
|
|
20
|
+
for key, item in value.items():
|
|
21
|
+
if key in {"source_ref", "dependent", "target"} and isinstance(item, str):
|
|
22
|
+
refs.add(item)
|
|
23
|
+
elif key.endswith("source_refs"):
|
|
24
|
+
collect(item)
|
|
25
|
+
else:
|
|
26
|
+
visit(item)
|
|
27
|
+
elif isinstance(value, list):
|
|
28
|
+
for item in value:
|
|
29
|
+
visit(item)
|
|
30
|
+
|
|
31
|
+
visit(payload)
|
|
32
|
+
expected_refs = set(expectation.required_source_refs)
|
|
33
|
+
expected_blocks = {
|
|
34
|
+
(sheet.name, block.kind, block.source_range)
|
|
35
|
+
for sheet in expectation.sheets
|
|
36
|
+
for block in sheet.blocks
|
|
37
|
+
}
|
|
38
|
+
actual_blocks = {
|
|
39
|
+
(block["sheet"], block["kind"], block["source_refs"][0].rsplit("!", 1)[1])
|
|
40
|
+
for block in payload["structures"]["blocks"]
|
|
41
|
+
if block["source_refs"]
|
|
42
|
+
}
|
|
43
|
+
encoded = bundle.to_json()
|
|
44
|
+
stable = WorkbookBundle.from_json(encoded).to_json() == encoded
|
|
45
|
+
return {
|
|
46
|
+
"bundle_source_ref_completeness": len(expected_refs & refs) / len(expected_refs)
|
|
47
|
+
if expected_refs
|
|
48
|
+
else 1.0,
|
|
49
|
+
"bundle_structure_recall": len(expected_blocks & actual_blocks) / len(expected_blocks)
|
|
50
|
+
if expected_blocks
|
|
51
|
+
else 1.0,
|
|
52
|
+
"bundle_roundtrip_stability": float(stable),
|
|
53
|
+
}
|
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections import Counter
|
|
4
|
+
from collections.abc import Iterable
|
|
5
|
+
from dataclasses import dataclass, fields, is_dataclass
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
from langparse.types import ParsedDocumentResult
|
|
9
|
+
from langparse.workbooks.quality.bundle import evaluate_bundle_contract
|
|
10
|
+
from langparse.workbooks.quality.facts import evaluate_fact_truth
|
|
11
|
+
from langparse.workbooks.quality.schema import WorkbookExpectation
|
|
12
|
+
from langparse.workbooks.types import SourceRef, WorkbookIR
|
|
13
|
+
|
|
14
|
+
WORKBOOK_QUALITY_METRIC_SCHEMA_VERSION = 2
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True)
|
|
18
|
+
class WorkbookQualityMetrics:
|
|
19
|
+
block_precision: float
|
|
20
|
+
block_recall: float
|
|
21
|
+
header_path_accuracy: float | None
|
|
22
|
+
row_role_f1: float | None
|
|
23
|
+
form_field_exact_match: float | None
|
|
24
|
+
matrix_axis_accuracy: float | None
|
|
25
|
+
continuation_precision: float | None
|
|
26
|
+
continuation_recall: float | None
|
|
27
|
+
source_ref_completeness: float
|
|
28
|
+
source_ref_validity_ratio: float
|
|
29
|
+
cell_coverage_ratio: float
|
|
30
|
+
fallback_rate: float
|
|
31
|
+
object_fact_precision: float | None
|
|
32
|
+
object_fact_recall: float | None
|
|
33
|
+
object_semantic_recall: float | None
|
|
34
|
+
formula_accuracy: float | None = None
|
|
35
|
+
named_range_accuracy: float | None = None
|
|
36
|
+
excel_table_accuracy: float | None = None
|
|
37
|
+
visibility_accuracy: float | None = None
|
|
38
|
+
dependency_precision: float | None = None
|
|
39
|
+
dependency_recall: float | None = None
|
|
40
|
+
bundle_source_ref_completeness: float = 0.0
|
|
41
|
+
bundle_structure_recall: float = 0.0
|
|
42
|
+
bundle_roundtrip_stability: float = 0.0
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def evaluate_workbook_result(
|
|
46
|
+
expectation: WorkbookExpectation,
|
|
47
|
+
parsed: ParsedDocumentResult,
|
|
48
|
+
) -> WorkbookQualityMetrics:
|
|
49
|
+
if not isinstance(parsed.structure, WorkbookIR):
|
|
50
|
+
raise TypeError("Workbook quality evaluation requires WorkbookIR")
|
|
51
|
+
workbook = parsed.structure
|
|
52
|
+
|
|
53
|
+
expected_blocks = [
|
|
54
|
+
(sheet.name, block.source_range, block.kind)
|
|
55
|
+
for sheet in expectation.sheets
|
|
56
|
+
for block in sheet.blocks
|
|
57
|
+
]
|
|
58
|
+
observed_blocks = [
|
|
59
|
+
(sheet.name, _block_range(block), block.kind)
|
|
60
|
+
for sheet in workbook.sheets
|
|
61
|
+
for block in sheet.blocks
|
|
62
|
+
if block.kind not in {"chart", "image"}
|
|
63
|
+
]
|
|
64
|
+
block_precision, block_recall = _precision_recall(expected_blocks, observed_blocks)
|
|
65
|
+
|
|
66
|
+
expected_headers = [
|
|
67
|
+
(sheet.name, block.source_range, header.coordinate, header.path)
|
|
68
|
+
for sheet in expectation.sheets
|
|
69
|
+
for block in sheet.blocks
|
|
70
|
+
for header in block.headers
|
|
71
|
+
]
|
|
72
|
+
observed_headers = [
|
|
73
|
+
(sheet.name, _block_range(block), column.coordinate, tuple(column.path))
|
|
74
|
+
for sheet in workbook.sheets
|
|
75
|
+
for block in sheet.blocks
|
|
76
|
+
if block.logical_table is not None
|
|
77
|
+
for column in block.logical_table.columns
|
|
78
|
+
]
|
|
79
|
+
|
|
80
|
+
expected_rows = [
|
|
81
|
+
(sheet.name, row.source_range, row.role)
|
|
82
|
+
for sheet in expectation.sheets
|
|
83
|
+
for block in sheet.blocks
|
|
84
|
+
for row in block.rows
|
|
85
|
+
]
|
|
86
|
+
observed_rows = [
|
|
87
|
+
(sheet.name, row.source_ref.range, row.role)
|
|
88
|
+
for sheet in workbook.sheets
|
|
89
|
+
for block in sheet.blocks
|
|
90
|
+
if block.logical_table is not None
|
|
91
|
+
for row in block.logical_table.rows
|
|
92
|
+
]
|
|
93
|
+
|
|
94
|
+
expected_fields = [
|
|
95
|
+
(sheet.name, block.source_range, label, _stable_value(value))
|
|
96
|
+
for sheet in expectation.sheets
|
|
97
|
+
for block in sheet.blocks
|
|
98
|
+
for label, value in block.form_fields
|
|
99
|
+
]
|
|
100
|
+
observed_fields = [
|
|
101
|
+
(sheet.name, _block_range(block), field.label, _stable_value(field.value))
|
|
102
|
+
for sheet in workbook.sheets
|
|
103
|
+
for block in sheet.blocks
|
|
104
|
+
if block.form is not None
|
|
105
|
+
for field in block.form.fields
|
|
106
|
+
]
|
|
107
|
+
|
|
108
|
+
expected_axes = [
|
|
109
|
+
(sheet.name, block.source_range, axis, value)
|
|
110
|
+
for sheet in expectation.sheets
|
|
111
|
+
for block in sheet.blocks
|
|
112
|
+
for axis, values in (
|
|
113
|
+
("row", block.matrix_axes.rows),
|
|
114
|
+
("column", block.matrix_axes.columns),
|
|
115
|
+
)
|
|
116
|
+
for value in values
|
|
117
|
+
]
|
|
118
|
+
observed_axes = [
|
|
119
|
+
(sheet.name, _block_range(block), axis, header.value)
|
|
120
|
+
for sheet in workbook.sheets
|
|
121
|
+
for block in sheet.blocks
|
|
122
|
+
if block.matrix is not None
|
|
123
|
+
for axis, headers in (
|
|
124
|
+
("row", block.matrix.row_headers),
|
|
125
|
+
("column", block.matrix.column_headers),
|
|
126
|
+
)
|
|
127
|
+
for header in headers
|
|
128
|
+
]
|
|
129
|
+
|
|
130
|
+
expected_continuations = list(expectation.continuations)
|
|
131
|
+
observed_continuations = [
|
|
132
|
+
tuple(ref.key for ref in continuation.source_refs)
|
|
133
|
+
for continuation in workbook.table_continuations
|
|
134
|
+
]
|
|
135
|
+
continuation_precision, continuation_recall = _optional_precision_recall(
|
|
136
|
+
expected_continuations,
|
|
137
|
+
observed_continuations,
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
available_refs = _collect_source_refs(workbook)
|
|
141
|
+
required_refs = set(expectation.required_source_refs)
|
|
142
|
+
source_ref_completeness = (
|
|
143
|
+
len(required_refs & available_refs) / len(required_refs) if required_refs else 1.0
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
expected_objects = [(item.sheet_name, item.kind, item.anchor) for item in expectation.objects]
|
|
147
|
+
observed_object_facts = [
|
|
148
|
+
(sheet.name, str(item.get("kind")), str(item.get("anchor")))
|
|
149
|
+
for sheet in (workbook.snapshot.sheets if workbook.snapshot is not None else [])
|
|
150
|
+
for item in sheet.objects
|
|
151
|
+
if item.get("kind") is not None and item.get("anchor") is not None
|
|
152
|
+
]
|
|
153
|
+
observed_semantic_objects = [
|
|
154
|
+
(sheet.name, block.kind, str(block.metadata.get("anchor")))
|
|
155
|
+
for sheet in workbook.sheets
|
|
156
|
+
for block in sheet.blocks
|
|
157
|
+
if block.kind in {"chart", "image"} and block.metadata.get("anchor") is not None
|
|
158
|
+
]
|
|
159
|
+
object_fact_precision, object_fact_recall = _optional_precision_recall(
|
|
160
|
+
expected_objects,
|
|
161
|
+
observed_object_facts,
|
|
162
|
+
)
|
|
163
|
+
_, object_semantic_recall = _optional_precision_recall(
|
|
164
|
+
expected_objects,
|
|
165
|
+
observed_semantic_objects,
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
block_count = sum(
|
|
169
|
+
block.kind not in {"chart", "image"} for sheet in workbook.sheets for block in sheet.blocks
|
|
170
|
+
)
|
|
171
|
+
fallback_count = sum(
|
|
172
|
+
block.kind == "unclassified" or not block.source_refs
|
|
173
|
+
for sheet in workbook.sheets
|
|
174
|
+
for block in sheet.blocks
|
|
175
|
+
)
|
|
176
|
+
diagnostics = parsed.diagnostics
|
|
177
|
+
return WorkbookQualityMetrics(
|
|
178
|
+
block_precision=block_precision,
|
|
179
|
+
block_recall=block_recall,
|
|
180
|
+
header_path_accuracy=_set_accuracy(expected_headers, observed_headers),
|
|
181
|
+
row_role_f1=_set_f1(expected_rows, observed_rows),
|
|
182
|
+
form_field_exact_match=_set_accuracy(expected_fields, observed_fields),
|
|
183
|
+
matrix_axis_accuracy=_set_accuracy(expected_axes, observed_axes),
|
|
184
|
+
continuation_precision=continuation_precision,
|
|
185
|
+
continuation_recall=continuation_recall,
|
|
186
|
+
source_ref_completeness=source_ref_completeness,
|
|
187
|
+
source_ref_validity_ratio=(
|
|
188
|
+
diagnostics.source_ref_validity_ratio if diagnostics is not None else 0.0
|
|
189
|
+
),
|
|
190
|
+
cell_coverage_ratio=diagnostics.coverage_ratio if diagnostics is not None else 0.0,
|
|
191
|
+
fallback_rate=fallback_count / block_count if block_count else 0.0,
|
|
192
|
+
object_fact_precision=object_fact_precision,
|
|
193
|
+
object_fact_recall=object_fact_recall,
|
|
194
|
+
object_semantic_recall=object_semantic_recall,
|
|
195
|
+
**evaluate_fact_truth(expectation.facts, workbook),
|
|
196
|
+
**evaluate_bundle_contract(expectation, parsed),
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def _precision_recall(expected: Iterable[Any], observed: Iterable[Any]) -> tuple[float, float]:
|
|
201
|
+
expected_counts = Counter(expected)
|
|
202
|
+
observed_counts = Counter(observed)
|
|
203
|
+
matched = sum((expected_counts & observed_counts).values())
|
|
204
|
+
expected_total = expected_counts.total()
|
|
205
|
+
observed_total = observed_counts.total()
|
|
206
|
+
precision = matched / observed_total if observed_total else float(not expected_total)
|
|
207
|
+
recall = matched / expected_total if expected_total else float(not observed_total)
|
|
208
|
+
return precision, recall
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _optional_precision_recall(
|
|
212
|
+
expected: Iterable[Any], observed: Iterable[Any]
|
|
213
|
+
) -> tuple[float | None, float | None]:
|
|
214
|
+
expected = tuple(expected)
|
|
215
|
+
observed = tuple(observed)
|
|
216
|
+
if not expected and not observed:
|
|
217
|
+
return None, None
|
|
218
|
+
return _precision_recall(expected, observed)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _set_accuracy(expected: Iterable[Any], observed: Iterable[Any]) -> float | None:
|
|
222
|
+
expected_counts = Counter(expected)
|
|
223
|
+
observed_counts = Counter(observed)
|
|
224
|
+
if not expected_counts and not observed_counts:
|
|
225
|
+
return None
|
|
226
|
+
return sum((expected_counts & observed_counts).values()) / max(
|
|
227
|
+
expected_counts.total(), observed_counts.total()
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _set_f1(expected: Iterable[Any], observed: Iterable[Any]) -> float | None:
|
|
232
|
+
expected = tuple(expected)
|
|
233
|
+
observed = tuple(observed)
|
|
234
|
+
if not expected and not observed:
|
|
235
|
+
return None
|
|
236
|
+
precision, recall = _precision_recall(expected, observed)
|
|
237
|
+
return 2 * precision * recall / (precision + recall) if precision + recall else 0.0
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _block_range(block: Any) -> str:
|
|
241
|
+
return block.source_refs[0].range if block.source_refs else "<missing-source-ref>"
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _stable_value(value: Any) -> str:
|
|
245
|
+
return repr(value)
|
|
246
|
+
|
|
247
|
+
|
|
248
|
+
def _collect_source_refs(value: Any) -> set[str]:
|
|
249
|
+
refs: set[str] = set()
|
|
250
|
+
|
|
251
|
+
def visit(item: Any) -> None:
|
|
252
|
+
if isinstance(item, SourceRef):
|
|
253
|
+
refs.add(item.key)
|
|
254
|
+
elif is_dataclass(item):
|
|
255
|
+
for item_field in fields(item):
|
|
256
|
+
if item_field.name != "snapshot":
|
|
257
|
+
visit(getattr(item, item_field.name))
|
|
258
|
+
elif isinstance(item, dict):
|
|
259
|
+
for nested in item.values():
|
|
260
|
+
visit(nested)
|
|
261
|
+
elif isinstance(item, (list, tuple, set)):
|
|
262
|
+
for nested in item:
|
|
263
|
+
visit(nested)
|
|
264
|
+
|
|
265
|
+
visit(value)
|
|
266
|
+
return refs
|