langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
"""Optional complete fact truth; absent truth is never scored as perfect."""
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Any
|
|
5
|
+
|
|
6
|
+
FACT_METRICS = frozenset(
|
|
7
|
+
{
|
|
8
|
+
"formula_accuracy",
|
|
9
|
+
"named_range_accuracy",
|
|
10
|
+
"excel_table_accuracy",
|
|
11
|
+
"visibility_accuracy",
|
|
12
|
+
"dependency_precision",
|
|
13
|
+
"dependency_recall",
|
|
14
|
+
}
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass(frozen=True)
|
|
19
|
+
class WorkbookFactTruth:
|
|
20
|
+
formulas: tuple[tuple[str, str], ...]
|
|
21
|
+
defined_names: tuple[tuple[str, str, str | None], ...]
|
|
22
|
+
tables: tuple[tuple[str, str, tuple[str, ...]], ...]
|
|
23
|
+
sheet_visibility: tuple[tuple[str, str], ...]
|
|
24
|
+
dependencies: tuple[tuple[str, str | None, str, str], ...]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def load_fact_truth(data: Any, sheets: set[str]) -> WorkbookFactTruth:
|
|
28
|
+
import re
|
|
29
|
+
|
|
30
|
+
keys = {"formulas", "defined_names", "tables", "sheet_visibility", "dependencies"}
|
|
31
|
+
if not isinstance(data, dict) or set(data) != keys:
|
|
32
|
+
raise ValueError("Fact truth keys do not match schema")
|
|
33
|
+
|
|
34
|
+
def rows(key, fields):
|
|
35
|
+
items = data[key]
|
|
36
|
+
if not isinstance(items, list):
|
|
37
|
+
raise ValueError(f"{key} truth must be a list")
|
|
38
|
+
for item in items:
|
|
39
|
+
if not isinstance(item, dict) or set(item) != set(fields):
|
|
40
|
+
raise ValueError(f"{key} truth keys do not match schema")
|
|
41
|
+
return items
|
|
42
|
+
|
|
43
|
+
def text(value):
|
|
44
|
+
if not isinstance(value, str) or not value:
|
|
45
|
+
raise ValueError("Fact truth text must be nonempty")
|
|
46
|
+
return value
|
|
47
|
+
|
|
48
|
+
def source(value):
|
|
49
|
+
from openpyxl.utils.cell import range_boundaries
|
|
50
|
+
|
|
51
|
+
value = text(value)
|
|
52
|
+
name, separator, cell_range = value.rpartition("!")
|
|
53
|
+
if (
|
|
54
|
+
not separator
|
|
55
|
+
or name not in sheets
|
|
56
|
+
or not re.fullmatch(r"[A-Z]+[1-9][0-9]*(?::[A-Z]+[1-9][0-9]*)?", cell_range)
|
|
57
|
+
):
|
|
58
|
+
raise ValueError("Fact truth source must identify a known sheet and finite range")
|
|
59
|
+
left, top, right, bottom = range_boundaries(cell_range)
|
|
60
|
+
if not (1 <= left <= right <= 16384 and 1 <= top <= bottom <= 1048576):
|
|
61
|
+
raise ValueError("Fact truth source is out of bounds")
|
|
62
|
+
return value
|
|
63
|
+
|
|
64
|
+
formulas = tuple(
|
|
65
|
+
(source(r["source_ref"]), text(r["formula"]))
|
|
66
|
+
for r in rows("formulas", ("source_ref", "formula"))
|
|
67
|
+
)
|
|
68
|
+
names = []
|
|
69
|
+
for r in rows("defined_names", ("name", "definition", "scope_sheet")):
|
|
70
|
+
scope = r["scope_sheet"]
|
|
71
|
+
if scope is not None and (not isinstance(scope, str) or scope not in sheets):
|
|
72
|
+
raise ValueError("Named range scope must identify a known sheet")
|
|
73
|
+
names.append((text(r["name"]), text(r["definition"]), scope))
|
|
74
|
+
tables = []
|
|
75
|
+
for r in rows("tables", ("name", "source_ref", "columns")):
|
|
76
|
+
if not isinstance(r["columns"], list):
|
|
77
|
+
raise ValueError("Table truth columns must be a list")
|
|
78
|
+
tables.append(
|
|
79
|
+
(text(r["name"]), source(r["source_ref"]), tuple(text(c) for c in r["columns"]))
|
|
80
|
+
)
|
|
81
|
+
visibility = []
|
|
82
|
+
for r in rows("sheet_visibility", ("sheet", "state")):
|
|
83
|
+
if r["sheet"] not in sheets or r["state"] not in {"visible", "hidden", "veryHidden"}:
|
|
84
|
+
raise ValueError("Invalid sheet visibility truth")
|
|
85
|
+
visibility.append((r["sheet"], r["state"]))
|
|
86
|
+
edges = []
|
|
87
|
+
for r in rows("dependencies", ("dependent", "target", "status", "reference")):
|
|
88
|
+
if r["status"] not in {
|
|
89
|
+
"resolved",
|
|
90
|
+
"external",
|
|
91
|
+
"unresolved",
|
|
92
|
+
"unsupported",
|
|
93
|
+
"dynamic",
|
|
94
|
+
"circular",
|
|
95
|
+
}:
|
|
96
|
+
raise ValueError("Invalid dependency status truth")
|
|
97
|
+
target = source(r["target"]) if r["target"] is not None else None
|
|
98
|
+
if r["status"] != "resolved" and target is not None:
|
|
99
|
+
raise ValueError("Unresolved dependency cannot have a target")
|
|
100
|
+
edges.append((source(r["dependent"]), target, r["status"], text(r["reference"])))
|
|
101
|
+
groups = (formulas, tuple(names), tuple(tables), tuple(visibility), tuple(edges))
|
|
102
|
+
if any(len(items) != len(set(items)) for items in groups):
|
|
103
|
+
raise ValueError("Duplicate fact truth")
|
|
104
|
+
return WorkbookFactTruth(*groups)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def evaluate_fact_truth(truth, workbook) -> dict[str, float | None]:
|
|
108
|
+
from collections import Counter
|
|
109
|
+
|
|
110
|
+
if truth is None:
|
|
111
|
+
return dict.fromkeys(FACT_METRICS)
|
|
112
|
+
snapshot = workbook.snapshot
|
|
113
|
+
facts = snapshot.reference_facts if snapshot else None
|
|
114
|
+
observed = (
|
|
115
|
+
[(f.source_ref.key, f.formula) for f in facts.formulas] if facts else [],
|
|
116
|
+
[(n.name, n.definition, n.scope_sheet) for n in facts.defined_names] if facts else [],
|
|
117
|
+
[(t.name, t.source_ref.key, tuple(t.columns)) for t in facts.tables] if facts else [],
|
|
118
|
+
[(s.name, s.visibility) for s in snapshot.sheets] if snapshot else [],
|
|
119
|
+
)
|
|
120
|
+
scores = {}
|
|
121
|
+
for name, expected, actual in zip(
|
|
122
|
+
("formula_accuracy", "named_range_accuracy", "excel_table_accuracy", "visibility_accuracy"),
|
|
123
|
+
(truth.formulas, truth.defined_names, truth.tables, truth.sheet_visibility),
|
|
124
|
+
observed,
|
|
125
|
+
strict=True,
|
|
126
|
+
):
|
|
127
|
+
want, got = Counter(expected), Counter(actual)
|
|
128
|
+
denominator = max(want.total(), got.total())
|
|
129
|
+
scores[name] = sum((want & got).values()) / denominator if denominator else 1.0
|
|
130
|
+
edges = (
|
|
131
|
+
[
|
|
132
|
+
(e.dependent.key, e.target.key if e.target else None, e.status, e.reference)
|
|
133
|
+
for e in workbook.lineage.edges
|
|
134
|
+
]
|
|
135
|
+
if workbook.lineage
|
|
136
|
+
else []
|
|
137
|
+
)
|
|
138
|
+
want, got = Counter(truth.dependencies), Counter(edges)
|
|
139
|
+
correct = sum((want & got).values())
|
|
140
|
+
scores["dependency_precision"] = correct / got.total() if got else (0.0 if want else 1.0)
|
|
141
|
+
scores["dependency_recall"] = correct / want.total() if want else (0.0 if got else 1.0)
|
|
142
|
+
return scores
|
|
@@ -0,0 +1,462 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import re
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from langparse.workbooks.quality.facts import FACT_METRICS, WorkbookFactTruth, load_fact_truth
|
|
11
|
+
|
|
12
|
+
_QUALITY_METRICS = FACT_METRICS | frozenset(
|
|
13
|
+
{
|
|
14
|
+
"bundle_source_ref_completeness",
|
|
15
|
+
"bundle_structure_recall",
|
|
16
|
+
"bundle_roundtrip_stability",
|
|
17
|
+
"block_precision",
|
|
18
|
+
"block_recall",
|
|
19
|
+
"header_path_accuracy",
|
|
20
|
+
"row_role_f1",
|
|
21
|
+
"form_field_exact_match",
|
|
22
|
+
"matrix_axis_accuracy",
|
|
23
|
+
"continuation_precision",
|
|
24
|
+
"continuation_recall",
|
|
25
|
+
"source_ref_completeness",
|
|
26
|
+
"source_ref_validity_ratio",
|
|
27
|
+
"cell_coverage_ratio",
|
|
28
|
+
"fallback_rate",
|
|
29
|
+
"object_fact_precision",
|
|
30
|
+
"object_fact_recall",
|
|
31
|
+
"object_semantic_recall",
|
|
32
|
+
}
|
|
33
|
+
)
|
|
34
|
+
_SOURCE_REF_PATTERN = re.compile(r"^[^!]+![A-Z]+[1-9][0-9]*(?::[A-Z]+[1-9][0-9]*)?$")
|
|
35
|
+
_CELL_RANGE_PATTERN = re.compile(r"^[A-Z]+[1-9][0-9]*(?::[A-Z]+[1-9][0-9]*)?$")
|
|
36
|
+
_BLOCK_KINDS = frozenset(
|
|
37
|
+
{"logical_table", "form", "matrix", "text", "unclassified", "chart", "image"}
|
|
38
|
+
)
|
|
39
|
+
_ROW_ROLES = frozenset(
|
|
40
|
+
{
|
|
41
|
+
"title",
|
|
42
|
+
"context",
|
|
43
|
+
"header",
|
|
44
|
+
"repeated_title",
|
|
45
|
+
"repeated_context",
|
|
46
|
+
"repeated_header",
|
|
47
|
+
"section_header",
|
|
48
|
+
"data",
|
|
49
|
+
"total",
|
|
50
|
+
"unknown",
|
|
51
|
+
}
|
|
52
|
+
)
|
|
53
|
+
_OBJECT_KINDS = frozenset({"chart", "image"})
|
|
54
|
+
_SAFE_TOKEN_PATTERN = re.compile(r"^[A-Za-z0-9_-]{1,128}$")
|
|
55
|
+
_SHA256_PATTERN = re.compile(r"^sha256:[0-9a-f]{64}$")
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class WorkbookQualityManifestError(ValueError):
|
|
59
|
+
"""Raised when a workbook quality manifest cannot be trusted."""
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass(frozen=True)
|
|
63
|
+
class HeaderTruth:
|
|
64
|
+
coordinate: str
|
|
65
|
+
path: tuple[str, ...]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass(frozen=True)
|
|
69
|
+
class RowTruth:
|
|
70
|
+
source_range: str
|
|
71
|
+
role: str
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
@dataclass(frozen=True)
|
|
75
|
+
class MatrixAxesTruth:
|
|
76
|
+
rows: tuple[str, ...]
|
|
77
|
+
columns: tuple[str, ...]
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
@dataclass(frozen=True)
|
|
81
|
+
class BlockTruth:
|
|
82
|
+
source_range: str
|
|
83
|
+
kind: str
|
|
84
|
+
headers: tuple[HeaderTruth, ...]
|
|
85
|
+
rows: tuple[RowTruth, ...]
|
|
86
|
+
form_fields: tuple[tuple[str, Any], ...]
|
|
87
|
+
matrix_axes: MatrixAxesTruth
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@dataclass(frozen=True)
|
|
91
|
+
class SheetTruth:
|
|
92
|
+
name: str
|
|
93
|
+
blocks: tuple[BlockTruth, ...]
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
@dataclass(frozen=True)
|
|
97
|
+
class ObjectTruth:
|
|
98
|
+
sheet_name: str
|
|
99
|
+
kind: str
|
|
100
|
+
anchor: str
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
@dataclass(frozen=True)
|
|
104
|
+
class WorkbookExpectation:
|
|
105
|
+
sheets: tuple[SheetTruth, ...]
|
|
106
|
+
continuations: tuple[tuple[str, ...], ...]
|
|
107
|
+
required_source_refs: tuple[str, ...]
|
|
108
|
+
objects: tuple[ObjectTruth, ...]
|
|
109
|
+
facts: WorkbookFactTruth | None = None
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
@dataclass(frozen=True)
|
|
113
|
+
class WorkbookQualitySample:
|
|
114
|
+
sample_id: str
|
|
115
|
+
path: str
|
|
116
|
+
sha256: str
|
|
117
|
+
expectation: WorkbookExpectation
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@dataclass(frozen=True)
|
|
121
|
+
class WorkbookQualityGate:
|
|
122
|
+
minimum: dict[str, float]
|
|
123
|
+
maximum: dict[str, float]
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
@dataclass(frozen=True)
|
|
127
|
+
class WorkbookQualityManifest:
|
|
128
|
+
schema_version: int
|
|
129
|
+
dataset_id: str
|
|
130
|
+
dataset_version: str
|
|
131
|
+
split: str
|
|
132
|
+
source_root: Path
|
|
133
|
+
quality_gate: WorkbookQualityGate
|
|
134
|
+
samples: tuple[WorkbookQualitySample, ...]
|
|
135
|
+
dataset_digest: str
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def load_workbook_quality_manifest(path: str | Path) -> WorkbookQualityManifest:
|
|
139
|
+
manifest_path = Path(path).resolve()
|
|
140
|
+
try:
|
|
141
|
+
data = json.loads(
|
|
142
|
+
manifest_path.read_text(encoding="utf-8"),
|
|
143
|
+
object_pairs_hook=_strict_object_pairs,
|
|
144
|
+
)
|
|
145
|
+
except json.JSONDecodeError as exc:
|
|
146
|
+
raise WorkbookQualityManifestError("Manifest is not valid JSON") from exc
|
|
147
|
+
_require_keys(
|
|
148
|
+
data,
|
|
149
|
+
{
|
|
150
|
+
"schema_version",
|
|
151
|
+
"dataset_id",
|
|
152
|
+
"dataset_version",
|
|
153
|
+
"split",
|
|
154
|
+
"source_root",
|
|
155
|
+
"quality_gate",
|
|
156
|
+
"samples",
|
|
157
|
+
},
|
|
158
|
+
"Manifest",
|
|
159
|
+
)
|
|
160
|
+
if type(data["schema_version"]) is not int or data["schema_version"] != 1:
|
|
161
|
+
raise WorkbookQualityManifestError("schema_version must be integer 1")
|
|
162
|
+
_safe_token(data["dataset_id"], "dataset_id")
|
|
163
|
+
_safe_token(data["dataset_version"], "dataset_version")
|
|
164
|
+
if not isinstance(data["split"], str) or data["split"] not in {
|
|
165
|
+
"tuning",
|
|
166
|
+
"holdout",
|
|
167
|
+
}:
|
|
168
|
+
raise WorkbookQualityManifestError("split must be tuning or holdout")
|
|
169
|
+
quality_gate = data["quality_gate"]
|
|
170
|
+
_require_keys(quality_gate, {"minimum", "maximum"}, "Quality gate")
|
|
171
|
+
if not isinstance(quality_gate["minimum"], dict) or not isinstance(
|
|
172
|
+
quality_gate["maximum"], dict
|
|
173
|
+
):
|
|
174
|
+
raise WorkbookQualityManifestError("Quality gate thresholds must be objects")
|
|
175
|
+
gate_names = set(quality_gate["minimum"]) | set(quality_gate["maximum"])
|
|
176
|
+
unknown_metrics = sorted(gate_names - _QUALITY_METRICS)
|
|
177
|
+
if unknown_metrics:
|
|
178
|
+
raise WorkbookQualityManifestError(f"Unknown quality metric: {', '.join(unknown_metrics)}")
|
|
179
|
+
for threshold in (*quality_gate["minimum"].values(), *quality_gate["maximum"].values()):
|
|
180
|
+
if type(threshold) not in {int, float} or not 0.0 <= threshold <= 1.0:
|
|
181
|
+
raise WorkbookQualityManifestError(
|
|
182
|
+
"Quality gate threshold must be a number between 0 and 1"
|
|
183
|
+
)
|
|
184
|
+
source_root_value = data["source_root"]
|
|
185
|
+
if (
|
|
186
|
+
not isinstance(source_root_value, str)
|
|
187
|
+
or not source_root_value
|
|
188
|
+
or Path(source_root_value).is_absolute()
|
|
189
|
+
or ".." in Path(source_root_value).parts
|
|
190
|
+
):
|
|
191
|
+
raise WorkbookQualityManifestError("source_root must be a safe relative path")
|
|
192
|
+
source_root = (manifest_path.parent / source_root_value).resolve()
|
|
193
|
+
if not source_root.is_dir():
|
|
194
|
+
raise WorkbookQualityManifestError("source_root directory does not exist")
|
|
195
|
+
if not isinstance(data["samples"], list):
|
|
196
|
+
raise WorkbookQualityManifestError("samples must be a list")
|
|
197
|
+
samples = tuple(_load_sample(source_root, item) for item in data["samples"])
|
|
198
|
+
sample_ids = [sample.sample_id for sample in samples]
|
|
199
|
+
if len(sample_ids) != len(set(sample_ids)):
|
|
200
|
+
raise WorkbookQualityManifestError("Duplicate sample_id")
|
|
201
|
+
if not samples:
|
|
202
|
+
raise WorkbookQualityManifestError("Manifest must contain at least one sample")
|
|
203
|
+
if not gate_names:
|
|
204
|
+
raise WorkbookQualityManifestError("Quality gate must contain at least one threshold")
|
|
205
|
+
return WorkbookQualityManifest(
|
|
206
|
+
schema_version=data["schema_version"],
|
|
207
|
+
dataset_id=data["dataset_id"],
|
|
208
|
+
dataset_version=data["dataset_version"],
|
|
209
|
+
split=data["split"],
|
|
210
|
+
source_root=source_root,
|
|
211
|
+
quality_gate=WorkbookQualityGate(
|
|
212
|
+
minimum=dict(data["quality_gate"]["minimum"]),
|
|
213
|
+
maximum=dict(data["quality_gate"]["maximum"]),
|
|
214
|
+
),
|
|
215
|
+
samples=samples,
|
|
216
|
+
dataset_digest=_canonical_digest(data),
|
|
217
|
+
)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _require_keys(data: object, expected: set[str], label: str) -> None:
|
|
221
|
+
if not isinstance(data, dict) or set(data) != expected:
|
|
222
|
+
raise WorkbookQualityManifestError(f"{label} keys do not match schema")
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def _safe_token(value: object, field_name: str) -> str:
|
|
226
|
+
if not isinstance(value, str) or _SAFE_TOKEN_PATTERN.fullmatch(value) is None:
|
|
227
|
+
raise WorkbookQualityManifestError(f"{field_name} must be a safe identifier")
|
|
228
|
+
return value
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _strict_object_pairs(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
232
|
+
result: dict[str, Any] = {}
|
|
233
|
+
for key, value in pairs:
|
|
234
|
+
if key in result:
|
|
235
|
+
raise WorkbookQualityManifestError(f"Duplicate JSON key: {key}")
|
|
236
|
+
result[key] = value
|
|
237
|
+
return result
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _canonical_digest(data: dict[str, Any]) -> str:
|
|
241
|
+
encoded = json.dumps(
|
|
242
|
+
data,
|
|
243
|
+
ensure_ascii=False,
|
|
244
|
+
sort_keys=True,
|
|
245
|
+
separators=(",", ":"),
|
|
246
|
+
).encode("utf-8")
|
|
247
|
+
return f"sha256:{hashlib.sha256(encoded).hexdigest()}"
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def _load_sample(source_root: Path, data: dict[str, Any]) -> WorkbookQualitySample:
|
|
251
|
+
_require_keys(data, {"sample_id", "path", "sha256", "expectation"}, "Sample")
|
|
252
|
+
sample_id = _safe_token(data["sample_id"], "sample_id")
|
|
253
|
+
sample_value = data["path"]
|
|
254
|
+
if not isinstance(sample_value, str) or not sample_value or Path(sample_value).is_absolute():
|
|
255
|
+
raise WorkbookQualityManifestError("Sample path must be a non-empty relative path")
|
|
256
|
+
if not isinstance(data["sha256"], str) or _SHA256_PATTERN.fullmatch(data["sha256"]) is None:
|
|
257
|
+
raise WorkbookQualityManifestError("Sample sha256 has invalid format")
|
|
258
|
+
sample_path = (source_root / sample_value).resolve()
|
|
259
|
+
if not sample_path.is_relative_to(source_root):
|
|
260
|
+
raise WorkbookQualityManifestError("Sample path resolves outside source_root")
|
|
261
|
+
if not sample_path.is_file():
|
|
262
|
+
raise WorkbookQualityManifestError("Sample file does not exist")
|
|
263
|
+
actual_hash = f"sha256:{hashlib.sha256(sample_path.read_bytes()).hexdigest()}"
|
|
264
|
+
if actual_hash != data["sha256"]:
|
|
265
|
+
raise WorkbookQualityManifestError("Sample file SHA-256 does not match manifest")
|
|
266
|
+
return WorkbookQualitySample(
|
|
267
|
+
sample_id=sample_id,
|
|
268
|
+
path=sample_value,
|
|
269
|
+
sha256=data["sha256"],
|
|
270
|
+
expectation=_load_expectation(data["expectation"]),
|
|
271
|
+
)
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
def _load_expectation(data: dict[str, Any]) -> WorkbookExpectation:
|
|
275
|
+
_require_keys(
|
|
276
|
+
data,
|
|
277
|
+
{"sheets", "continuations", "required_source_refs", "objects"}
|
|
278
|
+
| ({"facts"} if isinstance(data, dict) and "facts" in data else set()),
|
|
279
|
+
"Expectation",
|
|
280
|
+
)
|
|
281
|
+
sheets_data = _require_list(data["sheets"], "Expectation sheets")
|
|
282
|
+
continuations_data = _require_list(data["continuations"], "Expectation continuations")
|
|
283
|
+
required_refs_data = _require_list(
|
|
284
|
+
data["required_source_refs"], "Expectation required_source_refs"
|
|
285
|
+
)
|
|
286
|
+
objects_data = _require_list(data["objects"], "Expectation objects")
|
|
287
|
+
continuations: list[tuple[str, ...]] = []
|
|
288
|
+
for group in continuations_data:
|
|
289
|
+
refs = _require_list(group, "Continuation group")
|
|
290
|
+
if len(refs) < 2:
|
|
291
|
+
raise WorkbookQualityManifestError(
|
|
292
|
+
"Continuation group must contain at least two source refs"
|
|
293
|
+
)
|
|
294
|
+
if any(
|
|
295
|
+
not isinstance(ref, str) or _SOURCE_REF_PATTERN.fullmatch(ref) is None for ref in refs
|
|
296
|
+
):
|
|
297
|
+
raise WorkbookQualityManifestError("Invalid source ref")
|
|
298
|
+
if len(refs) != len(set(refs)):
|
|
299
|
+
raise WorkbookQualityManifestError("Continuation group contains duplicate source refs")
|
|
300
|
+
continuations.append(tuple(refs))
|
|
301
|
+
if len(continuations) != len(set(continuations)):
|
|
302
|
+
raise WorkbookQualityManifestError("Duplicate continuation group")
|
|
303
|
+
continuation_members: set[str] = set()
|
|
304
|
+
for group in continuations:
|
|
305
|
+
overlap = continuation_members.intersection(group)
|
|
306
|
+
if overlap:
|
|
307
|
+
raise WorkbookQualityManifestError(
|
|
308
|
+
"Continuation groups contain overlapping source refs"
|
|
309
|
+
)
|
|
310
|
+
continuation_members.update(group)
|
|
311
|
+
required_source_refs = tuple(required_refs_data)
|
|
312
|
+
continuation_refs = tuple(ref for group in continuations for ref in group)
|
|
313
|
+
if any(
|
|
314
|
+
not isinstance(ref, str) or _SOURCE_REF_PATTERN.fullmatch(ref) is None
|
|
315
|
+
for ref in (*required_source_refs, *continuation_refs)
|
|
316
|
+
):
|
|
317
|
+
raise WorkbookQualityManifestError("Invalid source ref")
|
|
318
|
+
sheets = tuple(_load_sheet(item) for item in sheets_data)
|
|
319
|
+
objects = tuple(_load_object(item) for item in objects_data)
|
|
320
|
+
sheet_names = [sheet.name for sheet in sheets]
|
|
321
|
+
if len(sheet_names) != len(set(sheet_names)):
|
|
322
|
+
raise WorkbookQualityManifestError("Duplicate sheet name")
|
|
323
|
+
block_truth_keys = [
|
|
324
|
+
(sheet.name, block.source_range, block.kind) for sheet in sheets for block in sheet.blocks
|
|
325
|
+
]
|
|
326
|
+
if len(block_truth_keys) != len(set(block_truth_keys)):
|
|
327
|
+
raise WorkbookQualityManifestError("Duplicate block truth")
|
|
328
|
+
expected_block_refs = {
|
|
329
|
+
f"{sheet.name}!{block.source_range}" for sheet in sheets for block in sheet.blocks
|
|
330
|
+
}
|
|
331
|
+
if any(ref not in expected_block_refs for group in continuations for ref in group):
|
|
332
|
+
raise WorkbookQualityManifestError(
|
|
333
|
+
"Continuation source ref must identify an expected block"
|
|
334
|
+
)
|
|
335
|
+
if any(item.sheet_name not in sheet_names for item in objects):
|
|
336
|
+
raise WorkbookQualityManifestError("Object sheet_name must identify an expected sheet")
|
|
337
|
+
if len(objects) != len(set(objects)):
|
|
338
|
+
raise WorkbookQualityManifestError("Duplicate object truth")
|
|
339
|
+
if len(required_source_refs) != len(set(required_source_refs)):
|
|
340
|
+
raise WorkbookQualityManifestError("Duplicate required_source_ref")
|
|
341
|
+
if any(_source_ref_sheet(ref) not in sheet_names for ref in required_source_refs):
|
|
342
|
+
raise WorkbookQualityManifestError("Required source ref must identify an expected sheet")
|
|
343
|
+
try:
|
|
344
|
+
facts = load_fact_truth(data["facts"], set(sheet_names)) if "facts" in data else None
|
|
345
|
+
except (ValueError, TypeError) as exc:
|
|
346
|
+
raise WorkbookQualityManifestError(str(exc)) from exc
|
|
347
|
+
return WorkbookExpectation(
|
|
348
|
+
sheets=sheets,
|
|
349
|
+
continuations=tuple(continuations),
|
|
350
|
+
required_source_refs=required_source_refs,
|
|
351
|
+
objects=objects,
|
|
352
|
+
facts=facts,
|
|
353
|
+
)
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _load_sheet(data: dict[str, Any]) -> SheetTruth:
|
|
357
|
+
_require_keys(data, {"name", "blocks"}, "Sheet")
|
|
358
|
+
name = _require_non_empty_string(data["name"], "Sheet name")
|
|
359
|
+
blocks = _require_list(data["blocks"], "Sheet blocks")
|
|
360
|
+
return SheetTruth(name=name, blocks=tuple(_load_block(item) for item in blocks))
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def _load_block(data: dict[str, Any]) -> BlockTruth:
|
|
364
|
+
_require_keys(
|
|
365
|
+
data,
|
|
366
|
+
{"source_range", "kind", "headers", "rows", "form_fields", "matrix_axes"},
|
|
367
|
+
"Block",
|
|
368
|
+
)
|
|
369
|
+
source_range = _require_cell_range(data["source_range"], "Block source_range")
|
|
370
|
+
if not isinstance(data["kind"], str) or data["kind"] not in _BLOCK_KINDS:
|
|
371
|
+
raise WorkbookQualityManifestError("Unknown block kind")
|
|
372
|
+
headers = _require_list(data["headers"], "Block headers")
|
|
373
|
+
rows = _require_list(data["rows"], "Block rows")
|
|
374
|
+
form_fields = _require_list(data["form_fields"], "Block form_fields")
|
|
375
|
+
matrix_axes = data["matrix_axes"]
|
|
376
|
+
_require_keys(matrix_axes, {"rows", "columns"}, "Matrix axes")
|
|
377
|
+
matrix_rows = _require_string_list(matrix_axes["rows"], "Matrix axes rows")
|
|
378
|
+
matrix_columns = _require_string_list(matrix_axes["columns"], "Matrix axes columns")
|
|
379
|
+
header_truth = tuple(_load_header(item) for item in headers)
|
|
380
|
+
if len(header_truth) != len(set(header_truth)):
|
|
381
|
+
raise WorkbookQualityManifestError("Duplicate header truth")
|
|
382
|
+
row_truth = tuple(_load_row(item) for item in rows)
|
|
383
|
+
if len(row_truth) != len(set(row_truth)):
|
|
384
|
+
raise WorkbookQualityManifestError("Duplicate row truth")
|
|
385
|
+
return BlockTruth(
|
|
386
|
+
source_range=source_range,
|
|
387
|
+
kind=data["kind"],
|
|
388
|
+
headers=header_truth,
|
|
389
|
+
rows=row_truth,
|
|
390
|
+
form_fields=tuple(_load_form_field(item) for item in form_fields),
|
|
391
|
+
matrix_axes=MatrixAxesTruth(
|
|
392
|
+
rows=matrix_rows,
|
|
393
|
+
columns=matrix_columns,
|
|
394
|
+
),
|
|
395
|
+
)
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def _load_header(data: dict[str, Any]) -> HeaderTruth:
|
|
399
|
+
_require_keys(data, {"coordinate", "path"}, "Header")
|
|
400
|
+
coordinate = _require_non_empty_string(data["coordinate"], "Header coordinate")
|
|
401
|
+
path = _require_string_list(data["path"], "Header path")
|
|
402
|
+
return HeaderTruth(coordinate=coordinate, path=path)
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def _load_row(data: dict[str, Any]) -> RowTruth:
|
|
406
|
+
_require_keys(data, {"source_range", "role"}, "Row")
|
|
407
|
+
role = _require_non_empty_string(data["role"], "Row role")
|
|
408
|
+
if role not in _ROW_ROLES:
|
|
409
|
+
raise WorkbookQualityManifestError("Unknown row role")
|
|
410
|
+
return RowTruth(
|
|
411
|
+
source_range=_require_cell_range(data["source_range"], "Row source_range"),
|
|
412
|
+
role=role,
|
|
413
|
+
)
|
|
414
|
+
|
|
415
|
+
|
|
416
|
+
def _load_form_field(data: dict[str, Any]) -> tuple[str, Any]:
|
|
417
|
+
_require_keys(data, {"label", "value"}, "Form field")
|
|
418
|
+
return _require_non_empty_string(data["label"], "Form field label"), data["value"]
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def _load_object(data: dict[str, Any]) -> ObjectTruth:
|
|
422
|
+
_require_keys(data, {"sheet_name", "kind", "anchor"}, "Object")
|
|
423
|
+
kind = _require_non_empty_string(data["kind"], "Object kind")
|
|
424
|
+
if kind not in _OBJECT_KINDS:
|
|
425
|
+
raise WorkbookQualityManifestError("Unknown object kind")
|
|
426
|
+
return ObjectTruth(
|
|
427
|
+
sheet_name=_require_non_empty_string(data["sheet_name"], "Object sheet_name"),
|
|
428
|
+
kind=kind,
|
|
429
|
+
anchor=_require_cell_range(data["anchor"], "Object anchor"),
|
|
430
|
+
)
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
def _require_list(value: object, label: str) -> list[Any]:
|
|
434
|
+
if not isinstance(value, list):
|
|
435
|
+
raise WorkbookQualityManifestError(f"{label} must be a list")
|
|
436
|
+
return value
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def _require_string_list(value: object, label: str) -> tuple[str, ...]:
|
|
440
|
+
items = _require_list(value, label)
|
|
441
|
+
if any(not isinstance(item, str) for item in items):
|
|
442
|
+
raise WorkbookQualityManifestError(f"{label} must contain only strings")
|
|
443
|
+
return tuple(items)
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
def _require_non_empty_string(value: object, label: str) -> str:
|
|
447
|
+
if not isinstance(value, str) or not value:
|
|
448
|
+
raise WorkbookQualityManifestError(f"{label} must be a non-empty string")
|
|
449
|
+
return value
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def _require_cell_range(value: object, label: str) -> str:
|
|
453
|
+
if not isinstance(value, str) or _CELL_RANGE_PATTERN.fullmatch(value) is None:
|
|
454
|
+
raise WorkbookQualityManifestError(f"{label} must be an A1 cell range")
|
|
455
|
+
return value
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
def _source_ref_sheet(value: str) -> str:
|
|
459
|
+
sheet_name = value.rsplit("!", 1)[0]
|
|
460
|
+
if len(sheet_name) >= 2 and sheet_name.startswith("'") and sheet_name.endswith("'"):
|
|
461
|
+
return sheet_name[1:-1].replace("''", "'")
|
|
462
|
+
return sheet_name
|
|
@@ -0,0 +1,73 @@
|
|
|
1
|
+
"""Typed source facts and evidence-bearing dependency results."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from typing import TYPE_CHECKING
|
|
7
|
+
|
|
8
|
+
if TYPE_CHECKING:
|
|
9
|
+
from langparse.workbooks.types import SourceRef
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass
|
|
13
|
+
class ReferenceDiagnostic:
|
|
14
|
+
code: str
|
|
15
|
+
message: str
|
|
16
|
+
source_refs: list[SourceRef] = field(default_factory=list)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass
|
|
20
|
+
class DefinedNameFact:
|
|
21
|
+
name: str
|
|
22
|
+
definition: str
|
|
23
|
+
scope_sheet: str | None = None
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class ExcelTableFact:
|
|
28
|
+
name: str
|
|
29
|
+
source_ref: SourceRef
|
|
30
|
+
columns: list[str] = field(default_factory=list)
|
|
31
|
+
header_rows: int = 1
|
|
32
|
+
totals_rows: int = 0
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class FormulaReference:
|
|
37
|
+
reference: str
|
|
38
|
+
status: str
|
|
39
|
+
targets: list[SourceRef] = field(default_factory=list)
|
|
40
|
+
external_workbook: str | None = None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
@dataclass
|
|
44
|
+
class FormulaFact:
|
|
45
|
+
source_ref: SourceRef
|
|
46
|
+
formula: str
|
|
47
|
+
references: list[FormulaReference] = field(default_factory=list)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass
|
|
51
|
+
class ExternalReferenceFact:
|
|
52
|
+
workbook: str
|
|
53
|
+
reference: str
|
|
54
|
+
source_ref: SourceRef
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass
|
|
58
|
+
class WorkbookReferenceFacts:
|
|
59
|
+
defined_names: list[DefinedNameFact] = field(default_factory=list)
|
|
60
|
+
tables: list[ExcelTableFact] = field(default_factory=list)
|
|
61
|
+
formulas: list[FormulaFact] = field(default_factory=list)
|
|
62
|
+
external_references: list[ExternalReferenceFact] = field(default_factory=list)
|
|
63
|
+
diagnostics: list[ReferenceDiagnostic] = field(default_factory=list)
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@dataclass
|
|
67
|
+
class DependencyEdge:
|
|
68
|
+
dependent: SourceRef
|
|
69
|
+
target: SourceRef | None
|
|
70
|
+
reference: str
|
|
71
|
+
status: str
|
|
72
|
+
source_refs: list[SourceRef] = field(default_factory=list)
|
|
73
|
+
external_workbook: str | None = None
|