langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,628 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import math
|
|
6
|
+
from collections.abc import Mapping
|
|
7
|
+
from dataclasses import fields
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
from openpyxl.utils import column_index_from_string, get_column_letter, range_boundaries
|
|
11
|
+
from openpyxl.utils.cell import coordinate_to_tuple
|
|
12
|
+
|
|
13
|
+
from langparse.workbooks.classification import RegionAssessment
|
|
14
|
+
from langparse.workbooks.types import CandidateRegion, CellSnapshot, SheetSnapshot, stable_id
|
|
15
|
+
|
|
16
|
+
from .ports import InvalidRegionAmbiguityCaseError, WorkbookModelResponseError
|
|
17
|
+
from .types import (
|
|
18
|
+
REGION_PRIVACY_VERSION,
|
|
19
|
+
REGION_PROMPT_VERSION,
|
|
20
|
+
REGION_RULE_VERSION,
|
|
21
|
+
REGION_SCHEMA_VERSION,
|
|
22
|
+
REGION_VALIDATOR_VERSION,
|
|
23
|
+
ModelIdentity,
|
|
24
|
+
ProviderReply,
|
|
25
|
+
RegionAmbiguityCase,
|
|
26
|
+
RegionCellCue,
|
|
27
|
+
RegionChoice,
|
|
28
|
+
RegionFeatureScalar,
|
|
29
|
+
RegionModelDecision,
|
|
30
|
+
WorkbookModelRequest,
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def build_region_case(
|
|
35
|
+
sheet: SheetSnapshot,
|
|
36
|
+
candidate: CandidateRegion,
|
|
37
|
+
assessment: RegionAssessment,
|
|
38
|
+
) -> RegionAmbiguityCase:
|
|
39
|
+
"""Project one ambiguous region into the privacy-preserving model contract."""
|
|
40
|
+
|
|
41
|
+
if not assessment.ambiguous:
|
|
42
|
+
raise InvalidRegionAmbiguityCaseError("assessment must be ambiguous")
|
|
43
|
+
if candidate.source_ref.sheet_name != sheet.name:
|
|
44
|
+
raise InvalidRegionAmbiguityCaseError("candidate source sheet does not match sheet")
|
|
45
|
+
if sheet.visibility != "visible":
|
|
46
|
+
raise InvalidRegionAmbiguityCaseError("hidden sheet content cannot be sent")
|
|
47
|
+
|
|
48
|
+
_validate_candidate_coordinates(sheet, candidate)
|
|
49
|
+
_reject_hidden_candidate_content(sheet, candidate)
|
|
50
|
+
if _candidate_envelope_has_formula(sheet, candidate):
|
|
51
|
+
raise InvalidRegionAmbiguityCaseError("formula candidate content cannot be sent")
|
|
52
|
+
_validate_choices(assessment.choices, assessment.deterministic.kind)
|
|
53
|
+
|
|
54
|
+
cells = tuple(
|
|
55
|
+
_cell_cue(cell)
|
|
56
|
+
for coordinate in _ordered_candidate_coordinates(candidate)
|
|
57
|
+
if (cell := sheet.cells.get(coordinate)) is not None
|
|
58
|
+
and _is_safe_visible_occupied_cell(sheet, cell)
|
|
59
|
+
)
|
|
60
|
+
feature_summary = _feature_summary(assessment)
|
|
61
|
+
choices = tuple(assessment.choices)
|
|
62
|
+
fallback_choice_id = next(
|
|
63
|
+
choice.choice_id for choice in choices if choice.kind == assessment.deterministic.kind
|
|
64
|
+
)
|
|
65
|
+
fact_digest = _digest(
|
|
66
|
+
{
|
|
67
|
+
"sheet_name": sheet.name,
|
|
68
|
+
"sheet_visibility": sheet.visibility,
|
|
69
|
+
"source_range": candidate.source_ref.range,
|
|
70
|
+
"cells": [_cue_payload(cell) for cell in cells],
|
|
71
|
+
"feature_summary": dict(feature_summary),
|
|
72
|
+
"choices": [_choice_payload(choice) for choice in choices],
|
|
73
|
+
"region_privacy_version": REGION_PRIVACY_VERSION,
|
|
74
|
+
"region_rule_version": REGION_RULE_VERSION,
|
|
75
|
+
"region_validator_version": REGION_VALIDATOR_VERSION,
|
|
76
|
+
}
|
|
77
|
+
)
|
|
78
|
+
return RegionAmbiguityCase(
|
|
79
|
+
case_id=stable_id("region_case", REGION_RULE_VERSION, candidate.source_ref.key),
|
|
80
|
+
sheet_name=sheet.name,
|
|
81
|
+
sheet_visibility=sheet.visibility,
|
|
82
|
+
source_range=candidate.source_ref.range,
|
|
83
|
+
fact_digest=fact_digest,
|
|
84
|
+
cells=cells,
|
|
85
|
+
feature_summary=feature_summary,
|
|
86
|
+
choices=choices,
|
|
87
|
+
fallback_choice_id=fallback_choice_id,
|
|
88
|
+
ambiguity_codes=tuple(assessment.ambiguity_codes),
|
|
89
|
+
)
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def build_model_request(case: RegionAmbiguityCase, identity: ModelIdentity) -> WorkbookModelRequest:
|
|
93
|
+
"""Build the one-case Phase 4A request with a canonical checksum."""
|
|
94
|
+
|
|
95
|
+
_validate_case(case)
|
|
96
|
+
_validate_model_identity(identity)
|
|
97
|
+
request_case = {
|
|
98
|
+
"case_id": case.case_id,
|
|
99
|
+
"sheet_name": case.sheet_name,
|
|
100
|
+
"sheet_visibility": case.sheet_visibility,
|
|
101
|
+
"source_range": case.source_range,
|
|
102
|
+
"fact_digest": case.fact_digest,
|
|
103
|
+
"feature_summary": dict(case.feature_summary),
|
|
104
|
+
"cells": [_cue_payload(cell) for cell in case.cells],
|
|
105
|
+
"choices": [_choice_payload(choice) for choice in case.choices],
|
|
106
|
+
"fallback_choice_id": case.fallback_choice_id,
|
|
107
|
+
"ambiguity_codes": list(case.ambiguity_codes),
|
|
108
|
+
}
|
|
109
|
+
envelope: dict[str, object] = {
|
|
110
|
+
"schema_version": REGION_SCHEMA_VERSION,
|
|
111
|
+
"prompt_version": REGION_PROMPT_VERSION,
|
|
112
|
+
"privacy_version": REGION_PRIVACY_VERSION,
|
|
113
|
+
"model_identity": {
|
|
114
|
+
"provider": identity.provider,
|
|
115
|
+
"model": identity.model,
|
|
116
|
+
"revision": identity.revision,
|
|
117
|
+
},
|
|
118
|
+
"cases": [request_case],
|
|
119
|
+
}
|
|
120
|
+
request_checksum = _digest(envelope)
|
|
121
|
+
body = _canonical_json({**envelope, "request_checksum": request_checksum})
|
|
122
|
+
return WorkbookModelRequest(
|
|
123
|
+
schema_version=REGION_SCHEMA_VERSION,
|
|
124
|
+
prompt_version=REGION_PROMPT_VERSION,
|
|
125
|
+
privacy_version=REGION_PRIVACY_VERSION,
|
|
126
|
+
request_checksum=request_checksum,
|
|
127
|
+
body=body,
|
|
128
|
+
case_ids=(case.case_id,),
|
|
129
|
+
choice_ids_by_case=((case.case_id, tuple(choice.choice_id for choice in case.choices)),),
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def decode_model_reply(
|
|
134
|
+
reply: ProviderReply,
|
|
135
|
+
request: WorkbookModelRequest,
|
|
136
|
+
*,
|
|
137
|
+
max_response_bytes: int,
|
|
138
|
+
) -> RegionModelDecision:
|
|
139
|
+
"""Decode a provider reply only when it exactly matches the local contract."""
|
|
140
|
+
|
|
141
|
+
reply = _copy_provider_reply(reply)
|
|
142
|
+
if type(max_response_bytes) is not int or max_response_bytes <= 0:
|
|
143
|
+
raise WorkbookModelResponseError("invalid local response byte limit")
|
|
144
|
+
if len(reply.body) > max_response_bytes:
|
|
145
|
+
raise WorkbookModelResponseError(f"response exceeds {max_response_bytes} bytes")
|
|
146
|
+
try:
|
|
147
|
+
payload = json.loads(
|
|
148
|
+
reply.body.decode("utf-8"),
|
|
149
|
+
object_pairs_hook=_reject_duplicate_json_members,
|
|
150
|
+
)
|
|
151
|
+
except (UnicodeDecodeError, json.JSONDecodeError) as error:
|
|
152
|
+
raise WorkbookModelResponseError("response is not valid JSON") from error
|
|
153
|
+
if not isinstance(payload, dict):
|
|
154
|
+
raise WorkbookModelResponseError("response must be a JSON object")
|
|
155
|
+
_exact_keys(
|
|
156
|
+
payload,
|
|
157
|
+
{"schema_version", "request_checksum", "decisions"},
|
|
158
|
+
"response",
|
|
159
|
+
)
|
|
160
|
+
if (
|
|
161
|
+
type(payload["schema_version"]) is not int
|
|
162
|
+
or payload["schema_version"] != REGION_SCHEMA_VERSION
|
|
163
|
+
):
|
|
164
|
+
raise WorkbookModelResponseError("unsupported response schema_version")
|
|
165
|
+
if not isinstance(payload["request_checksum"], str):
|
|
166
|
+
raise WorkbookModelResponseError("request_checksum must be a string")
|
|
167
|
+
if payload["request_checksum"] != request.request_checksum:
|
|
168
|
+
raise WorkbookModelResponseError("request checksum mismatch")
|
|
169
|
+
|
|
170
|
+
decisions = payload["decisions"]
|
|
171
|
+
if not isinstance(decisions, list):
|
|
172
|
+
raise WorkbookModelResponseError("decisions must be a list")
|
|
173
|
+
membership = _request_membership(request)
|
|
174
|
+
expected_case_ids = set(request.case_ids)
|
|
175
|
+
decoded: dict[str, RegionModelDecision] = {}
|
|
176
|
+
for raw_decision in decisions:
|
|
177
|
+
decision = _decode_decision(raw_decision, membership)
|
|
178
|
+
if decision.case_id not in expected_case_ids:
|
|
179
|
+
raise WorkbookModelResponseError("unknown case_id")
|
|
180
|
+
if decision.case_id in decoded:
|
|
181
|
+
raise WorkbookModelResponseError("duplicate decision")
|
|
182
|
+
decoded[decision.case_id] = decision
|
|
183
|
+
|
|
184
|
+
missing_case_ids = expected_case_ids - decoded.keys()
|
|
185
|
+
if missing_case_ids:
|
|
186
|
+
raise WorkbookModelResponseError("missing decision")
|
|
187
|
+
if len(decoded) != len(request.case_ids):
|
|
188
|
+
raise WorkbookModelResponseError("response must contain one decision per request case")
|
|
189
|
+
if len(request.case_ids) != 1:
|
|
190
|
+
raise WorkbookModelResponseError("Phase 4A requests require exactly one case")
|
|
191
|
+
return decoded[request.case_ids[0]]
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def response_checksum(body: bytes) -> str:
|
|
195
|
+
return _digest_bytes(body)
|
|
196
|
+
|
|
197
|
+
|
|
198
|
+
def _copy_provider_reply(reply: object) -> ProviderReply:
|
|
199
|
+
if not isinstance(reply, ProviderReply):
|
|
200
|
+
raise WorkbookModelResponseError("adapter returned an invalid reply")
|
|
201
|
+
try:
|
|
202
|
+
return ProviderReply(
|
|
203
|
+
body=reply.body,
|
|
204
|
+
provider_request_id=reply.provider_request_id,
|
|
205
|
+
usage=reply.usage,
|
|
206
|
+
)
|
|
207
|
+
except Exception as error:
|
|
208
|
+
raise WorkbookModelResponseError("adapter returned an invalid reply") from error
|
|
209
|
+
|
|
210
|
+
|
|
211
|
+
def _validate_model_identity(identity: object) -> None:
|
|
212
|
+
if (
|
|
213
|
+
not isinstance(identity, ModelIdentity)
|
|
214
|
+
or type(identity.provider) is not str
|
|
215
|
+
or not identity.provider
|
|
216
|
+
or type(identity.model) is not str
|
|
217
|
+
or not identity.model
|
|
218
|
+
or (identity.revision is not None and type(identity.revision) is not str)
|
|
219
|
+
):
|
|
220
|
+
raise WorkbookModelResponseError("adapter returned an invalid model identity")
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def _canonical_json(payload: dict[str, object]) -> bytes:
|
|
224
|
+
return json.dumps(
|
|
225
|
+
payload,
|
|
226
|
+
ensure_ascii=False,
|
|
227
|
+
sort_keys=True,
|
|
228
|
+
separators=(",", ":"),
|
|
229
|
+
).encode("utf-8")
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def _reject_duplicate_json_members(pairs: list[tuple[str, Any]]) -> dict[str, Any]:
|
|
233
|
+
payload: dict[str, Any] = {}
|
|
234
|
+
for key, value in pairs:
|
|
235
|
+
if key in payload:
|
|
236
|
+
raise WorkbookModelResponseError("response contains duplicate JSON member")
|
|
237
|
+
payload[key] = value
|
|
238
|
+
return payload
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _digest(payload: dict[str, object]) -> str:
|
|
242
|
+
return _digest_bytes(_canonical_json(payload))
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def _digest_bytes(body: bytes) -> str:
|
|
246
|
+
return f"sha256:{hashlib.sha256(body).hexdigest()}"
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
def _validate_candidate_coordinates(sheet: SheetSnapshot, candidate: CandidateRegion) -> None:
|
|
250
|
+
min_column, min_row, max_column, max_row = range_boundaries(candidate.source_ref.range)
|
|
251
|
+
for coordinate in candidate.cell_refs:
|
|
252
|
+
try:
|
|
253
|
+
row, column = coordinate_to_tuple(coordinate)
|
|
254
|
+
except ValueError as error:
|
|
255
|
+
raise InvalidRegionAmbiguityCaseError(
|
|
256
|
+
f"invalid candidate cell: {coordinate}"
|
|
257
|
+
) from error
|
|
258
|
+
if not min_row <= row <= max_row or not min_column <= column <= max_column:
|
|
259
|
+
raise InvalidRegionAmbiguityCaseError("candidate contains cell outside source range")
|
|
260
|
+
cell = sheet.cells.get(coordinate)
|
|
261
|
+
if cell is not None and cell.coordinate != coordinate:
|
|
262
|
+
raise InvalidRegionAmbiguityCaseError("candidate cell mapping coordinate mismatch")
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _reject_hidden_candidate_content(sheet: SheetSnapshot, candidate: CandidateRegion) -> None:
|
|
266
|
+
min_column, min_row, max_column, max_row = range_boundaries(candidate.source_ref.range)
|
|
267
|
+
if any(min_row <= row <= max_row for row in sheet.hidden_rows):
|
|
268
|
+
raise InvalidRegionAmbiguityCaseError("hidden candidate content cannot be sent")
|
|
269
|
+
if any(
|
|
270
|
+
min_column <= column_index_from_string(column) <= max_column
|
|
271
|
+
for column in sheet.hidden_columns
|
|
272
|
+
):
|
|
273
|
+
raise InvalidRegionAmbiguityCaseError("hidden candidate content cannot be sent")
|
|
274
|
+
for coordinate, cell in sheet.cells.items():
|
|
275
|
+
try:
|
|
276
|
+
row, column = coordinate_to_tuple(coordinate)
|
|
277
|
+
except ValueError as error:
|
|
278
|
+
raise InvalidRegionAmbiguityCaseError(
|
|
279
|
+
"invalid candidate cell mapping coordinate"
|
|
280
|
+
) from error
|
|
281
|
+
if min_row <= row <= max_row and min_column <= column <= max_column and cell.hidden:
|
|
282
|
+
raise InvalidRegionAmbiguityCaseError("hidden candidate content cannot be sent")
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def _candidate_envelope_has_formula(
|
|
286
|
+
sheet: SheetSnapshot,
|
|
287
|
+
candidate: CandidateRegion,
|
|
288
|
+
) -> bool:
|
|
289
|
+
min_column, min_row, max_column, max_row = range_boundaries(candidate.source_ref.range)
|
|
290
|
+
for coordinate, cell in sheet.cells.items():
|
|
291
|
+
try:
|
|
292
|
+
row, column = coordinate_to_tuple(coordinate)
|
|
293
|
+
except ValueError as error:
|
|
294
|
+
raise InvalidRegionAmbiguityCaseError(
|
|
295
|
+
"invalid candidate cell mapping coordinate"
|
|
296
|
+
) from error
|
|
297
|
+
if not (min_row <= row <= max_row and min_column <= column <= max_column):
|
|
298
|
+
continue
|
|
299
|
+
if cell.formula is not None:
|
|
300
|
+
return True
|
|
301
|
+
if cell.merge_anchor is not None:
|
|
302
|
+
anchor = sheet.cells.get(cell.merge_anchor)
|
|
303
|
+
if anchor is not None and anchor.formula is not None:
|
|
304
|
+
return True
|
|
305
|
+
return False
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def _validate_choices(choices: tuple[RegionChoice, ...], fallback_kind: str) -> None:
|
|
309
|
+
choice_ids = [choice.choice_id for choice in choices]
|
|
310
|
+
if len(set(choice_ids)) != len(choice_ids):
|
|
311
|
+
raise InvalidRegionAmbiguityCaseError("duplicate choice_id")
|
|
312
|
+
choice_kinds = {choice.kind for choice in choices}
|
|
313
|
+
if len(choice_kinds) < 2:
|
|
314
|
+
raise InvalidRegionAmbiguityCaseError(
|
|
315
|
+
"ambiguous assessment requires at least two choice kinds"
|
|
316
|
+
)
|
|
317
|
+
if fallback_kind not in choice_kinds:
|
|
318
|
+
raise InvalidRegionAmbiguityCaseError("fallback choice is not registered")
|
|
319
|
+
|
|
320
|
+
|
|
321
|
+
def _ordered_candidate_coordinates(candidate: CandidateRegion) -> tuple[str, ...]:
|
|
322
|
+
return tuple(sorted(set(candidate.cell_refs), key=coordinate_to_tuple))
|
|
323
|
+
|
|
324
|
+
|
|
325
|
+
def _is_safe_visible_occupied_cell(sheet: SheetSnapshot, cell: CellSnapshot) -> bool:
|
|
326
|
+
row, column = coordinate_to_tuple(cell.coordinate)
|
|
327
|
+
return (
|
|
328
|
+
cell.merge_anchor is None
|
|
329
|
+
and not cell.hidden
|
|
330
|
+
and row not in sheet.hidden_rows
|
|
331
|
+
and get_column_letter(column) not in sheet.hidden_columns
|
|
332
|
+
and any(
|
|
333
|
+
(
|
|
334
|
+
cell.raw_value is not None,
|
|
335
|
+
cell.display_value != "",
|
|
336
|
+
cell.formula is not None,
|
|
337
|
+
cell.comment is not None,
|
|
338
|
+
cell.hyperlink is not None,
|
|
339
|
+
)
|
|
340
|
+
)
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def _cell_cue(cell: CellSnapshot) -> RegionCellCue:
|
|
345
|
+
return RegionCellCue(
|
|
346
|
+
coordinate=cell.coordinate,
|
|
347
|
+
display_text=cell.display_value,
|
|
348
|
+
value_type=cell.data_type,
|
|
349
|
+
style_fingerprint=cell.style_id,
|
|
350
|
+
merge_anchor=cell.merge_anchor,
|
|
351
|
+
rowspan=cell.rowspan,
|
|
352
|
+
colspan=cell.colspan,
|
|
353
|
+
)
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _feature_summary(assessment: RegionAssessment) -> tuple[tuple[str, RegionFeatureScalar], ...]:
|
|
357
|
+
summary: list[tuple[str, RegionFeatureScalar]] = []
|
|
358
|
+
for feature in fields(assessment.deterministic.features):
|
|
359
|
+
value = getattr(assessment.deterministic.features, feature.name)
|
|
360
|
+
if value is None or type(value) in (str, int, float, bool):
|
|
361
|
+
summary.append((feature.name, value))
|
|
362
|
+
return tuple(summary)
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def _cue_payload(cell: RegionCellCue) -> dict[str, object]:
|
|
366
|
+
return {
|
|
367
|
+
"coordinate": cell.coordinate,
|
|
368
|
+
"display_text": cell.display_text,
|
|
369
|
+
"value_type": cell.value_type,
|
|
370
|
+
"style_fingerprint": cell.style_fingerprint,
|
|
371
|
+
"merge_anchor": cell.merge_anchor,
|
|
372
|
+
"rowspan": cell.rowspan,
|
|
373
|
+
"colspan": cell.colspan,
|
|
374
|
+
}
|
|
375
|
+
|
|
376
|
+
|
|
377
|
+
def _choice_payload(choice: RegionChoice) -> dict[str, object]:
|
|
378
|
+
return {
|
|
379
|
+
"choice_id": choice.choice_id,
|
|
380
|
+
"kind": choice.kind,
|
|
381
|
+
"local_score": choice.local_score,
|
|
382
|
+
"reason_codes": list(choice.reason_codes),
|
|
383
|
+
}
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
def _validate_case(
|
|
387
|
+
case: RegionAmbiguityCase,
|
|
388
|
+
*,
|
|
389
|
+
allow_hidden_sheet: bool = False,
|
|
390
|
+
) -> None:
|
|
391
|
+
if not isinstance(case, RegionAmbiguityCase):
|
|
392
|
+
raise InvalidRegionAmbiguityCaseError("invalid case shape")
|
|
393
|
+
if not _is_identifier(case.case_id):
|
|
394
|
+
raise InvalidRegionAmbiguityCaseError("invalid case identifier")
|
|
395
|
+
if not _is_required_string(case.sheet_name, max_length=128):
|
|
396
|
+
raise InvalidRegionAmbiguityCaseError("invalid case shape")
|
|
397
|
+
if not isinstance(case.sheet_visibility, str) or case.sheet_visibility not in {
|
|
398
|
+
"visible",
|
|
399
|
+
"hidden",
|
|
400
|
+
"veryHidden",
|
|
401
|
+
}:
|
|
402
|
+
raise InvalidRegionAmbiguityCaseError("invalid sheet visibility")
|
|
403
|
+
if not _is_required_string(case.source_range, max_length=64):
|
|
404
|
+
raise InvalidRegionAmbiguityCaseError("invalid case coordinate or source range")
|
|
405
|
+
if not _is_required_string(case.fact_digest, max_length=256):
|
|
406
|
+
raise InvalidRegionAmbiguityCaseError("invalid case shape")
|
|
407
|
+
_validate_case_feature_summary(case)
|
|
408
|
+
_validate_case_cells(case)
|
|
409
|
+
_validate_case_choices(case)
|
|
410
|
+
_validate_local_codes(case.ambiguity_codes, require_nonempty=True)
|
|
411
|
+
if not allow_hidden_sheet and case.sheet_visibility != "visible":
|
|
412
|
+
raise InvalidRegionAmbiguityCaseError("hidden sheet content cannot be sent")
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def _validate_case_feature_summary(case: RegionAmbiguityCase) -> None:
|
|
416
|
+
if not isinstance(case.feature_summary, tuple):
|
|
417
|
+
raise InvalidRegionAmbiguityCaseError("invalid feature summary")
|
|
418
|
+
keys: list[str] = []
|
|
419
|
+
for entry in case.feature_summary:
|
|
420
|
+
if not isinstance(entry, tuple) or len(entry) != 2:
|
|
421
|
+
raise InvalidRegionAmbiguityCaseError("invalid feature summary")
|
|
422
|
+
key, value = entry
|
|
423
|
+
if not _is_required_string(key, max_length=128) or not _is_valid_feature_scalar(value):
|
|
424
|
+
raise InvalidRegionAmbiguityCaseError("invalid feature summary")
|
|
425
|
+
keys.append(key)
|
|
426
|
+
if len(set(keys)) != len(keys):
|
|
427
|
+
raise InvalidRegionAmbiguityCaseError("invalid feature summary")
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def _validate_case_cells(case: RegionAmbiguityCase) -> None:
|
|
431
|
+
if not isinstance(case.cells, tuple):
|
|
432
|
+
raise InvalidRegionAmbiguityCaseError("invalid case shape")
|
|
433
|
+
try:
|
|
434
|
+
min_column, min_row, max_column, max_row = range_boundaries(case.source_range)
|
|
435
|
+
except (TypeError, ValueError) as error:
|
|
436
|
+
raise InvalidRegionAmbiguityCaseError("invalid case coordinate or source range") from error
|
|
437
|
+
if not _valid_excel_boundaries(min_column, min_row, max_column, max_row):
|
|
438
|
+
raise InvalidRegionAmbiguityCaseError("invalid case coordinate or source range")
|
|
439
|
+
coordinates: list[str] = []
|
|
440
|
+
for cell in case.cells:
|
|
441
|
+
if not isinstance(cell, RegionCellCue):
|
|
442
|
+
raise InvalidRegionAmbiguityCaseError("invalid case shape")
|
|
443
|
+
if not _valid_case_cell_fields(cell):
|
|
444
|
+
raise InvalidRegionAmbiguityCaseError("invalid case cell")
|
|
445
|
+
try:
|
|
446
|
+
row, column = coordinate_to_tuple(cell.coordinate)
|
|
447
|
+
except (TypeError, ValueError) as error:
|
|
448
|
+
raise InvalidRegionAmbiguityCaseError(
|
|
449
|
+
"invalid case coordinate or source range"
|
|
450
|
+
) from error
|
|
451
|
+
if not min_row <= row <= max_row or not min_column <= column <= max_column:
|
|
452
|
+
raise InvalidRegionAmbiguityCaseError("case cell outside source range")
|
|
453
|
+
if cell.merge_anchor is not None:
|
|
454
|
+
raise InvalidRegionAmbiguityCaseError("case cells must not contain merge children")
|
|
455
|
+
coordinates.append(cell.coordinate)
|
|
456
|
+
if len(set(coordinates)) != len(coordinates):
|
|
457
|
+
raise InvalidRegionAmbiguityCaseError("duplicate cell coordinate")
|
|
458
|
+
if coordinates != sorted(coordinates, key=coordinate_to_tuple):
|
|
459
|
+
raise InvalidRegionAmbiguityCaseError("case cells must use row/column order")
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
def _validate_case_choices(case: RegionAmbiguityCase) -> None:
|
|
463
|
+
if not isinstance(case.choices, tuple):
|
|
464
|
+
raise InvalidRegionAmbiguityCaseError("invalid case shape")
|
|
465
|
+
choice_ids: list[str] = []
|
|
466
|
+
choice_kinds: set[str] = set()
|
|
467
|
+
for choice in case.choices:
|
|
468
|
+
if not isinstance(choice, RegionChoice):
|
|
469
|
+
raise InvalidRegionAmbiguityCaseError("invalid case shape")
|
|
470
|
+
if (
|
|
471
|
+
not _is_identifier(choice.choice_id)
|
|
472
|
+
or not isinstance(choice.kind, str)
|
|
473
|
+
or choice.kind not in {"logical_table", "form", "matrix", "text", "unclassified"}
|
|
474
|
+
or not _is_finite_unit_float(choice.local_score)
|
|
475
|
+
):
|
|
476
|
+
raise InvalidRegionAmbiguityCaseError("invalid case choice")
|
|
477
|
+
_validate_local_codes(choice.reason_codes)
|
|
478
|
+
choice_ids.append(choice.choice_id)
|
|
479
|
+
choice_kinds.add(choice.kind)
|
|
480
|
+
if len(set(choice_ids)) != len(choice_ids):
|
|
481
|
+
raise InvalidRegionAmbiguityCaseError("duplicate choice_id")
|
|
482
|
+
if len(choice_kinds) < 2:
|
|
483
|
+
raise InvalidRegionAmbiguityCaseError(
|
|
484
|
+
"ambiguous assessment requires at least two choice kinds"
|
|
485
|
+
)
|
|
486
|
+
if not _is_identifier(case.fallback_choice_id) or case.fallback_choice_id not in set(
|
|
487
|
+
choice_ids
|
|
488
|
+
):
|
|
489
|
+
raise InvalidRegionAmbiguityCaseError("fallback choice is not registered")
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
def _valid_case_cell_fields(cell: RegionCellCue) -> bool:
|
|
493
|
+
return (
|
|
494
|
+
_is_required_string(cell.coordinate, max_length=32)
|
|
495
|
+
and isinstance(cell.display_text, str)
|
|
496
|
+
and isinstance(cell.value_type, str)
|
|
497
|
+
and isinstance(cell.style_fingerprint, str)
|
|
498
|
+
and isinstance(cell.rowspan, int)
|
|
499
|
+
and not isinstance(cell.rowspan, bool)
|
|
500
|
+
and cell.rowspan > 0
|
|
501
|
+
and isinstance(cell.colspan, int)
|
|
502
|
+
and not isinstance(cell.colspan, bool)
|
|
503
|
+
and cell.colspan > 0
|
|
504
|
+
)
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def _valid_excel_boundaries(
|
|
508
|
+
min_column: int | None,
|
|
509
|
+
min_row: int | None,
|
|
510
|
+
max_column: int | None,
|
|
511
|
+
max_row: int | None,
|
|
512
|
+
) -> bool:
|
|
513
|
+
return (
|
|
514
|
+
isinstance(min_column, int)
|
|
515
|
+
and isinstance(min_row, int)
|
|
516
|
+
and isinstance(max_column, int)
|
|
517
|
+
and isinstance(max_row, int)
|
|
518
|
+
and 1 <= min_column <= max_column <= 16_384
|
|
519
|
+
and 1 <= min_row <= max_row <= 1_048_576
|
|
520
|
+
)
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def _validate_local_codes(codes: object, *, require_nonempty: bool = False) -> None:
|
|
524
|
+
if (
|
|
525
|
+
not isinstance(codes, tuple)
|
|
526
|
+
or (require_nonempty and not codes)
|
|
527
|
+
or not all(_is_identifier(code) for code in codes)
|
|
528
|
+
):
|
|
529
|
+
raise InvalidRegionAmbiguityCaseError("invalid local codes")
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
def _is_identifier(value: object, *, max_length: int = 128) -> bool:
|
|
533
|
+
return (
|
|
534
|
+
isinstance(value, str)
|
|
535
|
+
and 0 < len(value) <= max_length
|
|
536
|
+
and value.isascii()
|
|
537
|
+
and all(character.isalnum() or character in "._-" for character in value)
|
|
538
|
+
)
|
|
539
|
+
|
|
540
|
+
|
|
541
|
+
def _is_required_string(value: object, *, max_length: int) -> bool:
|
|
542
|
+
return isinstance(value, str) and 0 < len(value) <= max_length
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def _is_valid_feature_scalar(value: object) -> bool:
|
|
546
|
+
if value is None or type(value) in (str, int, bool):
|
|
547
|
+
return True
|
|
548
|
+
return type(value) is float and math.isfinite(value)
|
|
549
|
+
|
|
550
|
+
|
|
551
|
+
def _is_finite_unit_float(value: object) -> bool:
|
|
552
|
+
return type(value) is float and math.isfinite(value) and 0 <= value <= 1
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def _is_valid_response_confidence(value: object) -> bool:
|
|
556
|
+
if type(value) is int:
|
|
557
|
+
return 0 <= value <= 1
|
|
558
|
+
return type(value) is float and math.isfinite(value) and 0 <= value <= 1
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def _exact_keys(value: dict[str, Any], expected: set[str], label: str) -> None:
|
|
562
|
+
unknown = set(value) - expected
|
|
563
|
+
if unknown:
|
|
564
|
+
raise WorkbookModelResponseError(f"unknown {label} fields")
|
|
565
|
+
missing = expected - set(value)
|
|
566
|
+
if missing:
|
|
567
|
+
raise WorkbookModelResponseError(f"missing {label} fields")
|
|
568
|
+
|
|
569
|
+
|
|
570
|
+
def _request_membership(request: WorkbookModelRequest) -> Mapping[str, tuple[str, ...]]:
|
|
571
|
+
membership = dict(request.choice_ids_by_case)
|
|
572
|
+
if set(membership) != set(request.case_ids) or len(membership) != len(request.case_ids):
|
|
573
|
+
raise WorkbookModelResponseError("invalid local request membership registry")
|
|
574
|
+
return membership
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
def _decode_decision(
|
|
578
|
+
value: object,
|
|
579
|
+
membership: Mapping[str, tuple[str, ...]],
|
|
580
|
+
) -> RegionModelDecision:
|
|
581
|
+
if not isinstance(value, dict):
|
|
582
|
+
raise WorkbookModelResponseError("decision must be an object")
|
|
583
|
+
if "status" not in value or not isinstance(value["status"], str):
|
|
584
|
+
raise WorkbookModelResponseError("decision status must be a string")
|
|
585
|
+
status = value["status"]
|
|
586
|
+
expected = {"case_id", "status", "confidence", "reason_codes"}
|
|
587
|
+
if status == "selected":
|
|
588
|
+
if "choice_id" not in value:
|
|
589
|
+
raise WorkbookModelResponseError("selected decision requires choice_id")
|
|
590
|
+
expected.add("choice_id")
|
|
591
|
+
elif status == "abstained" and "choice_id" in value:
|
|
592
|
+
raise WorkbookModelResponseError("abstained decision cannot include choice_id")
|
|
593
|
+
_exact_keys(value, expected, "decision")
|
|
594
|
+
case_id = value["case_id"]
|
|
595
|
+
if not isinstance(case_id, str):
|
|
596
|
+
raise WorkbookModelResponseError("case_id must be a string")
|
|
597
|
+
if status not in {"selected", "abstained"}:
|
|
598
|
+
raise WorkbookModelResponseError("unknown decision status")
|
|
599
|
+
confidence = value["confidence"]
|
|
600
|
+
if not _is_valid_response_confidence(confidence):
|
|
601
|
+
raise WorkbookModelResponseError("confidence must be finite and between 0 and 1")
|
|
602
|
+
reason_codes = value["reason_codes"]
|
|
603
|
+
if not isinstance(reason_codes, list) or not all(
|
|
604
|
+
isinstance(code, str) for code in reason_codes
|
|
605
|
+
):
|
|
606
|
+
raise WorkbookModelResponseError("reason_codes must be a list of strings")
|
|
607
|
+
if status == "abstained":
|
|
608
|
+
return RegionModelDecision(
|
|
609
|
+
case_id=case_id,
|
|
610
|
+
status="abstained",
|
|
611
|
+
choice_id=None,
|
|
612
|
+
reported_confidence=float(confidence),
|
|
613
|
+
reason_codes=tuple(reason_codes),
|
|
614
|
+
)
|
|
615
|
+
choice_id = value["choice_id"]
|
|
616
|
+
if not isinstance(choice_id, str):
|
|
617
|
+
raise WorkbookModelResponseError("choice_id must be a string")
|
|
618
|
+
if case_id not in membership:
|
|
619
|
+
raise WorkbookModelResponseError("unknown case_id")
|
|
620
|
+
if choice_id not in membership[case_id]:
|
|
621
|
+
raise WorkbookModelResponseError("unknown choice_id")
|
|
622
|
+
return RegionModelDecision(
|
|
623
|
+
case_id=case_id,
|
|
624
|
+
status="selected",
|
|
625
|
+
choice_id=choice_id,
|
|
626
|
+
reported_confidence=float(confidence),
|
|
627
|
+
reason_codes=tuple(reason_codes),
|
|
628
|
+
)
|