langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,419 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ import re
6
+ from collections.abc import Iterable
7
+ from dataclasses import dataclass
8
+ from pathlib import Path
9
+ from typing import Literal
10
+
11
+ from langparse.workbooks.modeling.types import RegionChoice
12
+
13
+ _SAFE_TOKEN_PATTERN = re.compile(r"^[a-zA-Z0-9_-]{1,128}$")
14
+ _HEX64_PATTERN = re.compile(r"^sha256:[0-9a-f]{64}$")
15
+ _RANGE_A1_PATTERN = re.compile(r"^[A-Za-z]+[1-9][0-9]*(?::[A-Za-z]+[1-9][0-9]*)?$")
16
+
17
+ VALID_SPLITS = frozenset({"tuning", "holdout"})
18
+ VALID_COHORTS = frozenset({"ambiguous", "clear_no_call"})
19
+ VALID_EXPECTED_KINDS = frozenset(
20
+ {"logical_table", "form", "matrix", "text", "unclassified", "unresolvable"}
21
+ )
22
+
23
+
24
+ class WorkbookEvaluationError(Exception):
25
+ """Base error for workbook ambiguity evaluation."""
26
+
27
+ def __init__(
28
+ self,
29
+ message: str,
30
+ *,
31
+ code: str = "evaluation_error",
32
+ evaluation_id: str | None = None,
33
+ ) -> None:
34
+ super().__init__(message)
35
+ self.code = code
36
+ self.evaluation_id = evaluation_id
37
+
38
+ def __str__(self) -> str:
39
+ if self.evaluation_id:
40
+ return f"[{self.code}] ({self.evaluation_id}): {super().__str__()}"
41
+ return f"[{self.code}]: {super().__str__()}"
42
+
43
+
44
+ class InvalidGoldenSetError(WorkbookEvaluationError, ValueError):
45
+ """Raised when golden set manifest, path, or hash is invalid."""
46
+
47
+
48
+ class GoldenSetDriftError(WorkbookEvaluationError, RuntimeError):
49
+ """Raised when runtime cases, facts, or choices deviate from truth."""
50
+
51
+
52
+ @dataclass(frozen=True)
53
+ class GoldenCase:
54
+ label_id: str
55
+ sheet_name: str
56
+ source_range: str
57
+ expected: Literal["logical_table", "form", "matrix", "text", "unclassified", "unresolvable"]
58
+ fact_digest: str
59
+ choices_digest: str
60
+
61
+
62
+ @dataclass(frozen=True)
63
+ class GoldenSample:
64
+ sample_id: str
65
+ path: str
66
+ sha256: str
67
+ cohort: Literal["ambiguous", "clear_no_call"]
68
+ cases: tuple[GoldenCase, ...]
69
+
70
+
71
+ @dataclass(frozen=True)
72
+ class GoldenSetManifest:
73
+ schema_version: int
74
+ dataset_id: str
75
+ dataset_version: str
76
+ split: Literal["tuning", "holdout"]
77
+ source_root: Path
78
+ samples: tuple[GoldenSample, ...]
79
+ dataset_digest: str
80
+
81
+
82
+ def _strict_json_object_pairs_hook(pairs: list[tuple[str, object]]) -> dict[str, object]:
83
+ result: dict[str, object] = {}
84
+ for key, value in pairs:
85
+ if key in result:
86
+ raise InvalidGoldenSetError(
87
+ f"Duplicate JSON key: {key}",
88
+ code="invalid_schema",
89
+ )
90
+ result[key] = value
91
+ return result
92
+
93
+
94
+ def compute_evaluation_id(
95
+ dataset_id: str,
96
+ dataset_version: str,
97
+ sample_id: str,
98
+ label_id: str,
99
+ ) -> str:
100
+ raw = f"{dataset_id}:{dataset_version}:{sample_id}:{label_id}".encode()
101
+ return f"sha256:{hashlib.sha256(raw).hexdigest()}"
102
+
103
+
104
+ def compute_sample_evaluation_id(
105
+ dataset_id: str,
106
+ dataset_version: str,
107
+ sample_id: str,
108
+ ) -> str:
109
+ raw = f"{dataset_id}:{dataset_version}:{sample_id}".encode()
110
+ return f"sha256:{hashlib.sha256(raw).hexdigest()}"
111
+
112
+
113
+ def compute_choices_digest(choices: Iterable[RegionChoice]) -> str:
114
+ payload = [
115
+ {
116
+ "choice_id": choice.choice_id,
117
+ "kind": choice.kind,
118
+ "local_score": round(float(choice.local_score), 6),
119
+ "reason_codes": list(choice.reason_codes),
120
+ }
121
+ for choice in choices
122
+ ]
123
+ serialized = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode("utf-8")
124
+ return f"sha256:{hashlib.sha256(serialized).hexdigest()}"
125
+
126
+
127
+ def compute_dataset_digest(
128
+ schema_version: int,
129
+ dataset_id: str,
130
+ dataset_version: str,
131
+ split: str,
132
+ samples: tuple[GoldenSample, ...],
133
+ ) -> str:
134
+ payload = {
135
+ "schema_version": schema_version,
136
+ "dataset_id": dataset_id,
137
+ "dataset_version": dataset_version,
138
+ "split": split,
139
+ "samples": [
140
+ {
141
+ "sample_id": sample.sample_id,
142
+ "path": sample.path,
143
+ "sha256": sample.sha256,
144
+ "cohort": sample.cohort,
145
+ "cases": [
146
+ {
147
+ "label_id": case.label_id,
148
+ "sheet_name": case.sheet_name,
149
+ "source_range": case.source_range,
150
+ "expected": case.expected,
151
+ "fact_digest": case.fact_digest,
152
+ "choices_digest": case.choices_digest,
153
+ }
154
+ for case in sample.cases
155
+ ],
156
+ }
157
+ for sample in samples
158
+ ],
159
+ }
160
+ serialized = json.dumps(payload, sort_keys=True, separators=(",", ":")).encode("utf-8")
161
+ return f"sha256:{hashlib.sha256(serialized).hexdigest()}"
162
+
163
+
164
+ def validate_output_dir_isolation(source_root: Path, output_dir: Path) -> None:
165
+ resolved_root = source_root.resolve()
166
+ resolved_output = output_dir.resolve()
167
+ if resolved_output == resolved_root or resolved_output.is_relative_to(resolved_root):
168
+ raise InvalidGoldenSetError(
169
+ "Output directory must not be inside or equal to source root",
170
+ code="input_output_overlap",
171
+ )
172
+
173
+
174
+ def _validate_safe_token(token: object, field_name: str) -> str:
175
+ if not isinstance(token, str) or not _SAFE_TOKEN_PATTERN.match(token):
176
+ raise InvalidGoldenSetError(
177
+ f"Invalid identifier for {field_name}: must be ASCII safe token (1..128 chars)",
178
+ code="invalid_schema",
179
+ )
180
+ return token
181
+
182
+
183
+ def _validate_sha256(hash_str: object, field_name: str) -> str:
184
+ if not isinstance(hash_str, str) or not _HEX64_PATTERN.match(hash_str):
185
+ raise InvalidGoldenSetError(
186
+ f"Invalid sha256 format for {field_name}: must be 'sha256:' followed by 64 lowercase hex characters",
187
+ code="invalid_schema",
188
+ )
189
+ return hash_str
190
+
191
+
192
+ def _compute_file_sha256(file_path: Path) -> str:
193
+ hasher = hashlib.sha256()
194
+ with file_path.open("rb") as f:
195
+ while chunk := f.read(65536):
196
+ hasher.update(chunk)
197
+ return f"sha256:{hasher.hexdigest()}"
198
+
199
+
200
+ def load_golden_set_manifest(manifest_path: str | Path) -> GoldenSetManifest:
201
+ path = Path(manifest_path).resolve()
202
+ if not path.is_file():
203
+ raise InvalidGoldenSetError("Manifest file does not exist", code="file_not_found")
204
+
205
+ try:
206
+ content = path.read_text(encoding="utf-8")
207
+ data = json.loads(content, object_pairs_hook=_strict_json_object_pairs_hook)
208
+ except json.JSONDecodeError as exc:
209
+ raise InvalidGoldenSetError("Manifest is not valid JSON", code="invalid_schema") from exc
210
+
211
+ if not isinstance(data, dict):
212
+ raise InvalidGoldenSetError("Manifest root must be an object", code="invalid_schema")
213
+
214
+ expected_manifest_keys = {
215
+ "schema_version",
216
+ "dataset_id",
217
+ "dataset_version",
218
+ "split",
219
+ "source_root",
220
+ "samples",
221
+ }
222
+ if set(data.keys()) != expected_manifest_keys:
223
+ raise InvalidGoldenSetError(
224
+ "Manifest keys mismatch expected schema",
225
+ code="invalid_schema",
226
+ )
227
+
228
+ schema_version = data["schema_version"]
229
+ if type(schema_version) is not int or schema_version != 1 or isinstance(schema_version, bool):
230
+ raise InvalidGoldenSetError(
231
+ "schema_version must be integer 1",
232
+ code="invalid_schema",
233
+ )
234
+
235
+ dataset_id = _validate_safe_token(data["dataset_id"], "dataset_id")
236
+ dataset_version = _validate_safe_token(data["dataset_version"], "dataset_version")
237
+
238
+ split = data["split"]
239
+ if split not in VALID_SPLITS:
240
+ raise InvalidGoldenSetError(
241
+ f"split must be one of {sorted(VALID_SPLITS)}",
242
+ code="invalid_schema",
243
+ )
244
+
245
+ source_root_str = data["source_root"]
246
+ if not isinstance(source_root_str, str) or not source_root_str.strip():
247
+ raise InvalidGoldenSetError("source_root must be a non-empty string", code="invalid_schema")
248
+ if Path(source_root_str).is_absolute():
249
+ raise InvalidGoldenSetError(
250
+ "source_root must be relative to the manifest",
251
+ code="path_traversal",
252
+ )
253
+
254
+ source_root = (path.parent / source_root_str).resolve()
255
+ if not source_root.is_dir():
256
+ raise InvalidGoldenSetError("source_root directory does not exist", code="file_not_found")
257
+
258
+ samples_raw = data["samples"]
259
+ if not isinstance(samples_raw, list):
260
+ raise InvalidGoldenSetError("samples must be a list", code="invalid_schema")
261
+
262
+ expected_sample_keys = {"sample_id", "path", "sha256", "cohort", "cases"}
263
+ expected_case_keys = {
264
+ "label_id",
265
+ "sheet_name",
266
+ "source_range",
267
+ "expected",
268
+ "fact_digest",
269
+ "choices_digest",
270
+ }
271
+
272
+ seen_sample_ids: set[str] = set()
273
+ samples: list[GoldenSample] = []
274
+
275
+ for sample_dict in samples_raw:
276
+ if not isinstance(sample_dict, dict) or set(sample_dict.keys()) != expected_sample_keys:
277
+ raise InvalidGoldenSetError(
278
+ "Sample entry keys mismatch schema",
279
+ code="invalid_schema",
280
+ )
281
+
282
+ sample_id = _validate_safe_token(sample_dict["sample_id"], "sample_id")
283
+ if sample_id in seen_sample_ids:
284
+ raise InvalidGoldenSetError(
285
+ f"Duplicate sample_id: {sample_id}",
286
+ code="duplicate_id",
287
+ )
288
+ seen_sample_ids.add(sample_id)
289
+
290
+ sample_rel_path = sample_dict["path"]
291
+ if not isinstance(sample_rel_path, str) or not sample_rel_path.strip():
292
+ raise InvalidGoldenSetError(
293
+ "Sample path must be a non-empty string", code="invalid_schema"
294
+ )
295
+
296
+ if Path(sample_rel_path).is_absolute() or ".." in Path(sample_rel_path).parts:
297
+ raise InvalidGoldenSetError(
298
+ "Sample path must be relative and not contain '..'",
299
+ code="path_traversal",
300
+ )
301
+
302
+ full_sample_path = (source_root / sample_rel_path).resolve()
303
+ if not full_sample_path.is_relative_to(source_root):
304
+ raise InvalidGoldenSetError(
305
+ "Sample path resolves outside source_root",
306
+ code="path_traversal",
307
+ )
308
+
309
+ if not full_sample_path.is_file():
310
+ raise InvalidGoldenSetError("Sample file does not exist", code="file_not_found")
311
+
312
+ expected_sha256 = _validate_sha256(sample_dict["sha256"], "sample sha256")
313
+ actual_sha256 = _compute_file_sha256(full_sample_path)
314
+ if actual_sha256 != expected_sha256:
315
+ raise InvalidGoldenSetError(
316
+ "Sample file SHA-256 does not match manifest",
317
+ code="hash_mismatch",
318
+ )
319
+
320
+ cohort = sample_dict["cohort"]
321
+ if cohort not in VALID_COHORTS:
322
+ raise InvalidGoldenSetError(
323
+ f"cohort must be one of {sorted(VALID_COHORTS)}",
324
+ code="invalid_schema",
325
+ )
326
+
327
+ cases_raw = sample_dict["cases"]
328
+ if not isinstance(cases_raw, list):
329
+ raise InvalidGoldenSetError("cases must be a list", code="invalid_schema")
330
+
331
+ if cohort == "clear_no_call" and len(cases_raw) != 0:
332
+ raise InvalidGoldenSetError(
333
+ "clear_no_call cohort must have empty cases",
334
+ code="invalid_schema",
335
+ )
336
+ if cohort == "ambiguous" and len(cases_raw) == 0:
337
+ raise InvalidGoldenSetError(
338
+ "ambiguous cohort must have at least one case",
339
+ code="invalid_schema",
340
+ )
341
+
342
+ seen_label_ids: set[str] = set()
343
+ cases: list[GoldenCase] = []
344
+ for case_dict in cases_raw:
345
+ if not isinstance(case_dict, dict) or set(case_dict.keys()) != expected_case_keys:
346
+ raise InvalidGoldenSetError(
347
+ "Case entry keys mismatch schema",
348
+ code="invalid_schema",
349
+ )
350
+
351
+ label_id = _validate_safe_token(case_dict["label_id"], "label_id")
352
+ if label_id in seen_label_ids:
353
+ raise InvalidGoldenSetError(
354
+ f"Duplicate label_id in sample: {label_id}",
355
+ code="duplicate_id",
356
+ )
357
+ seen_label_ids.add(label_id)
358
+
359
+ sheet_name = case_dict["sheet_name"]
360
+ if not isinstance(sheet_name, str) or not sheet_name.strip():
361
+ raise InvalidGoldenSetError(
362
+ "sheet_name must be non-empty string", code="invalid_schema"
363
+ )
364
+
365
+ source_range = case_dict["source_range"]
366
+ if not isinstance(source_range, str) or not _RANGE_A1_PATTERN.match(source_range):
367
+ raise InvalidGoldenSetError(
368
+ "source_range must be standard A1 format", code="invalid_schema"
369
+ )
370
+
371
+ expected = case_dict["expected"]
372
+ if expected not in VALID_EXPECTED_KINDS:
373
+ raise InvalidGoldenSetError(
374
+ f"expected must be one of {sorted(VALID_EXPECTED_KINDS)}",
375
+ code="invalid_schema",
376
+ )
377
+
378
+ fact_digest = _validate_sha256(case_dict["fact_digest"], "fact_digest")
379
+ choices_digest = _validate_sha256(case_dict["choices_digest"], "choices_digest")
380
+
381
+ cases.append(
382
+ GoldenCase(
383
+ label_id=label_id,
384
+ sheet_name=sheet_name,
385
+ source_range=source_range,
386
+ expected=expected,
387
+ fact_digest=fact_digest,
388
+ choices_digest=choices_digest,
389
+ )
390
+ )
391
+
392
+ samples.append(
393
+ GoldenSample(
394
+ sample_id=sample_id,
395
+ path=sample_rel_path,
396
+ sha256=expected_sha256,
397
+ cohort=cohort,
398
+ cases=tuple(cases),
399
+ )
400
+ )
401
+
402
+ samples_tuple = tuple(samples)
403
+ dataset_digest = compute_dataset_digest(
404
+ schema_version=schema_version,
405
+ dataset_id=dataset_id,
406
+ dataset_version=dataset_version,
407
+ split=split,
408
+ samples=samples_tuple,
409
+ )
410
+
411
+ return GoldenSetManifest(
412
+ schema_version=schema_version,
413
+ dataset_id=dataset_id,
414
+ dataset_version=dataset_version,
415
+ split=split,
416
+ source_root=source_root,
417
+ samples=samples_tuple,
418
+ dataset_digest=dataset_digest,
419
+ )
@@ -0,0 +1,14 @@
1
+ """Explicit row labels shared by region detection and table interpretation."""
2
+
3
+ import re
4
+
5
+ _SECTION_LABEL = re.compile(r"^[一二三四五六七八九十百]+[、..]\s*\S")
6
+ _TOTAL_LABEL = re.compile(r"(?:合计|小计|总计)(?:[::]|[==].*|[((\[].*[))\]])?$")
7
+
8
+
9
+ def is_section_label(value: str) -> bool:
10
+ return bool(_SECTION_LABEL.match(value.strip()))
11
+
12
+
13
+ def is_total_label(value: str) -> bool:
14
+ return bool(_TOTAL_LABEL.search(re.sub(r"\s+", "", value)))
@@ -0,0 +1,117 @@
1
+ """Independent range-level dependency graph with bounded, non-expanding queries."""
2
+
3
+ import re
4
+ from dataclasses import dataclass, field
5
+
6
+ from langparse.workbooks.reference_types import DependencyEdge, ReferenceDiagnostic
7
+ from langparse.workbooks.types import SourceRef, WorkbookSnapshot
8
+
9
+
10
+ def _bounds(value):
11
+ if not isinstance(value, str):
12
+ raise ValueError("Expected a finite A1 range")
13
+ parts = value.replace("$", "").upper().split(":")
14
+ if len(parts) not in {1, 2}:
15
+ raise ValueError(f"Expected finite A1 range, got {value!r}")
16
+ result = []
17
+ for part in (parts[0], parts[-1]):
18
+ match = re.fullmatch(r"([A-Z]+)([1-9][0-9]*)", part)
19
+ if not match:
20
+ raise ValueError(f"Expected finite A1 range, got {value!r}")
21
+ column = 0
22
+ for char in match[1]:
23
+ column = column * 26 + ord(char) - 64
24
+ result.append((column, int(match[2])))
25
+ left, top = result[0]
26
+ right, bottom = result[1]
27
+ if not (1 <= left <= right <= 16384 and 1 <= top <= bottom <= 1048576):
28
+ raise ValueError(f"Range is outside Excel bounds: {value!r}")
29
+ return left, top, right, bottom
30
+
31
+
32
+ def overlaps(first: SourceRef, second: SourceRef) -> bool:
33
+ if first.sheet_name.casefold() != second.sheet_name.casefold():
34
+ return False
35
+ a, b, c, d = _bounds(first.range)
36
+ e, f, g, h = _bounds(second.range)
37
+ return a <= g and e <= c and b <= h and f <= d
38
+
39
+
40
+ @dataclass
41
+ class WorkbookLineage:
42
+ edges: list[DependencyEdge] = field(default_factory=list)
43
+ diagnostics: list[ReferenceDiagnostic] = field(default_factory=list)
44
+
45
+ def dependencies_of(self, source: SourceRef) -> list[DependencyEdge]:
46
+ return [edge for edge in self.edges if overlaps(edge.dependent, source)]
47
+
48
+ def dependents_of(self, source: SourceRef) -> list[DependencyEdge]:
49
+ return [edge for edge in self.edges if edge.target and overlaps(edge.target, source)]
50
+
51
+
52
+ def build_lineage(snapshot: WorkbookSnapshot) -> WorkbookLineage:
53
+ facts = snapshot.reference_facts
54
+ result = WorkbookLineage(diagnostics=list(facts.diagnostics))
55
+ for formula in facts.formulas:
56
+ for reference in formula.references:
57
+ if reference.status == "resolved" and not reference.targets:
58
+ continue
59
+ for target in reference.targets or [None]:
60
+ result.edges.append(
61
+ DependencyEdge(
62
+ formula.source_ref,
63
+ target,
64
+ reference.reference,
65
+ reference.status,
66
+ [formula.source_ref, *([target] if target else [])],
67
+ reference.external_workbook,
68
+ )
69
+ )
70
+ # Iterative DFS avoids recursion limits on long formula chains.
71
+ nodes = {f.source_ref.key: f.source_ref for f in facts.formulas}
72
+ adjacency = {key: [] for key in nodes}
73
+ single_cells = {
74
+ (ref.sheet_name.casefold(), ref.range.replace("$", "").upper()): key
75
+ for key, ref in nodes.items()
76
+ }
77
+ for edge in result.edges:
78
+ if edge.target:
79
+ target_range = edge.target.range.replace("$", "").upper()
80
+ if ":" not in target_range:
81
+ key = single_cells.get((edge.target.sheet_name.casefold(), target_range))
82
+ if key is not None:
83
+ adjacency[edge.dependent.key].append(key)
84
+ else:
85
+ adjacency[edge.dependent.key].extend(
86
+ key for key, ref in nodes.items() if overlaps(edge.target, ref)
87
+ )
88
+ visited, active, reported = set(), set(), set()
89
+ for start in nodes:
90
+ if start in visited:
91
+ continue
92
+ stack = [(start, iter(adjacency[start]))]
93
+ active.add(start)
94
+ while stack:
95
+ node, children = stack[-1]
96
+ child = next(children, None)
97
+ if child is None:
98
+ visited.add(node)
99
+ active.remove(node)
100
+ stack.pop()
101
+ elif child in active:
102
+ path = [item[0] for item in stack]
103
+ cycle = path[path.index(child) :]
104
+ identity = frozenset(cycle)
105
+ if identity not in reported:
106
+ reported.add(identity)
107
+ result.diagnostics.append(
108
+ ReferenceDiagnostic(
109
+ "circular_reference",
110
+ " -> ".join([*cycle, child]),
111
+ [nodes[key] for key in cycle],
112
+ )
113
+ )
114
+ elif child not in visited:
115
+ active.add(child)
116
+ stack.append((child, iter(adjacency[child])))
117
+ return result
@@ -0,0 +1,52 @@
1
+ from .config import WorkbookModelConfig, resolve_workbook_model_config
2
+ from .openai_adapter import OpenAIWorkbookStructureAdapter
3
+ from .policy import WorkbookDisambiguation
4
+ from .ports import (
5
+ InvalidRegionAmbiguityCaseError,
6
+ RequiredWorkbookDisambiguationError,
7
+ WorkbookModelConfigurationError,
8
+ WorkbookModelError,
9
+ WorkbookModelResponseError,
10
+ WorkbookStructureModelAdapter,
11
+ )
12
+ from .types import (
13
+ ModelCallAudit,
14
+ ModelIdentity,
15
+ ProviderReply,
16
+ RegionAmbiguityCase,
17
+ RegionCellCue,
18
+ RegionChoice,
19
+ RegionFeatureScalar,
20
+ RegionModelDecision,
21
+ RegionResolution,
22
+ RegionResolutionBatch,
23
+ WorkbookModelMode,
24
+ WorkbookModelPolicy,
25
+ WorkbookModelRequest,
26
+ )
27
+
28
+ __all__ = [
29
+ "InvalidRegionAmbiguityCaseError",
30
+ "ModelCallAudit",
31
+ "ModelIdentity",
32
+ "OpenAIWorkbookStructureAdapter",
33
+ "ProviderReply",
34
+ "RegionAmbiguityCase",
35
+ "RegionCellCue",
36
+ "RegionChoice",
37
+ "RegionFeatureScalar",
38
+ "RegionModelDecision",
39
+ "RegionResolution",
40
+ "RegionResolutionBatch",
41
+ "RequiredWorkbookDisambiguationError",
42
+ "WorkbookDisambiguation",
43
+ "WorkbookModelConfig",
44
+ "WorkbookModelConfigurationError",
45
+ "WorkbookModelError",
46
+ "WorkbookModelMode",
47
+ "WorkbookModelPolicy",
48
+ "WorkbookModelRequest",
49
+ "WorkbookModelResponseError",
50
+ "WorkbookStructureModelAdapter",
51
+ "resolve_workbook_model_config",
52
+ ]
@@ -0,0 +1,20 @@
1
+ from __future__ import annotations
2
+
3
+ from threading import RLock
4
+
5
+
6
+ class MemoryDecisionCache:
7
+ """Process-local storage for response envelopes that passed contract validation."""
8
+
9
+ def __init__(self) -> None:
10
+ self._responses: dict[str, bytes] = {}
11
+ self._lock = RLock()
12
+
13
+ def get(self, key: str) -> bytes | None:
14
+ with self._lock:
15
+ return self._responses.get(key)
16
+
17
+ def put(self, key: str, body: bytes) -> None:
18
+ copied = bytes(body)
19
+ with self._lock:
20
+ self._responses[key] = copied