langparse 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. langparse/__init__.py +55 -0
  2. langparse/autoparser.py +25 -0
  3. langparse/chunkers/__init__.py +12 -0
  4. langparse/chunkers/blocks.py +151 -0
  5. langparse/chunkers/profiles.py +53 -0
  6. langparse/chunkers/registry.py +38 -0
  7. langparse/chunkers/semantic.py +242 -0
  8. langparse/chunkers/text.py +96 -0
  9. langparse/chunkers/workbook.py +942 -0
  10. langparse/cli.py +329 -0
  11. langparse/config.py +169 -0
  12. langparse/core/__init__.py +0 -0
  13. langparse/core/chunker.py +16 -0
  14. langparse/core/engine.py +37 -0
  15. langparse/core/parser.py +35 -0
  16. langparse/core/rendering.py +49 -0
  17. langparse/engines/__init__.py +1 -0
  18. langparse/engines/pdf/__init__.py +1 -0
  19. langparse/engines/pdf/deepdoc/__init__.py +55 -0
  20. langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
  21. langparse/engines/pdf/deepdoc/model_loader.py +101 -0
  22. langparse/engines/pdf/deepdoc/ocr.py +641 -0
  23. langparse/engines/pdf/deepdoc/operators.py +684 -0
  24. langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
  25. langparse/engines/pdf/deepdoc/postprocess.py +339 -0
  26. langparse/engines/pdf/deepdoc/recognizer.py +418 -0
  27. langparse/engines/pdf/deepdoc/rendering.py +210 -0
  28. langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
  29. langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
  30. langparse/engines/pdf/deepdoc/utils.py +36 -0
  31. langparse/engines/pdf/deepdoc_engine.py +164 -0
  32. langparse/engines/pdf/mineru.py +259 -0
  33. langparse/engines/pdf/mineru_client.py +318 -0
  34. langparse/engines/pdf/mineru_service.py +225 -0
  35. langparse/engines/pdf/ocr.py +101 -0
  36. langparse/engines/pdf/other.py +20 -0
  37. langparse/engines/pdf/simple.py +134 -0
  38. langparse/engines/pdf/vision_llm.py +27 -0
  39. langparse/errors.py +70 -0
  40. langparse/logging.py +27 -0
  41. langparse/metrics.py +129 -0
  42. langparse/parsers/__init__.py +0 -0
  43. langparse/parsers/docx_parser.py +114 -0
  44. langparse/parsers/excel_parser.py +220 -0
  45. langparse/parsers/markdown_parser.py +34 -0
  46. langparse/parsers/pdf_parser.py +31 -0
  47. langparse/parsers/registry.py +48 -0
  48. langparse/parsers/sniff.py +72 -0
  49. langparse/progress.py +77 -0
  50. langparse/py.typed +0 -0
  51. langparse/services/__init__.py +11 -0
  52. langparse/services/batch_service.py +339 -0
  53. langparse/services/benchmark_service.py +202 -0
  54. langparse/services/fidelity.py +154 -0
  55. langparse/services/output_paths.py +86 -0
  56. langparse/services/parse_service.py +523 -0
  57. langparse/services/quality.py +65 -0
  58. langparse/services/workbook_ambiguity_benchmark.py +563 -0
  59. langparse/services/workbook_quality_benchmark.py +230 -0
  60. langparse/types.py +97 -0
  61. langparse/workbooks/__init__.py +103 -0
  62. langparse/workbooks/adapters.py +474 -0
  63. langparse/workbooks/assembly.py +993 -0
  64. langparse/workbooks/blocks.py +209 -0
  65. langparse/workbooks/bundle-v1.schema.json +71 -0
  66. langparse/workbooks/bundle.py +341 -0
  67. langparse/workbooks/classification.py +393 -0
  68. langparse/workbooks/continuation.py +577 -0
  69. langparse/workbooks/evaluation/__init__.py +45 -0
  70. langparse/workbooks/evaluation/evaluator.py +381 -0
  71. langparse/workbooks/evaluation/schema.py +419 -0
  72. langparse/workbooks/labels.py +14 -0
  73. langparse/workbooks/lineage.py +117 -0
  74. langparse/workbooks/modeling/__init__.py +52 -0
  75. langparse/workbooks/modeling/cache.py +20 -0
  76. langparse/workbooks/modeling/config.py +87 -0
  77. langparse/workbooks/modeling/contract.py +628 -0
  78. langparse/workbooks/modeling/disambiguation.py +800 -0
  79. langparse/workbooks/modeling/openai_adapter.py +192 -0
  80. langparse/workbooks/modeling/policy.py +79 -0
  81. langparse/workbooks/modeling/ports.py +44 -0
  82. langparse/workbooks/modeling/pricing.py +17 -0
  83. langparse/workbooks/modeling/types.py +251 -0
  84. langparse/workbooks/objects.py +229 -0
  85. langparse/workbooks/quality/__init__.py +23 -0
  86. langparse/workbooks/quality/bundle.py +53 -0
  87. langparse/workbooks/quality/evaluator.py +266 -0
  88. langparse/workbooks/quality/facts.py +142 -0
  89. langparse/workbooks/quality/schema.py +462 -0
  90. langparse/workbooks/reference_types.py +73 -0
  91. langparse/workbooks/references.py +178 -0
  92. langparse/workbooks/regions.py +932 -0
  93. langparse/workbooks/rendering.py +222 -0
  94. langparse/workbooks/tables.py +477 -0
  95. langparse/workbooks/types.py +257 -0
  96. langparse-0.1.0.dist-info/METADATA +790 -0
  97. langparse-0.1.0.dist-info/RECORD +101 -0
  98. langparse-0.1.0.dist-info/WHEEL +5 -0
  99. langparse-0.1.0.dist-info/entry_points.txt +2 -0
  100. langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
  101. langparse-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,563 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ import json
5
+ import os
6
+ import shutil
7
+ import uuid
8
+ from dataclasses import asdict, dataclass
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+ from langparse.workbooks.adapters import OOXMLWorkbookAdapter
13
+ from langparse.workbooks.assembly import assemble_workbook
14
+ from langparse.workbooks.evaluation.evaluator import (
15
+ CaseEvaluationDetail,
16
+ CaseObservation,
17
+ CaseTruth,
18
+ WorkbookEvaluationMetrics,
19
+ assess_production_readiness,
20
+ classify_case_evaluation,
21
+ evaluate_workbook_ambiguity,
22
+ )
23
+ from langparse.workbooks.evaluation.schema import (
24
+ GoldenSetDriftError,
25
+ WorkbookEvaluationError,
26
+ compute_choices_digest,
27
+ compute_evaluation_id,
28
+ load_golden_set_manifest,
29
+ validate_output_dir_isolation,
30
+ )
31
+ from langparse.workbooks.modeling.policy import WorkbookDisambiguation
32
+ from langparse.workbooks.modeling.ports import WorkbookStructureModelAdapter
33
+ from langparse.workbooks.modeling.types import (
34
+ REGION_PRIVACY_VERSION,
35
+ REGION_PROMPT_VERSION,
36
+ REGION_RULE_VERSION,
37
+ REGION_SCHEMA_VERSION,
38
+ REGION_VALIDATOR_VERSION,
39
+ ModelIdentity,
40
+ ProviderReply,
41
+ RegionChoice,
42
+ WorkbookModelRequest,
43
+ )
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class _CapturedCase:
48
+ case_id: str
49
+ sheet_name: str
50
+ source_range: str
51
+ fact_digest: str
52
+ fallback_choice_id: str
53
+ choices: tuple[RegionChoice, ...]
54
+
55
+
56
+ class _EvaluationCaptureAdapter(WorkbookStructureModelAdapter):
57
+ """Zero-network in-memory capture adapter for golden set evaluation."""
58
+
59
+ def __init__(self) -> None:
60
+ self.captured_cases: list[_CapturedCase] = []
61
+ self._identity = ModelIdentity(
62
+ provider="evaluation",
63
+ model="capture",
64
+ revision="1",
65
+ )
66
+
67
+ @property
68
+ def identity(self) -> ModelIdentity:
69
+ return self._identity
70
+
71
+ def complete(
72
+ self,
73
+ request: WorkbookModelRequest,
74
+ *,
75
+ timeout_seconds: float,
76
+ ) -> ProviderReply:
77
+ payload = json.loads(request.body.decode("utf-8"))
78
+ for raw_case in payload.get("cases", []):
79
+ choices = tuple(
80
+ RegionChoice(
81
+ choice_id=c["choice_id"],
82
+ kind=c["kind"],
83
+ local_score=float(c["local_score"]),
84
+ reason_codes=tuple(c["reason_codes"]),
85
+ )
86
+ for c in raw_case["choices"]
87
+ )
88
+ self.captured_cases.append(
89
+ _CapturedCase(
90
+ case_id=raw_case["case_id"],
91
+ sheet_name=raw_case["sheet_name"],
92
+ source_range=raw_case["source_range"],
93
+ fact_digest=raw_case["fact_digest"],
94
+ fallback_choice_id=raw_case["fallback_choice_id"],
95
+ choices=choices,
96
+ )
97
+ )
98
+
99
+ decisions = [
100
+ {
101
+ "case_id": case_id,
102
+ "status": "abstained",
103
+ "confidence": 0.0,
104
+ "reason_codes": ["evaluation_capture"],
105
+ }
106
+ for case_id in request.case_ids
107
+ ]
108
+ reply_dict = {
109
+ "schema_version": request.schema_version,
110
+ "request_checksum": request.request_checksum,
111
+ "decisions": decisions,
112
+ }
113
+ reply_bytes = json.dumps(reply_dict, separators=(",", ":")).encode("utf-8")
114
+ return ProviderReply(
115
+ body=reply_bytes,
116
+ provider_request_id="eval-capture",
117
+ usage={"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0},
118
+ )
119
+
120
+
121
+ _UNRESOLVED_AUDIT_OUTCOMES = frozenset(
122
+ {
123
+ "case_limit_exceeded",
124
+ "case_unavailable",
125
+ "cell_limit_exceeded",
126
+ "hidden_sheet",
127
+ "kill_switch_activated",
128
+ "limit_exceeded",
129
+ "quota_exceeded",
130
+ "request_too_large",
131
+ }
132
+ )
133
+
134
+
135
+ def _observation_from_final_audit(
136
+ case: _CapturedCase,
137
+ audit: dict[str, object] | None,
138
+ ) -> tuple[str, str | None]:
139
+ """Translate only the production assembly's final audit into evaluation evidence."""
140
+
141
+ if audit is None:
142
+ return "unresolved", None
143
+ outcome = audit.get("outcome")
144
+ if outcome == "accepted":
145
+ selected_choice_id = audit.get("selected_choice_id")
146
+ selected_kind = next(
147
+ (choice.kind for choice in case.choices if choice.choice_id == selected_choice_id),
148
+ None,
149
+ )
150
+ return ("selected", selected_kind) if selected_kind is not None else ("failed", None)
151
+ if outcome == "abstained":
152
+ return "abstained", None
153
+ if outcome in _UNRESOLVED_AUDIT_OUTCOMES:
154
+ return "unresolved", None
155
+ return "failed", None
156
+
157
+
158
+ @dataclass(frozen=True)
159
+ class WorkbookEvaluationReport:
160
+ run_digest: str
161
+ output_path: Path
162
+ metrics: WorkbookEvaluationMetrics
163
+ results: tuple[CaseEvaluationDetail, ...]
164
+ summary: dict[str, Any]
165
+
166
+
167
+ def _compute_sha256(path: Path) -> str:
168
+ hasher = hashlib.sha256()
169
+ with path.open("rb") as f:
170
+ while chunk := f.read(65536):
171
+ hasher.update(chunk)
172
+ return f"sha256:{hasher.hexdigest()}"
173
+
174
+
175
+ def _get_stat_tuple(path: Path) -> os.stat_result:
176
+ return path.stat()
177
+
178
+
179
+ def _render_summary_markdown(summary: dict[str, Any]) -> str:
180
+ lines = [
181
+ f"# Workbook Ambiguity Evaluation Summary ({summary['dataset_id']})",
182
+ "",
183
+ f"- **Dataset Version**: {summary['dataset_version']}",
184
+ f"- **Split**: {summary['split']}",
185
+ f"- **Status**: {summary['status']}",
186
+ f"- **Effectiveness Evidence**: {summary['effectiveness_evidence']}",
187
+ f"- **Production Ready**: {summary.get('production_ready', False)}",
188
+ f"- **Verdict Reasons**: {', '.join(summary.get('verdict_reasons', [])) or 'none'}",
189
+ f"- **Run Digest**: `{summary['run_digest']}`",
190
+ f"- **Dataset Digest**: `{summary['dataset_digest']}`",
191
+ f"- **Model Prompt Version**: `{summary['model_contract']['prompt_version']}`",
192
+ "",
193
+ "## Metrics",
194
+ "",
195
+ "| Metric | Value |",
196
+ "| :--- | :--- |",
197
+ f"| Ambiguous Cases | {summary['ambiguous_case_count']} |",
198
+ f"| Resolvable Cases | {summary['resolvable_case_count']} |",
199
+ f"| Baseline Correct | {summary['baseline_correct_count']} |",
200
+ f"| Baseline Wrong | {summary['baseline_wrong_count']} |",
201
+ f"| Baseline Accuracy | {summary['baseline_accuracy']} |",
202
+ f"| Model Selected | {summary['model_selected_count']} |",
203
+ f"| Model Correct Acceptance | {summary['model_correct_acceptance_count']} |",
204
+ f"| Model Wrong Acceptance | {summary['model_wrong_acceptance_count']} |",
205
+ f"| Model Selection Accuracy | {summary['model_selection_accuracy']} |",
206
+ f"| Wrong Acceptance Rate | {summary['wrong_acceptance_rate']} |",
207
+ f"| Model Coverage | {summary['model_coverage']} |",
208
+ f"| Fixed Errors | {summary['fixed_error_count']} |",
209
+ f"| Introduced Errors | {summary['introduced_error_count']} |",
210
+ f"| Net Correct Delta | {summary['net_correct_delta']} |",
211
+ f"| Clear Samples | {summary['clear_sample_count']} |",
212
+ f"| Clear Unexpected Calls | {summary['clear_unexpected_call_count']} |",
213
+ f"| Clear Call Rate | {summary['clear_call_rate']} |",
214
+ "",
215
+ ]
216
+ return "\n".join(lines)
217
+
218
+
219
+ def _report_directories_match(existing: Path, candidate: Path) -> bool:
220
+ existing_files = {
221
+ path.relative_to(existing): path
222
+ for path in existing.rglob("*")
223
+ if path.is_file() and not path.is_symlink()
224
+ }
225
+ candidate_files = {
226
+ path.relative_to(candidate): path
227
+ for path in candidate.rglob("*")
228
+ if path.is_file() and not path.is_symlink()
229
+ }
230
+ if existing_files.keys() != candidate_files.keys():
231
+ return False
232
+ return all(
233
+ existing_files[relative].read_bytes() == candidate_files[relative].read_bytes()
234
+ for relative in existing_files
235
+ )
236
+
237
+
238
+ def _safe_model_identity_payload(
239
+ adapter: WorkbookStructureModelAdapter | None,
240
+ ) -> dict[str, str | None] | None:
241
+ if adapter is None:
242
+ return None
243
+ identity = adapter.identity
244
+ if not isinstance(identity, ModelIdentity):
245
+ return None
246
+ values = (identity.provider, identity.model, identity.revision)
247
+ if any(
248
+ value is not None
249
+ and (
250
+ type(value) is not str
251
+ or not 0 < len(value) <= 64
252
+ or not value.isascii()
253
+ or not all(character.isalnum() or character in "._-/" for character in value)
254
+ )
255
+ for value in values
256
+ ):
257
+ return None
258
+ return {
259
+ "provider": identity.provider,
260
+ "model": identity.model,
261
+ "revision": identity.revision,
262
+ }
263
+
264
+
265
+ class WorkbookAmbiguityBenchmarkService:
266
+ """Service to evaluate workbook model disambiguation against golden set baselines."""
267
+
268
+ def __init__(self, adapter: OOXMLWorkbookAdapter | None = None) -> None:
269
+ self._ooxml_adapter = adapter or OOXMLWorkbookAdapter()
270
+
271
+ def run(
272
+ self,
273
+ manifest_path: str | Path,
274
+ *,
275
+ output_dir: str | Path,
276
+ markdown: bool = True,
277
+ adapter: WorkbookStructureModelAdapter | None = None,
278
+ model: str | None = None,
279
+ api_key: str | None = None,
280
+ base_url: str | None = None,
281
+ ) -> WorkbookEvaluationReport:
282
+ manifest = load_golden_set_manifest(manifest_path)
283
+ output_dir_path = Path(output_dir).resolve()
284
+ validate_output_dir_isolation(manifest.source_root, output_dir_path)
285
+
286
+ from langparse.workbooks.modeling.openai_adapter import (
287
+ OpenAIWorkbookStructureAdapter,
288
+ )
289
+
290
+ if adapter is None and model is not None:
291
+ adapter = OpenAIWorkbookStructureAdapter.from_env(
292
+ cli_model=model,
293
+ cli_api_key=api_key,
294
+ cli_base_url=base_url,
295
+ )
296
+
297
+ provider_evidence = type(adapter) is OpenAIWorkbookStructureAdapter
298
+
299
+ # Pre-execution stat and hash check
300
+ initial_stats: dict[str, tuple[os.stat_result, str]] = {}
301
+ for sample in manifest.samples:
302
+ full_path = (manifest.source_root / sample.path).resolve()
303
+ initial_stats[sample.sample_id] = (
304
+ _get_stat_tuple(full_path),
305
+ _compute_sha256(full_path),
306
+ )
307
+
308
+ truths: list[CaseTruth] = []
309
+ observations: list[CaseObservation] = [] if adapter is not None else None
310
+ clear_sample_count = 0
311
+ clear_unexpected_call_count = 0
312
+
313
+ for sample in manifest.samples:
314
+ full_path = (manifest.source_root / sample.path).resolve()
315
+ capture_adapter = _EvaluationCaptureAdapter()
316
+ snapshot = self._ooxml_adapter.snapshot(full_path)
317
+ assemble_workbook(
318
+ snapshot,
319
+ disambiguation=WorkbookDisambiguation.auto(capture_adapter),
320
+ )
321
+
322
+ if sample.cohort == "clear_no_call":
323
+ clear_sample_count += 1
324
+ if len(capture_adapter.captured_cases) > 0:
325
+ clear_unexpected_call_count += 1
326
+ continue
327
+
328
+ # ambiguous cohort matching
329
+ runtime_cases_by_loc = {
330
+ (c.sheet_name, c.source_range): c for c in capture_adapter.captured_cases
331
+ }
332
+ manifest_cases_by_loc = {(c.sheet_name, c.source_range): c for c in sample.cases}
333
+
334
+ # Drift checks
335
+ missing_locs = set(manifest_cases_by_loc.keys()) - set(runtime_cases_by_loc.keys())
336
+ if missing_locs:
337
+ raise GoldenSetDriftError(
338
+ "Runtime case missing from snapshot",
339
+ code="case_missing",
340
+ )
341
+
342
+ unlabeled_locs = set(runtime_cases_by_loc.keys()) - set(manifest_cases_by_loc.keys())
343
+ if unlabeled_locs:
344
+ raise GoldenSetDriftError(
345
+ "Runtime generated unlabeled ambiguity case",
346
+ code="case_unlabeled",
347
+ )
348
+
349
+ for loc, golden_case in manifest_cases_by_loc.items():
350
+ runtime_case = runtime_cases_by_loc[loc]
351
+ if runtime_case.fact_digest != golden_case.fact_digest:
352
+ raise GoldenSetDriftError(
353
+ "Case fact digest changed",
354
+ code="facts_changed",
355
+ )
356
+
357
+ runtime_choices_digest = compute_choices_digest(runtime_case.choices)
358
+ if runtime_choices_digest != golden_case.choices_digest:
359
+ raise GoldenSetDriftError(
360
+ "Case choices digest changed",
361
+ code="choices_changed",
362
+ )
363
+
364
+ # Extract baseline kind from fallback choice
365
+ fallback_choice = next(
366
+ (
367
+ c
368
+ for c in runtime_case.choices
369
+ if c.choice_id == runtime_case.fallback_choice_id
370
+ ),
371
+ None,
372
+ )
373
+ baseline_kind = fallback_choice.kind if fallback_choice is not None else None
374
+
375
+ expected_kind = (
376
+ None if golden_case.expected == "unresolvable" else golden_case.expected
377
+ )
378
+ eval_id = compute_evaluation_id(
379
+ manifest.dataset_id,
380
+ manifest.dataset_version,
381
+ sample.sample_id,
382
+ golden_case.label_id,
383
+ )
384
+ truths.append(
385
+ CaseTruth(
386
+ evaluation_id=eval_id,
387
+ cohort="ambiguous",
388
+ expected_kind=expected_kind,
389
+ baseline_kind=baseline_kind,
390
+ )
391
+ )
392
+
393
+ # If adapter provided, run live observation pass
394
+ if adapter is not None:
395
+ _, live_diagnostics = assemble_workbook(
396
+ snapshot,
397
+ disambiguation=WorkbookDisambiguation.auto(adapter),
398
+ )
399
+ audits_by_case_id = {
400
+ audit["case_id"]: audit
401
+ for audit in live_diagnostics.model_calls
402
+ if isinstance(audit.get("case_id"), str)
403
+ }
404
+ for loc, golden_case in manifest_cases_by_loc.items():
405
+ runtime_case = runtime_cases_by_loc[loc]
406
+ eval_id = compute_evaluation_id(
407
+ manifest.dataset_id,
408
+ manifest.dataset_version,
409
+ sample.sample_id,
410
+ golden_case.label_id,
411
+ )
412
+ status, selected_kind = _observation_from_final_audit(
413
+ runtime_case,
414
+ audits_by_case_id.get(runtime_case.case_id),
415
+ )
416
+ observations.append(
417
+ CaseObservation(
418
+ evaluation_id=eval_id,
419
+ evidence_class="provider" if provider_evidence else "harness_only",
420
+ status=status,
421
+ selected_kind=selected_kind,
422
+ )
423
+ )
424
+
425
+ # Post-execution stat and hash verification
426
+ for sample in manifest.samples:
427
+ full_path = (manifest.source_root / sample.path).resolve()
428
+ current_stat = _get_stat_tuple(full_path)
429
+ current_hash = _compute_sha256(full_path)
430
+ initial_stat, initial_hash = initial_stats[sample.sample_id]
431
+ if current_stat != initial_stat or current_hash != initial_hash:
432
+ raise GoldenSetDriftError(
433
+ "Source file mutated during evaluation execution",
434
+ code="source_file_mutated",
435
+ )
436
+
437
+ metrics = evaluate_workbook_ambiguity(
438
+ tuple(truths),
439
+ observations=tuple(observations) if observations is not None else None,
440
+ clear_sample_count=clear_sample_count,
441
+ clear_unexpected_call_count=clear_unexpected_call_count,
442
+ )
443
+
444
+ if observations is not None:
445
+ obs_map = {o.evaluation_id: o for o in observations}
446
+ results = tuple(
447
+ classify_case_evaluation(truth, obs_map[truth.evaluation_id]) for truth in truths
448
+ )
449
+ evidence_class = "provider" if metrics.effectiveness_evidence else "harness_only"
450
+ else:
451
+ results = tuple(classify_case_evaluation(truth, None) for truth in truths)
452
+ evidence_class = "none"
453
+
454
+ # Compute deterministic run digest
455
+ model_identity = _safe_model_identity_payload(adapter)
456
+ model_contract = {
457
+ "schema_version": REGION_SCHEMA_VERSION,
458
+ "prompt_version": REGION_PROMPT_VERSION,
459
+ "privacy_version": REGION_PRIVACY_VERSION,
460
+ "rule_version": REGION_RULE_VERSION,
461
+ "validator_version": REGION_VALIDATOR_VERSION,
462
+ }
463
+ run_payload = {
464
+ "dataset_digest": manifest.dataset_digest,
465
+ "evidence_class": evidence_class,
466
+ "model_identity": model_identity,
467
+ "model_contract": model_contract,
468
+ "metrics": asdict(metrics),
469
+ "results": [asdict(r) for r in results],
470
+ }
471
+ serialized_run = json.dumps(run_payload, sort_keys=True, separators=(",", ":")).encode(
472
+ "utf-8"
473
+ )
474
+ run_digest = f"sha256:{hashlib.sha256(serialized_run).hexdigest()}"
475
+
476
+ production_ready, verdict_reasons = assess_production_readiness(
477
+ metrics,
478
+ split=manifest.split,
479
+ operational_evidence=False,
480
+ )
481
+ summary_dict = {
482
+ "schema_version": 1,
483
+ "dataset_id": manifest.dataset_id,
484
+ "dataset_version": manifest.dataset_version,
485
+ "split": manifest.split,
486
+ "dataset_digest": manifest.dataset_digest,
487
+ "run_digest": run_digest,
488
+ "status": "valid",
489
+ "effectiveness_evidence": metrics.effectiveness_evidence,
490
+ "model_identity": model_identity,
491
+ "model_contract": model_contract,
492
+ "production_ready": production_ready,
493
+ "verdict_reasons": list(verdict_reasons),
494
+ **asdict(metrics),
495
+ }
496
+
497
+ # Atomic publishing
498
+ output_dir_path.mkdir(parents=True, exist_ok=True)
499
+ tmp_run_dir = output_dir_path / f".tmp-{uuid.uuid4().hex}"
500
+ tmp_run_dir.mkdir(parents=True, exist_ok=False)
501
+
502
+ try:
503
+ # Write results.jsonl (sanitized rows)
504
+ results_path = tmp_run_dir / "workbook-ambiguity-results.jsonl"
505
+ with results_path.open("w", encoding="utf-8") as f:
506
+ for res in results:
507
+ row = {
508
+ "run_digest": run_digest,
509
+ "evaluation_id": res.evaluation_id,
510
+ "cohort": res.cohort,
511
+ "expected_kind": res.expected_kind,
512
+ "baseline_kind": res.baseline_kind,
513
+ "baseline_correct": res.baseline_correct,
514
+ "observation_status": res.observation_status,
515
+ "observed_kind": res.observed_kind,
516
+ "transition": res.transition,
517
+ }
518
+ f.write(json.dumps(row, separators=(",", ":")) + "\n")
519
+
520
+ # Write summary.json
521
+ summary_path = tmp_run_dir / "workbook-ambiguity-summary.json"
522
+ summary_path.write_text(
523
+ json.dumps(summary_dict, indent=2, sort_keys=True),
524
+ encoding="utf-8",
525
+ )
526
+
527
+ # Write summary.md if requested
528
+ if markdown:
529
+ summary_md_path = tmp_run_dir / "workbook-ambiguity-summary.md"
530
+ summary_md_path.write_text(
531
+ _render_summary_markdown(summary_dict),
532
+ encoding="utf-8",
533
+ )
534
+
535
+ final_run_dir = output_dir_path / run_digest.removeprefix("sha256:")
536
+
537
+ if final_run_dir.exists():
538
+ if _report_directories_match(final_run_dir, tmp_run_dir):
539
+ shutil.rmtree(tmp_run_dir, ignore_errors=True)
540
+ return WorkbookEvaluationReport(
541
+ run_digest=run_digest,
542
+ output_path=final_run_dir,
543
+ metrics=metrics,
544
+ results=results,
545
+ summary=summary_dict,
546
+ )
547
+ shutil.rmtree(tmp_run_dir, ignore_errors=True)
548
+ raise WorkbookEvaluationError(
549
+ "Output run directory conflict with different content",
550
+ code="report_conflict",
551
+ )
552
+
553
+ os.replace(tmp_run_dir, final_run_dir)
554
+ return WorkbookEvaluationReport(
555
+ run_digest=run_digest,
556
+ output_path=final_run_dir,
557
+ metrics=metrics,
558
+ results=results,
559
+ summary=summary_dict,
560
+ )
561
+ except Exception:
562
+ shutil.rmtree(tmp_run_dir, ignore_errors=True)
563
+ raise