langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,563 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import json
|
|
5
|
+
import os
|
|
6
|
+
import shutil
|
|
7
|
+
import uuid
|
|
8
|
+
from dataclasses import asdict, dataclass
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any
|
|
11
|
+
|
|
12
|
+
from langparse.workbooks.adapters import OOXMLWorkbookAdapter
|
|
13
|
+
from langparse.workbooks.assembly import assemble_workbook
|
|
14
|
+
from langparse.workbooks.evaluation.evaluator import (
|
|
15
|
+
CaseEvaluationDetail,
|
|
16
|
+
CaseObservation,
|
|
17
|
+
CaseTruth,
|
|
18
|
+
WorkbookEvaluationMetrics,
|
|
19
|
+
assess_production_readiness,
|
|
20
|
+
classify_case_evaluation,
|
|
21
|
+
evaluate_workbook_ambiguity,
|
|
22
|
+
)
|
|
23
|
+
from langparse.workbooks.evaluation.schema import (
|
|
24
|
+
GoldenSetDriftError,
|
|
25
|
+
WorkbookEvaluationError,
|
|
26
|
+
compute_choices_digest,
|
|
27
|
+
compute_evaluation_id,
|
|
28
|
+
load_golden_set_manifest,
|
|
29
|
+
validate_output_dir_isolation,
|
|
30
|
+
)
|
|
31
|
+
from langparse.workbooks.modeling.policy import WorkbookDisambiguation
|
|
32
|
+
from langparse.workbooks.modeling.ports import WorkbookStructureModelAdapter
|
|
33
|
+
from langparse.workbooks.modeling.types import (
|
|
34
|
+
REGION_PRIVACY_VERSION,
|
|
35
|
+
REGION_PROMPT_VERSION,
|
|
36
|
+
REGION_RULE_VERSION,
|
|
37
|
+
REGION_SCHEMA_VERSION,
|
|
38
|
+
REGION_VALIDATOR_VERSION,
|
|
39
|
+
ModelIdentity,
|
|
40
|
+
ProviderReply,
|
|
41
|
+
RegionChoice,
|
|
42
|
+
WorkbookModelRequest,
|
|
43
|
+
)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class _CapturedCase:
|
|
48
|
+
case_id: str
|
|
49
|
+
sheet_name: str
|
|
50
|
+
source_range: str
|
|
51
|
+
fact_digest: str
|
|
52
|
+
fallback_choice_id: str
|
|
53
|
+
choices: tuple[RegionChoice, ...]
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class _EvaluationCaptureAdapter(WorkbookStructureModelAdapter):
|
|
57
|
+
"""Zero-network in-memory capture adapter for golden set evaluation."""
|
|
58
|
+
|
|
59
|
+
def __init__(self) -> None:
|
|
60
|
+
self.captured_cases: list[_CapturedCase] = []
|
|
61
|
+
self._identity = ModelIdentity(
|
|
62
|
+
provider="evaluation",
|
|
63
|
+
model="capture",
|
|
64
|
+
revision="1",
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
@property
|
|
68
|
+
def identity(self) -> ModelIdentity:
|
|
69
|
+
return self._identity
|
|
70
|
+
|
|
71
|
+
def complete(
|
|
72
|
+
self,
|
|
73
|
+
request: WorkbookModelRequest,
|
|
74
|
+
*,
|
|
75
|
+
timeout_seconds: float,
|
|
76
|
+
) -> ProviderReply:
|
|
77
|
+
payload = json.loads(request.body.decode("utf-8"))
|
|
78
|
+
for raw_case in payload.get("cases", []):
|
|
79
|
+
choices = tuple(
|
|
80
|
+
RegionChoice(
|
|
81
|
+
choice_id=c["choice_id"],
|
|
82
|
+
kind=c["kind"],
|
|
83
|
+
local_score=float(c["local_score"]),
|
|
84
|
+
reason_codes=tuple(c["reason_codes"]),
|
|
85
|
+
)
|
|
86
|
+
for c in raw_case["choices"]
|
|
87
|
+
)
|
|
88
|
+
self.captured_cases.append(
|
|
89
|
+
_CapturedCase(
|
|
90
|
+
case_id=raw_case["case_id"],
|
|
91
|
+
sheet_name=raw_case["sheet_name"],
|
|
92
|
+
source_range=raw_case["source_range"],
|
|
93
|
+
fact_digest=raw_case["fact_digest"],
|
|
94
|
+
fallback_choice_id=raw_case["fallback_choice_id"],
|
|
95
|
+
choices=choices,
|
|
96
|
+
)
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
decisions = [
|
|
100
|
+
{
|
|
101
|
+
"case_id": case_id,
|
|
102
|
+
"status": "abstained",
|
|
103
|
+
"confidence": 0.0,
|
|
104
|
+
"reason_codes": ["evaluation_capture"],
|
|
105
|
+
}
|
|
106
|
+
for case_id in request.case_ids
|
|
107
|
+
]
|
|
108
|
+
reply_dict = {
|
|
109
|
+
"schema_version": request.schema_version,
|
|
110
|
+
"request_checksum": request.request_checksum,
|
|
111
|
+
"decisions": decisions,
|
|
112
|
+
}
|
|
113
|
+
reply_bytes = json.dumps(reply_dict, separators=(",", ":")).encode("utf-8")
|
|
114
|
+
return ProviderReply(
|
|
115
|
+
body=reply_bytes,
|
|
116
|
+
provider_request_id="eval-capture",
|
|
117
|
+
usage={"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0},
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
_UNRESOLVED_AUDIT_OUTCOMES = frozenset(
|
|
122
|
+
{
|
|
123
|
+
"case_limit_exceeded",
|
|
124
|
+
"case_unavailable",
|
|
125
|
+
"cell_limit_exceeded",
|
|
126
|
+
"hidden_sheet",
|
|
127
|
+
"kill_switch_activated",
|
|
128
|
+
"limit_exceeded",
|
|
129
|
+
"quota_exceeded",
|
|
130
|
+
"request_too_large",
|
|
131
|
+
}
|
|
132
|
+
)
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def _observation_from_final_audit(
|
|
136
|
+
case: _CapturedCase,
|
|
137
|
+
audit: dict[str, object] | None,
|
|
138
|
+
) -> tuple[str, str | None]:
|
|
139
|
+
"""Translate only the production assembly's final audit into evaluation evidence."""
|
|
140
|
+
|
|
141
|
+
if audit is None:
|
|
142
|
+
return "unresolved", None
|
|
143
|
+
outcome = audit.get("outcome")
|
|
144
|
+
if outcome == "accepted":
|
|
145
|
+
selected_choice_id = audit.get("selected_choice_id")
|
|
146
|
+
selected_kind = next(
|
|
147
|
+
(choice.kind for choice in case.choices if choice.choice_id == selected_choice_id),
|
|
148
|
+
None,
|
|
149
|
+
)
|
|
150
|
+
return ("selected", selected_kind) if selected_kind is not None else ("failed", None)
|
|
151
|
+
if outcome == "abstained":
|
|
152
|
+
return "abstained", None
|
|
153
|
+
if outcome in _UNRESOLVED_AUDIT_OUTCOMES:
|
|
154
|
+
return "unresolved", None
|
|
155
|
+
return "failed", None
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
@dataclass(frozen=True)
|
|
159
|
+
class WorkbookEvaluationReport:
|
|
160
|
+
run_digest: str
|
|
161
|
+
output_path: Path
|
|
162
|
+
metrics: WorkbookEvaluationMetrics
|
|
163
|
+
results: tuple[CaseEvaluationDetail, ...]
|
|
164
|
+
summary: dict[str, Any]
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _compute_sha256(path: Path) -> str:
|
|
168
|
+
hasher = hashlib.sha256()
|
|
169
|
+
with path.open("rb") as f:
|
|
170
|
+
while chunk := f.read(65536):
|
|
171
|
+
hasher.update(chunk)
|
|
172
|
+
return f"sha256:{hasher.hexdigest()}"
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _get_stat_tuple(path: Path) -> os.stat_result:
|
|
176
|
+
return path.stat()
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _render_summary_markdown(summary: dict[str, Any]) -> str:
|
|
180
|
+
lines = [
|
|
181
|
+
f"# Workbook Ambiguity Evaluation Summary ({summary['dataset_id']})",
|
|
182
|
+
"",
|
|
183
|
+
f"- **Dataset Version**: {summary['dataset_version']}",
|
|
184
|
+
f"- **Split**: {summary['split']}",
|
|
185
|
+
f"- **Status**: {summary['status']}",
|
|
186
|
+
f"- **Effectiveness Evidence**: {summary['effectiveness_evidence']}",
|
|
187
|
+
f"- **Production Ready**: {summary.get('production_ready', False)}",
|
|
188
|
+
f"- **Verdict Reasons**: {', '.join(summary.get('verdict_reasons', [])) or 'none'}",
|
|
189
|
+
f"- **Run Digest**: `{summary['run_digest']}`",
|
|
190
|
+
f"- **Dataset Digest**: `{summary['dataset_digest']}`",
|
|
191
|
+
f"- **Model Prompt Version**: `{summary['model_contract']['prompt_version']}`",
|
|
192
|
+
"",
|
|
193
|
+
"## Metrics",
|
|
194
|
+
"",
|
|
195
|
+
"| Metric | Value |",
|
|
196
|
+
"| :--- | :--- |",
|
|
197
|
+
f"| Ambiguous Cases | {summary['ambiguous_case_count']} |",
|
|
198
|
+
f"| Resolvable Cases | {summary['resolvable_case_count']} |",
|
|
199
|
+
f"| Baseline Correct | {summary['baseline_correct_count']} |",
|
|
200
|
+
f"| Baseline Wrong | {summary['baseline_wrong_count']} |",
|
|
201
|
+
f"| Baseline Accuracy | {summary['baseline_accuracy']} |",
|
|
202
|
+
f"| Model Selected | {summary['model_selected_count']} |",
|
|
203
|
+
f"| Model Correct Acceptance | {summary['model_correct_acceptance_count']} |",
|
|
204
|
+
f"| Model Wrong Acceptance | {summary['model_wrong_acceptance_count']} |",
|
|
205
|
+
f"| Model Selection Accuracy | {summary['model_selection_accuracy']} |",
|
|
206
|
+
f"| Wrong Acceptance Rate | {summary['wrong_acceptance_rate']} |",
|
|
207
|
+
f"| Model Coverage | {summary['model_coverage']} |",
|
|
208
|
+
f"| Fixed Errors | {summary['fixed_error_count']} |",
|
|
209
|
+
f"| Introduced Errors | {summary['introduced_error_count']} |",
|
|
210
|
+
f"| Net Correct Delta | {summary['net_correct_delta']} |",
|
|
211
|
+
f"| Clear Samples | {summary['clear_sample_count']} |",
|
|
212
|
+
f"| Clear Unexpected Calls | {summary['clear_unexpected_call_count']} |",
|
|
213
|
+
f"| Clear Call Rate | {summary['clear_call_rate']} |",
|
|
214
|
+
"",
|
|
215
|
+
]
|
|
216
|
+
return "\n".join(lines)
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _report_directories_match(existing: Path, candidate: Path) -> bool:
|
|
220
|
+
existing_files = {
|
|
221
|
+
path.relative_to(existing): path
|
|
222
|
+
for path in existing.rglob("*")
|
|
223
|
+
if path.is_file() and not path.is_symlink()
|
|
224
|
+
}
|
|
225
|
+
candidate_files = {
|
|
226
|
+
path.relative_to(candidate): path
|
|
227
|
+
for path in candidate.rglob("*")
|
|
228
|
+
if path.is_file() and not path.is_symlink()
|
|
229
|
+
}
|
|
230
|
+
if existing_files.keys() != candidate_files.keys():
|
|
231
|
+
return False
|
|
232
|
+
return all(
|
|
233
|
+
existing_files[relative].read_bytes() == candidate_files[relative].read_bytes()
|
|
234
|
+
for relative in existing_files
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
def _safe_model_identity_payload(
|
|
239
|
+
adapter: WorkbookStructureModelAdapter | None,
|
|
240
|
+
) -> dict[str, str | None] | None:
|
|
241
|
+
if adapter is None:
|
|
242
|
+
return None
|
|
243
|
+
identity = adapter.identity
|
|
244
|
+
if not isinstance(identity, ModelIdentity):
|
|
245
|
+
return None
|
|
246
|
+
values = (identity.provider, identity.model, identity.revision)
|
|
247
|
+
if any(
|
|
248
|
+
value is not None
|
|
249
|
+
and (
|
|
250
|
+
type(value) is not str
|
|
251
|
+
or not 0 < len(value) <= 64
|
|
252
|
+
or not value.isascii()
|
|
253
|
+
or not all(character.isalnum() or character in "._-/" for character in value)
|
|
254
|
+
)
|
|
255
|
+
for value in values
|
|
256
|
+
):
|
|
257
|
+
return None
|
|
258
|
+
return {
|
|
259
|
+
"provider": identity.provider,
|
|
260
|
+
"model": identity.model,
|
|
261
|
+
"revision": identity.revision,
|
|
262
|
+
}
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
class WorkbookAmbiguityBenchmarkService:
|
|
266
|
+
"""Service to evaluate workbook model disambiguation against golden set baselines."""
|
|
267
|
+
|
|
268
|
+
def __init__(self, adapter: OOXMLWorkbookAdapter | None = None) -> None:
|
|
269
|
+
self._ooxml_adapter = adapter or OOXMLWorkbookAdapter()
|
|
270
|
+
|
|
271
|
+
def run(
|
|
272
|
+
self,
|
|
273
|
+
manifest_path: str | Path,
|
|
274
|
+
*,
|
|
275
|
+
output_dir: str | Path,
|
|
276
|
+
markdown: bool = True,
|
|
277
|
+
adapter: WorkbookStructureModelAdapter | None = None,
|
|
278
|
+
model: str | None = None,
|
|
279
|
+
api_key: str | None = None,
|
|
280
|
+
base_url: str | None = None,
|
|
281
|
+
) -> WorkbookEvaluationReport:
|
|
282
|
+
manifest = load_golden_set_manifest(manifest_path)
|
|
283
|
+
output_dir_path = Path(output_dir).resolve()
|
|
284
|
+
validate_output_dir_isolation(manifest.source_root, output_dir_path)
|
|
285
|
+
|
|
286
|
+
from langparse.workbooks.modeling.openai_adapter import (
|
|
287
|
+
OpenAIWorkbookStructureAdapter,
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
if adapter is None and model is not None:
|
|
291
|
+
adapter = OpenAIWorkbookStructureAdapter.from_env(
|
|
292
|
+
cli_model=model,
|
|
293
|
+
cli_api_key=api_key,
|
|
294
|
+
cli_base_url=base_url,
|
|
295
|
+
)
|
|
296
|
+
|
|
297
|
+
provider_evidence = type(adapter) is OpenAIWorkbookStructureAdapter
|
|
298
|
+
|
|
299
|
+
# Pre-execution stat and hash check
|
|
300
|
+
initial_stats: dict[str, tuple[os.stat_result, str]] = {}
|
|
301
|
+
for sample in manifest.samples:
|
|
302
|
+
full_path = (manifest.source_root / sample.path).resolve()
|
|
303
|
+
initial_stats[sample.sample_id] = (
|
|
304
|
+
_get_stat_tuple(full_path),
|
|
305
|
+
_compute_sha256(full_path),
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
truths: list[CaseTruth] = []
|
|
309
|
+
observations: list[CaseObservation] = [] if adapter is not None else None
|
|
310
|
+
clear_sample_count = 0
|
|
311
|
+
clear_unexpected_call_count = 0
|
|
312
|
+
|
|
313
|
+
for sample in manifest.samples:
|
|
314
|
+
full_path = (manifest.source_root / sample.path).resolve()
|
|
315
|
+
capture_adapter = _EvaluationCaptureAdapter()
|
|
316
|
+
snapshot = self._ooxml_adapter.snapshot(full_path)
|
|
317
|
+
assemble_workbook(
|
|
318
|
+
snapshot,
|
|
319
|
+
disambiguation=WorkbookDisambiguation.auto(capture_adapter),
|
|
320
|
+
)
|
|
321
|
+
|
|
322
|
+
if sample.cohort == "clear_no_call":
|
|
323
|
+
clear_sample_count += 1
|
|
324
|
+
if len(capture_adapter.captured_cases) > 0:
|
|
325
|
+
clear_unexpected_call_count += 1
|
|
326
|
+
continue
|
|
327
|
+
|
|
328
|
+
# ambiguous cohort matching
|
|
329
|
+
runtime_cases_by_loc = {
|
|
330
|
+
(c.sheet_name, c.source_range): c for c in capture_adapter.captured_cases
|
|
331
|
+
}
|
|
332
|
+
manifest_cases_by_loc = {(c.sheet_name, c.source_range): c for c in sample.cases}
|
|
333
|
+
|
|
334
|
+
# Drift checks
|
|
335
|
+
missing_locs = set(manifest_cases_by_loc.keys()) - set(runtime_cases_by_loc.keys())
|
|
336
|
+
if missing_locs:
|
|
337
|
+
raise GoldenSetDriftError(
|
|
338
|
+
"Runtime case missing from snapshot",
|
|
339
|
+
code="case_missing",
|
|
340
|
+
)
|
|
341
|
+
|
|
342
|
+
unlabeled_locs = set(runtime_cases_by_loc.keys()) - set(manifest_cases_by_loc.keys())
|
|
343
|
+
if unlabeled_locs:
|
|
344
|
+
raise GoldenSetDriftError(
|
|
345
|
+
"Runtime generated unlabeled ambiguity case",
|
|
346
|
+
code="case_unlabeled",
|
|
347
|
+
)
|
|
348
|
+
|
|
349
|
+
for loc, golden_case in manifest_cases_by_loc.items():
|
|
350
|
+
runtime_case = runtime_cases_by_loc[loc]
|
|
351
|
+
if runtime_case.fact_digest != golden_case.fact_digest:
|
|
352
|
+
raise GoldenSetDriftError(
|
|
353
|
+
"Case fact digest changed",
|
|
354
|
+
code="facts_changed",
|
|
355
|
+
)
|
|
356
|
+
|
|
357
|
+
runtime_choices_digest = compute_choices_digest(runtime_case.choices)
|
|
358
|
+
if runtime_choices_digest != golden_case.choices_digest:
|
|
359
|
+
raise GoldenSetDriftError(
|
|
360
|
+
"Case choices digest changed",
|
|
361
|
+
code="choices_changed",
|
|
362
|
+
)
|
|
363
|
+
|
|
364
|
+
# Extract baseline kind from fallback choice
|
|
365
|
+
fallback_choice = next(
|
|
366
|
+
(
|
|
367
|
+
c
|
|
368
|
+
for c in runtime_case.choices
|
|
369
|
+
if c.choice_id == runtime_case.fallback_choice_id
|
|
370
|
+
),
|
|
371
|
+
None,
|
|
372
|
+
)
|
|
373
|
+
baseline_kind = fallback_choice.kind if fallback_choice is not None else None
|
|
374
|
+
|
|
375
|
+
expected_kind = (
|
|
376
|
+
None if golden_case.expected == "unresolvable" else golden_case.expected
|
|
377
|
+
)
|
|
378
|
+
eval_id = compute_evaluation_id(
|
|
379
|
+
manifest.dataset_id,
|
|
380
|
+
manifest.dataset_version,
|
|
381
|
+
sample.sample_id,
|
|
382
|
+
golden_case.label_id,
|
|
383
|
+
)
|
|
384
|
+
truths.append(
|
|
385
|
+
CaseTruth(
|
|
386
|
+
evaluation_id=eval_id,
|
|
387
|
+
cohort="ambiguous",
|
|
388
|
+
expected_kind=expected_kind,
|
|
389
|
+
baseline_kind=baseline_kind,
|
|
390
|
+
)
|
|
391
|
+
)
|
|
392
|
+
|
|
393
|
+
# If adapter provided, run live observation pass
|
|
394
|
+
if adapter is not None:
|
|
395
|
+
_, live_diagnostics = assemble_workbook(
|
|
396
|
+
snapshot,
|
|
397
|
+
disambiguation=WorkbookDisambiguation.auto(adapter),
|
|
398
|
+
)
|
|
399
|
+
audits_by_case_id = {
|
|
400
|
+
audit["case_id"]: audit
|
|
401
|
+
for audit in live_diagnostics.model_calls
|
|
402
|
+
if isinstance(audit.get("case_id"), str)
|
|
403
|
+
}
|
|
404
|
+
for loc, golden_case in manifest_cases_by_loc.items():
|
|
405
|
+
runtime_case = runtime_cases_by_loc[loc]
|
|
406
|
+
eval_id = compute_evaluation_id(
|
|
407
|
+
manifest.dataset_id,
|
|
408
|
+
manifest.dataset_version,
|
|
409
|
+
sample.sample_id,
|
|
410
|
+
golden_case.label_id,
|
|
411
|
+
)
|
|
412
|
+
status, selected_kind = _observation_from_final_audit(
|
|
413
|
+
runtime_case,
|
|
414
|
+
audits_by_case_id.get(runtime_case.case_id),
|
|
415
|
+
)
|
|
416
|
+
observations.append(
|
|
417
|
+
CaseObservation(
|
|
418
|
+
evaluation_id=eval_id,
|
|
419
|
+
evidence_class="provider" if provider_evidence else "harness_only",
|
|
420
|
+
status=status,
|
|
421
|
+
selected_kind=selected_kind,
|
|
422
|
+
)
|
|
423
|
+
)
|
|
424
|
+
|
|
425
|
+
# Post-execution stat and hash verification
|
|
426
|
+
for sample in manifest.samples:
|
|
427
|
+
full_path = (manifest.source_root / sample.path).resolve()
|
|
428
|
+
current_stat = _get_stat_tuple(full_path)
|
|
429
|
+
current_hash = _compute_sha256(full_path)
|
|
430
|
+
initial_stat, initial_hash = initial_stats[sample.sample_id]
|
|
431
|
+
if current_stat != initial_stat or current_hash != initial_hash:
|
|
432
|
+
raise GoldenSetDriftError(
|
|
433
|
+
"Source file mutated during evaluation execution",
|
|
434
|
+
code="source_file_mutated",
|
|
435
|
+
)
|
|
436
|
+
|
|
437
|
+
metrics = evaluate_workbook_ambiguity(
|
|
438
|
+
tuple(truths),
|
|
439
|
+
observations=tuple(observations) if observations is not None else None,
|
|
440
|
+
clear_sample_count=clear_sample_count,
|
|
441
|
+
clear_unexpected_call_count=clear_unexpected_call_count,
|
|
442
|
+
)
|
|
443
|
+
|
|
444
|
+
if observations is not None:
|
|
445
|
+
obs_map = {o.evaluation_id: o for o in observations}
|
|
446
|
+
results = tuple(
|
|
447
|
+
classify_case_evaluation(truth, obs_map[truth.evaluation_id]) for truth in truths
|
|
448
|
+
)
|
|
449
|
+
evidence_class = "provider" if metrics.effectiveness_evidence else "harness_only"
|
|
450
|
+
else:
|
|
451
|
+
results = tuple(classify_case_evaluation(truth, None) for truth in truths)
|
|
452
|
+
evidence_class = "none"
|
|
453
|
+
|
|
454
|
+
# Compute deterministic run digest
|
|
455
|
+
model_identity = _safe_model_identity_payload(adapter)
|
|
456
|
+
model_contract = {
|
|
457
|
+
"schema_version": REGION_SCHEMA_VERSION,
|
|
458
|
+
"prompt_version": REGION_PROMPT_VERSION,
|
|
459
|
+
"privacy_version": REGION_PRIVACY_VERSION,
|
|
460
|
+
"rule_version": REGION_RULE_VERSION,
|
|
461
|
+
"validator_version": REGION_VALIDATOR_VERSION,
|
|
462
|
+
}
|
|
463
|
+
run_payload = {
|
|
464
|
+
"dataset_digest": manifest.dataset_digest,
|
|
465
|
+
"evidence_class": evidence_class,
|
|
466
|
+
"model_identity": model_identity,
|
|
467
|
+
"model_contract": model_contract,
|
|
468
|
+
"metrics": asdict(metrics),
|
|
469
|
+
"results": [asdict(r) for r in results],
|
|
470
|
+
}
|
|
471
|
+
serialized_run = json.dumps(run_payload, sort_keys=True, separators=(",", ":")).encode(
|
|
472
|
+
"utf-8"
|
|
473
|
+
)
|
|
474
|
+
run_digest = f"sha256:{hashlib.sha256(serialized_run).hexdigest()}"
|
|
475
|
+
|
|
476
|
+
production_ready, verdict_reasons = assess_production_readiness(
|
|
477
|
+
metrics,
|
|
478
|
+
split=manifest.split,
|
|
479
|
+
operational_evidence=False,
|
|
480
|
+
)
|
|
481
|
+
summary_dict = {
|
|
482
|
+
"schema_version": 1,
|
|
483
|
+
"dataset_id": manifest.dataset_id,
|
|
484
|
+
"dataset_version": manifest.dataset_version,
|
|
485
|
+
"split": manifest.split,
|
|
486
|
+
"dataset_digest": manifest.dataset_digest,
|
|
487
|
+
"run_digest": run_digest,
|
|
488
|
+
"status": "valid",
|
|
489
|
+
"effectiveness_evidence": metrics.effectiveness_evidence,
|
|
490
|
+
"model_identity": model_identity,
|
|
491
|
+
"model_contract": model_contract,
|
|
492
|
+
"production_ready": production_ready,
|
|
493
|
+
"verdict_reasons": list(verdict_reasons),
|
|
494
|
+
**asdict(metrics),
|
|
495
|
+
}
|
|
496
|
+
|
|
497
|
+
# Atomic publishing
|
|
498
|
+
output_dir_path.mkdir(parents=True, exist_ok=True)
|
|
499
|
+
tmp_run_dir = output_dir_path / f".tmp-{uuid.uuid4().hex}"
|
|
500
|
+
tmp_run_dir.mkdir(parents=True, exist_ok=False)
|
|
501
|
+
|
|
502
|
+
try:
|
|
503
|
+
# Write results.jsonl (sanitized rows)
|
|
504
|
+
results_path = tmp_run_dir / "workbook-ambiguity-results.jsonl"
|
|
505
|
+
with results_path.open("w", encoding="utf-8") as f:
|
|
506
|
+
for res in results:
|
|
507
|
+
row = {
|
|
508
|
+
"run_digest": run_digest,
|
|
509
|
+
"evaluation_id": res.evaluation_id,
|
|
510
|
+
"cohort": res.cohort,
|
|
511
|
+
"expected_kind": res.expected_kind,
|
|
512
|
+
"baseline_kind": res.baseline_kind,
|
|
513
|
+
"baseline_correct": res.baseline_correct,
|
|
514
|
+
"observation_status": res.observation_status,
|
|
515
|
+
"observed_kind": res.observed_kind,
|
|
516
|
+
"transition": res.transition,
|
|
517
|
+
}
|
|
518
|
+
f.write(json.dumps(row, separators=(",", ":")) + "\n")
|
|
519
|
+
|
|
520
|
+
# Write summary.json
|
|
521
|
+
summary_path = tmp_run_dir / "workbook-ambiguity-summary.json"
|
|
522
|
+
summary_path.write_text(
|
|
523
|
+
json.dumps(summary_dict, indent=2, sort_keys=True),
|
|
524
|
+
encoding="utf-8",
|
|
525
|
+
)
|
|
526
|
+
|
|
527
|
+
# Write summary.md if requested
|
|
528
|
+
if markdown:
|
|
529
|
+
summary_md_path = tmp_run_dir / "workbook-ambiguity-summary.md"
|
|
530
|
+
summary_md_path.write_text(
|
|
531
|
+
_render_summary_markdown(summary_dict),
|
|
532
|
+
encoding="utf-8",
|
|
533
|
+
)
|
|
534
|
+
|
|
535
|
+
final_run_dir = output_dir_path / run_digest.removeprefix("sha256:")
|
|
536
|
+
|
|
537
|
+
if final_run_dir.exists():
|
|
538
|
+
if _report_directories_match(final_run_dir, tmp_run_dir):
|
|
539
|
+
shutil.rmtree(tmp_run_dir, ignore_errors=True)
|
|
540
|
+
return WorkbookEvaluationReport(
|
|
541
|
+
run_digest=run_digest,
|
|
542
|
+
output_path=final_run_dir,
|
|
543
|
+
metrics=metrics,
|
|
544
|
+
results=results,
|
|
545
|
+
summary=summary_dict,
|
|
546
|
+
)
|
|
547
|
+
shutil.rmtree(tmp_run_dir, ignore_errors=True)
|
|
548
|
+
raise WorkbookEvaluationError(
|
|
549
|
+
"Output run directory conflict with different content",
|
|
550
|
+
code="report_conflict",
|
|
551
|
+
)
|
|
552
|
+
|
|
553
|
+
os.replace(tmp_run_dir, final_run_dir)
|
|
554
|
+
return WorkbookEvaluationReport(
|
|
555
|
+
run_digest=run_digest,
|
|
556
|
+
output_path=final_run_dir,
|
|
557
|
+
metrics=metrics,
|
|
558
|
+
results=results,
|
|
559
|
+
summary=summary_dict,
|
|
560
|
+
)
|
|
561
|
+
except Exception:
|
|
562
|
+
shutil.rmtree(tmp_run_dir, ignore_errors=True)
|
|
563
|
+
raise
|