langparse 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- langparse/__init__.py +55 -0
- langparse/autoparser.py +25 -0
- langparse/chunkers/__init__.py +12 -0
- langparse/chunkers/blocks.py +151 -0
- langparse/chunkers/profiles.py +53 -0
- langparse/chunkers/registry.py +38 -0
- langparse/chunkers/semantic.py +242 -0
- langparse/chunkers/text.py +96 -0
- langparse/chunkers/workbook.py +942 -0
- langparse/cli.py +329 -0
- langparse/config.py +169 -0
- langparse/core/__init__.py +0 -0
- langparse/core/chunker.py +16 -0
- langparse/core/engine.py +37 -0
- langparse/core/parser.py +35 -0
- langparse/core/rendering.py +49 -0
- langparse/engines/__init__.py +1 -0
- langparse/engines/pdf/__init__.py +1 -0
- langparse/engines/pdf/deepdoc/__init__.py +55 -0
- langparse/engines/pdf/deepdoc/layout_recognizer.py +235 -0
- langparse/engines/pdf/deepdoc/model_loader.py +101 -0
- langparse/engines/pdf/deepdoc/ocr.py +641 -0
- langparse/engines/pdf/deepdoc/operators.py +684 -0
- langparse/engines/pdf/deepdoc/pdf_parser.py +1894 -0
- langparse/engines/pdf/deepdoc/postprocess.py +339 -0
- langparse/engines/pdf/deepdoc/recognizer.py +418 -0
- langparse/engines/pdf/deepdoc/rendering.py +210 -0
- langparse/engines/pdf/deepdoc/table_structure_recognizer.py +559 -0
- langparse/engines/pdf/deepdoc/tokenizer.py +30 -0
- langparse/engines/pdf/deepdoc/utils.py +36 -0
- langparse/engines/pdf/deepdoc_engine.py +164 -0
- langparse/engines/pdf/mineru.py +259 -0
- langparse/engines/pdf/mineru_client.py +318 -0
- langparse/engines/pdf/mineru_service.py +225 -0
- langparse/engines/pdf/ocr.py +101 -0
- langparse/engines/pdf/other.py +20 -0
- langparse/engines/pdf/simple.py +134 -0
- langparse/engines/pdf/vision_llm.py +27 -0
- langparse/errors.py +70 -0
- langparse/logging.py +27 -0
- langparse/metrics.py +129 -0
- langparse/parsers/__init__.py +0 -0
- langparse/parsers/docx_parser.py +114 -0
- langparse/parsers/excel_parser.py +220 -0
- langparse/parsers/markdown_parser.py +34 -0
- langparse/parsers/pdf_parser.py +31 -0
- langparse/parsers/registry.py +48 -0
- langparse/parsers/sniff.py +72 -0
- langparse/progress.py +77 -0
- langparse/py.typed +0 -0
- langparse/services/__init__.py +11 -0
- langparse/services/batch_service.py +339 -0
- langparse/services/benchmark_service.py +202 -0
- langparse/services/fidelity.py +154 -0
- langparse/services/output_paths.py +86 -0
- langparse/services/parse_service.py +523 -0
- langparse/services/quality.py +65 -0
- langparse/services/workbook_ambiguity_benchmark.py +563 -0
- langparse/services/workbook_quality_benchmark.py +230 -0
- langparse/types.py +97 -0
- langparse/workbooks/__init__.py +103 -0
- langparse/workbooks/adapters.py +474 -0
- langparse/workbooks/assembly.py +993 -0
- langparse/workbooks/blocks.py +209 -0
- langparse/workbooks/bundle-v1.schema.json +71 -0
- langparse/workbooks/bundle.py +341 -0
- langparse/workbooks/classification.py +393 -0
- langparse/workbooks/continuation.py +577 -0
- langparse/workbooks/evaluation/__init__.py +45 -0
- langparse/workbooks/evaluation/evaluator.py +381 -0
- langparse/workbooks/evaluation/schema.py +419 -0
- langparse/workbooks/labels.py +14 -0
- langparse/workbooks/lineage.py +117 -0
- langparse/workbooks/modeling/__init__.py +52 -0
- langparse/workbooks/modeling/cache.py +20 -0
- langparse/workbooks/modeling/config.py +87 -0
- langparse/workbooks/modeling/contract.py +628 -0
- langparse/workbooks/modeling/disambiguation.py +800 -0
- langparse/workbooks/modeling/openai_adapter.py +192 -0
- langparse/workbooks/modeling/policy.py +79 -0
- langparse/workbooks/modeling/ports.py +44 -0
- langparse/workbooks/modeling/pricing.py +17 -0
- langparse/workbooks/modeling/types.py +251 -0
- langparse/workbooks/objects.py +229 -0
- langparse/workbooks/quality/__init__.py +23 -0
- langparse/workbooks/quality/bundle.py +53 -0
- langparse/workbooks/quality/evaluator.py +266 -0
- langparse/workbooks/quality/facts.py +142 -0
- langparse/workbooks/quality/schema.py +462 -0
- langparse/workbooks/reference_types.py +73 -0
- langparse/workbooks/references.py +178 -0
- langparse/workbooks/regions.py +932 -0
- langparse/workbooks/rendering.py +222 -0
- langparse/workbooks/tables.py +477 -0
- langparse/workbooks/types.py +257 -0
- langparse-0.1.0.dist-info/METADATA +790 -0
- langparse-0.1.0.dist-info/RECORD +101 -0
- langparse-0.1.0.dist-info/WHEEL +5 -0
- langparse-0.1.0.dist-info/entry_points.txt +2 -0
- langparse-0.1.0.dist-info/licenses/LICENSE +192 -0
- langparse-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,381 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
from typing import Literal
|
|
5
|
+
|
|
6
|
+
from .schema import InvalidGoldenSetError
|
|
7
|
+
|
|
8
|
+
RegionKind = Literal["logical_table", "form", "matrix", "text", "unclassified"]
|
|
9
|
+
_REGION_KINDS = frozenset({"logical_table", "form", "matrix", "text", "unclassified"})
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class CaseTruth:
|
|
14
|
+
evaluation_id: str
|
|
15
|
+
cohort: Literal["ambiguous"]
|
|
16
|
+
expected_kind: RegionKind | None # None indicates unresolvable
|
|
17
|
+
baseline_kind: RegionKind | None
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class CaseObservation:
|
|
22
|
+
evaluation_id: str
|
|
23
|
+
evidence_class: Literal["harness_only", "provider"]
|
|
24
|
+
status: Literal["selected", "abstained", "unresolved", "failed"]
|
|
25
|
+
selected_kind: RegionKind | None
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@dataclass(frozen=True)
|
|
29
|
+
class CaseEvaluationDetail:
|
|
30
|
+
evaluation_id: str
|
|
31
|
+
cohort: Literal["ambiguous"]
|
|
32
|
+
expected_kind: RegionKind | None
|
|
33
|
+
baseline_kind: RegionKind | None
|
|
34
|
+
baseline_correct: bool
|
|
35
|
+
observation_status: Literal["selected", "abstained", "unresolved", "failed"] | None
|
|
36
|
+
observed_kind: RegionKind | None
|
|
37
|
+
transition: str | None
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(frozen=True)
|
|
41
|
+
class WorkbookEvaluationMetrics:
|
|
42
|
+
# 17 Raw counts
|
|
43
|
+
ambiguous_case_count: int
|
|
44
|
+
resolvable_case_count: int
|
|
45
|
+
unresolvable_case_count: int
|
|
46
|
+
baseline_correct_count: int
|
|
47
|
+
baseline_wrong_count: int
|
|
48
|
+
model_selected_count: int | None = None
|
|
49
|
+
model_correct_acceptance_count: int | None = None
|
|
50
|
+
model_wrong_acceptance_count: int | None = None
|
|
51
|
+
model_abstained_count: int | None = None
|
|
52
|
+
model_unresolved_count: int | None = None
|
|
53
|
+
model_failed_count: int | None = None
|
|
54
|
+
fixed_error_count: int | None = None
|
|
55
|
+
introduced_error_count: int | None = None
|
|
56
|
+
unchanged_correct_count: int | None = None
|
|
57
|
+
unchanged_wrong_count: int | None = None
|
|
58
|
+
clear_sample_count: int = 0
|
|
59
|
+
clear_unexpected_call_count: int = 0
|
|
60
|
+
|
|
61
|
+
# 9 Derived rates/deltas (None if denominator is 0)
|
|
62
|
+
baseline_accuracy: float | None = None
|
|
63
|
+
model_selection_accuracy: float | None = None
|
|
64
|
+
wrong_acceptance_rate: float | None = None
|
|
65
|
+
model_coverage: float | None = None
|
|
66
|
+
abstain_rate: float | None = None
|
|
67
|
+
unresolved_rate: float | None = None
|
|
68
|
+
failure_rate: float | None = None
|
|
69
|
+
clear_call_rate: float | None = None
|
|
70
|
+
net_correct_delta: int | None = None
|
|
71
|
+
|
|
72
|
+
# Evidence grading
|
|
73
|
+
effectiveness_evidence: bool = False
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def classify_case_evaluation(
|
|
77
|
+
truth: CaseTruth,
|
|
78
|
+
observation: CaseObservation | None,
|
|
79
|
+
) -> CaseEvaluationDetail:
|
|
80
|
+
is_resolvable = truth.expected_kind is not None
|
|
81
|
+
baseline_correct = is_resolvable and (truth.baseline_kind == truth.expected_kind)
|
|
82
|
+
|
|
83
|
+
if observation is None:
|
|
84
|
+
return CaseEvaluationDetail(
|
|
85
|
+
evaluation_id=truth.evaluation_id,
|
|
86
|
+
cohort=truth.cohort,
|
|
87
|
+
expected_kind=truth.expected_kind,
|
|
88
|
+
baseline_kind=truth.baseline_kind,
|
|
89
|
+
baseline_correct=baseline_correct,
|
|
90
|
+
observation_status=None,
|
|
91
|
+
observed_kind=None,
|
|
92
|
+
transition=None,
|
|
93
|
+
)
|
|
94
|
+
|
|
95
|
+
transition: str | None = None
|
|
96
|
+
if observation.status == "selected":
|
|
97
|
+
if is_resolvable:
|
|
98
|
+
model_correct = observation.selected_kind == truth.expected_kind
|
|
99
|
+
if not baseline_correct and model_correct:
|
|
100
|
+
transition = "fixed_error"
|
|
101
|
+
elif baseline_correct and not model_correct:
|
|
102
|
+
transition = "introduced_error"
|
|
103
|
+
elif baseline_correct and model_correct:
|
|
104
|
+
transition = "unchanged_correct"
|
|
105
|
+
else:
|
|
106
|
+
transition = "unchanged_wrong"
|
|
107
|
+
else:
|
|
108
|
+
# Unresolvable truth selected by model is an introduced error / wrong acceptance
|
|
109
|
+
transition = "introduced_error"
|
|
110
|
+
else:
|
|
111
|
+
transition = observation.status
|
|
112
|
+
|
|
113
|
+
return CaseEvaluationDetail(
|
|
114
|
+
evaluation_id=truth.evaluation_id,
|
|
115
|
+
cohort=truth.cohort,
|
|
116
|
+
expected_kind=truth.expected_kind,
|
|
117
|
+
baseline_kind=truth.baseline_kind,
|
|
118
|
+
baseline_correct=baseline_correct,
|
|
119
|
+
observation_status=observation.status,
|
|
120
|
+
observed_kind=observation.selected_kind,
|
|
121
|
+
transition=transition,
|
|
122
|
+
)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def evaluate_workbook_ambiguity(
|
|
126
|
+
truths: tuple[CaseTruth, ...],
|
|
127
|
+
observations: tuple[CaseObservation, ...] | None = None,
|
|
128
|
+
*,
|
|
129
|
+
clear_sample_count: int = 0,
|
|
130
|
+
clear_unexpected_call_count: int = 0,
|
|
131
|
+
) -> WorkbookEvaluationMetrics:
|
|
132
|
+
if (
|
|
133
|
+
type(clear_sample_count) is not int
|
|
134
|
+
or type(clear_unexpected_call_count) is not int
|
|
135
|
+
or clear_sample_count < 0
|
|
136
|
+
or clear_unexpected_call_count < 0
|
|
137
|
+
or clear_unexpected_call_count > clear_sample_count
|
|
138
|
+
):
|
|
139
|
+
raise InvalidGoldenSetError(
|
|
140
|
+
"Clear-sample counts are inconsistent",
|
|
141
|
+
code="invalid_metric_input",
|
|
142
|
+
)
|
|
143
|
+
truth_ids: set[str] = set()
|
|
144
|
+
for truth in truths:
|
|
145
|
+
if (
|
|
146
|
+
not isinstance(truth, CaseTruth)
|
|
147
|
+
or truth.cohort != "ambiguous"
|
|
148
|
+
or truth.expected_kind not in _REGION_KINDS | {None}
|
|
149
|
+
or truth.baseline_kind not in _REGION_KINDS | {None}
|
|
150
|
+
):
|
|
151
|
+
raise InvalidGoldenSetError("Invalid truth shape", code="invalid_truth")
|
|
152
|
+
if truth.evaluation_id in truth_ids:
|
|
153
|
+
raise InvalidGoldenSetError(
|
|
154
|
+
f"Duplicate truth evaluation_id: {truth.evaluation_id}",
|
|
155
|
+
code="duplicate_id",
|
|
156
|
+
)
|
|
157
|
+
truth_ids.add(truth.evaluation_id)
|
|
158
|
+
|
|
159
|
+
obs_by_id: dict[str, CaseObservation] = {}
|
|
160
|
+
if observations is not None:
|
|
161
|
+
for obs in observations:
|
|
162
|
+
if (
|
|
163
|
+
not isinstance(obs, CaseObservation)
|
|
164
|
+
or obs.evidence_class not in {"harness_only", "provider"}
|
|
165
|
+
or obs.status not in {"selected", "abstained", "unresolved", "failed"}
|
|
166
|
+
or (obs.status == "selected" and obs.selected_kind not in _REGION_KINDS)
|
|
167
|
+
or (obs.status != "selected" and obs.selected_kind is not None)
|
|
168
|
+
):
|
|
169
|
+
raise InvalidGoldenSetError(
|
|
170
|
+
"Invalid observation shape",
|
|
171
|
+
code="invalid_observation",
|
|
172
|
+
)
|
|
173
|
+
if obs.evaluation_id in obs_by_id:
|
|
174
|
+
raise InvalidGoldenSetError(
|
|
175
|
+
f"Duplicate observation evaluation_id: {obs.evaluation_id}",
|
|
176
|
+
code="duplicate_id",
|
|
177
|
+
)
|
|
178
|
+
if obs.evaluation_id not in truth_ids:
|
|
179
|
+
raise InvalidGoldenSetError(
|
|
180
|
+
f"Observation evaluation_id not found in truth: {obs.evaluation_id}",
|
|
181
|
+
code="evaluation_id_mismatch",
|
|
182
|
+
)
|
|
183
|
+
obs_by_id[obs.evaluation_id] = obs
|
|
184
|
+
|
|
185
|
+
if len(obs_by_id) != len(truth_ids):
|
|
186
|
+
raise InvalidGoldenSetError(
|
|
187
|
+
"Observations count does not match truths count",
|
|
188
|
+
code="evaluation_id_mismatch",
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
ambiguous_case_count = len(truths)
|
|
192
|
+
resolvable_case_count = 0
|
|
193
|
+
unresolvable_case_count = 0
|
|
194
|
+
baseline_correct_count = 0
|
|
195
|
+
baseline_wrong_count = 0
|
|
196
|
+
|
|
197
|
+
model_selected_count = 0
|
|
198
|
+
model_correct_acceptance_count = 0
|
|
199
|
+
model_wrong_acceptance_count = 0
|
|
200
|
+
model_abstained_count = 0
|
|
201
|
+
model_unresolved_count = 0
|
|
202
|
+
model_failed_count = 0
|
|
203
|
+
|
|
204
|
+
fixed_error_count = 0
|
|
205
|
+
introduced_error_count = 0
|
|
206
|
+
unchanged_correct_count = 0
|
|
207
|
+
unchanged_wrong_count = 0
|
|
208
|
+
|
|
209
|
+
has_observations = observations is not None
|
|
210
|
+
has_provider_only_evidence = has_observations
|
|
211
|
+
|
|
212
|
+
for truth in truths:
|
|
213
|
+
is_resolvable = truth.expected_kind is not None
|
|
214
|
+
if is_resolvable:
|
|
215
|
+
resolvable_case_count += 1
|
|
216
|
+
if truth.baseline_kind == truth.expected_kind:
|
|
217
|
+
baseline_correct_count += 1
|
|
218
|
+
else:
|
|
219
|
+
baseline_wrong_count += 1
|
|
220
|
+
else:
|
|
221
|
+
unresolvable_case_count += 1
|
|
222
|
+
baseline_wrong_count += 1
|
|
223
|
+
|
|
224
|
+
if has_observations:
|
|
225
|
+
obs = obs_by_id[truth.evaluation_id]
|
|
226
|
+
if obs.evidence_class != "provider":
|
|
227
|
+
has_provider_only_evidence = False
|
|
228
|
+
|
|
229
|
+
if obs.status == "selected":
|
|
230
|
+
model_selected_count += 1
|
|
231
|
+
if is_resolvable:
|
|
232
|
+
if obs.selected_kind == truth.expected_kind:
|
|
233
|
+
model_correct_acceptance_count += 1
|
|
234
|
+
if truth.baseline_kind != truth.expected_kind:
|
|
235
|
+
fixed_error_count += 1
|
|
236
|
+
else:
|
|
237
|
+
unchanged_correct_count += 1
|
|
238
|
+
else:
|
|
239
|
+
model_wrong_acceptance_count += 1
|
|
240
|
+
if truth.baseline_kind == truth.expected_kind:
|
|
241
|
+
introduced_error_count += 1
|
|
242
|
+
else:
|
|
243
|
+
unchanged_wrong_count += 1
|
|
244
|
+
else:
|
|
245
|
+
# Unresolvable truth selected is wrong acceptance
|
|
246
|
+
model_wrong_acceptance_count += 1
|
|
247
|
+
introduced_error_count += 1
|
|
248
|
+
elif obs.status == "abstained":
|
|
249
|
+
model_abstained_count += 1
|
|
250
|
+
elif obs.status == "unresolved":
|
|
251
|
+
model_unresolved_count += 1
|
|
252
|
+
elif obs.status == "failed":
|
|
253
|
+
model_failed_count += 1
|
|
254
|
+
|
|
255
|
+
# Calculate derived metrics (null if denominator is 0)
|
|
256
|
+
baseline_accuracy = (
|
|
257
|
+
baseline_correct_count / resolvable_case_count if resolvable_case_count > 0 else None
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
if has_observations:
|
|
261
|
+
model_selection_accuracy = (
|
|
262
|
+
model_correct_acceptance_count / model_selected_count
|
|
263
|
+
if model_selected_count > 0
|
|
264
|
+
else None
|
|
265
|
+
)
|
|
266
|
+
wrong_acceptance_rate = (
|
|
267
|
+
model_wrong_acceptance_count / ambiguous_case_count
|
|
268
|
+
if ambiguous_case_count > 0
|
|
269
|
+
else None
|
|
270
|
+
)
|
|
271
|
+
model_coverage = (
|
|
272
|
+
model_selected_count / ambiguous_case_count if ambiguous_case_count > 0 else None
|
|
273
|
+
)
|
|
274
|
+
abstain_rate = (
|
|
275
|
+
model_abstained_count / ambiguous_case_count if ambiguous_case_count > 0 else None
|
|
276
|
+
)
|
|
277
|
+
unresolved_rate = (
|
|
278
|
+
model_unresolved_count / ambiguous_case_count if ambiguous_case_count > 0 else None
|
|
279
|
+
)
|
|
280
|
+
failure_rate = (
|
|
281
|
+
model_failed_count / ambiguous_case_count if ambiguous_case_count > 0 else None
|
|
282
|
+
)
|
|
283
|
+
net_correct_delta = fixed_error_count - introduced_error_count
|
|
284
|
+
effectiveness_evidence = has_provider_only_evidence
|
|
285
|
+
else:
|
|
286
|
+
model_selection_accuracy = None
|
|
287
|
+
wrong_acceptance_rate = None
|
|
288
|
+
model_coverage = None
|
|
289
|
+
abstain_rate = None
|
|
290
|
+
unresolved_rate = None
|
|
291
|
+
failure_rate = None
|
|
292
|
+
net_correct_delta = None
|
|
293
|
+
effectiveness_evidence = False
|
|
294
|
+
|
|
295
|
+
clear_call_rate = (
|
|
296
|
+
clear_unexpected_call_count / clear_sample_count if clear_sample_count > 0 else None
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
return WorkbookEvaluationMetrics(
|
|
300
|
+
ambiguous_case_count=ambiguous_case_count,
|
|
301
|
+
resolvable_case_count=resolvable_case_count,
|
|
302
|
+
unresolvable_case_count=unresolvable_case_count,
|
|
303
|
+
baseline_correct_count=baseline_correct_count,
|
|
304
|
+
baseline_wrong_count=baseline_wrong_count,
|
|
305
|
+
baseline_accuracy=baseline_accuracy,
|
|
306
|
+
model_selected_count=model_selected_count,
|
|
307
|
+
model_correct_acceptance_count=model_correct_acceptance_count,
|
|
308
|
+
model_wrong_acceptance_count=model_wrong_acceptance_count,
|
|
309
|
+
model_abstained_count=model_abstained_count,
|
|
310
|
+
model_unresolved_count=model_unresolved_count,
|
|
311
|
+
model_failed_count=model_failed_count,
|
|
312
|
+
model_selection_accuracy=model_selection_accuracy,
|
|
313
|
+
wrong_acceptance_rate=wrong_acceptance_rate,
|
|
314
|
+
model_coverage=model_coverage,
|
|
315
|
+
abstain_rate=abstain_rate,
|
|
316
|
+
unresolved_rate=unresolved_rate,
|
|
317
|
+
failure_rate=failure_rate,
|
|
318
|
+
fixed_error_count=fixed_error_count,
|
|
319
|
+
introduced_error_count=introduced_error_count,
|
|
320
|
+
unchanged_correct_count=unchanged_correct_count,
|
|
321
|
+
unchanged_wrong_count=unchanged_wrong_count,
|
|
322
|
+
net_correct_delta=net_correct_delta,
|
|
323
|
+
clear_sample_count=clear_sample_count,
|
|
324
|
+
clear_unexpected_call_count=clear_unexpected_call_count,
|
|
325
|
+
clear_call_rate=clear_call_rate,
|
|
326
|
+
effectiveness_evidence=effectiveness_evidence,
|
|
327
|
+
)
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def assess_production_readiness(
|
|
331
|
+
metrics: WorkbookEvaluationMetrics,
|
|
332
|
+
*,
|
|
333
|
+
split: str | None = None,
|
|
334
|
+
operational_evidence: bool = False,
|
|
335
|
+
minimum_ambiguous_cases: int = 30,
|
|
336
|
+
) -> tuple[bool, tuple[str, ...]]:
|
|
337
|
+
"""Assess whether evaluation metrics satisfy production rollout criteria.
|
|
338
|
+
|
|
339
|
+
All gates are unconditional: omitting ``split`` or ``operational_evidence``
|
|
340
|
+
defaults to the conservative (blocking) position.
|
|
341
|
+
"""
|
|
342
|
+
reasons: list[str] = []
|
|
343
|
+
|
|
344
|
+
# Gate 1: holdout split required (tuning / None / unknown all fail)
|
|
345
|
+
if split != "holdout":
|
|
346
|
+
reasons.append("holdout_evidence_required")
|
|
347
|
+
|
|
348
|
+
# Gate 2: minimum sample size
|
|
349
|
+
if metrics.ambiguous_case_count < minimum_ambiguous_cases:
|
|
350
|
+
reasons.append("minimum_ambiguous_cases_not_met")
|
|
351
|
+
|
|
352
|
+
# Gate 3: operational staging evidence
|
|
353
|
+
if not operational_evidence:
|
|
354
|
+
reasons.append("operational_evidence_missing")
|
|
355
|
+
|
|
356
|
+
# Gate 4: provider effectiveness evidence
|
|
357
|
+
if not metrics.effectiveness_evidence:
|
|
358
|
+
reasons.append("effectiveness_evidence_missing")
|
|
359
|
+
|
|
360
|
+
# Gate 5: net improvement
|
|
361
|
+
if metrics.net_correct_delta is None or metrics.net_correct_delta <= 0:
|
|
362
|
+
reasons.append("net_correct_delta_non_positive")
|
|
363
|
+
|
|
364
|
+
# Gate 6: wrong acceptance threshold
|
|
365
|
+
if metrics.wrong_acceptance_rate is not None and metrics.wrong_acceptance_rate > 0.05:
|
|
366
|
+
reasons.append("wrong_acceptance_rate_exceeds_threshold")
|
|
367
|
+
|
|
368
|
+
# Gate 7: clear samples must not trigger model calls
|
|
369
|
+
if metrics.clear_call_rate is not None and metrics.clear_call_rate > 0.0:
|
|
370
|
+
reasons.append("clear_sample_unexpected_calls")
|
|
371
|
+
|
|
372
|
+
# Gate 8: introduced errors must be fewer than fixed
|
|
373
|
+
if (
|
|
374
|
+
metrics.fixed_error_count is not None
|
|
375
|
+
and metrics.introduced_error_count is not None
|
|
376
|
+
and metrics.introduced_error_count >= metrics.fixed_error_count
|
|
377
|
+
):
|
|
378
|
+
reasons.append("introduced_errors_exceed_fixed_errors")
|
|
379
|
+
|
|
380
|
+
is_ready = len(reasons) == 0
|
|
381
|
+
return is_ready, tuple(reasons)
|