clastogen 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- clastogen/__init__.py +35 -0
- clastogen/core/injection.py +59 -0
- clastogen/core/mutator.py +140 -0
- clastogen/core/sprt.py +63 -0
- clastogen/exceptions.py +9 -0
- clastogen/models.py +189 -0
- clastogen/plugin.py +492 -0
- clastogen/py.typed +0 -0
- clastogen/report.py +250 -0
- clastogen/scoring.py +25 -0
- clastogen/stats.py +249 -0
- clastogen/types.py +29 -0
- clastogen-0.2.0.dist-info/METADATA +308 -0
- clastogen-0.2.0.dist-info/RECORD +17 -0
- clastogen-0.2.0.dist-info/WHEEL +4 -0
- clastogen-0.2.0.dist-info/entry_points.txt +2 -0
- clastogen-0.2.0.dist-info/licenses/LICENSE +21 -0
clastogen/__init__.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Clastogen: Mutation testing & statistical assertion framework for LLMs and Agents."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import version
|
|
4
|
+
|
|
5
|
+
from clastogen.core.injection import override_prompt
|
|
6
|
+
from clastogen.core.mutator import PromptMutator
|
|
7
|
+
from clastogen.core.sprt import SPRT
|
|
8
|
+
from clastogen.models import Mutant, PassRateResult, RegressionResult, SPRTConfig, SPRTResult
|
|
9
|
+
from clastogen.stats import (
|
|
10
|
+
assert_no_regression,
|
|
11
|
+
assert_pass_rate,
|
|
12
|
+
compute_wilson_interval,
|
|
13
|
+
evaluate_pass_rate,
|
|
14
|
+
evaluate_regression,
|
|
15
|
+
)
|
|
16
|
+
from clastogen.types import Decision
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"SPRT",
|
|
20
|
+
"Decision",
|
|
21
|
+
"Mutant",
|
|
22
|
+
"PassRateResult",
|
|
23
|
+
"PromptMutator",
|
|
24
|
+
"RegressionResult",
|
|
25
|
+
"SPRTConfig",
|
|
26
|
+
"SPRTResult",
|
|
27
|
+
"assert_no_regression",
|
|
28
|
+
"assert_pass_rate",
|
|
29
|
+
"compute_wilson_interval",
|
|
30
|
+
"evaluate_pass_rate",
|
|
31
|
+
"evaluate_regression",
|
|
32
|
+
"override_prompt",
|
|
33
|
+
]
|
|
34
|
+
|
|
35
|
+
__version__ = version("clastogen")
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import importlib
|
|
4
|
+
import logging
|
|
5
|
+
import sys
|
|
6
|
+
import types
|
|
7
|
+
from collections.abc import Generator
|
|
8
|
+
from contextlib import ExitStack, contextmanager
|
|
9
|
+
from functools import reduce
|
|
10
|
+
from unittest.mock import patch
|
|
11
|
+
|
|
12
|
+
logger = logging.getLogger(__name__)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def resolve_target(target_spec: str) -> tuple[object, str]:
|
|
16
|
+
"""Resolves 'module:attr' or 'module:Class.attr' (dots after the colon walk nested attributes) into (owner, attr)."""
|
|
17
|
+
module_name, sep, attr_path = target_spec.partition(":")
|
|
18
|
+
if not sep:
|
|
19
|
+
raise ValueError(f"Target '{target_spec}' must be 'module:attr' or 'module:Class.attr'")
|
|
20
|
+
*owner_path, attr = attr_path.split(".")
|
|
21
|
+
owner = reduce(getattr, owner_path, importlib.import_module(module_name))
|
|
22
|
+
if not hasattr(owner, attr):
|
|
23
|
+
raise AttributeError(f"Object {owner!r} has no attribute '{attr}'")
|
|
24
|
+
return owner, attr
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _aliasing_modules(owner: object, attr: str, original: object) -> list[types.ModuleType]:
|
|
28
|
+
"""Modules other than owner that bound the same string under the same name, e.g. via `from agent import PROMPT`."""
|
|
29
|
+
if not isinstance(original, str):
|
|
30
|
+
return []
|
|
31
|
+
return [
|
|
32
|
+
mod
|
|
33
|
+
for mod in list(sys.modules.values())
|
|
34
|
+
if isinstance(mod, types.ModuleType) and mod is not owner and vars(mod).get(attr) is original
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
@contextmanager
|
|
39
|
+
def override_prompt(target_spec: str, mutated_text: str) -> Generator[str, None, None]:
|
|
40
|
+
"""Temporarily replaces the target attribute, and every same-name module alias of it, with mutated_text.
|
|
41
|
+
|
|
42
|
+
unittest.mock.patch.object restores descriptors such as @classmethod and removes shadow attributes
|
|
43
|
+
on classes that only inherited the target.
|
|
44
|
+
"""
|
|
45
|
+
if not isinstance(mutated_text, str):
|
|
46
|
+
raise TypeError(f"mutated_text must be a string, got {type(mutated_text).__name__}")
|
|
47
|
+
|
|
48
|
+
owner, attr = resolve_target(target_spec)
|
|
49
|
+
original = getattr(owner, attr)
|
|
50
|
+
if original is None:
|
|
51
|
+
raise TypeError(f"Target '{target_spec}' resolved to None, expected a prompt string.")
|
|
52
|
+
|
|
53
|
+
with ExitStack() as stack:
|
|
54
|
+
stack.enter_context(patch.object(owner, attr, mutated_text))
|
|
55
|
+
aliases = _aliasing_modules(owner, attr, original)
|
|
56
|
+
for mod in aliases:
|
|
57
|
+
stack.enter_context(patch.object(mod, attr, mutated_text))
|
|
58
|
+
logger.debug("Injected prompt into target '%s' (%d module alias(es))", target_spec, len(aliases))
|
|
59
|
+
yield mutated_text
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Sequence
|
|
5
|
+
from decimal import Decimal
|
|
6
|
+
from itertools import zip_longest
|
|
7
|
+
|
|
8
|
+
from clastogen.models import Mutant
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class PromptMutator:
|
|
12
|
+
"""Extracts load-bearing imperative constraints and generates targeted prompt mutants."""
|
|
13
|
+
|
|
14
|
+
RULE_PATTERN = re.compile(
|
|
15
|
+
r"\b(?:never|do not|don't|must not|mustn't|should not|shouldn't|cannot|can't|always|must|shall not|may not|avoid|only|requires?|required)\b",
|
|
16
|
+
re.IGNORECASE,
|
|
17
|
+
)
|
|
18
|
+
|
|
19
|
+
SENTENCE_SPLIT_PATTERN = re.compile(r"\n+|(?<!\be\.g)(?<!\bi\.e)(?<!\betc)[.?!]\s+")
|
|
20
|
+
|
|
21
|
+
NEGATION_REPLACEMENTS: tuple[tuple[re.Pattern[str], str], ...] = (
|
|
22
|
+
(re.compile(r"\bdo not\b", re.IGNORECASE), "do"),
|
|
23
|
+
(re.compile(r"\bdon't\b", re.IGNORECASE), "do"),
|
|
24
|
+
(re.compile(r"\bmust not\b", re.IGNORECASE), "must"),
|
|
25
|
+
(re.compile(r"\bmustn't\b", re.IGNORECASE), "must"),
|
|
26
|
+
(re.compile(r"\bnever\b", re.IGNORECASE), "always"),
|
|
27
|
+
(re.compile(r"\bcannot\b", re.IGNORECASE), "can"),
|
|
28
|
+
(re.compile(r"\bcan't\b", re.IGNORECASE), "can"),
|
|
29
|
+
(re.compile(r"\bshould not\b", re.IGNORECASE), "should"),
|
|
30
|
+
(re.compile(r"\bshouldn't\b", re.IGNORECASE), "should"),
|
|
31
|
+
(re.compile(r"\bshall not\b", re.IGNORECASE), "shall"),
|
|
32
|
+
(re.compile(r"\bmay not\b", re.IGNORECASE), "may"),
|
|
33
|
+
(re.compile(r"\balways\b", re.IGNORECASE), "never"),
|
|
34
|
+
(re.compile(r"\bmust\b", re.IGNORECASE), "must not"),
|
|
35
|
+
(re.compile(r"\bavoid\b", re.IGNORECASE), "prefer"),
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
NUMBER_PATTERN = re.compile(r"\b\d+(?:,\d{3})*(?:\.\d+)?\b")
|
|
39
|
+
|
|
40
|
+
def __init__(self, target_symbol: str = "SYSTEM_PROMPT") -> None:
|
|
41
|
+
self.target_symbol = target_symbol
|
|
42
|
+
|
|
43
|
+
def extract_candidate_rules(self, prompt: str) -> list[str]:
|
|
44
|
+
"""Extracts candidate load-bearing sentences using linear string splitting and keyword search."""
|
|
45
|
+
if not isinstance(prompt, str):
|
|
46
|
+
raise TypeError(f"prompt must be a string, got {type(prompt).__name__}")
|
|
47
|
+
return [
|
|
48
|
+
cleaned
|
|
49
|
+
for line in self.SENTENCE_SPLIT_PATTERN.split(prompt)
|
|
50
|
+
if (cleaned := re.sub(r"^[-*]\s+", "", line.strip()))
|
|
51
|
+
and len(cleaned) > 10
|
|
52
|
+
and self.RULE_PATTERN.search(cleaned)
|
|
53
|
+
]
|
|
54
|
+
|
|
55
|
+
def _delete_mutant(self, prompt: str, rule: str) -> Mutant | None:
|
|
56
|
+
"""Removes rule with its bullet and trailing punctuation; None when the prompt is unchanged."""
|
|
57
|
+
clean_pat = re.compile(rf"[ \t]*(?:[-*]\s+)?{re.escape(rule)}[.?!]?([ \t]*)(\n?)")
|
|
58
|
+
|
|
59
|
+
def join(m: re.Match[str]) -> str:
|
|
60
|
+
if m.start() == 0 or prompt[m.start() - 1] == "\n":
|
|
61
|
+
return ""
|
|
62
|
+
return m[2] or ("" if m.end() == len(prompt) else " ")
|
|
63
|
+
|
|
64
|
+
mutated = clean_pat.sub(join, prompt).strip()
|
|
65
|
+
mutated = re.sub(r"\n\s*\.\s*\n", "\n", mutated)
|
|
66
|
+
mutated = re.sub(r"\n{3,}", "\n\n", mutated)
|
|
67
|
+
if mutated == prompt:
|
|
68
|
+
return None
|
|
69
|
+
return Mutant.create(
|
|
70
|
+
target_symbol=self.target_symbol,
|
|
71
|
+
operator_name="delete_constraint",
|
|
72
|
+
original_snippet=rule,
|
|
73
|
+
mutated_snippet="[DELETED]",
|
|
74
|
+
mutated_prompt=mutated,
|
|
75
|
+
description=f"Deleted load-bearing constraint: '{rule[:50]}...'",
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
def _invert_mutant(self, prompt: str, rule: str) -> Mutant | None:
|
|
79
|
+
"""Flips the first matching negation or obligation keyword in rule; None when none applies."""
|
|
80
|
+
for pattern, replacement in self.NEGATION_REPLACEMENTS:
|
|
81
|
+
if pattern.search(rule):
|
|
82
|
+
inverted = pattern.sub(replacement.upper(), rule, count=1)
|
|
83
|
+
mutated = prompt.replace(rule, inverted)
|
|
84
|
+
if mutated != prompt:
|
|
85
|
+
return Mutant.create(
|
|
86
|
+
target_symbol=self.target_symbol,
|
|
87
|
+
operator_name="invert_negation",
|
|
88
|
+
original_snippet=rule,
|
|
89
|
+
mutated_snippet=inverted,
|
|
90
|
+
mutated_prompt=mutated,
|
|
91
|
+
description=f"Inverted constraint: '{rule[:40]}' -> '{inverted[:40]}'",
|
|
92
|
+
)
|
|
93
|
+
return None
|
|
94
|
+
|
|
95
|
+
def _threshold_mutant(self, prompt: str, rule: str) -> Mutant | None:
|
|
96
|
+
"""Multiplies the first number in rule by 10 (a $50 limit becomes $500); None when rule has no number."""
|
|
97
|
+
match = self.NUMBER_PATTERN.search(rule)
|
|
98
|
+
if match is None:
|
|
99
|
+
return None
|
|
100
|
+
raw = match.group()
|
|
101
|
+
scaled = Decimal(raw.replace(",", "")) * 10
|
|
102
|
+
number = f"{scaled:,}" if "," in raw else str(scaled)
|
|
103
|
+
changed = rule[: match.start()] + number + rule[match.end() :]
|
|
104
|
+
mutated = prompt.replace(rule, changed)
|
|
105
|
+
if mutated == prompt:
|
|
106
|
+
return None
|
|
107
|
+
return Mutant.create(
|
|
108
|
+
target_symbol=self.target_symbol,
|
|
109
|
+
operator_name="change_threshold",
|
|
110
|
+
original_snippet=rule,
|
|
111
|
+
mutated_snippet=changed,
|
|
112
|
+
mutated_prompt=mutated,
|
|
113
|
+
description=f"Changed threshold: '{raw}' -> '{number}' in '{rule[:40]}'",
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
def generate_mutants(self, prompt: str, max_mutants: int) -> Sequence[Mutant]:
|
|
117
|
+
"""Generates a bounded sequence of mutants (constraint deletions, negation inversions, threshold changes)."""
|
|
118
|
+
if not isinstance(prompt, str):
|
|
119
|
+
raise TypeError(f"prompt must be a string, got {type(prompt).__name__}")
|
|
120
|
+
if max_mutants < 1:
|
|
121
|
+
raise ValueError(f"max_mutants must be >= 1, got {max_mutants}")
|
|
122
|
+
|
|
123
|
+
candidates = list(dict.fromkeys(self.extract_candidate_rules(prompt)))
|
|
124
|
+
if not candidates:
|
|
125
|
+
return []
|
|
126
|
+
|
|
127
|
+
needed_rules = max(1, (max_mutants + 1) // 2)
|
|
128
|
+
if len(candidates) > needed_rules:
|
|
129
|
+
step = len(candidates) / needed_rules
|
|
130
|
+
selected_rules = [candidates[int(i * step)] for i in range(needed_rules)]
|
|
131
|
+
else:
|
|
132
|
+
selected_rules = candidates
|
|
133
|
+
|
|
134
|
+
operators = (self._delete_mutant, self._invert_mutant, self._threshold_mutant)
|
|
135
|
+
per_rule = [
|
|
136
|
+
[m for op in operators[i % 3 :] + operators[: i % 3] if (m := op(prompt, rule))]
|
|
137
|
+
for i, rule in enumerate(selected_rules)
|
|
138
|
+
]
|
|
139
|
+
mutants = [m for group in zip_longest(*per_rule) for m in group if m]
|
|
140
|
+
return mutants[:max_mutants]
|
clastogen/core/sprt.py
ADDED
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
import inspect
|
|
2
|
+
import logging
|
|
3
|
+
import math
|
|
4
|
+
from collections.abc import Callable, Sequence
|
|
5
|
+
|
|
6
|
+
from clastogen.models import SPRTConfig, SPRTResult
|
|
7
|
+
from clastogen.types import Decision
|
|
8
|
+
|
|
9
|
+
logger = logging.getLogger(__name__)
|
|
10
|
+
|
|
11
|
+
# Unmutated runs used to estimate p0; the Laplace estimate with n=10 keeps false KILLs on unchanged mutants low.
|
|
12
|
+
BASELINE_RUNS = 10
|
|
13
|
+
MIN_BASELINE_RATE = 0.80
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def check_outcome(raw: object) -> bool:
|
|
17
|
+
"""Returns raw if it is a bool; rejects awaitables and other types with TypeError."""
|
|
18
|
+
if inspect.isawaitable(raw):
|
|
19
|
+
raise TypeError("evaluator returned an awaitable/coroutine; expected synchronous callable returning bool")
|
|
20
|
+
if not isinstance(raw, bool):
|
|
21
|
+
raise TypeError(f"evaluator must return bool, got {type(raw).__name__}")
|
|
22
|
+
return raw
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def estimate_p0(outcomes: Sequence[bool]) -> float:
|
|
26
|
+
"""Estimates the baseline pass probability with Laplace's rule of succession, (s + 1) / (n + 2)."""
|
|
27
|
+
return (sum(outcomes) + 1) / (len(outcomes) + 2)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def config_from_baseline(outcomes: Sequence[bool], *, delta: float, max_steps: int) -> SPRTConfig | None:
|
|
31
|
+
"""Builds the SPRT config from baseline outcomes, or None when the raw pass rate is below MIN_BASELINE_RATE."""
|
|
32
|
+
if sum(outcomes) / len(outcomes) < MIN_BASELINE_RATE:
|
|
33
|
+
return None
|
|
34
|
+
return SPRTConfig.from_absolute_drop(p0=estimate_p0(outcomes), delta=delta, max_steps=max_steps)
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class SPRT:
|
|
38
|
+
"""Wald's Sequential Probability Ratio Test engine for Bernoulli trials with early stopping."""
|
|
39
|
+
|
|
40
|
+
def __init__(self, config: SPRTConfig) -> None:
|
|
41
|
+
self.config = config
|
|
42
|
+
self.threshold_a = math.log((1.0 - config.beta) / config.alpha)
|
|
43
|
+
self.threshold_b = math.log(config.beta / (1.0 - config.alpha))
|
|
44
|
+
self.llr_pass = math.log(config.p1 / config.p0)
|
|
45
|
+
self.llr_fail = math.log((1.0 - config.p1) / (1.0 - config.p0))
|
|
46
|
+
|
|
47
|
+
def run_evaluator(self, evaluator: Callable[[], bool]) -> SPRTResult:
|
|
48
|
+
"""Executes evaluator dynamically until a decision is reached or max_steps is exhausted."""
|
|
49
|
+
cumulative_llr = 0.0
|
|
50
|
+
decision = Decision.INCONCLUSIVE
|
|
51
|
+
step = 0
|
|
52
|
+
while step < self.config.max_steps:
|
|
53
|
+
step += 1
|
|
54
|
+
cumulative_llr += self.llr_pass if check_outcome(evaluator()) else self.llr_fail
|
|
55
|
+
if cumulative_llr >= self.threshold_a:
|
|
56
|
+
decision = Decision.KILLED
|
|
57
|
+
break
|
|
58
|
+
if cumulative_llr <= self.threshold_b:
|
|
59
|
+
decision = Decision.SURVIVED
|
|
60
|
+
break
|
|
61
|
+
|
|
62
|
+
logger.debug("SPRT %s after %d step(s) (LLR=%.3f)", decision.value, step, cumulative_llr)
|
|
63
|
+
return SPRTResult(decision=decision, sample_count=step, cumulative_llr=cumulative_llr)
|
clastogen/exceptions.py
ADDED
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""Raised inside a mutation trial and translated by the plugin into SKIPPED or ERROR statuses."""
|
|
2
|
+
|
|
3
|
+
|
|
4
|
+
class TrialSkipped(Exception):
|
|
5
|
+
"""The test called pytest.skip during a trial."""
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class TrialError(Exception):
|
|
9
|
+
"""A trial could not produce a pass/fail outcome; the message is the recorded reason."""
|
clastogen/models.py
ADDED
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
|
|
6
|
+
from clastogen.types import Decision, MutantStatus
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
@dataclass(frozen=True)
|
|
10
|
+
class Mutant:
|
|
11
|
+
"""Stable, content-addressed mutant specification."""
|
|
12
|
+
|
|
13
|
+
id: str
|
|
14
|
+
target_symbol: str
|
|
15
|
+
operator_name: str
|
|
16
|
+
original_snippet: str
|
|
17
|
+
mutated_snippet: str
|
|
18
|
+
mutated_prompt: str
|
|
19
|
+
description: str
|
|
20
|
+
|
|
21
|
+
@classmethod
|
|
22
|
+
def create(
|
|
23
|
+
cls,
|
|
24
|
+
target_symbol: str,
|
|
25
|
+
operator_name: str,
|
|
26
|
+
original_snippet: str,
|
|
27
|
+
mutated_snippet: str,
|
|
28
|
+
mutated_prompt: str,
|
|
29
|
+
description: str,
|
|
30
|
+
) -> Mutant:
|
|
31
|
+
"""Constructs a mutant with a stable 12-character SHA-256 hash ID."""
|
|
32
|
+
content_key = f"{operator_name}:{target_symbol}:{original_snippet}->{mutated_snippet}"
|
|
33
|
+
stable_id = hashlib.sha256(content_key.encode("utf-8")).hexdigest()[:12]
|
|
34
|
+
return cls(
|
|
35
|
+
id=stable_id,
|
|
36
|
+
target_symbol=target_symbol,
|
|
37
|
+
operator_name=operator_name,
|
|
38
|
+
original_snippet=original_snippet,
|
|
39
|
+
mutated_snippet=mutated_snippet,
|
|
40
|
+
mutated_prompt=mutated_prompt,
|
|
41
|
+
description=description,
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(frozen=True)
|
|
46
|
+
class MutantExecution:
|
|
47
|
+
"""Record of a single mutant evaluation by a test; sample_count and llr are None unless the SPRT ran."""
|
|
48
|
+
|
|
49
|
+
test_id: str
|
|
50
|
+
target: str
|
|
51
|
+
mutant_id: str
|
|
52
|
+
description: str
|
|
53
|
+
status: MutantStatus
|
|
54
|
+
sample_count: int | None = None
|
|
55
|
+
llr: float | None = None
|
|
56
|
+
error: str | None = None
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass(frozen=True)
|
|
60
|
+
class BaselineRecord:
|
|
61
|
+
"""Outcome of the unmutated baseline runs for a marked test; p0 is the Laplace estimate fed to the SPRT; error is set when a trial could not run cleanly."""
|
|
62
|
+
|
|
63
|
+
test_id: str
|
|
64
|
+
target: str
|
|
65
|
+
successes: int
|
|
66
|
+
runs: int
|
|
67
|
+
p0: float
|
|
68
|
+
stable: bool
|
|
69
|
+
error: str | None = None
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
@dataclass(frozen=True)
|
|
73
|
+
class MutationSummary:
|
|
74
|
+
"""Order-independent aggregate of all mutant evaluations in a session."""
|
|
75
|
+
|
|
76
|
+
results: tuple[MutantExecution, ...]
|
|
77
|
+
counts: dict[MutantStatus, int]
|
|
78
|
+
score: float | None
|
|
79
|
+
|
|
80
|
+
@property
|
|
81
|
+
def total(self) -> int:
|
|
82
|
+
return len(self.results)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@dataclass
|
|
86
|
+
class ClastogenPluginState:
|
|
87
|
+
"""Type-safe state stored in pytest config.stash."""
|
|
88
|
+
|
|
89
|
+
results: list[MutantExecution] = field(default_factory=list)
|
|
90
|
+
baselines: list[BaselineRecord] = field(default_factory=list)
|
|
91
|
+
summary: MutationSummary | None = None
|
|
92
|
+
killed_mutants: set[str] = field(default_factory=set)
|
|
93
|
+
suppressed_mutants: set[str] = field(default_factory=set)
|
|
94
|
+
marked_tests_count: int = 0
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
@dataclass(frozen=True)
|
|
98
|
+
class ClastogenParams:
|
|
99
|
+
"""Arguments of @pytest.mark.clastogen; p0=None measures the baseline instead of assuming it."""
|
|
100
|
+
|
|
101
|
+
target: str
|
|
102
|
+
max_mutants: int = 5
|
|
103
|
+
delta: float = 0.30
|
|
104
|
+
p0: float | None = None
|
|
105
|
+
max_steps: int = 20
|
|
106
|
+
|
|
107
|
+
def __post_init__(self) -> None:
|
|
108
|
+
for name in ("max_mutants", "max_steps"):
|
|
109
|
+
value = getattr(self, name)
|
|
110
|
+
if isinstance(value, bool) or not isinstance(value, int) or value < 1:
|
|
111
|
+
raise ValueError(f"{name} must be an integer >= 1, got {value!r}")
|
|
112
|
+
for name in ("delta", "p0"):
|
|
113
|
+
value = getattr(self, name)
|
|
114
|
+
if name == "p0" and value is None:
|
|
115
|
+
continue
|
|
116
|
+
if isinstance(value, bool) or not isinstance(value, int | float) or not 0.0 < value < 1.0:
|
|
117
|
+
raise ValueError(f"{name} must be a number in (0, 1), got {value!r}")
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
@dataclass(frozen=True)
|
|
121
|
+
class SPRTConfig:
|
|
122
|
+
"""Configuration for Wald's Sequential Probability Ratio Test."""
|
|
123
|
+
|
|
124
|
+
alpha: float = 0.05
|
|
125
|
+
beta: float = 0.10
|
|
126
|
+
p0: float = 0.90
|
|
127
|
+
p1: float = 0.60
|
|
128
|
+
max_steps: int = 20
|
|
129
|
+
|
|
130
|
+
def __post_init__(self) -> None:
|
|
131
|
+
if not (0.0 < self.alpha < 1.0):
|
|
132
|
+
raise ValueError(f"alpha must be in (0, 1), got {self.alpha}")
|
|
133
|
+
if not (0.0 < self.beta < 1.0):
|
|
134
|
+
raise ValueError(f"beta must be in (0, 1), got {self.beta}")
|
|
135
|
+
if not (0.0 < self.p1 < self.p0 < 1.0):
|
|
136
|
+
raise ValueError(f"Requirement 0 < p1 < p0 < 1 violated: p0={self.p0}, p1={self.p1}")
|
|
137
|
+
if isinstance(self.max_steps, bool) or not isinstance(self.max_steps, int) or self.max_steps < 1:
|
|
138
|
+
raise ValueError(f"max_steps must be an integer >= 1, got {self.max_steps}")
|
|
139
|
+
|
|
140
|
+
@classmethod
|
|
141
|
+
def from_absolute_drop(
|
|
142
|
+
cls,
|
|
143
|
+
p0: float,
|
|
144
|
+
delta: float,
|
|
145
|
+
alpha: float = 0.05,
|
|
146
|
+
beta: float = 0.10,
|
|
147
|
+
max_steps: int = 20,
|
|
148
|
+
) -> SPRTConfig:
|
|
149
|
+
"""Derives p1 by subtracting delta from p0, clamping p0 to [0.02, 0.99] to prevent division by zero."""
|
|
150
|
+
clamped_p0 = min(max(p0, 0.02), 0.99)
|
|
151
|
+
p1 = max(clamped_p0 - delta, 0.01)
|
|
152
|
+
if p1 >= clamped_p0:
|
|
153
|
+
raise ValueError(f"Drop delta={delta} leaves p1={p1} >= p0={clamped_p0}")
|
|
154
|
+
return cls(alpha=alpha, beta=beta, p0=clamped_p0, p1=p1, max_steps=max_steps)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
@dataclass(frozen=True)
|
|
158
|
+
class SPRTResult:
|
|
159
|
+
decision: Decision
|
|
160
|
+
sample_count: int
|
|
161
|
+
cumulative_llr: float
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
@dataclass(frozen=True, slots=True)
|
|
165
|
+
class PassRateResult:
|
|
166
|
+
"""Outcome of a statistical pass rate assertion."""
|
|
167
|
+
|
|
168
|
+
passed: bool
|
|
169
|
+
observed_rate: float
|
|
170
|
+
sample_count: int
|
|
171
|
+
success_count: int
|
|
172
|
+
ci_lower: float
|
|
173
|
+
ci_upper: float
|
|
174
|
+
confidence: float
|
|
175
|
+
decided: bool
|
|
176
|
+
description: str | None = None
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
@dataclass(frozen=True, slots=True)
|
|
180
|
+
class RegressionResult:
|
|
181
|
+
"""Outcome of a one-sided test that a candidate evaluator regressed relative to a baseline."""
|
|
182
|
+
|
|
183
|
+
regressed: bool
|
|
184
|
+
p_value: float
|
|
185
|
+
test_name: str
|
|
186
|
+
baseline_rate: float
|
|
187
|
+
candidate_rate: float
|
|
188
|
+
sample_count: int
|
|
189
|
+
description: str | None = None
|