clastogen 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
clastogen/__init__.py ADDED
@@ -0,0 +1,35 @@
1
+ """Clastogen: Mutation testing & statistical assertion framework for LLMs and Agents."""
2
+
3
+ from importlib.metadata import version
4
+
5
+ from clastogen.core.injection import override_prompt
6
+ from clastogen.core.mutator import PromptMutator
7
+ from clastogen.core.sprt import SPRT
8
+ from clastogen.models import Mutant, PassRateResult, RegressionResult, SPRTConfig, SPRTResult
9
+ from clastogen.stats import (
10
+ assert_no_regression,
11
+ assert_pass_rate,
12
+ compute_wilson_interval,
13
+ evaluate_pass_rate,
14
+ evaluate_regression,
15
+ )
16
+ from clastogen.types import Decision
17
+
18
+ __all__ = [
19
+ "SPRT",
20
+ "Decision",
21
+ "Mutant",
22
+ "PassRateResult",
23
+ "PromptMutator",
24
+ "RegressionResult",
25
+ "SPRTConfig",
26
+ "SPRTResult",
27
+ "assert_no_regression",
28
+ "assert_pass_rate",
29
+ "compute_wilson_interval",
30
+ "evaluate_pass_rate",
31
+ "evaluate_regression",
32
+ "override_prompt",
33
+ ]
34
+
35
+ __version__ = version("clastogen")
@@ -0,0 +1,59 @@
1
+ from __future__ import annotations
2
+
3
+ import importlib
4
+ import logging
5
+ import sys
6
+ import types
7
+ from collections.abc import Generator
8
+ from contextlib import ExitStack, contextmanager
9
+ from functools import reduce
10
+ from unittest.mock import patch
11
+
12
+ logger = logging.getLogger(__name__)
13
+
14
+
15
+ def resolve_target(target_spec: str) -> tuple[object, str]:
16
+ """Resolves 'module:attr' or 'module:Class.attr' (dots after the colon walk nested attributes) into (owner, attr)."""
17
+ module_name, sep, attr_path = target_spec.partition(":")
18
+ if not sep:
19
+ raise ValueError(f"Target '{target_spec}' must be 'module:attr' or 'module:Class.attr'")
20
+ *owner_path, attr = attr_path.split(".")
21
+ owner = reduce(getattr, owner_path, importlib.import_module(module_name))
22
+ if not hasattr(owner, attr):
23
+ raise AttributeError(f"Object {owner!r} has no attribute '{attr}'")
24
+ return owner, attr
25
+
26
+
27
+ def _aliasing_modules(owner: object, attr: str, original: object) -> list[types.ModuleType]:
28
+ """Modules other than owner that bound the same string under the same name, e.g. via `from agent import PROMPT`."""
29
+ if not isinstance(original, str):
30
+ return []
31
+ return [
32
+ mod
33
+ for mod in list(sys.modules.values())
34
+ if isinstance(mod, types.ModuleType) and mod is not owner and vars(mod).get(attr) is original
35
+ ]
36
+
37
+
38
+ @contextmanager
39
+ def override_prompt(target_spec: str, mutated_text: str) -> Generator[str, None, None]:
40
+ """Temporarily replaces the target attribute, and every same-name module alias of it, with mutated_text.
41
+
42
+ unittest.mock.patch.object restores descriptors such as @classmethod and removes shadow attributes
43
+ on classes that only inherited the target.
44
+ """
45
+ if not isinstance(mutated_text, str):
46
+ raise TypeError(f"mutated_text must be a string, got {type(mutated_text).__name__}")
47
+
48
+ owner, attr = resolve_target(target_spec)
49
+ original = getattr(owner, attr)
50
+ if original is None:
51
+ raise TypeError(f"Target '{target_spec}' resolved to None, expected a prompt string.")
52
+
53
+ with ExitStack() as stack:
54
+ stack.enter_context(patch.object(owner, attr, mutated_text))
55
+ aliases = _aliasing_modules(owner, attr, original)
56
+ for mod in aliases:
57
+ stack.enter_context(patch.object(mod, attr, mutated_text))
58
+ logger.debug("Injected prompt into target '%s' (%d module alias(es))", target_spec, len(aliases))
59
+ yield mutated_text
@@ -0,0 +1,140 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+ from collections.abc import Sequence
5
+ from decimal import Decimal
6
+ from itertools import zip_longest
7
+
8
+ from clastogen.models import Mutant
9
+
10
+
11
+ class PromptMutator:
12
+ """Extracts load-bearing imperative constraints and generates targeted prompt mutants."""
13
+
14
+ RULE_PATTERN = re.compile(
15
+ r"\b(?:never|do not|don't|must not|mustn't|should not|shouldn't|cannot|can't|always|must|shall not|may not|avoid|only|requires?|required)\b",
16
+ re.IGNORECASE,
17
+ )
18
+
19
+ SENTENCE_SPLIT_PATTERN = re.compile(r"\n+|(?<!\be\.g)(?<!\bi\.e)(?<!\betc)[.?!]\s+")
20
+
21
+ NEGATION_REPLACEMENTS: tuple[tuple[re.Pattern[str], str], ...] = (
22
+ (re.compile(r"\bdo not\b", re.IGNORECASE), "do"),
23
+ (re.compile(r"\bdon't\b", re.IGNORECASE), "do"),
24
+ (re.compile(r"\bmust not\b", re.IGNORECASE), "must"),
25
+ (re.compile(r"\bmustn't\b", re.IGNORECASE), "must"),
26
+ (re.compile(r"\bnever\b", re.IGNORECASE), "always"),
27
+ (re.compile(r"\bcannot\b", re.IGNORECASE), "can"),
28
+ (re.compile(r"\bcan't\b", re.IGNORECASE), "can"),
29
+ (re.compile(r"\bshould not\b", re.IGNORECASE), "should"),
30
+ (re.compile(r"\bshouldn't\b", re.IGNORECASE), "should"),
31
+ (re.compile(r"\bshall not\b", re.IGNORECASE), "shall"),
32
+ (re.compile(r"\bmay not\b", re.IGNORECASE), "may"),
33
+ (re.compile(r"\balways\b", re.IGNORECASE), "never"),
34
+ (re.compile(r"\bmust\b", re.IGNORECASE), "must not"),
35
+ (re.compile(r"\bavoid\b", re.IGNORECASE), "prefer"),
36
+ )
37
+
38
+ NUMBER_PATTERN = re.compile(r"\b\d+(?:,\d{3})*(?:\.\d+)?\b")
39
+
40
+ def __init__(self, target_symbol: str = "SYSTEM_PROMPT") -> None:
41
+ self.target_symbol = target_symbol
42
+
43
+ def extract_candidate_rules(self, prompt: str) -> list[str]:
44
+ """Extracts candidate load-bearing sentences using linear string splitting and keyword search."""
45
+ if not isinstance(prompt, str):
46
+ raise TypeError(f"prompt must be a string, got {type(prompt).__name__}")
47
+ return [
48
+ cleaned
49
+ for line in self.SENTENCE_SPLIT_PATTERN.split(prompt)
50
+ if (cleaned := re.sub(r"^[-*]\s+", "", line.strip()))
51
+ and len(cleaned) > 10
52
+ and self.RULE_PATTERN.search(cleaned)
53
+ ]
54
+
55
+ def _delete_mutant(self, prompt: str, rule: str) -> Mutant | None:
56
+ """Removes rule with its bullet and trailing punctuation; None when the prompt is unchanged."""
57
+ clean_pat = re.compile(rf"[ \t]*(?:[-*]\s+)?{re.escape(rule)}[.?!]?([ \t]*)(\n?)")
58
+
59
+ def join(m: re.Match[str]) -> str:
60
+ if m.start() == 0 or prompt[m.start() - 1] == "\n":
61
+ return ""
62
+ return m[2] or ("" if m.end() == len(prompt) else " ")
63
+
64
+ mutated = clean_pat.sub(join, prompt).strip()
65
+ mutated = re.sub(r"\n\s*\.\s*\n", "\n", mutated)
66
+ mutated = re.sub(r"\n{3,}", "\n\n", mutated)
67
+ if mutated == prompt:
68
+ return None
69
+ return Mutant.create(
70
+ target_symbol=self.target_symbol,
71
+ operator_name="delete_constraint",
72
+ original_snippet=rule,
73
+ mutated_snippet="[DELETED]",
74
+ mutated_prompt=mutated,
75
+ description=f"Deleted load-bearing constraint: '{rule[:50]}...'",
76
+ )
77
+
78
+ def _invert_mutant(self, prompt: str, rule: str) -> Mutant | None:
79
+ """Flips the first matching negation or obligation keyword in rule; None when none applies."""
80
+ for pattern, replacement in self.NEGATION_REPLACEMENTS:
81
+ if pattern.search(rule):
82
+ inverted = pattern.sub(replacement.upper(), rule, count=1)
83
+ mutated = prompt.replace(rule, inverted)
84
+ if mutated != prompt:
85
+ return Mutant.create(
86
+ target_symbol=self.target_symbol,
87
+ operator_name="invert_negation",
88
+ original_snippet=rule,
89
+ mutated_snippet=inverted,
90
+ mutated_prompt=mutated,
91
+ description=f"Inverted constraint: '{rule[:40]}' -> '{inverted[:40]}'",
92
+ )
93
+ return None
94
+
95
+ def _threshold_mutant(self, prompt: str, rule: str) -> Mutant | None:
96
+ """Multiplies the first number in rule by 10 (a $50 limit becomes $500); None when rule has no number."""
97
+ match = self.NUMBER_PATTERN.search(rule)
98
+ if match is None:
99
+ return None
100
+ raw = match.group()
101
+ scaled = Decimal(raw.replace(",", "")) * 10
102
+ number = f"{scaled:,}" if "," in raw else str(scaled)
103
+ changed = rule[: match.start()] + number + rule[match.end() :]
104
+ mutated = prompt.replace(rule, changed)
105
+ if mutated == prompt:
106
+ return None
107
+ return Mutant.create(
108
+ target_symbol=self.target_symbol,
109
+ operator_name="change_threshold",
110
+ original_snippet=rule,
111
+ mutated_snippet=changed,
112
+ mutated_prompt=mutated,
113
+ description=f"Changed threshold: '{raw}' -> '{number}' in '{rule[:40]}'",
114
+ )
115
+
116
+ def generate_mutants(self, prompt: str, max_mutants: int) -> Sequence[Mutant]:
117
+ """Generates a bounded sequence of mutants (constraint deletions, negation inversions, threshold changes)."""
118
+ if not isinstance(prompt, str):
119
+ raise TypeError(f"prompt must be a string, got {type(prompt).__name__}")
120
+ if max_mutants < 1:
121
+ raise ValueError(f"max_mutants must be >= 1, got {max_mutants}")
122
+
123
+ candidates = list(dict.fromkeys(self.extract_candidate_rules(prompt)))
124
+ if not candidates:
125
+ return []
126
+
127
+ needed_rules = max(1, (max_mutants + 1) // 2)
128
+ if len(candidates) > needed_rules:
129
+ step = len(candidates) / needed_rules
130
+ selected_rules = [candidates[int(i * step)] for i in range(needed_rules)]
131
+ else:
132
+ selected_rules = candidates
133
+
134
+ operators = (self._delete_mutant, self._invert_mutant, self._threshold_mutant)
135
+ per_rule = [
136
+ [m for op in operators[i % 3 :] + operators[: i % 3] if (m := op(prompt, rule))]
137
+ for i, rule in enumerate(selected_rules)
138
+ ]
139
+ mutants = [m for group in zip_longest(*per_rule) for m in group if m]
140
+ return mutants[:max_mutants]
clastogen/core/sprt.py ADDED
@@ -0,0 +1,63 @@
1
+ import inspect
2
+ import logging
3
+ import math
4
+ from collections.abc import Callable, Sequence
5
+
6
+ from clastogen.models import SPRTConfig, SPRTResult
7
+ from clastogen.types import Decision
8
+
9
+ logger = logging.getLogger(__name__)
10
+
11
+ # Unmutated runs used to estimate p0; the Laplace estimate with n=10 keeps false KILLs on unchanged mutants low.
12
+ BASELINE_RUNS = 10
13
+ MIN_BASELINE_RATE = 0.80
14
+
15
+
16
+ def check_outcome(raw: object) -> bool:
17
+ """Returns raw if it is a bool; rejects awaitables and other types with TypeError."""
18
+ if inspect.isawaitable(raw):
19
+ raise TypeError("evaluator returned an awaitable/coroutine; expected synchronous callable returning bool")
20
+ if not isinstance(raw, bool):
21
+ raise TypeError(f"evaluator must return bool, got {type(raw).__name__}")
22
+ return raw
23
+
24
+
25
+ def estimate_p0(outcomes: Sequence[bool]) -> float:
26
+ """Estimates the baseline pass probability with Laplace's rule of succession, (s + 1) / (n + 2)."""
27
+ return (sum(outcomes) + 1) / (len(outcomes) + 2)
28
+
29
+
30
+ def config_from_baseline(outcomes: Sequence[bool], *, delta: float, max_steps: int) -> SPRTConfig | None:
31
+ """Builds the SPRT config from baseline outcomes, or None when the raw pass rate is below MIN_BASELINE_RATE."""
32
+ if sum(outcomes) / len(outcomes) < MIN_BASELINE_RATE:
33
+ return None
34
+ return SPRTConfig.from_absolute_drop(p0=estimate_p0(outcomes), delta=delta, max_steps=max_steps)
35
+
36
+
37
+ class SPRT:
38
+ """Wald's Sequential Probability Ratio Test engine for Bernoulli trials with early stopping."""
39
+
40
+ def __init__(self, config: SPRTConfig) -> None:
41
+ self.config = config
42
+ self.threshold_a = math.log((1.0 - config.beta) / config.alpha)
43
+ self.threshold_b = math.log(config.beta / (1.0 - config.alpha))
44
+ self.llr_pass = math.log(config.p1 / config.p0)
45
+ self.llr_fail = math.log((1.0 - config.p1) / (1.0 - config.p0))
46
+
47
+ def run_evaluator(self, evaluator: Callable[[], bool]) -> SPRTResult:
48
+ """Executes evaluator dynamically until a decision is reached or max_steps is exhausted."""
49
+ cumulative_llr = 0.0
50
+ decision = Decision.INCONCLUSIVE
51
+ step = 0
52
+ while step < self.config.max_steps:
53
+ step += 1
54
+ cumulative_llr += self.llr_pass if check_outcome(evaluator()) else self.llr_fail
55
+ if cumulative_llr >= self.threshold_a:
56
+ decision = Decision.KILLED
57
+ break
58
+ if cumulative_llr <= self.threshold_b:
59
+ decision = Decision.SURVIVED
60
+ break
61
+
62
+ logger.debug("SPRT %s after %d step(s) (LLR=%.3f)", decision.value, step, cumulative_llr)
63
+ return SPRTResult(decision=decision, sample_count=step, cumulative_llr=cumulative_llr)
@@ -0,0 +1,9 @@
1
+ """Raised inside a mutation trial and translated by the plugin into SKIPPED or ERROR statuses."""
2
+
3
+
4
+ class TrialSkipped(Exception):
5
+ """The test called pytest.skip during a trial."""
6
+
7
+
8
+ class TrialError(Exception):
9
+ """A trial could not produce a pass/fail outcome; the message is the recorded reason."""
clastogen/models.py ADDED
@@ -0,0 +1,189 @@
1
+ from __future__ import annotations
2
+
3
+ import hashlib
4
+ from dataclasses import dataclass, field
5
+
6
+ from clastogen.types import Decision, MutantStatus
7
+
8
+
9
+ @dataclass(frozen=True)
10
+ class Mutant:
11
+ """Stable, content-addressed mutant specification."""
12
+
13
+ id: str
14
+ target_symbol: str
15
+ operator_name: str
16
+ original_snippet: str
17
+ mutated_snippet: str
18
+ mutated_prompt: str
19
+ description: str
20
+
21
+ @classmethod
22
+ def create(
23
+ cls,
24
+ target_symbol: str,
25
+ operator_name: str,
26
+ original_snippet: str,
27
+ mutated_snippet: str,
28
+ mutated_prompt: str,
29
+ description: str,
30
+ ) -> Mutant:
31
+ """Constructs a mutant with a stable 12-character SHA-256 hash ID."""
32
+ content_key = f"{operator_name}:{target_symbol}:{original_snippet}->{mutated_snippet}"
33
+ stable_id = hashlib.sha256(content_key.encode("utf-8")).hexdigest()[:12]
34
+ return cls(
35
+ id=stable_id,
36
+ target_symbol=target_symbol,
37
+ operator_name=operator_name,
38
+ original_snippet=original_snippet,
39
+ mutated_snippet=mutated_snippet,
40
+ mutated_prompt=mutated_prompt,
41
+ description=description,
42
+ )
43
+
44
+
45
+ @dataclass(frozen=True)
46
+ class MutantExecution:
47
+ """Record of a single mutant evaluation by a test; sample_count and llr are None unless the SPRT ran."""
48
+
49
+ test_id: str
50
+ target: str
51
+ mutant_id: str
52
+ description: str
53
+ status: MutantStatus
54
+ sample_count: int | None = None
55
+ llr: float | None = None
56
+ error: str | None = None
57
+
58
+
59
+ @dataclass(frozen=True)
60
+ class BaselineRecord:
61
+ """Outcome of the unmutated baseline runs for a marked test; p0 is the Laplace estimate fed to the SPRT; error is set when a trial could not run cleanly."""
62
+
63
+ test_id: str
64
+ target: str
65
+ successes: int
66
+ runs: int
67
+ p0: float
68
+ stable: bool
69
+ error: str | None = None
70
+
71
+
72
+ @dataclass(frozen=True)
73
+ class MutationSummary:
74
+ """Order-independent aggregate of all mutant evaluations in a session."""
75
+
76
+ results: tuple[MutantExecution, ...]
77
+ counts: dict[MutantStatus, int]
78
+ score: float | None
79
+
80
+ @property
81
+ def total(self) -> int:
82
+ return len(self.results)
83
+
84
+
85
+ @dataclass
86
+ class ClastogenPluginState:
87
+ """Type-safe state stored in pytest config.stash."""
88
+
89
+ results: list[MutantExecution] = field(default_factory=list)
90
+ baselines: list[BaselineRecord] = field(default_factory=list)
91
+ summary: MutationSummary | None = None
92
+ killed_mutants: set[str] = field(default_factory=set)
93
+ suppressed_mutants: set[str] = field(default_factory=set)
94
+ marked_tests_count: int = 0
95
+
96
+
97
+ @dataclass(frozen=True)
98
+ class ClastogenParams:
99
+ """Arguments of @pytest.mark.clastogen; p0=None measures the baseline instead of assuming it."""
100
+
101
+ target: str
102
+ max_mutants: int = 5
103
+ delta: float = 0.30
104
+ p0: float | None = None
105
+ max_steps: int = 20
106
+
107
+ def __post_init__(self) -> None:
108
+ for name in ("max_mutants", "max_steps"):
109
+ value = getattr(self, name)
110
+ if isinstance(value, bool) or not isinstance(value, int) or value < 1:
111
+ raise ValueError(f"{name} must be an integer >= 1, got {value!r}")
112
+ for name in ("delta", "p0"):
113
+ value = getattr(self, name)
114
+ if name == "p0" and value is None:
115
+ continue
116
+ if isinstance(value, bool) or not isinstance(value, int | float) or not 0.0 < value < 1.0:
117
+ raise ValueError(f"{name} must be a number in (0, 1), got {value!r}")
118
+
119
+
120
+ @dataclass(frozen=True)
121
+ class SPRTConfig:
122
+ """Configuration for Wald's Sequential Probability Ratio Test."""
123
+
124
+ alpha: float = 0.05
125
+ beta: float = 0.10
126
+ p0: float = 0.90
127
+ p1: float = 0.60
128
+ max_steps: int = 20
129
+
130
+ def __post_init__(self) -> None:
131
+ if not (0.0 < self.alpha < 1.0):
132
+ raise ValueError(f"alpha must be in (0, 1), got {self.alpha}")
133
+ if not (0.0 < self.beta < 1.0):
134
+ raise ValueError(f"beta must be in (0, 1), got {self.beta}")
135
+ if not (0.0 < self.p1 < self.p0 < 1.0):
136
+ raise ValueError(f"Requirement 0 < p1 < p0 < 1 violated: p0={self.p0}, p1={self.p1}")
137
+ if isinstance(self.max_steps, bool) or not isinstance(self.max_steps, int) or self.max_steps < 1:
138
+ raise ValueError(f"max_steps must be an integer >= 1, got {self.max_steps}")
139
+
140
+ @classmethod
141
+ def from_absolute_drop(
142
+ cls,
143
+ p0: float,
144
+ delta: float,
145
+ alpha: float = 0.05,
146
+ beta: float = 0.10,
147
+ max_steps: int = 20,
148
+ ) -> SPRTConfig:
149
+ """Derives p1 by subtracting delta from p0, clamping p0 to [0.02, 0.99] to prevent division by zero."""
150
+ clamped_p0 = min(max(p0, 0.02), 0.99)
151
+ p1 = max(clamped_p0 - delta, 0.01)
152
+ if p1 >= clamped_p0:
153
+ raise ValueError(f"Drop delta={delta} leaves p1={p1} >= p0={clamped_p0}")
154
+ return cls(alpha=alpha, beta=beta, p0=clamped_p0, p1=p1, max_steps=max_steps)
155
+
156
+
157
+ @dataclass(frozen=True)
158
+ class SPRTResult:
159
+ decision: Decision
160
+ sample_count: int
161
+ cumulative_llr: float
162
+
163
+
164
+ @dataclass(frozen=True, slots=True)
165
+ class PassRateResult:
166
+ """Outcome of a statistical pass rate assertion."""
167
+
168
+ passed: bool
169
+ observed_rate: float
170
+ sample_count: int
171
+ success_count: int
172
+ ci_lower: float
173
+ ci_upper: float
174
+ confidence: float
175
+ decided: bool
176
+ description: str | None = None
177
+
178
+
179
+ @dataclass(frozen=True, slots=True)
180
+ class RegressionResult:
181
+ """Outcome of a one-sided test that a candidate evaluator regressed relative to a baseline."""
182
+
183
+ regressed: bool
184
+ p_value: float
185
+ test_name: str
186
+ baseline_rate: float
187
+ candidate_rate: float
188
+ sample_count: int
189
+ description: str | None = None