crapkit 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (59) hide show
  1. crapkit/__init__.py +2 -0
  2. crapkit/__main__.py +5 -0
  3. crapkit/_pygdefer.py +86 -0
  4. crapkit/analyze.py +375 -0
  5. crapkit/cache.py +58 -0
  6. crapkit/churn.py +113 -0
  7. crapkit/churn_cache.py +108 -0
  8. crapkit/churn_log.py +286 -0
  9. crapkit/cli/__init__.py +316 -0
  10. crapkit/cli/_shared.py +130 -0
  11. crapkit/cli/admin.py +650 -0
  12. crapkit/cli/analyses.py +144 -0
  13. crapkit/cli/parser.py +384 -0
  14. crapkit/cli/queue.py +926 -0
  15. crapkit/cli/ratchet_cmds.py +172 -0
  16. crapkit/cli/reports.py +459 -0
  17. crapkit/cli/scoring.py +500 -0
  18. crapkit/cli/verifying.py +580 -0
  19. crapkit/config.py +289 -0
  20. crapkit/coupling.py +89 -0
  21. crapkit/coverage_istanbul.py +225 -0
  22. crapkit/coverage_py.py +87 -0
  23. crapkit/covstream.py +320 -0
  24. crapkit/diffparse.py +98 -0
  25. crapkit/digest.py +191 -0
  26. crapkit/discover.py +365 -0
  27. crapkit/doctor.py +308 -0
  28. crapkit/dup.py +179 -0
  29. crapkit/errors.py +18 -0
  30. crapkit/gitio.py +504 -0
  31. crapkit/hook.py +167 -0
  32. crapkit/junitparse.py +87 -0
  33. crapkit/lanes.py +373 -0
  34. crapkit/lizardcognitive.py +238 -0
  35. crapkit/mcp_server.py +167 -0
  36. crapkit/merge.py +77 -0
  37. crapkit/mutate.py +96 -0
  38. crapkit/mutate_pool.py +152 -0
  39. crapkit/override.py +94 -0
  40. crapkit/packet.py +343 -0
  41. crapkit/ratchet.py +236 -0
  42. crapkit/ratchet_report.py +135 -0
  43. crapkit/sarif.py +82 -0
  44. crapkit/sarifio.py +49 -0
  45. crapkit/scaffold.py +361 -0
  46. crapkit/score.py +255 -0
  47. crapkit/snapshot.py +51 -0
  48. crapkit/store.py +1066 -0
  49. crapkit/uncovered.py +131 -0
  50. crapkit/universe.py +157 -0
  51. crapkit/verify.py +194 -0
  52. crapkit/watch.py +112 -0
  53. crapkit/worklist.py +290 -0
  54. crapkit-0.2.0.dist-info/METADATA +802 -0
  55. crapkit-0.2.0.dist-info/RECORD +59 -0
  56. crapkit-0.2.0.dist-info/WHEEL +5 -0
  57. crapkit-0.2.0.dist-info/entry_points.txt +2 -0
  58. crapkit-0.2.0.dist-info/licenses/LICENSE +21 -0
  59. crapkit-0.2.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,238 @@
1
+ """Cognitive complexity (Sonar spec) as a lizard token-stream extension.
2
+
3
+ Rides lizard's language-aware tokenizers, so TS/TSX/JS/Python all pay the same
4
+ rules with no second parse and no new dependency:
5
+ +1 and +nesting for if / ternary / switch / loops / catch-except
6
+ +1 flat for else / elif (an else-if chain costs one per link, no deepening)
7
+ +1 per boolean-operator run, +1 each time the operator alternates
8
+ +1 for a labeled break/continue or goto, +1 once for direct recursion
9
+ try / finally / case labels / with are free; nesting rises inside the
10
+ block structures listed above.
11
+
12
+ Attribution follows lizard's function splitting (a nested arrow's tokens are
13
+ the arrow's), exactly as ccn is attributed today. Ternary branches do not
14
+ deepen nesting (a structure inside a ternary arm is rare enough to accept).
15
+
16
+ The whitepaper's worked examples in tests/unit/test_cognitive.py are the spec.
17
+ """
18
+ from __future__ import annotations
19
+
20
+ _COUNTING = frozenset({"if", "for", "foreach", "while", "do", "catch", "except", "switch"})
21
+ _BOOL_OPS = frozenset({"&&", "||", "??", "and", "or"})
22
+ _RUN_RESETS = frozenset({";", ",", "{", "}"})
23
+
24
+
25
+ class _FnState:
26
+ __slots__ = ("total", "stack", "brace_depth", "line_indent", "at_line_start",
27
+ "pending", "else_pending", "question_pending", "bool_op", "name",
28
+ "recursed", "body_started", "prev", "label_check")
29
+
30
+ def __init__(self, name: str):
31
+ self.total = 0
32
+ self.stack = [] # (entry_brace_depth) or python header indents
33
+ self.brace_depth = 0
34
+ self.line_indent = 0
35
+ self.at_line_start = True
36
+ self.pending = False # a counting structure awaits its '{'
37
+ self.else_pending = False
38
+ self.question_pending = False
39
+ self.bool_op = None
40
+ self.name = name
41
+ self.recursed = False
42
+ self.body_started = False
43
+ self.prev = ""
44
+ self.label_check = False # just saw break/continue
45
+
46
+
47
+ class LizardExtension:
48
+ """One instance per analysis pass; state is per lizard FunctionInfo."""
49
+
50
+ FUNCTION_INFO = {"cognitive_complexity": {"caption": " Cog "}}
51
+
52
+ def __call__(self, tokens, reader):
53
+ # Keyed on the FunctionInfo itself, never on id(fn): the map would hold no
54
+ # reference, a finished function's address would be recycled under a later
55
+ # one, and that one would inherit a stranger's running total. Measured on
56
+ # the consumer repo: 377-379 rows moved between two runs of the same commit.
57
+ states: dict[object, _FnState] = {}
58
+ is_python = type(reader).__name__.lower().startswith("python")
59
+ for token in tokens:
60
+ fn = reader.context.current_function
61
+ state = states.get(fn)
62
+ if state is None:
63
+ state = states[fn] = _FnState(getattr(fn, "name", ""))
64
+ _step(state, token, is_python)
65
+ fn.cognitive_complexity = state.total
66
+ yield token
67
+
68
+
69
+ def _step(state: _FnState, token: str, is_python: bool) -> None:
70
+ if not token.strip():
71
+ _line_event(state, token, is_python)
72
+ return
73
+ if token.startswith(("#", "//", "/*")):
74
+ return # a comment token must never read as code, whatever it contains
75
+ if is_python and state.at_line_start:
76
+ _python_dedent(state)
77
+ if _resolve_lookbehinds(state, token, is_python):
78
+ state.prev = token
79
+ state.at_line_start = False
80
+ return
81
+ _consume(state, token, is_python)
82
+ state.prev = token
83
+ state.at_line_start = False
84
+
85
+
86
+ def _line_event(state: _FnState, token: str, is_python: bool) -> None:
87
+ """Whitespace arrives split ('\\n' then ' '): the newline opens the line,
88
+ later whitespace extends its indent, and the dedent settles only when the
89
+ first real token of the line arrives."""
90
+ if "\n" in token:
91
+ state.at_line_start = True
92
+ state.bool_op = None
93
+ state.line_indent = len(token) - token.rfind("\n") - 1
94
+ state.body_started = state.body_started or is_python
95
+ elif state.at_line_start:
96
+ state.line_indent += len(token)
97
+
98
+
99
+ def _python_dedent(state: _FnState) -> None:
100
+ while state.stack and state.line_indent <= state.stack[-1]:
101
+ state.stack.pop()
102
+
103
+
104
+ def _resolve_lookbehinds(state: _FnState, token: str, is_python: bool) -> bool:
105
+ """Signals needing one token of hindsight. True = this token is consumed."""
106
+ if state.question_pending:
107
+ _resolve_question(state, token, is_python)
108
+ if state.label_check:
109
+ _resolve_label(state, token)
110
+ if state.else_pending:
111
+ state.else_pending = False
112
+ state.pending = True
113
+ # else-if: the else already paid the flat +1; this if only opens the block
114
+ return token == "if"
115
+ return False
116
+
117
+
118
+ def _resolve_question(state: _FnState, token: str, is_python: bool) -> None:
119
+ state.question_pending = False
120
+ if token not in (".", ":", ")"): # optional chaining / optional type / trailing
121
+ state.total += 1 + _nesting(state, is_python)
122
+
123
+
124
+ def _resolve_label(state: _FnState, token: str) -> None:
125
+ state.label_check = False
126
+ if token not in (";", "}", ")") and token.strip():
127
+ state.total += 1 # break/continue TO A LABEL
128
+
129
+
130
+ def _nesting(state: _FnState, is_python: bool) -> int:
131
+ return len(state.stack)
132
+
133
+
134
+ def _consume(state: _FnState, token: str, is_python: bool) -> None:
135
+ if token in _RUN_RESETS:
136
+ state.bool_op = None
137
+ if token in ("{", "}"):
138
+ _brace(state, token)
139
+ return
140
+ if is_python and not state.body_started:
141
+ return # tokens of the def header line never count
142
+ if token in _BOOL_OPS:
143
+ _bool_op(state, token)
144
+ return
145
+ _keywords(state, token, is_python)
146
+
147
+
148
+ def _brace(state: _FnState, token: str) -> None:
149
+ if token == "{":
150
+ _open_brace(state)
151
+ else:
152
+ _close_brace(state)
153
+
154
+
155
+ def _open_brace(state: _FnState) -> None:
156
+ if state.pending:
157
+ state.stack.append(state.brace_depth)
158
+ state.pending = False
159
+ state.brace_depth += 1
160
+
161
+
162
+ def _close_brace(state: _FnState) -> None:
163
+ state.brace_depth -= 1
164
+ if state.stack and state.stack[-1] == state.brace_depth:
165
+ state.stack.pop()
166
+
167
+
168
+ def _bool_op(state: _FnState, token: str) -> None:
169
+ op = {"and": "&&", "or": "||"}.get(token, token)
170
+ if op != state.bool_op:
171
+ state.total += 1
172
+ state.bool_op = op
173
+
174
+
175
+ def _keywords(state: _FnState, token: str, is_python: bool) -> None:
176
+ if token == "if":
177
+ _if_token(state, is_python)
178
+ elif token in ("else", "elif"):
179
+ _else_token(state, token, is_python)
180
+ elif token in _COUNTING:
181
+ _structure_token(state, token, is_python)
182
+ else:
183
+ _jumps_and_recursion(state, token, is_python)
184
+
185
+
186
+ def _structure_token(state: _FnState, token: str, is_python: bool) -> None:
187
+ if token == "while" and state.prev == "}":
188
+ return # the closing half of do-while; the do already paid
189
+ _structure(state, token, is_python)
190
+
191
+
192
+ def _jumps_and_recursion(state: _FnState, token: str, is_python: bool) -> None:
193
+ if token == "?":
194
+ state.question_pending = True
195
+ elif token in ("break", "continue"):
196
+ state.label_check = not is_python
197
+ elif token == "goto":
198
+ state.total += 1
199
+ elif _is_recursion(state, token):
200
+ state.recursed = True
201
+ state.total += 1
202
+
203
+
204
+ def _is_recursion(state: _FnState, token: str) -> bool:
205
+ return (token == state.name and bool(state.name) and not state.recursed
206
+ and (state.body_started or state.brace_depth > 0))
207
+
208
+
209
+ def _if_token(state: _FnState, is_python: bool) -> None:
210
+ if is_python and not state.at_line_start:
211
+ state.total += 1 + _nesting(state, is_python) # ternary expression form
212
+ return
213
+ state.total += 1 + _nesting(state, is_python)
214
+ _push_structure(state, is_python)
215
+
216
+
217
+ def _else_token(state: _FnState, token: str, is_python: bool) -> None:
218
+ if is_python and not state.at_line_start:
219
+ return # the else arm of a ternary expression is part of its +1
220
+ state.total += 1
221
+ if token == "elif":
222
+ _push_structure(state, is_python)
223
+ else:
224
+ state.else_pending = not is_python
225
+ if is_python:
226
+ _push_structure(state, is_python)
227
+
228
+
229
+ def _structure(state: _FnState, token: str, is_python: bool) -> None:
230
+ state.total += 1 + _nesting(state, is_python)
231
+ _push_structure(state, is_python)
232
+
233
+
234
+ def _push_structure(state: _FnState, is_python: bool) -> None:
235
+ if is_python:
236
+ state.stack.append(state.line_indent)
237
+ else:
238
+ state.pending = True
crapkit/mcp_server.py ADDED
@@ -0,0 +1,167 @@
1
+ """Minimal MCP stdio server. JSON-RPC 2.0, newline-delimited, no SDK dependency.
2
+
3
+ Read-side only: every tool shells to the CLI's own --json surface, so the MCP
4
+ view can never drift from what the CLI reports, and nothing here writes a
5
+ baseline, a ratchet, or a mutant. Runs that mutate state stay in the CLI.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import subprocess
11
+ import sys
12
+ from pathlib import Path
13
+
14
+ PROTOCOL_VERSION = "2024-11-05"
15
+
16
+ _REPO = {"repo": {"type": "string", "description": "repo root (default: the server's)"}}
17
+
18
+ TOOLS: tuple[dict, ...] = (
19
+ {"name": "next_item", "argv": ["next-item"], "json_flag": False, "positional": (),
20
+ "flags": {"top": "--top", "exclude": "--exclude"},
21
+ "description": "The highest-risk function(s) to refactor next, with scores and effort estimates",
22
+ "properties": {"top": {"type": "integer"}, "exclude": {"type": "array", "items": {"type": "string"}}}},
23
+ {"name": "worklist", "argv": ["worklist"], "json_flag": True, "positional": (),
24
+ "flags": {"top": "--top"},
25
+ "description": "Risk-ranked decomposition queue (ccn x recency-weighted churn)",
26
+ "properties": {"top": {"type": "integer"}}},
27
+ {"name": "runs", "argv": ["runs"], "json_flag": True, "positional": (), "flags": {},
28
+ "description": "Run history: id, kind, verdict, commit, lane set", "properties": {}},
29
+ {"name": "brief", "argv": ["brief"], "json_flag": True, "positional": ("path", "name"),
30
+ "flags": {},
31
+ "description": "One function's whole context: scored row, ratchet mark, uncovered lines, "
32
+ "duplication twins, churn and change-coupling partners",
33
+ "properties": {"path": {"type": "string", "description": "repo-relative source file"},
34
+ "name": {"type": "string",
35
+ "description": "the bare identifier (classify) or the whole "
36
+ "long_name next_item printed "
37
+ "(classify( score , late )); both resolve"}}},
38
+ {"name": "explain", "argv": ["explain"], "json_flag": False, "positional": ("path", "name"),
39
+ "flags": {},
40
+ "description": "One function's score trajectory across runs plus its ratchet mark",
41
+ "properties": {"path": {"type": "string"}, "name": {"type": "string"}}},
42
+ {"name": "doctor", "argv": ["doctor"], "json_flag": False, "positional": (), "flags": {},
43
+ "description": "Config/repo agreement check: typo keys, empty scopes, missing lane cwds",
44
+ "properties": {}},
45
+ {"name": "coupling", "argv": ["coupling"], "json_flag": True, "positional": (),
46
+ "flags": {"min_support": "--min-support", "min_confidence": "--min-confidence"},
47
+ "description": "File pairs that keep landing in the same commits (hidden dependencies)",
48
+ "properties": {"min_support": {"type": "integer"}, "min_confidence": {"type": "number"}}},
49
+ {"name": "duplication", "argv": ["duplication"], "json_flag": True, "positional": (),
50
+ "flags": {"similarity": "--similarity"},
51
+ "description": "Near-duplicate functions by normalized line shingles",
52
+ "properties": {"similarity": {"type": "number"}}},
53
+ {"name": "ratchet_report", "argv": ["ratchet", "report"], "json_flag": True, "positional": (),
54
+ "flags": {},
55
+ "description": "Debt burn-down: open mark ages and repayment velocity from git history",
56
+ "properties": {}},
57
+ )
58
+
59
+
60
+ def tool_listing() -> list[dict]:
61
+ return [{"name": t["name"], "description": t["description"],
62
+ "inputSchema": {"type": "object", "properties": {**_REPO, **t["properties"]}}}
63
+ for t in TOOLS]
64
+
65
+
66
+ def _flag_args(flags: dict, arguments: dict) -> list[str]:
67
+ out: list[str] = []
68
+ for key, flag in flags.items():
69
+ value = arguments.get(key)
70
+ if value is None:
71
+ continue
72
+ for v in (value if isinstance(value, list) else [value]):
73
+ out += [flag, str(v)]
74
+ return out
75
+
76
+
77
+ def build_argv(tool: dict, arguments: dict) -> list[str]:
78
+ argv = list(tool["argv"])
79
+ argv += [str(arguments[p]) for p in tool["positional"]]
80
+ argv += _flag_args(tool["flags"], arguments)
81
+ if tool["json_flag"]:
82
+ argv.append("--json")
83
+ return argv
84
+
85
+
86
+ def _result(text: str, *, is_error: bool) -> dict:
87
+ return {"content": [{"type": "text", "text": text}], "isError": is_error}
88
+
89
+
90
+ def _no_config_result(repo: str) -> dict:
91
+ """What every tool answers in a directory crapkit has never measured.
92
+
93
+ A tool result, never a JSON-RPC error: the client keeps the session, and the
94
+ caller reads one sentence that says what to run instead of a transport
95
+ failure. isError stays true so nothing reads an unmeasured directory as a
96
+ repo with nothing to report.
97
+ """
98
+ return _result(f"no crapkit.toml in {repo} — nothing measured here. "
99
+ "Run `crapkit init` in the repo you want scored, or pass this tool a "
100
+ "`repo` argument (or start the server with --repo) pointing at one.",
101
+ is_error=True)
102
+
103
+
104
+ def _run_cli(tool: dict, arguments: dict, repo: str) -> dict:
105
+ """One tool, run as the CLI command it maps to. A command that printed
106
+ nothing answers with its stderr, so a refusal reaches the caller as text."""
107
+ argv = build_argv(tool, arguments) + ["--repo", repo]
108
+ proc = subprocess.run([sys.executable, "-m", "crapkit", *argv],
109
+ capture_output=True, text=True, timeout=600)
110
+ text = proc.stdout if proc.stdout.strip() else proc.stderr
111
+ return _result(text, is_error=proc.returncode != 0)
112
+
113
+
114
+ def _call_tool(root: Path, name: str, arguments: dict) -> dict:
115
+ """Name lookup, then the repo the call names, then the run."""
116
+ tool = next((t for t in TOOLS if t["name"] == name), None)
117
+ if tool is None:
118
+ return _result(f"unknown tool {name!r}", is_error=True)
119
+ repo = arguments.get("repo") or str(root)
120
+ if not (Path(repo) / "crapkit.toml").is_file():
121
+ return _no_config_result(repo)
122
+ return _run_cli(tool, arguments, repo)
123
+
124
+
125
+ def _respond(msg_id, result=None, error=None) -> dict:
126
+ resp = {"jsonrpc": "2.0", "id": msg_id}
127
+ resp["error" if error else "result"] = error if error else result
128
+ return resp
129
+
130
+
131
+ def _handle(root: Path, msg: dict) -> dict | None:
132
+ method = msg.get("method", "")
133
+ if "id" not in msg:
134
+ return None # a notification (e.g. notifications/initialized) needs no reply
135
+ if method == "initialize":
136
+ return _respond(msg["id"], {
137
+ "protocolVersion": PROTOCOL_VERSION,
138
+ "capabilities": {"tools": {}},
139
+ "serverInfo": {"name": "crapkit", "version": _version()}})
140
+ if method == "tools/list":
141
+ return _respond(msg["id"], {"tools": tool_listing()})
142
+ if method == "tools/call":
143
+ params = msg.get("params", {})
144
+ return _respond(msg["id"], _call_tool(root, params.get("name", ""),
145
+ params.get("arguments", {})))
146
+ return _respond(msg["id"], error={"code": -32601, "message": f"unknown method {method!r}"})
147
+
148
+
149
+ def _version() -> str:
150
+ from . import __version__
151
+ return __version__
152
+
153
+
154
+ def serve(root: Path) -> int:
155
+ """Newline-delimited JSON-RPC over stdio until EOF."""
156
+ for line in sys.stdin:
157
+ if not line.strip():
158
+ continue
159
+ try:
160
+ msg = json.loads(line)
161
+ except ValueError:
162
+ continue
163
+ resp = _handle(root, msg)
164
+ if resp is not None:
165
+ sys.stdout.write(json.dumps(resp) + "\n")
166
+ sys.stdout.flush()
167
+ return 0
crapkit/merge.py ADDED
@@ -0,0 +1,77 @@
1
+ """Merge the standard and modified lizard passes into one record per function.
2
+
3
+ Pure. Join key is (path, start, end, long_name) — spans, never bare names,
4
+ because anonymous and nested functions collide on names within a file.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ from typing import NamedTuple
9
+
10
+
11
+ class RawFn(NamedTuple):
12
+ path: str
13
+ long_name: str
14
+ start: int
15
+ end: int
16
+ ccn: int
17
+ nloc: int
18
+ params: int
19
+ nesting: int
20
+ cognitive: int = 0
21
+
22
+
23
+ class FunctionRecord(NamedTuple):
24
+ path: str
25
+ long_name: str
26
+ start: int
27
+ end: int
28
+ ccn_std: int
29
+ ccn_mod: int
30
+ ccn: int
31
+ nloc: int
32
+ params: int
33
+ nesting: int
34
+ cognitive: int = 0 # Sonar-spec, from the standard pass; reporting only
35
+
36
+
37
+ def _key(fn: RawFn) -> tuple[str, int, int, str]:
38
+ return (fn.path, fn.start, fn.end, fn.long_name)
39
+
40
+
41
+ def _group_by_key(fns: list[RawFn]) -> dict[tuple, list[RawFn]]:
42
+ # Ordered lists, not one entry per key: two anonymous callbacks on one line
43
+ # share a key, and a plain dict would keep only the last twin.
44
+ grouped: dict[tuple, list[RawFn]] = {}
45
+ for f in fns:
46
+ grouped.setdefault(_key(f), []).append(f)
47
+ return grouped
48
+
49
+
50
+ def _key_disagreements(std_by_key: dict[tuple, list[RawFn]],
51
+ mod_by_key: dict[tuple, list[RawFn]]) -> list[tuple]:
52
+ # Keys one pass has and the other lacks; failing that, keys both have at
53
+ # differing multiplicity. Empty means the two passes line up exactly.
54
+ return sorted(set(std_by_key) ^ set(mod_by_key)) or sorted(
55
+ k for k, v in std_by_key.items() if len(v) != len(mod_by_key.get(k, ()))
56
+ )
57
+
58
+
59
+ def merge_passes(standard: list[RawFn], modified: list[RawFn]) -> list[FunctionRecord]:
60
+ mod_by_key = _group_by_key(modified)
61
+ diff = _key_disagreements(_group_by_key(standard), mod_by_key)
62
+ if diff:
63
+ raise ValueError(f"lizard pass mismatch: {len(diff)} function key(s) differ between passes: {diff[:5]}")
64
+ # Both passes emit twins in the same parse order, so pairing within a key
65
+ # is by position.
66
+ taken: dict[tuple, int] = {}
67
+ records = []
68
+ for fn in standard:
69
+ key = _key(fn)
70
+ mod = mod_by_key[key][taken.get(key, 0)]
71
+ taken[key] = taken.get(key, 0) + 1
72
+ records.append(FunctionRecord(
73
+ path=fn.path, long_name=fn.long_name, start=fn.start, end=fn.end,
74
+ ccn_std=fn.ccn, ccn_mod=mod.ccn, ccn=min(fn.ccn, mod.ccn),
75
+ nloc=fn.nloc, params=fn.params, nesting=fn.nesting, cognitive=fn.cognitive,
76
+ ))
77
+ return records
crapkit/mutate.py ADDED
@@ -0,0 +1,96 @@
1
+ """Line-based mutant generation. Pure: text + changed lines in, mutants out.
2
+
3
+ Diff-scoped by design: mutating a whole repo is a research project, mutating
4
+ the lines a change touched is a review step. Operators flip comparisons,
5
+ boolean connectives, and boolean literals — the mutants that catch a test
6
+ suite asserting nothing. String literals are masked and comment lines skipped:
7
+ a survivor in dead text would erode trust in the survivor list.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import re
12
+ from typing import NamedTuple
13
+
14
+
15
+ class Mutant(NamedTuple):
16
+ path: str
17
+ line: int # 1-indexed
18
+ original: str # the whole original line
19
+ mutated: str # the whole mutated line
20
+ op: str # "a -> b"
21
+
22
+
23
+ # source token -> mutation targets. Negation flips plus boundary shifts —
24
+ # the boundary mutant (> vs >=) is the one an off-by-one test hole misses.
25
+ # Word-ish operators bind on spaces to dodge identifiers like Sand/oreo.
26
+ _OPS = {
27
+ "python": {"==": ("!=",), "!=": ("==",), "<=": ("<", ">"), ">=": (">", "<"),
28
+ "<": ("<=", ">="), ">": (">=", "<="),
29
+ " and ": (" or ",), " or ": (" and ",), "True": ("False",), "False": ("True",)},
30
+ "typescript": {"===": ("!==",), "!==": ("===",), "==": ("!=",), "!=": ("==",),
31
+ "<=": ("<", ">"), ">=": (">", "<"), "<": ("<=", ">="), ">": (">=", "<="),
32
+ "&&": ("||",), "||": ("&&",), "true": ("false",), "false": ("true",)},
33
+ }
34
+ # a short token matching INSIDE one of these is not that operator (== in ===,
35
+ # > in => arrows, < in <=): skip the occurrence entirely
36
+ _PROTECT = ("===", "!==", "==", "!=", "<=", ">=", "=>", "->")
37
+ _COMMENT_PREFIXES = ("#", "//", "/*", "*")
38
+ _STRING_RE = re.compile(r"'[^']*'|\"[^\"]*\"|`[^`]*`")
39
+
40
+
41
+ def _masked(line: str) -> str:
42
+ """String literal contents blanked, length preserved, so operator offsets
43
+ found on the mask apply to the real line."""
44
+ return _STRING_RE.sub(lambda m: " " * len(m.group()), line)
45
+
46
+
47
+ def _occurrences(mask: str, needle: str) -> list[int]:
48
+ out, at = [], mask.find(needle)
49
+ while at != -1:
50
+ out.append(at)
51
+ at = mask.find(needle, at + 1)
52
+ return out
53
+
54
+
55
+ def _covered_by_longer(mask: str, at: int, token: str) -> bool:
56
+ for p in _PROTECT:
57
+ if len(p) <= len(token):
58
+ continue
59
+ for start in range(max(0, at - len(p) + 1), at + 1):
60
+ if mask.startswith(p, start):
61
+ return True
62
+ return False
63
+
64
+
65
+ def _line_mutations(line: str, ops: dict) -> list[tuple[str, str]]:
66
+ """(mutated_line, op_label) for every operator occurrence outside strings."""
67
+ mask = _masked(line)
68
+ out = []
69
+ for source, targets in ops.items():
70
+ for at in _occurrences(mask, source):
71
+ if _covered_by_longer(mask, at, source):
72
+ continue
73
+ for target in targets:
74
+ out.append((line[:at] + target + line[at + len(source):],
75
+ f"{source.strip()} -> {target.strip()}"))
76
+ return out
77
+
78
+
79
+ def file_mutants(text: str, changed_lines: set[int] | None, language: str) -> list[Mutant]:
80
+ ops = _OPS.get(language, _OPS["typescript"])
81
+ mutants = []
82
+ for line_no, line in enumerate(text.splitlines(), start=1):
83
+ if changed_lines is not None and line_no not in changed_lines:
84
+ continue
85
+ if line.strip().startswith(_COMMENT_PREFIXES):
86
+ continue
87
+ for mutated, op in _line_mutations(line, ops):
88
+ mutants.append(Mutant("", line_no, line, mutated, op))
89
+ return mutants
90
+
91
+
92
+ def apply_mutant(text: str, mutant: Mutant) -> str:
93
+ lines = text.splitlines(keepends=True)
94
+ eol = "\n" if lines[mutant.line - 1].endswith("\n") else ""
95
+ lines[mutant.line - 1] = mutant.mutated + eol
96
+ return "".join(lines)