qaas-python 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. qaas/adapters/__init__.py +19 -0
  2. qaas/adapters/tracker.py +1783 -0
  3. qaas/adapters/vcs.py +555 -0
  4. qaas/cli.py +1757 -0
  5. qaas/config.py +409 -0
  6. qaas/defaults/config/agents/api.yaml +18 -0
  7. qaas/defaults/config/agents/architect.yaml +21 -0
  8. qaas/defaults/config/agents/auditor.yaml +19 -0
  9. qaas/defaults/config/agents/browser.yaml +15 -0
  10. qaas/defaults/config/agents/dba.yaml +20 -0
  11. qaas/defaults/config/agents/fixer.yaml +55 -0
  12. qaas/defaults/config/agents/guide.yaml +23 -0
  13. qaas/defaults/config/agents/load.yaml +26 -0
  14. qaas/defaults/config/agents/mapper.yaml +19 -0
  15. qaas/defaults/config/agents/reporter.yaml +19 -0
  16. qaas/defaults/config/agents/reproducer.yaml +21 -0
  17. qaas/defaults/config/agents/reviewer.yaml +18 -0
  18. qaas/defaults/config/agents/socket.yaml +23 -0
  19. qaas/defaults/config/agents/triage.yaml +20 -0
  20. qaas/defaults/config/agents/verifier.yaml +20 -0
  21. qaas/defaults/config/system.yaml +64 -0
  22. qaas/discover.py +242 -0
  23. qaas/envelope.py +318 -0
  24. qaas/envfile.py +100 -0
  25. qaas/guardrails.py +589 -0
  26. qaas/mcp/__init__.py +0 -0
  27. qaas/mcp/context.py +78 -0
  28. qaas/mcp/contract_diff.py +1011 -0
  29. qaas/mcp/defect_memory.py +495 -0
  30. qaas/mcp/env_control.py +925 -0
  31. qaas/mcp/envelope_server.py +463 -0
  32. qaas/mcp/test_runner.py +842 -0
  33. qaas/mcp/tracker.py +420 -0
  34. qaas/mcp/vcs.py +501 -0
  35. qaas/paths.py +317 -0
  36. qaas/plugin/.claude-plugin/plugin.json +9 -0
  37. qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
  38. qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
  39. qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
  40. qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
  41. qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
  42. qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
  43. qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
  44. qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
  45. qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
  46. qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
  47. qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
  48. qaas/plugin/skills/flake-detection/SKILL.md +39 -0
  49. qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
  50. qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
  51. qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
  52. qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
  53. qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
  54. qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
  55. qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
  56. qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
  57. qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
  58. qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
  59. qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
  60. qaas/plugin/skills/routing-rules/SKILL.md +34 -0
  61. qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
  62. qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
  63. qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
  64. qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
  65. qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
  66. qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
  67. qaas/prompts/API.md +44 -0
  68. qaas/prompts/ARCHITECT.md +80 -0
  69. qaas/prompts/AUDITOR.md +62 -0
  70. qaas/prompts/BROWSER.md +46 -0
  71. qaas/prompts/DBA.md +59 -0
  72. qaas/prompts/FIXER.md +55 -0
  73. qaas/prompts/GUIDE.md +94 -0
  74. qaas/prompts/LOAD.md +109 -0
  75. qaas/prompts/MAPPER.md +46 -0
  76. qaas/prompts/REPORTER.md +61 -0
  77. qaas/prompts/REPRODUCER.md +43 -0
  78. qaas/prompts/REVIEWER.md +53 -0
  79. qaas/prompts/SOCKET.md +100 -0
  80. qaas/prompts/TRIAGE.md +45 -0
  81. qaas/prompts/VERIFIER.md +41 -0
  82. qaas/prompts/_shared.md +45 -0
  83. qaas/registry.py +496 -0
  84. qaas/router.py +581 -0
  85. qaas/runner.py +210 -0
  86. qaas/scorecard.py +448 -0
  87. qaas/sdk_compat.py +52 -0
  88. qaas/store.py +323 -0
  89. qaas/target.py +287 -0
  90. qaas/tasks.py +438 -0
  91. qaas/trace.py +342 -0
  92. qaas_python-0.0.1.dist-info/METADATA +429 -0
  93. qaas_python-0.0.1.dist-info/RECORD +96 -0
  94. qaas_python-0.0.1.dist-info/WHEEL +4 -0
  95. qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
  96. qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,842 @@
1
+ """The `test_runner` MCP server — structured outcomes, never scraped CLI text.
2
+
3
+ An agent that reads pytest's terminal output makes two kinds of mistake: it
4
+ misreads a summary line, and it argues with itself about what "2 failed, 1
5
+ passed" implies for the one test it cares about. Both disappear if the tool
6
+ returns a list of `{nodeid, outcome, duration_s, message}` and the agent reads a
7
+ field. §5.2 asks for this server for exactly that reason.
8
+
9
+ Parsing strategy, in order of preference:
10
+
11
+ * `pytest-json-report` if it is installed. Exact durations, exact longrepr.
12
+ * The terminal output otherwise, run with `-v -rfE --durations=0` so the
13
+ facts we need are on lines with a stable shape. Robust beats clever here:
14
+ outcomes come from the per-test progress lines, failure messages from the
15
+ short summary, durations from the durations table, and anything unparseable
16
+ degrades to a total from the exit code rather than to a wrong answer.
17
+
18
+ Every subprocess is argv-only and time-boxed. A caller never supplies a command
19
+ string, and a run that exceeds its timeout returns what it managed to collect,
20
+ flagged, instead of hanging the agent's turn.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import asyncio
26
+ import importlib.util
27
+ import json
28
+ import os
29
+ import re
30
+ import subprocess
31
+ import sys
32
+ import tempfile
33
+ import time
34
+ from collections import Counter
35
+ from dataclasses import dataclass
36
+ from pathlib import Path
37
+ from typing import Any
38
+
39
+ from claude_agent_sdk import create_sdk_mcp_server, tool
40
+
41
+ from qaas.mcp.context import ToolContext, err, ok
42
+
43
+ DEFAULT_TIMEOUT_S = 300
44
+ MAX_TIMEOUT_S = 900
45
+
46
+ # §10 says flake rate is measured, not guessed, but a 200-run loop is a budget
47
+ # incident. Twenty runs already resolves a 5%-flaky test most of the time.
48
+ MAX_FLAKE_RUNS = 20
49
+
50
+ # And a wall clock over the whole investigation: 20 runs of the 900s per-run
51
+ # cap is five hours in one tool call, which nothing else in the system bounds.
52
+ MAX_FLAKE_TOTAL_S = 1_800
53
+
54
+ # How much raw output to hand back when parsing found nothing useful. Enough to
55
+ # diagnose a collection error, not enough to flood the agent's context.
56
+ OUTPUT_TAIL = 6_000
57
+
58
+ # pytest exit codes we treat specially (see pytest.ExitCode).
59
+ _EXIT_NO_TESTS = 5
60
+ _EXIT_USAGE_ERROR = 4
61
+
62
+ _OUTCOMES = {
63
+ "PASSED": "passed",
64
+ "FAILED": "failed",
65
+ "ERROR": "error",
66
+ "SKIPPED": "skipped",
67
+ "XFAIL": "xfailed",
68
+ "XPASS": "xpassed",
69
+ }
70
+
71
+ # `tests/test_a.py::test_x PASSED [ 50%]` — the -v progress line.
72
+ _PROGRESS_RE = re.compile(
73
+ r"^(?P<nodeid>\S+)\s+(?P<outcome>PASSED|FAILED|ERROR|SKIPPED|XFAIL|XPASS)\b"
74
+ )
75
+ # `FAILED tests/test_a.py::test_x - AssertionError: ...` — the -rfE summary line.
76
+ _SUMMARY_RE = re.compile(
77
+ r"^(?P<outcome>FAILED|ERROR)\s+(?P<nodeid>\S+)(?:\s+-\s+(?P<message>.*))?$"
78
+ )
79
+ # `0.01s call tests/test_a.py::test_x` — the --durations table.
80
+ _DURATION_RE = re.compile(r"^(?P<seconds>\d+\.\d+)s\s+(?:call|setup|teardown)\s+(?P<nodeid>\S+)$")
81
+
82
+ # pytest reports an unknown nodeid as a usage error, not as "no tests ran", so
83
+ # the two have to be told apart by the message rather than by the exit code.
84
+ _NOT_FOUND_RE = re.compile(r"^ERROR: not found: ", re.MULTILINE)
85
+ _UNKNOWN_OPTION_RE = re.compile(r"unrecognized (arguments|option)", re.IGNORECASE)
86
+
87
+ _FAILURE_HEADER_RE = re.compile(r"^=+ (FAILURES|ERRORS) =+$", re.MULTILINE)
88
+ _SUMMARY_HEADER_RE = re.compile(r"^=+ (short test summary info|warnings summary) =+", re.MULTILINE)
89
+
90
+
91
+ # ---------------------------------------------------------------------------
92
+ # subprocess plumbing
93
+ # ---------------------------------------------------------------------------
94
+
95
+
96
+ @dataclass
97
+ class _Completed:
98
+ argv: list[str]
99
+ returncode: int
100
+ stdout: str
101
+ stderr: str
102
+ duration_s: float
103
+ timed_out: bool
104
+ started: bool = True
105
+
106
+
107
+ def _decode(raw: str | bytes | None) -> str:
108
+ if raw is None:
109
+ return ""
110
+ return raw if isinstance(raw, str) else raw.decode(errors="replace")
111
+
112
+
113
+ def _run(argv: list[str], cwd: Path, timeout_s: int, env_extra: dict[str, str] | None = None) -> _Completed:
114
+ """Run argv under a hard timeout, returning partial output if it expires.
115
+
116
+ `subprocess.run` kills the child and hands the partial streams back on the
117
+ exception, which is the difference between "the suite hung, here is how far
118
+ it got" and an agent staring at nothing.
119
+ """
120
+ env = dict(os.environ, **(env_extra or {}))
121
+ # A wide terminal keeps pytest from wrapping nodeids across lines, which is
122
+ # the one thing that would break the progress-line parser.
123
+ env["COLUMNS"] = "250"
124
+ # The parent process is usually pytest itself (this server is exercised from
125
+ # a test suite); its addopts must not leak into the child's run.
126
+ env.pop("PYTEST_ADDOPTS", None)
127
+ env.pop("PYTEST_CURRENT_TEST", None)
128
+
129
+ started = time.monotonic()
130
+ try:
131
+ proc = subprocess.run(
132
+ argv, cwd=cwd, env=env, capture_output=True, text=True, timeout=timeout_s, check=False
133
+ )
134
+ except subprocess.TimeoutExpired as exc:
135
+ return _Completed(
136
+ argv, -1, _decode(exc.stdout), _decode(exc.stderr), time.monotonic() - started, True
137
+ )
138
+ except OSError as exc:
139
+ return _Completed(argv, -1, "", f"could not start {argv[0]}: {exc}", 0.0, False, started=False)
140
+ return _Completed(
141
+ argv, proc.returncode, proc.stdout, proc.stderr, time.monotonic() - started, False
142
+ )
143
+
144
+
145
+ async def _run_async(argv: list[str], cwd: Path, timeout_s: int, env_extra: dict[str, str] | None = None) -> _Completed:
146
+ """Off the event loop: a 300s suite must not block the other tools."""
147
+ return await asyncio.to_thread(_run, argv, cwd, timeout_s, env_extra)
148
+
149
+
150
+ def _resolve_cwd(ctx: ToolContext, raw: str | None) -> tuple[Path | None, str | None]:
151
+ """Working directory for a run: the repo root, or a directory inside it.
152
+
153
+ Resolved before the containment check so `..` and symlinks cannot walk out
154
+ of the checkout the run is supposed to be confined to.
155
+ """
156
+ root = ctx.target_root.resolve()
157
+ if not raw:
158
+ return root, None
159
+ candidate = Path(raw)
160
+ resolved = (candidate if candidate.is_absolute() else root / candidate).resolve()
161
+ if resolved != root and not resolved.is_relative_to(root):
162
+ return None, f"cwd '{raw}' resolves to {resolved}, outside the repository ({root})."
163
+ if not resolved.is_dir():
164
+ return None, f"cwd '{raw}' is not a directory."
165
+ return resolved, None
166
+
167
+
168
+ def _selector_refusal(ctx: ToolContext, cwd: Path, selector: str) -> str | None:
169
+ """Why this selector may not reach pytest, if it may not.
170
+
171
+ `cwd` was resolved and contained two lines above every call site; the
172
+ selector beside it was not, and it is the argument that decides what runs.
173
+ Two holes, both reachable from one tool call:
174
+
175
+ * Selectors are appended to pytest's argv with no `--` separator, so one
176
+ beginning with `-` is parsed as an *option*. `-p`, `-c`, `--rootdir=`
177
+ and `-o addopts=...` each load code of the caller's choosing.
178
+ * A path-shaped selector was passed to the collector unchecked, so
179
+ `run_suite({"selector": "../outside"})` collected and **executed**
180
+ modules outside the target root. Under `qaas run --repo <url>` the
181
+ sibling of that root is `.qaas/targets/`, holding every other clone.
182
+
183
+ Resolution happens before the containment test, so `..` and a symlink out of
184
+ the checkout are caught by the same check — the reasoning `_resolve_cwd`
185
+ already records.
186
+ """
187
+ if selector.startswith("-"):
188
+ return (
189
+ f"selector '{selector}' may not start with '-': pytest reads it as an "
190
+ "option rather than a test to run. Name a path, a nodeid, or a -k "
191
+ "expression without a leading dash."
192
+ )
193
+ root = ctx.target_root.resolve()
194
+ resolved = _selector_head(cwd, selector)
195
+ if resolved.exists() and not resolved.is_relative_to(root):
196
+ return (
197
+ f"selector '{selector}' resolves to {resolved}, outside the repository "
198
+ f"({root}). Tests are run from inside the checkout, never beside it."
199
+ )
200
+ return None
201
+
202
+
203
+ def _selector_head(cwd: Path, selector: str) -> Path:
204
+ """The filesystem part of a selector ('tests/x.py::test_y' -> 'tests/x.py')."""
205
+ head = Path(selector.split("::", 1)[0])
206
+ return (head if head.is_absolute() else cwd / head).resolve()
207
+
208
+
209
+ def _timeout(args: dict[str, Any]) -> tuple[int, str | None]:
210
+ raw = args.get("timeout_s")
211
+ if raw is None:
212
+ return DEFAULT_TIMEOUT_S, None
213
+ try:
214
+ value = int(raw)
215
+ except (TypeError, ValueError):
216
+ return DEFAULT_TIMEOUT_S, f"timeout_s must be a number, got {raw!r}."
217
+ if value < 1:
218
+ return DEFAULT_TIMEOUT_S, "timeout_s must be at least 1 second."
219
+ if value > MAX_TIMEOUT_S:
220
+ return DEFAULT_TIMEOUT_S, (
221
+ f"timeout_s {value} exceeds the {MAX_TIMEOUT_S}s cap. A test that needs "
222
+ "longer than fifteen minutes is a finding in itself, not a longer wait."
223
+ )
224
+ return value, None
225
+
226
+
227
+ # ---------------------------------------------------------------------------
228
+ # pytest invocation and parsing
229
+ # ---------------------------------------------------------------------------
230
+
231
+
232
+ def _json_report_available() -> bool:
233
+ """Whether the child interpreter (which is this one) has the plugin."""
234
+ return importlib.util.find_spec("pytest_jsonreport") is not None
235
+
236
+
237
+ def _base_argv(selectors: list[str], json_report_path: Path | None) -> list[str]:
238
+ argv = [sys.executable, "-m", "pytest", "-p", "no:cacheprovider", "--tb=short"]
239
+ if json_report_path is not None:
240
+ argv += ["--json-report", f"--json-report-file={json_report_path}", "-q"]
241
+ else:
242
+ argv += ["-v", "-rfE", "--durations=0", "--durations-min=0"]
243
+ return argv + selectors
244
+
245
+
246
+ def _parse_json_report(path: Path) -> list[dict[str, Any]] | None:
247
+ """Per-test rows from pytest-json-report, or None if it wrote nothing usable."""
248
+ try:
249
+ report = json.loads(path.read_text(encoding="utf-8"))
250
+ except (OSError, ValueError):
251
+ return None
252
+ raw_tests = report.get("tests")
253
+ if not isinstance(raw_tests, list):
254
+ return None
255
+
256
+ rows: list[dict[str, Any]] = []
257
+ for entry in raw_tests:
258
+ phases = [entry.get(p) for p in ("setup", "call", "teardown")]
259
+ duration = sum(p.get("duration", 0.0) for p in phases if isinstance(p, dict))
260
+ message = None
261
+ for phase in phases:
262
+ if isinstance(phase, dict) and phase.get("outcome") in {"failed", "error"}:
263
+ message = _shorten(_stringify_longrepr(phase.get("longrepr")))
264
+ break
265
+ rows.append(
266
+ {
267
+ "nodeid": entry.get("nodeid", "?"),
268
+ "outcome": entry.get("outcome", "unknown"),
269
+ "duration_s": round(duration, 4),
270
+ "message": message,
271
+ }
272
+ )
273
+ return rows
274
+
275
+
276
+ def _stringify_longrepr(longrepr: Any) -> str:
277
+ """pytest-json-report emits a string, or a dict when tracebacks are structured."""
278
+ if isinstance(longrepr, str):
279
+ return longrepr
280
+ if isinstance(longrepr, dict):
281
+ crash = longrepr.get("crash")
282
+ if isinstance(crash, dict) and crash.get("message"):
283
+ return str(crash["message"])
284
+ return json.dumps(longrepr)[:2_000]
285
+ return "" if longrepr is None else str(longrepr)
286
+
287
+
288
+ def _shorten(text: str, limit: int = 1_200) -> str | None:
289
+ text = (text or "").strip()
290
+ if not text:
291
+ return None
292
+ return text if len(text) <= limit else text[:limit] + "\n… (truncated)"
293
+
294
+
295
+ def _parse_terminal(stdout: str) -> list[dict[str, Any]]:
296
+ """Per-test rows from pytest's terminal output.
297
+
298
+ Three independent line shapes, merged: outcomes from the progress lines
299
+ (the only source that names every test, skips included), messages from the
300
+ short summary, durations from the durations table. Any one of them missing
301
+ degrades a field, not the whole result.
302
+ """
303
+ outcomes: dict[str, str] = {}
304
+ order: list[str] = []
305
+ messages: dict[str, str] = {}
306
+ durations: dict[str, float] = {}
307
+
308
+ for line in stdout.splitlines():
309
+ line = line.rstrip()
310
+ summary = _SUMMARY_RE.match(line)
311
+ if summary:
312
+ nodeid = summary.group("nodeid").rstrip(":")
313
+ message = (summary.group("message") or "").strip()
314
+ if message:
315
+ messages[nodeid] = message
316
+ outcomes.setdefault(nodeid, _OUTCOMES[summary.group("outcome")])
317
+ if nodeid not in order:
318
+ order.append(nodeid)
319
+ continue
320
+
321
+ progress = _PROGRESS_RE.match(line)
322
+ if progress:
323
+ nodeid = progress.group("nodeid")
324
+ if nodeid not in outcomes:
325
+ order.append(nodeid)
326
+ outcomes[nodeid] = _OUTCOMES[progress.group("outcome")]
327
+ continue
328
+
329
+ duration = _DURATION_RE.match(line.strip())
330
+ if duration:
331
+ nodeid = duration.group("nodeid")
332
+ durations[nodeid] = durations.get(nodeid, 0.0) + float(duration.group("seconds"))
333
+
334
+ return [
335
+ {
336
+ "nodeid": nodeid,
337
+ "outcome": outcomes.get(nodeid, "unknown"),
338
+ "duration_s": round(durations[nodeid], 4) if nodeid in durations else None,
339
+ "message": _shorten(messages.get(nodeid, "")),
340
+ }
341
+ for nodeid in order
342
+ ]
343
+
344
+
345
+ def _failure_detail(stdout: str) -> str | None:
346
+ """The FAILURES/ERRORS section verbatim — what `run_single` is actually for."""
347
+ header = _FAILURE_HEADER_RE.search(stdout)
348
+ if not header:
349
+ return None
350
+ tail = stdout[header.start():]
351
+ end = _SUMMARY_HEADER_RE.search(tail, 1)
352
+ section = tail[: end.start()] if end else tail
353
+ return _shorten(section, 8_000)
354
+
355
+
356
+ def _totals(rows: list[dict[str, Any]]) -> dict[str, int]:
357
+ counts = Counter(row["outcome"] for row in rows)
358
+ return {"total": len(rows), **{outcome: counts[outcome] for outcome in sorted(counts)}}
359
+
360
+
361
+ async def _pytest(cwd: Path, selectors: list[str], timeout_s: int) -> tuple[_Completed, list[dict[str, Any]], str]:
362
+ """Run pytest and return (process, per-test rows, parser used)."""
363
+ use_json = _json_report_available()
364
+ with tempfile.TemporaryDirectory(prefix="qaas-pytest-") as tmp:
365
+ report_path = Path(tmp) / "report.json" if use_json else None
366
+ proc = await _run_async(_base_argv(selectors, report_path), cwd, timeout_s)
367
+
368
+ # An unknown option (an older pytest without --durations-min, say) is a
369
+ # usage error, not a test failure. Retry once with the minimal flag set
370
+ # rather than reporting a suite that never ran.
371
+ if proc.returncode == _EXIT_USAGE_ERROR and not use_json and _UNKNOWN_OPTION_RE.search(
372
+ proc.stdout + proc.stderr
373
+ ):
374
+ argv = [sys.executable, "-m", "pytest", "-p", "no:cacheprovider", "--tb=short", "-v", "-rfE"]
375
+ proc = await _run_async(argv + selectors, cwd, timeout_s)
376
+
377
+ rows: list[dict[str, Any]] | None = None
378
+ if report_path is not None:
379
+ rows = _parse_json_report(report_path)
380
+ parser = "json-report" if rows is not None else "terminal"
381
+ if rows is None:
382
+ rows = _parse_terminal(proc.stdout)
383
+ return proc, rows, parser
384
+
385
+
386
+ def _matched_nothing(proc: _Completed) -> bool:
387
+ """Whether the selector picked no test at all, however pytest said so."""
388
+ if proc.returncode == _EXIT_NO_TESTS:
389
+ return True
390
+ return proc.returncode == _EXIT_USAGE_ERROR and bool(
391
+ _NOT_FOUND_RE.search(proc.stdout + proc.stderr)
392
+ )
393
+
394
+
395
+ def _outcome_of(rows: list[dict[str, Any]], test_id: str, proc: _Completed) -> str:
396
+ """One test's outcome, falling back to the exit code if parsing missed it."""
397
+ for row in rows:
398
+ if row["nodeid"] == test_id or row["nodeid"].endswith(test_id):
399
+ return row["outcome"]
400
+ if proc.timed_out:
401
+ return "timeout"
402
+ if _matched_nothing(proc):
403
+ return "not_collected"
404
+ if proc.returncode == 0:
405
+ return "passed"
406
+ return "failed"
407
+
408
+
409
+ def _tail(proc: _Completed) -> str:
410
+ combined = (proc.stdout + ("\n" + proc.stderr if proc.stderr else "")).strip()
411
+ return combined[-OUTPUT_TAIL:]
412
+
413
+
414
+ # ---------------------------------------------------------------------------
415
+ # the server
416
+ # ---------------------------------------------------------------------------
417
+
418
+
419
+ def build_tools(ctx: ToolContext) -> list:
420
+ """The test-runner tools, bound to one agent's run context.
421
+
422
+ Split from `build` so tests can call the handlers directly without standing
423
+ up an MCP transport.
424
+ """
425
+
426
+ _CWD_SCHEMA = {
427
+ "cwd": {"type": "string", "description": "Directory to run in. Defaults to the repo root; must stay inside it."},
428
+ "timeout_s": {"type": "number", "description": f"Seconds before the run is killed. Default {DEFAULT_TIMEOUT_S}, cap {MAX_TIMEOUT_S}."},
429
+ }
430
+
431
+ @tool(
432
+ "run_suite",
433
+ "Run the test suite, optionally narrowed by a selector, and get structured "
434
+ "per-test outcomes back. Read the fields; do not parse the summary text.",
435
+ {
436
+ "type": "object",
437
+ "properties": {
438
+ "selector": {
439
+ "type": "string",
440
+ "description": "A path ('tests/api'), a nodeid, or a -k expression ('order and not slow').",
441
+ },
442
+ **_CWD_SCHEMA,
443
+ },
444
+ },
445
+ )
446
+ async def run_suite(args: dict[str, Any]) -> dict[str, Any]:
447
+ cwd, cwd_error = _resolve_cwd(ctx, args.get("cwd"))
448
+ if cwd_error:
449
+ return err(cwd_error)
450
+ timeout_s, timeout_error = _timeout(args)
451
+ if timeout_error:
452
+ return err(timeout_error)
453
+
454
+ selector = (args.get("selector") or "").strip()
455
+ selectors: list[str] = []
456
+ if selector:
457
+ # A selector that names something on disk is a path; anything else is
458
+ # a -k expression. Guessing wrong wastes a run, so the check is a
459
+ # filesystem question, not a syntax one.
460
+ refusal = _selector_refusal(ctx, cwd, selector)
461
+ if refusal:
462
+ return err(refusal)
463
+ head = _selector_head(cwd, selector)
464
+ selectors = [selector] if head.exists() else ["-k", selector]
465
+
466
+ proc, rows, parser = await _pytest(cwd, selectors, timeout_s)
467
+ if not proc.started:
468
+ return err(proc.stderr)
469
+ if _matched_nothing(proc) and not rows:
470
+ return err(
471
+ f"No tests matched {selector or 'the default collection'} in {cwd}. "
472
+ "Check the selector against the files that exist."
473
+ )
474
+
475
+ totals = _totals(rows)
476
+ structured = {
477
+ "tests": rows,
478
+ "totals": totals,
479
+ "exit_code": proc.returncode,
480
+ "timed_out": proc.timed_out,
481
+ "duration_s": round(proc.duration_s, 3),
482
+ "parser": parser,
483
+ "cwd": str(cwd),
484
+ "selector": selector or None,
485
+ }
486
+ if proc.timed_out or not rows:
487
+ structured["output_tail"] = _tail(proc)
488
+
489
+ headline = ", ".join(f"{count} {name}" for name, count in totals.items() if name != "total")
490
+ note = f" TIMED OUT after {timeout_s}s; these are partial results." if proc.timed_out else ""
491
+ return ok(
492
+ f"{totals['total']} tests: {headline or 'none run'} in {proc.duration_s:.1f}s.{note}",
493
+ **structured,
494
+ )
495
+
496
+ @tool(
497
+ "run_single",
498
+ "Run one test by nodeid and get its full failure output. Use this to confirm "
499
+ "a repro, not to browse the suite.",
500
+ {
501
+ "type": "object",
502
+ "required": ["test_id"],
503
+ "properties": {
504
+ "test_id": {"type": "string", "description": "e.g. 'tests/api/test_orders.py::test_returns_500'"},
505
+ **_CWD_SCHEMA,
506
+ },
507
+ },
508
+ )
509
+ async def run_single(args: dict[str, Any]) -> dict[str, Any]:
510
+ cwd, cwd_error = _resolve_cwd(ctx, args.get("cwd"))
511
+ if cwd_error:
512
+ return err(cwd_error)
513
+ timeout_s, timeout_error = _timeout(args)
514
+ if timeout_error:
515
+ return err(timeout_error)
516
+
517
+ test_id = str(args["test_id"]).strip()
518
+ if not test_id:
519
+ return err("test_id is required.")
520
+ refusal = _selector_refusal(ctx, cwd, test_id)
521
+ if refusal:
522
+ return err(refusal)
523
+
524
+ proc, rows, parser = await _pytest(cwd, [test_id], timeout_s)
525
+ if not proc.started:
526
+ return err(proc.stderr)
527
+ if _matched_nothing(proc):
528
+ return err(
529
+ f"'{test_id}' matched no test in {cwd}. Nodeids look like "
530
+ "'path/to/test_file.py::test_name'."
531
+ )
532
+
533
+ outcome = _outcome_of(rows, test_id, proc)
534
+ row = next((r for r in rows if r["nodeid"] == test_id or r["nodeid"].endswith(test_id)), None)
535
+ detail = _failure_detail(proc.stdout)
536
+ structured = {
537
+ "nodeid": test_id,
538
+ "outcome": outcome,
539
+ "duration_s": (row or {}).get("duration_s"),
540
+ "message": (row or {}).get("message"),
541
+ "output": detail or (_tail(proc) if outcome != "passed" else None),
542
+ "exit_code": proc.returncode,
543
+ "timed_out": proc.timed_out,
544
+ "parser": parser,
545
+ }
546
+ return ok(f"{test_id}: {outcome} in {proc.duration_s:.1f}s.", **structured)
547
+
548
+ @tool(
549
+ "run_n_times",
550
+ "Run one test repeatedly and measure its flake rate — the share of runs whose "
551
+ "outcome differs from the majority. A non-zero rate means the defect is flaky, "
552
+ "which is a different finding from a defect that always reproduces (§10).",
553
+ {
554
+ "type": "object",
555
+ "required": ["test_id", "n"],
556
+ "properties": {
557
+ "test_id": {"type": "string"},
558
+ "n": {"type": "number", "description": f"Number of runs, 1-{MAX_FLAKE_RUNS}."},
559
+ **_CWD_SCHEMA,
560
+ },
561
+ },
562
+ )
563
+ async def run_n_times(args: dict[str, Any]) -> dict[str, Any]:
564
+ cwd, cwd_error = _resolve_cwd(ctx, args.get("cwd"))
565
+ if cwd_error:
566
+ return err(cwd_error)
567
+ timeout_s, timeout_error = _timeout(args)
568
+ if timeout_error:
569
+ return err(timeout_error)
570
+
571
+ test_id = str(args["test_id"]).strip()
572
+ if not test_id:
573
+ return err("test_id is required.")
574
+ refusal = _selector_refusal(ctx, cwd, test_id)
575
+ if refusal:
576
+ return err(refusal)
577
+ try:
578
+ n = int(args["n"])
579
+ except (TypeError, ValueError):
580
+ return err(f"n must be a whole number, got {args['n']!r}.")
581
+ if n < 1:
582
+ return err("n must be at least 1.")
583
+ if n > MAX_FLAKE_RUNS:
584
+ return err(
585
+ f"n={n} exceeds the {MAX_FLAKE_RUNS}-run cap. Twenty runs resolve a "
586
+ "5% flake most of the time; more is a budget problem, not better evidence."
587
+ )
588
+
589
+ # `n` runs of `timeout_s` each is 20 x 900s = five hours inside a single
590
+ # tool call, which no budget or turn limit sees. The per-run timeout
591
+ # bounds a hang; only an aggregate bounds the loop.
592
+ deadline = asyncio.get_running_loop().time() + min(MAX_FLAKE_TOTAL_S, n * timeout_s)
593
+
594
+ outcomes: list[str] = []
595
+ messages: list[str] = []
596
+ for attempt in range(n):
597
+ remaining = deadline - asyncio.get_running_loop().time()
598
+ if attempt and remaining <= 0:
599
+ messages.append(
600
+ f"Stopped after {attempt} of {n} runs: the {MAX_FLAKE_TOTAL_S}s budget for "
601
+ "one flake investigation was reached. Judge the flake on these."
602
+ )
603
+ break
604
+ proc, rows, _ = await _pytest(cwd, [test_id], max(1, int(min(timeout_s, remaining))))
605
+ if not proc.started:
606
+ return err(proc.stderr)
607
+ if attempt == 0 and _matched_nothing(proc):
608
+ return err(f"'{test_id}' matched no test in {cwd}.")
609
+ outcome = _outcome_of(rows, test_id, proc)
610
+ outcomes.append(outcome)
611
+ row = next((r for r in rows if r["nodeid"] == test_id or r["nodeid"].endswith(test_id)), None)
612
+ if row and row.get("message"):
613
+ messages.append(f"run {attempt + 1}: {row['message']}")
614
+
615
+ counts = Counter(outcomes)
616
+ majority, majority_count = counts.most_common(1)[0]
617
+ flake_rate = round((n - majority_count) / n, 4)
618
+ passed = counts.get("passed", 0)
619
+
620
+ verdict = (
621
+ f"stable ({majority})" if flake_rate == 0
622
+ else f"FLAKY: {flake_rate:.0%} of runs disagreed with the majority ({majority})"
623
+ )
624
+ return ok(
625
+ f"{test_id} over {n} runs — {passed} passed, {n - passed} not passed. {verdict}.",
626
+ runs=n,
627
+ passed=passed,
628
+ failed=n - passed,
629
+ flake_rate=flake_rate,
630
+ majority_outcome=majority,
631
+ outcomes=outcomes,
632
+ counts=dict(counts),
633
+ messages=messages[:5],
634
+ )
635
+
636
+ @tool(
637
+ "affected_tests",
638
+ "Heuristic: given changed source paths, the test files most likely to cover them. "
639
+ "It is a heuristic, not a coverage-derived answer — treat the ranking as a place to "
640
+ "start, and run the full suite before concluding nothing broke.",
641
+ {
642
+ "type": "object",
643
+ "required": ["paths"],
644
+ "properties": {
645
+ "paths": {"type": "array", "items": {"type": "string"}, "description": "Repo-relative changed paths."},
646
+ "cwd": _CWD_SCHEMA["cwd"],
647
+ },
648
+ },
649
+ )
650
+ async def affected_tests(args: dict[str, Any]) -> dict[str, Any]:
651
+ cwd, cwd_error = _resolve_cwd(ctx, args.get("cwd"))
652
+ if cwd_error:
653
+ return err(cwd_error)
654
+ raw_paths = args.get("paths")
655
+ if not isinstance(raw_paths, list) or not raw_paths:
656
+ return err("paths must be a non-empty array of repo-relative paths.")
657
+
658
+ candidates = await asyncio.to_thread(_collect_test_files, cwd)
659
+ if not candidates:
660
+ return err(f"No test files found under {cwd}.")
661
+
662
+ scored = await asyncio.to_thread(_score_tests, cwd, [str(p) for p in raw_paths], candidates)
663
+ if not scored:
664
+ return ok(
665
+ "No test file looks related to those paths. That is itself worth reporting: "
666
+ "the change may be untested.",
667
+ affected=[],
668
+ heuristic=True,
669
+ )
670
+ return ok(
671
+ "Likely covering tests, best first: "
672
+ + ", ".join(item["test_file"] for item in scored[:10]),
673
+ affected=scored[:25],
674
+ heuristic=True,
675
+ searched=len(candidates),
676
+ )
677
+
678
+ @tool(
679
+ "get_coverage",
680
+ "Per-file line coverage, measured by running the suite under coverage.py. "
681
+ "Returns an error if coverage is not installed rather than an estimate.",
682
+ {
683
+ "type": "object",
684
+ "properties": {
685
+ "paths": {"type": "array", "items": {"type": "string"}, "description": "Limit measurement to these source paths."},
686
+ "selector": {"type": "string", "description": "Optional -k expression or path to narrow the suite."},
687
+ **_CWD_SCHEMA,
688
+ },
689
+ },
690
+ )
691
+ async def get_coverage(args: dict[str, Any]) -> dict[str, Any]:
692
+ if importlib.util.find_spec("coverage") is None:
693
+ return err(
694
+ "coverage is not installed in this environment, so there is no coverage "
695
+ "number to report. Install `coverage` (or add it to the dev extras) and "
696
+ "call again; do not estimate coverage from reading the code."
697
+ )
698
+
699
+ cwd, cwd_error = _resolve_cwd(ctx, args.get("cwd"))
700
+ if cwd_error:
701
+ return err(cwd_error)
702
+ timeout_s, timeout_error = _timeout(args)
703
+ if timeout_error:
704
+ return err(timeout_error)
705
+
706
+ paths = [str(p) for p in (args.get("paths") or [])]
707
+ selector = (args.get("selector") or "").strip()
708
+ selectors: list[str] = []
709
+ if selector:
710
+ refusal = _selector_refusal(ctx, cwd, selector)
711
+ if refusal:
712
+ return err(refusal)
713
+ head = _selector_head(cwd, selector)
714
+ selectors = [selector] if head.exists() else ["-k", selector]
715
+
716
+ with tempfile.TemporaryDirectory(prefix="qaas-coverage-") as tmp:
717
+ data_file = Path(tmp) / ".coverage"
718
+ json_file = Path(tmp) / "coverage.json"
719
+ env_extra = {"COVERAGE_FILE": str(data_file)}
720
+ run_argv = [sys.executable, "-m", "coverage", "run"]
721
+ if paths:
722
+ run_argv.append("--source=" + ",".join(paths))
723
+ run_argv += ["-m", "pytest", "-q", "-p", "no:cacheprovider", *selectors]
724
+
725
+ run = await _run_async(run_argv, cwd, timeout_s, env_extra)
726
+ if not run.started:
727
+ return err(run.stderr)
728
+ if run.timed_out:
729
+ return err(f"The coverage run exceeded {timeout_s}s and was killed. Narrow it with `selector`.")
730
+
731
+ report = await _run_async(
732
+ [sys.executable, "-m", "coverage", "json", "-o", str(json_file)],
733
+ cwd,
734
+ min(timeout_s, 120),
735
+ env_extra,
736
+ )
737
+ if not json_file.exists():
738
+ return err(
739
+ "coverage produced no report: "
740
+ + (_tail(report) or _tail(run) or "no output")
741
+ )
742
+ try:
743
+ data = json.loads(json_file.read_text(encoding="utf-8"))
744
+ except ValueError as exc:
745
+ return err(f"coverage report was not valid JSON: {exc}")
746
+
747
+ files = {
748
+ name: round(info.get("summary", {}).get("percent_covered", 0.0), 2)
749
+ for name, info in (data.get("files") or {}).items()
750
+ }
751
+ if paths:
752
+ wanted = tuple(paths)
753
+ files = {n: pct for n, pct in files.items() if n.startswith(wanted)} or files
754
+ overall = round((data.get("totals") or {}).get("percent_covered", 0.0), 2)
755
+ lowest = sorted(files.items(), key=lambda kv: kv[1])[:5]
756
+
757
+ return ok(
758
+ f"Overall line coverage {overall}% across {len(files)} files. "
759
+ + ("Lowest: " + ", ".join(f"{n} {p}%" for n, p in lowest) if lowest else ""),
760
+ overall_percent=overall,
761
+ files=files,
762
+ tests_exit_code=run.returncode,
763
+ )
764
+
765
+ return [run_suite, run_single, run_n_times, affected_tests, get_coverage]
766
+
767
+
768
+ # ---------------------------------------------------------------------------
769
+ # affected_tests heuristic
770
+ # ---------------------------------------------------------------------------
771
+
772
+ _SKIP_DIRS = {".git", ".venv", "venv", "node_modules", "__pycache__", ".tox", ".mypy_cache", ".qaas"}
773
+
774
+
775
+ def _collect_test_files(root: Path) -> list[Path]:
776
+ found: list[Path] = []
777
+ for path in root.rglob("*.py"):
778
+ if any(part in _SKIP_DIRS for part in path.parts):
779
+ continue
780
+ if path.name.startswith("test_") or path.name.endswith("_test.py"):
781
+ found.append(path)
782
+ return found
783
+
784
+
785
+ def _score_tests(root: Path, changed: list[str], candidates: list[Path]) -> list[dict[str, Any]]:
786
+ """Rank test files by three independent, cheap signals.
787
+
788
+ Name correspondence is the strongest (`foo.py` -> `test_foo.py` is a
789
+ convention people actually follow), an import of the changed module is next,
790
+ and sharing a directory is the weak tie-breaker that catches package-level
791
+ test layouts. Deliberately no AST or coverage database: this runs before the
792
+ agent knows which suite to run, so it must be fast and never wrong-by-crash.
793
+ """
794
+ scores: dict[Path, int] = {}
795
+ reasons: dict[Path, list[str]] = {}
796
+
797
+ for raw in changed:
798
+ source = Path(raw)
799
+ stem = source.stem
800
+ module_dir = source.parent.as_posix()
801
+ for candidate in candidates:
802
+ score = 0
803
+ why: list[str] = []
804
+ name = candidate.name
805
+ if name in {f"test_{stem}.py", f"{stem}_test.py"}:
806
+ score += 100
807
+ why.append(f"name matches {source.name}")
808
+ elif stem and stem in name:
809
+ score += 25
810
+ why.append(f"filename mentions '{stem}'")
811
+
812
+ if stem:
813
+ try:
814
+ text = candidate.read_text(errors="ignore")
815
+ except OSError:
816
+ text = ""
817
+ if re.search(rf"\b(import|from)\b[^\n]*\b{re.escape(stem)}\b", text):
818
+ score += 40
819
+ why.append(f"imports '{stem}'")
820
+
821
+ if module_dir and module_dir not in {".", ""} and module_dir in candidate.as_posix():
822
+ score += 10
823
+ why.append(f"shares directory {module_dir}")
824
+
825
+ if score:
826
+ scores[candidate] = scores.get(candidate, 0) + score
827
+ reasons.setdefault(candidate, []).extend(w for w in why if w not in reasons.get(candidate, []))
828
+
829
+ ranked = sorted(scores.items(), key=lambda kv: (-kv[1], str(kv[0])))
830
+ return [
831
+ {
832
+ "test_file": path.relative_to(root).as_posix() if path.is_relative_to(root) else str(path),
833
+ "score": score,
834
+ "why": reasons.get(path, []),
835
+ }
836
+ for path, score in ranked
837
+ ]
838
+
839
+
840
+ def build(ctx: ToolContext):
841
+ """Construct the test_runner MCP server bound to one agent's run context."""
842
+ return create_sdk_mcp_server(name="test_runner", version="1.0.0", tools=build_tools(ctx))