qaas-python 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. qaas/adapters/__init__.py +19 -0
  2. qaas/adapters/tracker.py +1783 -0
  3. qaas/adapters/vcs.py +555 -0
  4. qaas/cli.py +1757 -0
  5. qaas/config.py +409 -0
  6. qaas/defaults/config/agents/api.yaml +18 -0
  7. qaas/defaults/config/agents/architect.yaml +21 -0
  8. qaas/defaults/config/agents/auditor.yaml +19 -0
  9. qaas/defaults/config/agents/browser.yaml +15 -0
  10. qaas/defaults/config/agents/dba.yaml +20 -0
  11. qaas/defaults/config/agents/fixer.yaml +55 -0
  12. qaas/defaults/config/agents/guide.yaml +23 -0
  13. qaas/defaults/config/agents/load.yaml +26 -0
  14. qaas/defaults/config/agents/mapper.yaml +19 -0
  15. qaas/defaults/config/agents/reporter.yaml +19 -0
  16. qaas/defaults/config/agents/reproducer.yaml +21 -0
  17. qaas/defaults/config/agents/reviewer.yaml +18 -0
  18. qaas/defaults/config/agents/socket.yaml +23 -0
  19. qaas/defaults/config/agents/triage.yaml +20 -0
  20. qaas/defaults/config/agents/verifier.yaml +20 -0
  21. qaas/defaults/config/system.yaml +64 -0
  22. qaas/discover.py +242 -0
  23. qaas/envelope.py +318 -0
  24. qaas/envfile.py +100 -0
  25. qaas/guardrails.py +589 -0
  26. qaas/mcp/__init__.py +0 -0
  27. qaas/mcp/context.py +78 -0
  28. qaas/mcp/contract_diff.py +1011 -0
  29. qaas/mcp/defect_memory.py +495 -0
  30. qaas/mcp/env_control.py +925 -0
  31. qaas/mcp/envelope_server.py +463 -0
  32. qaas/mcp/test_runner.py +842 -0
  33. qaas/mcp/tracker.py +420 -0
  34. qaas/mcp/vcs.py +501 -0
  35. qaas/paths.py +317 -0
  36. qaas/plugin/.claude-plugin/plugin.json +9 -0
  37. qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
  38. qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
  39. qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
  40. qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
  41. qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
  42. qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
  43. qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
  44. qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
  45. qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
  46. qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
  47. qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
  48. qaas/plugin/skills/flake-detection/SKILL.md +39 -0
  49. qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
  50. qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
  51. qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
  52. qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
  53. qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
  54. qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
  55. qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
  56. qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
  57. qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
  58. qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
  59. qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
  60. qaas/plugin/skills/routing-rules/SKILL.md +34 -0
  61. qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
  62. qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
  63. qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
  64. qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
  65. qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
  66. qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
  67. qaas/prompts/API.md +44 -0
  68. qaas/prompts/ARCHITECT.md +80 -0
  69. qaas/prompts/AUDITOR.md +62 -0
  70. qaas/prompts/BROWSER.md +46 -0
  71. qaas/prompts/DBA.md +59 -0
  72. qaas/prompts/FIXER.md +55 -0
  73. qaas/prompts/GUIDE.md +94 -0
  74. qaas/prompts/LOAD.md +109 -0
  75. qaas/prompts/MAPPER.md +46 -0
  76. qaas/prompts/REPORTER.md +61 -0
  77. qaas/prompts/REPRODUCER.md +43 -0
  78. qaas/prompts/REVIEWER.md +53 -0
  79. qaas/prompts/SOCKET.md +100 -0
  80. qaas/prompts/TRIAGE.md +45 -0
  81. qaas/prompts/VERIFIER.md +41 -0
  82. qaas/prompts/_shared.md +45 -0
  83. qaas/registry.py +496 -0
  84. qaas/router.py +581 -0
  85. qaas/runner.py +210 -0
  86. qaas/scorecard.py +448 -0
  87. qaas/sdk_compat.py +52 -0
  88. qaas/store.py +323 -0
  89. qaas/target.py +287 -0
  90. qaas/tasks.py +438 -0
  91. qaas/trace.py +342 -0
  92. qaas_python-0.0.1.dist-info/METADATA +429 -0
  93. qaas_python-0.0.1.dist-info/RECORD +96 -0
  94. qaas_python-0.0.1.dist-info/WHEEL +4 -0
  95. qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
  96. qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/sdk_compat.py ADDED
@@ -0,0 +1,52 @@
1
+ """Pinned facts about the installed Claude Agent SDK.
2
+
3
+ These were wrong in the published docs at the time of writing: the docs give
4
+ hook events as camelCase (`preToolUse`) and `HookMatcher(event=, handler=)`,
5
+ while the installed SDK uses PascalCase events and `HookMatcher(matcher=, hooks=)`.
6
+ Rather than trust either, this module reads the truth out of the installed
7
+ package at import time and fails loudly if it changes under us.
8
+
9
+ Guardrails do not depend on hooks working (they live in `can_use_tool`), so a
10
+ drift here degrades observability, not enforcement — but it should still be a
11
+ noisy failure rather than a silent one.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import typing
17
+
18
+ from claude_agent_sdk import types as _sdk_types
19
+
20
+ _EVENTS: tuple[str, ...] = tuple(
21
+ typing.get_args(arg)[0] for arg in typing.get_args(_sdk_types.HookEvent)
22
+ )
23
+
24
+ PRE_TOOL_USE = "PreToolUse"
25
+ POST_TOOL_USE = "PostToolUse"
26
+ SUBAGENT_START = "SubagentStart"
27
+ STOP = "Stop"
28
+
29
+ _REQUIRED = (PRE_TOOL_USE, POST_TOOL_USE, STOP)
30
+
31
+
32
+ def check() -> None:
33
+ """Raise if the SDK no longer exposes the hook events we register."""
34
+ missing = [e for e in _REQUIRED if e not in _EVENTS]
35
+ if missing:
36
+ raise RuntimeError(
37
+ f"claude-agent-sdk no longer exposes hook events {missing}; "
38
+ f"it offers {list(_EVENTS)}. Update qaas.sdk_compat and guardrails."
39
+ )
40
+
41
+
42
+ def mcp_tool_name(server: str, tool: str) -> str:
43
+ """The name an MCP tool is exposed under: `mcp__<server>__<tool>`."""
44
+ return f"mcp__{server}__{tool}"
45
+
46
+
47
+ def mcp_server_wildcard(server: str) -> str:
48
+ """Allowlist pattern covering every tool on one server."""
49
+ return f"mcp__{server}"
50
+
51
+
52
+ check()
qaas/store.py ADDED
@@ -0,0 +1,323 @@
1
+ """Run state on disk: the ledger, the artifact store, the versioned system map.
2
+
3
+ Everything an agent produces lands here. The ledger is the audit trail the
4
+ architecture asks for (§2, §8) — every agent invocation, every cost, every
5
+ guardrail denial, appended and never rewritten.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import os
12
+ import re
13
+ import shutil
14
+ import uuid
15
+ from datetime import datetime, timezone
16
+ from enum import StrEnum
17
+ from pathlib import Path
18
+ from typing import Any, Iterator
19
+
20
+ from pydantic import BaseModel, ConfigDict, Field
21
+
22
+ from qaas.envelope import DefectEnvelope
23
+
24
+ DEFAULT_ROOT = Path(".qaas")
25
+
26
+
27
+ def _utcnow() -> datetime:
28
+ return datetime.now(timezone.utc)
29
+
30
+
31
+ class LedgerKind(StrEnum):
32
+ """Every kind of line the ledger may contain.
33
+
34
+ This was a bare `str` whose comment named 7 of the 28 kinds actually
35
+ written, which left the ledger unreadable by anything but grep: nothing
36
+ could enumerate what a run might contain, and a typo at a `store.log()`
37
+ call site invented a 29th kind that no reader would ever look for. It is a
38
+ closed set now, so a misspelling fails at the write instead of vanishing.
39
+
40
+ `StrEnum` and not a plain `Enum` on purpose: members *are* their strings, so
41
+ `entry.kind == "denial"` still holds, `model_dump_json()` still writes
42
+ `"kind":"denial"`, and every ledger already on disk still parses. Adding a
43
+ kind means adding a member here -- deliberately a visible act, because the
44
+ router reads several of these back for control flow (`_latest_verdict`,
45
+ `_latest_review`, `_branch_written_since`), so a rename is a breaking
46
+ change to a wire format, not a rename.
47
+ """
48
+
49
+ # run lifecycle (router)
50
+ RUN_STARTED = "run_started"
51
+ RUN_FINISHED = "run_finished"
52
+ SKIPPED = "skipped"
53
+ ESCALATION = "escalation"
54
+
55
+ # agent lifecycle (runner, store)
56
+ AGENT_STARTED = "agent_started"
57
+ AGENT_FINISHED = "agent_finished"
58
+ AGENT_ERROR = "agent_error"
59
+ SKILLS_MISSING = "skills_missing"
60
+
61
+ # tool traffic and its refusals (guardrails, registry)
62
+ TOOL_CALL = "tool_call"
63
+ TOOL_ERROR = "tool_error"
64
+ DENIAL = "denial"
65
+ STOP_BLOCKED = "stop_blocked"
66
+ CONTRACT_UNMET = "contract_unmet"
67
+
68
+ # findings and the evidence behind them
69
+ ENVELOPE = "envelope"
70
+ REPRODUCTION = "reproduction"
71
+ CONTRACT_TEST = "contract_test"
72
+ SYSTEM_MAP = "system_map"
73
+ DEFECT_MEMORY = "defect_memory"
74
+ REGRESSION = "regression"
75
+
76
+ # the file/verify/fix loop
77
+ TICKET = "ticket"
78
+ VERDICT = "verdict"
79
+ VERIFIED = "verified"
80
+ REOPENED = "reopened"
81
+ REVIEW = "review"
82
+ REVIEW_ROUND_TRIP = "review_round_trip"
83
+
84
+ # side effects on the world outside the run
85
+ VCS = "vcs"
86
+ ENV = "env"
87
+ DRY_RUN = "dry_run"
88
+
89
+
90
+ class LedgerEntry(BaseModel):
91
+ """One line in the run ledger. Append-only."""
92
+
93
+ model_config = ConfigDict(extra="forbid")
94
+
95
+ at: datetime = Field(default_factory=_utcnow)
96
+ kind: LedgerKind
97
+ agent: str | None = None
98
+ detail: dict[str, Any] = Field(default_factory=dict)
99
+
100
+
101
+ class AgentResult(BaseModel):
102
+ """What one agent invocation cost and produced."""
103
+
104
+ model_config = ConfigDict(extra="forbid")
105
+
106
+ agent: str
107
+ subtype: str = "success"
108
+ cost_usd: float = 0.0
109
+ num_turns: int = 0
110
+ duration_s: float = 0.0
111
+ envelope_ids: list[str] = Field(default_factory=list)
112
+ error: str | None = None
113
+
114
+
115
+ class RunStore:
116
+ """The filesystem home of a single run.
117
+
118
+ Layout:
119
+ .qaas/runs/<run_id>/ledger.jsonl
120
+ envelopes/<envelope_id>.json
121
+ artifacts/<name>
122
+ results/<AGENT>.json
123
+ .qaas/system-map/<version>.json (shared across runs)
124
+ .qaas/system-map/latest (pointer file)
125
+ """
126
+
127
+ def __init__(self, run_id: str, root: Path | str = DEFAULT_ROOT, *, create: bool = True):
128
+ self.run_id = run_id
129
+ self.root = Path(root)
130
+ self.dir = self.root / "runs" / run_id
131
+ # `create=False` for the read-only commands. Constructing a store used to
132
+ # mkdir unconditionally, so `qaas show <typo>` left a permanent empty run
133
+ # on disk that then appeared in `qaas runs` forever -- a reader that
134
+ # writes, and the one thing an audit trail must not have.
135
+ if create:
136
+ for sub in ("envelopes", "artifacts", "results"):
137
+ (self.dir / sub).mkdir(parents=True, exist_ok=True)
138
+ #: agent -> files it has written this run. See `touched_files`.
139
+ self._touched: dict[str, set[str]] = {}
140
+
141
+ @classmethod
142
+ def new(cls, root: Path | str = DEFAULT_ROOT, prefix: str = "run") -> "RunStore":
143
+ run_id = f"{prefix}-{_utcnow():%Y%m%dT%H%M%S}-{uuid.uuid4().hex[:6]}"
144
+ return cls(run_id, root)
145
+
146
+ # -- ledger -----------------------------------------------------------
147
+
148
+ @property
149
+ def ledger_path(self) -> Path:
150
+ return self.dir / "ledger.jsonl"
151
+
152
+ def log(self, kind: LedgerKind | str, agent: str | None = None, **detail: Any) -> LedgerEntry:
153
+ # `str` stays in the signature because ~60 call sites pass a literal and
154
+ # reading `store.log("denial", ...)` at the call site beats reading
155
+ # `store.log(LedgerKind.DENIAL, ...)`. Pydantic converts and, crucially,
156
+ # rejects: an unknown kind raises here rather than appending a line no
157
+ # reader will ever ask for.
158
+ entry = LedgerEntry(kind=kind, agent=agent, detail=detail)
159
+ with self.ledger_path.open("a", encoding="utf-8") as fh:
160
+ fh.write(entry.model_dump_json() + "\n")
161
+ return entry
162
+
163
+ def ledger(self, kind: LedgerKind | str | None = None) -> Iterator[LedgerEntry]:
164
+ if not self.ledger_path.exists():
165
+ return
166
+ for line in self.ledger_path.read_text(encoding="utf-8").splitlines():
167
+ if not line.strip():
168
+ continue
169
+ entry = LedgerEntry.model_validate_json(line)
170
+ if kind is None or entry.kind == kind:
171
+ yield entry
172
+
173
+ # -- envelopes --------------------------------------------------------
174
+
175
+ def put_envelope(self, envelope: DefectEnvelope) -> Path:
176
+ """Persist an envelope, stamping its fingerprint if absent."""
177
+ if envelope.dedupe.fingerprint is None:
178
+ envelope = envelope.with_fingerprint()
179
+ path = self.dir / "envelopes" / f"{envelope.id}.json"
180
+ path.write_text(envelope.to_json(), encoding="utf-8")
181
+ self.log(
182
+ "envelope",
183
+ agent=envelope.discovered_by,
184
+ envelope_id=envelope.id,
185
+ domain=envelope.domain.value,
186
+ severity=envelope.severity.value,
187
+ confidence=envelope.confidence,
188
+ fingerprint=envelope.dedupe.fingerprint,
189
+ )
190
+ return path
191
+
192
+ def envelopes(self) -> list[DefectEnvelope]:
193
+ paths = sorted((self.dir / "envelopes").glob("*.json"))
194
+ return [DefectEnvelope.from_json(p.read_text(encoding="utf-8")) for p in paths]
195
+
196
+ def get_envelope(self, envelope_id: str) -> DefectEnvelope | None:
197
+ path = self.dir / "envelopes" / f"{envelope_id}.json"
198
+ return DefectEnvelope.from_json(path.read_text(encoding="utf-8")) if path.exists() else None
199
+
200
+ def touched_files(self, agent: str) -> set[str]:
201
+ """The distinct files one agent has written this run (§8.2's denominator).
202
+
203
+ Held here rather than on `ToolContext` because a context is built per
204
+ dispatch and a run outlives many of them.
205
+ """
206
+ return self._touched.setdefault(agent, set())
207
+
208
+ # -- artifacts --------------------------------------------------------
209
+
210
+ def _artifact_path(self, name: str) -> tuple[str, Path]:
211
+ """A flat, contained filename for an agent-supplied artifact name.
212
+
213
+ `resolve_artifact` does resolve-then-contain, which is the right shape.
214
+ These two did string replacement instead — `/` and `..` to `_` — which is
215
+ a denylist, and denylists are wrong here for the ordinary reason: a
216
+ backslash was not in it, and neither is anything else nobody thought of.
217
+ The name is agent-supplied, so this keeps the flattening (an artifact
218
+ store is deliberately one directory deep) and then *checks* the result
219
+ rather than trusting the substitution.
220
+ """
221
+ flat = re.sub(r"[^A-Za-z0-9._-]", "_", name).lstrip(".") or "artifact"
222
+ base = (self.dir / "artifacts").resolve()
223
+ path = (base / flat).resolve()
224
+ if path.parent != base:
225
+ raise ValueError(f"artifact name escapes the store: {name!r}")
226
+ return flat, path
227
+
228
+ def put_artifact(self, name: str, content: str | bytes) -> str:
229
+ """Store evidence and return the artifact:// uri that references it."""
230
+ safe, path = self._artifact_path(name)
231
+ if isinstance(content, bytes):
232
+ path.write_bytes(content)
233
+ else:
234
+ path.write_text(content, encoding="utf-8")
235
+ return f"artifact://{self.run_id}/{safe}"
236
+
237
+ def copy_artifact(self, name: str, source: Path | str) -> str:
238
+ safe, path = self._artifact_path(name)
239
+ shutil.copy2(source, path)
240
+ return f"artifact://{self.run_id}/{safe}"
241
+
242
+ def resolve_artifact(self, uri: str) -> Path:
243
+ """artifact://<run_id>/<name> -> a real path. Raises if it escapes the store."""
244
+ if not uri.startswith("artifact://"):
245
+ raise ValueError(f"not an artifact uri: {uri}")
246
+ run_id, _, name = uri[len("artifact://"):].partition("/")
247
+ path = (self.root / "runs" / run_id / "artifacts" / name).resolve()
248
+ base = (self.root / "runs" / run_id / "artifacts").resolve()
249
+ if not path.is_relative_to(base):
250
+ raise ValueError(f"artifact uri escapes the store: {uri}")
251
+ return path
252
+
253
+ # -- per-agent results ------------------------------------------------
254
+
255
+ def put_result(self, result: AgentResult) -> None:
256
+ # One file per invocation, not per agent. REPRODUCER runs once per finding and
257
+ # FIXER once per review round trip, so a per-agent filename silently
258
+ # keeps only the last one — and the persisted cost of a run then
259
+ # under-reports by however much the repeated agents actually spent.
260
+ existing = len(list((self.dir / "results").glob(f"{result.agent}-*.json")))
261
+ path = self.dir / "results" / f"{result.agent}-{existing + 1:02d}.json"
262
+ path.write_text(result.model_dump_json(indent=2), encoding="utf-8")
263
+ self.log(
264
+ "agent_finished",
265
+ agent=result.agent,
266
+ subtype=result.subtype,
267
+ cost_usd=result.cost_usd,
268
+ num_turns=result.num_turns,
269
+ envelopes=len(result.envelope_ids),
270
+ error=result.error,
271
+ )
272
+
273
+ def results(self) -> list[AgentResult]:
274
+ paths = sorted((self.dir / "results").glob("*.json"))
275
+ return [AgentResult.model_validate_json(p.read_text(encoding="utf-8")) for p in paths]
276
+
277
+ def total_cost_usd(self) -> float:
278
+ return sum(r.cost_usd for r in self.results())
279
+
280
+
281
+ class SystemMapStore:
282
+ """Versioned Mapper output, shared across runs.
283
+
284
+ Agents pin a map version for the length of a run so a bad map cannot
285
+ half-propagate mid-run (§10, context poisoning).
286
+ """
287
+
288
+ def __init__(self, root: Path | str = DEFAULT_ROOT):
289
+ self.dir = Path(root) / "system-map"
290
+ self.dir.mkdir(parents=True, exist_ok=True)
291
+
292
+ def put(self, payload: dict[str, Any]) -> str:
293
+ # The suffix is not decoration: a bare second-resolution timestamp lets
294
+ # two maps written in the same second collide, which would silently
295
+ # rewrite a version another run had already pinned.
296
+ version = f"{_utcnow():%Y%m%dT%H%M%S}-{uuid.uuid4().hex[:6]}"
297
+ path = self.dir / f"{version}.json"
298
+ if path.exists():
299
+ raise RuntimeError(f"system map version {version} already exists")
300
+ path.write_text(json.dumps(payload, indent=2, sort_keys=True), encoding="utf-8")
301
+ tmp = self.dir / "latest.tmp"
302
+ tmp.write_text(version, encoding="utf-8")
303
+ os.replace(tmp, self.dir / "latest")
304
+ return version
305
+
306
+ def latest_version(self) -> str | None:
307
+ pointer = self.dir / "latest"
308
+ return pointer.read_text(encoding="utf-8").strip() if pointer.exists() else None
309
+
310
+ def get(self, version: str | None = None) -> dict[str, Any] | None:
311
+ version = version or self.latest_version()
312
+ if not version:
313
+ return None
314
+ path = self.dir / f"{version}.json"
315
+ return json.loads(path.read_text(encoding="utf-8")) if path.exists() else None
316
+
317
+ def versions(self) -> list[str]:
318
+ return sorted(p.stem for p in self.dir.glob("*.json"))
319
+
320
+
321
+ def list_runs(root: Path | str = DEFAULT_ROOT) -> list[str]:
322
+ runs = Path(root) / "runs"
323
+ return sorted((p.name for p in runs.iterdir() if p.is_dir()), reverse=True) if runs.exists() else []
qaas/target.py ADDED
@@ -0,0 +1,287 @@
1
+ """What the system is pointed at.
2
+
3
+ Everything the agents need to know about an application they have never seen:
4
+ where its code lives, how to reach it, who its users are. Without this the
5
+ system can only ever run against the app it was built alongside — which is the
6
+ difference between a demo and a tool.
7
+
8
+ A profile is deliberately small and mostly optional. An agent can discover a
9
+ great deal on its own; what it cannot discover is anything requiring a
10
+ credential, a URL that is not in the repository, or a judgement about which of
11
+ three directories is "the backend". Those are what a profile supplies.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ import os
17
+ import re
18
+ from pathlib import Path
19
+ from typing import Any, Literal
20
+
21
+ import yaml
22
+ from pydantic import BaseModel, ConfigDict, Field, model_validator
23
+
24
+ from qaas.paths import project_root
25
+
26
+ # Directories that are never product code. Excluded by default so a profile does
27
+ # not have to restate them and an agent does not waste a turn reading vendored
28
+ # dependencies.
29
+ DEFAULT_EXCLUDES = [
30
+ "node_modules", "vendor", "dist", "build", ".venv", "venv", "__pycache__",
31
+ ".git", ".next", "target", "coverage", ".pytest_cache", "site-packages",
32
+ ]
33
+
34
+
35
+ class Layout(BaseModel):
36
+ """Where things are. Every field is a hint; agents verify what they find."""
37
+
38
+ model_config = ConfigDict(extra="forbid")
39
+
40
+ backend: list[str] = Field(default_factory=list)
41
+ frontend: list[str] = Field(default_factory=list)
42
+ tests: list[str] = Field(default_factory=list)
43
+ migrations: list[str] = Field(default_factory=list)
44
+ spec: str | None = Field(default=None, description="OpenAPI document, if one exists.")
45
+ ownership: str | None = Field(default=None, description="CODEOWNERS or equivalent.")
46
+ docs: list[str] = Field(default_factory=list)
47
+ exclude: list[str] = Field(default_factory=lambda: list(DEFAULT_EXCLUDES))
48
+
49
+ def described(self) -> str:
50
+ """A prose sketch of the layout for an agent's task prompt."""
51
+ bits = []
52
+ for label, paths in (
53
+ ("backend", self.backend), ("frontend", self.frontend),
54
+ ("tests", self.tests), ("migrations", self.migrations),
55
+ ("docs", self.docs),
56
+ ):
57
+ if paths:
58
+ bits.append(f"{label}: {', '.join(paths)}")
59
+ if self.spec:
60
+ bits.append(f"API spec: {self.spec}")
61
+ if self.ownership:
62
+ bits.append(f"ownership: {self.ownership}")
63
+ return "; ".join(bits) or "not recorded — discover it yourself"
64
+
65
+
66
+ class Role(BaseModel):
67
+ """One account an agent can act as.
68
+
69
+ Credentials are read from the environment, never stored here: a profile is
70
+ committed to a repository and a password in it is a leak, not a convenience.
71
+ """
72
+
73
+ model_config = ConfigDict(extra="forbid")
74
+
75
+ username: str
76
+ password_env: str = "QAAS_PASSWORD"
77
+ description: str = ""
78
+
79
+ def password(self) -> str | None:
80
+ return os.environ.get(self.password_env)
81
+
82
+
83
+ class Auth(BaseModel):
84
+ """How an agent gets a credential for the running app."""
85
+
86
+ model_config = ConfigDict(extra="forbid")
87
+
88
+ mode: Literal["none", "login", "token"] = "none"
89
+ login_endpoint: str | None = Field(
90
+ default=None, description="e.g. 'POST /v1/auth/login'"
91
+ )
92
+ username_field: str = "email"
93
+ password_field: str = "password"
94
+ token_path: str = "access_token"
95
+ token_env: str | None = Field(
96
+ default=None, description="For mode=token: env var holding a bearer token."
97
+ )
98
+ roles: dict[str, Role] = Field(default_factory=dict)
99
+
100
+ @model_validator(mode="after")
101
+ def _coherent(self) -> "Auth":
102
+ if self.mode == "login" and not self.login_endpoint:
103
+ raise ValueError("auth.mode is 'login' but no login_endpoint is set")
104
+ if self.mode == "login" and not self.roles:
105
+ raise ValueError("auth.mode is 'login' but no roles are defined")
106
+ if self.mode == "token" and not self.token_env:
107
+ raise ValueError("auth.mode is 'token' but no token_env is set")
108
+ return self
109
+
110
+ def missing_secrets(self) -> list[str]:
111
+ """Env vars this profile needs that are not set. Checked before a run."""
112
+ missing = []
113
+ if self.mode == "token" and self.token_env and not os.environ.get(self.token_env):
114
+ missing.append(self.token_env)
115
+ if self.mode == "login":
116
+ for role in self.roles.values():
117
+ if not role.password() and role.password_env not in missing:
118
+ missing.append(role.password_env)
119
+ return missing
120
+
121
+
122
+ class Environment(BaseModel):
123
+ """How to get a running instance of the application.
124
+
125
+ Three modes, because real projects differ and pretending otherwise is what
126
+ makes a tool unusable:
127
+
128
+ compose — this system brings the app up and owns its lifecycle.
129
+ external — the app is already running somewhere (staging, a dev server).
130
+ Agents may read and exercise it but never reset or reseed it.
131
+ none — there is no reachable instance. Discovery is static only:
132
+ code, spec and schema. Most first runs against a real repo
133
+ start here, and that is a perfectly useful mode.
134
+ """
135
+
136
+ model_config = ConfigDict(extra="forbid")
137
+
138
+ mode: Literal["compose", "external", "none"] = "none"
139
+ api_url: str | None = None
140
+ web_url: str | None = None
141
+ health_path: str = "/health"
142
+
143
+ compose_file: str | None = None
144
+ services: list[str] = Field(default_factory=list)
145
+ seed_sql: str | None = None
146
+ db_service: str | None = None
147
+ db_user: str | None = None
148
+ db_name: str | None = None
149
+
150
+ startup_timeout_s: int = 180
151
+
152
+ @property
153
+ def is_managed(self) -> bool:
154
+ """Whether this system may create, reset and destroy the environment."""
155
+ return self.mode == "compose"
156
+
157
+ @property
158
+ def is_reachable(self) -> bool:
159
+ return self.mode in {"compose", "external"}
160
+
161
+ @model_validator(mode="after")
162
+ def _coherent(self) -> "Environment":
163
+ if self.mode == "compose" and not self.compose_file:
164
+ raise ValueError("environment.mode is 'compose' but no compose_file is set")
165
+ if self.mode == "external" and not (self.api_url or self.web_url):
166
+ raise ValueError(
167
+ "environment.mode is 'external' but neither api_url nor web_url is set"
168
+ )
169
+ return self
170
+
171
+
172
+ class TargetProfile(BaseModel):
173
+ """One application this system can be pointed at."""
174
+
175
+ model_config = ConfigDict(extra="forbid")
176
+
177
+ name: str
178
+ root: str = Field(description="Path to the repository, relative to cwd or absolute.")
179
+ description: str = ""
180
+ repo_url: str | None = None
181
+ default_branch: str = "main"
182
+
183
+ layout: Layout = Field(default_factory=Layout)
184
+ environment: Environment = Field(default_factory=Environment)
185
+ auth: Auth = Field(default_factory=Auth)
186
+
187
+ #: Golden ledger for calibration. Absent for real applications — you only
188
+ #: have one for an app whose defects you planted yourself.
189
+ ledger: str | None = None
190
+
191
+ @model_validator(mode="after")
192
+ def _name_is_a_slug(self) -> "TargetProfile":
193
+ if not re.fullmatch(r"[a-z0-9][a-z0-9-]{0,40}", self.name):
194
+ raise ValueError("target name must be a lowercase slug, e.g. 'my-app'")
195
+ return self
196
+
197
+ def root_path(self, base: Path | None = None) -> Path:
198
+ """Where the application under test actually is, on this machine.
199
+
200
+ This is *the* answer to "what am I testing", and it is deliberately not
201
+ the process cwd. `qaas run --repo <url>` clones into
202
+ `.qaas/targets/<slug>`, so the target is routinely somewhere the qaas
203
+ project is not; an absolute `root` in a profile is honoured as written.
204
+ """
205
+ root = Path(self.root)
206
+ if root.is_absolute():
207
+ return root
208
+ return (base if base is not None else project_root()) / root
209
+
210
+ def readiness(self, base: Path | None = None) -> list[str]:
211
+ """Everything that would stop a run right now. Empty means ready."""
212
+ problems: list[str] = []
213
+ root = self.root_path(base)
214
+ if not root.exists():
215
+ problems.append(f"repository root does not exist: {root}")
216
+ elif not root.is_dir():
217
+ problems.append(f"repository root is not a directory: {root}")
218
+
219
+ for missing in self.auth.missing_secrets():
220
+ problems.append(f"environment variable {missing} is not set")
221
+
222
+ if self.environment.mode == "compose":
223
+ compose = root / (self.environment.compose_file or "")
224
+ if not compose.exists():
225
+ problems.append(f"compose file not found: {compose}")
226
+ if self.layout.spec:
227
+ spec = root / self.layout.spec
228
+ if not spec.exists():
229
+ problems.append(f"API spec not found: {spec}")
230
+ return problems
231
+
232
+ def capabilities(self) -> dict[str, bool]:
233
+ """What this profile makes possible. Drives which agents can usefully run."""
234
+ return {
235
+ "static_analysis": True,
236
+ "spec_diff": self.layout.spec is not None,
237
+ "live_api": self.environment.is_reachable and self.environment.api_url is not None,
238
+ "live_ui": self.environment.is_reachable and self.environment.web_url is not None,
239
+ "reset_state": self.environment.is_managed,
240
+ "impersonate": self.auth.mode != "none",
241
+ "scored": self.ledger is not None,
242
+ }
243
+
244
+
245
+ def load_target(name: str, targets_dir: Path | str = "config/targets") -> TargetProfile:
246
+ path = Path(targets_dir) / f"{name}.yaml"
247
+ if not path.exists():
248
+ available = sorted(p.stem for p in Path(targets_dir).glob("*.yaml"))
249
+ raise FileNotFoundError(
250
+ f"no target profile '{name}' at {path}. "
251
+ f"Available: {', '.join(available) or 'none'}. "
252
+ "Create one with `qaas init <path-to-repo>`."
253
+ )
254
+ raw: dict[str, Any] = yaml.safe_load(path.read_text(encoding="utf-8")) or {}
255
+ raw.setdefault("name", name)
256
+ return TargetProfile.model_validate(raw)
257
+
258
+
259
+ # `list_targets(one_dir)` lived here and is gone. Listing profiles from a single
260
+ # directory is the bug that hid `<project>/config/targets/` the moment anything
261
+ # wrote into `.qaas/config/targets/`; profiles layer across every config
262
+ # directory, and `config.target_files(dirs)` is the one place that knows it.
263
+
264
+
265
+ def agent_usable(agent_name: str, caps: dict[str, bool]) -> bool:
266
+ """Whether an agent can do useful work with the capabilities available.
267
+
268
+ Lives here, beside `capabilities()`, because it has two callers that must
269
+ agree: `qaas doctor` reports it, and the router acts on it. They did not
270
+ agree for a while -- doctor would say "agents that cannot: BROWSER" and then
271
+ a run would dispatch BROWSER anyway and spend its whole budget looking for a
272
+ browser that was never there. Being told an agent cannot work and then
273
+ watching it run is worse than not being told.
274
+
275
+ Everything except BROWSER can contribute from static analysis alone, at
276
+ lower confidence. BROWSER without a reachable UI has nothing to do at all.
277
+ """
278
+ if agent_name in ("BROWSER", "GUIDE"):
279
+ # Both drive a browser. GUIDE's whole method is navigating the product
280
+ # as a person would; with nothing to navigate it has no job at all.
281
+ return caps.get("live_ui", False)
282
+ if agent_name == "LOAD":
283
+ # Performance work needs something to measure. LOAD can read query and
284
+ # rendering code statically, but a latency claim about an application it
285
+ # never called is a guess, and this system does not ship guesses.
286
+ return caps.get("live_api", False) or caps.get("live_ui", False)
287
+ return True