qaas-python 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. qaas/adapters/__init__.py +19 -0
  2. qaas/adapters/tracker.py +1783 -0
  3. qaas/adapters/vcs.py +555 -0
  4. qaas/cli.py +1757 -0
  5. qaas/config.py +409 -0
  6. qaas/defaults/config/agents/api.yaml +18 -0
  7. qaas/defaults/config/agents/architect.yaml +21 -0
  8. qaas/defaults/config/agents/auditor.yaml +19 -0
  9. qaas/defaults/config/agents/browser.yaml +15 -0
  10. qaas/defaults/config/agents/dba.yaml +20 -0
  11. qaas/defaults/config/agents/fixer.yaml +55 -0
  12. qaas/defaults/config/agents/guide.yaml +23 -0
  13. qaas/defaults/config/agents/load.yaml +26 -0
  14. qaas/defaults/config/agents/mapper.yaml +19 -0
  15. qaas/defaults/config/agents/reporter.yaml +19 -0
  16. qaas/defaults/config/agents/reproducer.yaml +21 -0
  17. qaas/defaults/config/agents/reviewer.yaml +18 -0
  18. qaas/defaults/config/agents/socket.yaml +23 -0
  19. qaas/defaults/config/agents/triage.yaml +20 -0
  20. qaas/defaults/config/agents/verifier.yaml +20 -0
  21. qaas/defaults/config/system.yaml +64 -0
  22. qaas/discover.py +242 -0
  23. qaas/envelope.py +318 -0
  24. qaas/envfile.py +100 -0
  25. qaas/guardrails.py +589 -0
  26. qaas/mcp/__init__.py +0 -0
  27. qaas/mcp/context.py +78 -0
  28. qaas/mcp/contract_diff.py +1011 -0
  29. qaas/mcp/defect_memory.py +495 -0
  30. qaas/mcp/env_control.py +925 -0
  31. qaas/mcp/envelope_server.py +463 -0
  32. qaas/mcp/test_runner.py +842 -0
  33. qaas/mcp/tracker.py +420 -0
  34. qaas/mcp/vcs.py +501 -0
  35. qaas/paths.py +317 -0
  36. qaas/plugin/.claude-plugin/plugin.json +9 -0
  37. qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
  38. qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
  39. qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
  40. qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
  41. qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
  42. qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
  43. qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
  44. qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
  45. qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
  46. qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
  47. qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
  48. qaas/plugin/skills/flake-detection/SKILL.md +39 -0
  49. qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
  50. qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
  51. qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
  52. qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
  53. qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
  54. qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
  55. qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
  56. qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
  57. qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
  58. qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
  59. qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
  60. qaas/plugin/skills/routing-rules/SKILL.md +34 -0
  61. qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
  62. qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
  63. qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
  64. qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
  65. qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
  66. qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
  67. qaas/prompts/API.md +44 -0
  68. qaas/prompts/ARCHITECT.md +80 -0
  69. qaas/prompts/AUDITOR.md +62 -0
  70. qaas/prompts/BROWSER.md +46 -0
  71. qaas/prompts/DBA.md +59 -0
  72. qaas/prompts/FIXER.md +55 -0
  73. qaas/prompts/GUIDE.md +94 -0
  74. qaas/prompts/LOAD.md +109 -0
  75. qaas/prompts/MAPPER.md +46 -0
  76. qaas/prompts/REPORTER.md +61 -0
  77. qaas/prompts/REPRODUCER.md +43 -0
  78. qaas/prompts/REVIEWER.md +53 -0
  79. qaas/prompts/SOCKET.md +100 -0
  80. qaas/prompts/TRIAGE.md +45 -0
  81. qaas/prompts/VERIFIER.md +41 -0
  82. qaas/prompts/_shared.md +45 -0
  83. qaas/registry.py +496 -0
  84. qaas/router.py +581 -0
  85. qaas/runner.py +210 -0
  86. qaas/scorecard.py +448 -0
  87. qaas/sdk_compat.py +52 -0
  88. qaas/store.py +323 -0
  89. qaas/target.py +287 -0
  90. qaas/tasks.py +438 -0
  91. qaas/trace.py +342 -0
  92. qaas_python-0.0.1.dist-info/METADATA +429 -0
  93. qaas_python-0.0.1.dist-info/RECORD +96 -0
  94. qaas_python-0.0.1.dist-info/WHEEL +4 -0
  95. qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
  96. qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/runner.py ADDED
@@ -0,0 +1,210 @@
1
+ """Running one agent: build its options, stream its turn, record what it cost.
2
+
3
+ Every invocation is its own `query()`. What comes back that matters is not the
4
+ agent's prose — that is a summary for the log — but what it wrote through its
5
+ tools, plus the cost and turn count the ledger needs.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import time
11
+ from dataclasses import dataclass
12
+ from typing import Any, AsyncIterator, Callable
13
+
14
+ from claude_agent_sdk import (
15
+ AssistantMessage,
16
+ ClaudeAgentOptions,
17
+ ResultMessage,
18
+ SystemMessage,
19
+ TextBlock,
20
+ ToolUseBlock,
21
+ query,
22
+ )
23
+
24
+ from qaas.config import AgentSpec
25
+ from qaas.mcp.context import ToolContext
26
+ from qaas.registry import build_options
27
+ from qaas.store import AgentResult
28
+
29
+
30
+ @dataclass
31
+ class RunOutcome:
32
+ """What one agent invocation produced, beyond its side effects."""
33
+
34
+ result: AgentResult
35
+ final_text: str = ""
36
+ tool_calls: int = 0
37
+
38
+ @property
39
+ def ok(self) -> bool:
40
+ return self.result.subtype == "success" and self.result.error is None
41
+
42
+
43
+ def _check_skills_loaded(spec: AgentSpec, ctx: ToolContext, message: Any, emit) -> None:
44
+ """Say something when the skills an agent declared did not load.
45
+
46
+ This exists because the failure has no symptom. Skills used to be found
47
+ through `setting_sources=["project"]`, resolved against the agent's cwd, so
48
+ a user whose repository had no `.claude/skills/` got none of them -- no
49
+ error, no warning, findings still produced, every procedure missing. It was
50
+ invisible for the life of the project and only surfaced when someone tried
51
+ to install the package.
52
+
53
+ The CLI's init message lists what it loaded. Comparing it against what was
54
+ asked for costs nothing and makes the next regression loud. A mismatch is
55
+ recorded and reported rather than raised: an agent with three of its four
56
+ skills is degraded, not broken, and killing the run would lose the work.
57
+ """
58
+ declared = list(spec.skills)
59
+ if not declared:
60
+ return
61
+ data = getattr(message, "data", None) or {}
62
+ loaded = {str(n) for n in (data.get("slash_commands") or [])}
63
+ if not loaded:
64
+ return # nothing reported; do not cry wolf about a shape we do not know
65
+ missing = [
66
+ name for name in declared
67
+ if not any(c == name or c.endswith(f":{name}") for c in loaded)
68
+ ]
69
+ if not missing:
70
+ return
71
+ ctx.store.log(
72
+ "skills_missing", agent=spec.name, declared=declared, missing=missing,
73
+ cwd=str(data.get("cwd") or ""),
74
+ )
75
+ emit("skills_missing", agent=spec.name, missing=missing)
76
+
77
+
78
+ #: How much of the task goes inline in the ledger line. Enough to tell two
79
+ #: REPRODUCER invocations apart at a glance; the artifact holds the rest.
80
+ TASK_PREVIEW_CHARS = 300
81
+
82
+
83
+ def _record_task(ctx: ToolContext, agent: str, task: str) -> dict[str, Any]:
84
+ """Persist the instruction an agent was actually given, and reference it.
85
+
86
+ `agent_started` recorded `task_chars=len(task)` -- the *length* of the
87
+ prompt. So the one thing needed to explain why an agent did what it did, or
88
+ to replay it, was the one thing the ledger threw away; REPRODUCER runs once per
89
+ finding and its five lines were distinguishable only by character count.
90
+
91
+ The task goes to the artifact store rather than inline because a task is
92
+ kilobytes and `qaas trace` has to stay readable. A preview stays on the line
93
+ so the common case needs no second lookup.
94
+
95
+ Never raises: an unwritable artifact store must not stop the agent from
96
+ running. Provenance degrades to the preview.
97
+ """
98
+ detail: dict[str, Any] = {"task_preview": task[:TASK_PREVIEW_CHARS]}
99
+ try:
100
+ # Numbered off what is already on disk, not off a ToolContext counter:
101
+ # the router builds a fresh context per dispatch, so an in-memory
102
+ # counter would restart at 1 and each REPRODUCER invocation would overwrite
103
+ # the previous one's task. This is the bug `put_result` already had.
104
+ existing = len(list((ctx.store.dir / "artifacts").glob(f"task-{agent}-*.md")))
105
+ detail["task_uri"] = ctx.store.put_artifact(f"task-{agent}-{existing + 1:02d}.md", task)
106
+ except OSError:
107
+ pass
108
+ return detail
109
+
110
+
111
+ async def run_agent(
112
+ spec: AgentSpec,
113
+ ctx: ToolContext,
114
+ task: str,
115
+ *,
116
+ options: ClaudeAgentOptions | None = None,
117
+ max_budget_usd: float | None = None,
118
+ on_event: Callable[[str, dict[str, Any]], None] | None = None,
119
+ ) -> RunOutcome:
120
+ """Invoke one agent and record the outcome.
121
+
122
+ Failures are captured, not raised. One agent falling over should cost the run
123
+ that agent's findings, not the whole run — the router decides whether to
124
+ retry, skip, or escalate.
125
+ """
126
+ # Building the options can fail on its own — a missing prompt file
127
+ # (FileNotFoundError), an MCP server the config names but nothing provides
128
+ # (UnknownServer), a declared server whose ${VAR} is unset
129
+ # (MissingServerEnv). Outside the try below, those propagated out of a
130
+ # function whose contract is "failures are captured, not raised": the
131
+ # router's `_gather` calls `asyncio.gather` without `return_exceptions`
132
+ # and `run()` catches only BudgetExceeded, so one such agent aborted the
133
+ # whole run with no `run_finished` line — while its sibling discovery
134
+ # agents, which gather does not cancel, kept going and kept spending with
135
+ # nothing recording their cost.
136
+ try:
137
+ options = options or build_options(spec, ctx)
138
+ except Exception as exc: # noqa: BLE001 — same contract as the query below
139
+ error = f"{type(exc).__name__}: {exc}"
140
+ ctx.store.log("agent_error", agent=spec.name, error=error)
141
+ result = AgentResult(agent=spec.name, subtype="failure", error=error)
142
+ ctx.store.put_result(result)
143
+ return RunOutcome(result=result)
144
+
145
+ if max_budget_usd is not None:
146
+ options.max_budget_usd = max_budget_usd
147
+ started = time.monotonic()
148
+ ctx.store.log(
149
+ "agent_started", agent=spec.name, model=spec.model, task_chars=len(task),
150
+ **_record_task(ctx, spec.name, task),
151
+ )
152
+
153
+ before = {e.id for e in ctx.store.envelopes()}
154
+ final_text = ""
155
+ tool_calls = 0
156
+ subtype = "success"
157
+ error: str | None = None
158
+ cost = 0.0
159
+ turns = 0
160
+
161
+ def emit(kind: str, **detail: Any) -> None:
162
+ if on_event:
163
+ on_event(kind, detail)
164
+
165
+ try:
166
+ async for message in query(prompt=task, options=options):
167
+ if isinstance(message, SystemMessage) and message.subtype == "init":
168
+ _check_skills_loaded(spec, ctx, message, emit)
169
+ elif isinstance(message, AssistantMessage):
170
+ for block in message.content:
171
+ if isinstance(block, TextBlock):
172
+ final_text = block.text
173
+ elif isinstance(block, ToolUseBlock):
174
+ tool_calls += 1
175
+ emit("tool", agent=spec.name, tool=block.name)
176
+ elif isinstance(message, ResultMessage):
177
+ subtype = message.subtype or "success"
178
+ cost = message.total_cost_usd or 0.0
179
+ turns = message.num_turns or 0
180
+ if message.is_error:
181
+ error = _error_text(message)
182
+ if isinstance(message.result, str):
183
+ final_text = message.result
184
+ except Exception as exc: # noqa: BLE001 — the router decides what a failure means
185
+ subtype = "failure"
186
+ error = f"{type(exc).__name__}: {exc}"
187
+ ctx.store.log("agent_error", agent=spec.name, error=error)
188
+
189
+ produced = [e.id for e in ctx.store.envelopes() if e.id not in before]
190
+ result = AgentResult(
191
+ agent=spec.name,
192
+ subtype=subtype,
193
+ cost_usd=cost,
194
+ num_turns=turns,
195
+ duration_s=round(time.monotonic() - started, 2),
196
+ envelope_ids=produced,
197
+ error=error,
198
+ )
199
+ ctx.store.put_result(result)
200
+ emit("finished", agent=spec.name, cost=cost, envelopes=len(produced), subtype=subtype)
201
+ return RunOutcome(result=result, final_text=final_text.strip(), tool_calls=tool_calls)
202
+
203
+
204
+ def _error_text(message: ResultMessage) -> str:
205
+ """A usable error string out of whichever field this SDK version populated."""
206
+ for attr in ("errors", "terminal_reason", "stop_reason", "api_error_status"):
207
+ value = getattr(message, attr, None)
208
+ if value:
209
+ return f"{attr}={value}"
210
+ return f"result subtype={message.subtype}"
qaas/scorecard.py ADDED
@@ -0,0 +1,448 @@
1
+ """Scoring a run against the golden ledger.
2
+
3
+ This is the module that keeps the project honest. Everything else in the system
4
+ can look like it works — agents run, envelopes appear, tickets get filed — while
5
+ the findings are noise. The scorecard is what says otherwise, in numbers.
6
+
7
+ Matching is deliberately deterministic. Using a model to judge whether a finding
8
+ matches a seeded defect would make the score depend on the same class of system
9
+ being measured, and a generous judge would flatter the result exactly when the
10
+ result least deserves it.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import re
16
+ from dataclasses import dataclass, field
17
+ from pathlib import Path
18
+ from typing import Any, Iterable
19
+
20
+ import yaml
21
+
22
+ from qaas.envelope import DefectEnvelope, Domain, normalize_path, Severity
23
+
24
+ MATCH_THRESHOLD = 0.5
25
+
26
+
27
+ @dataclass(frozen=True)
28
+ class GoldenDefect:
29
+ """One seeded defect, as recorded in target-app/defects.yaml."""
30
+
31
+ id: str
32
+ domain: str
33
+ defect_class: str
34
+ severity: str
35
+ title: str
36
+ detail: str
37
+ endpoint: str | None = None
38
+ ui_route: str | None = None
39
+ paths: tuple[str, ...] = ()
40
+ keywords: tuple[str, ...] = ()
41
+ phase: int = 1
42
+ security_relevant: bool = False
43
+ # True for defects the system found rather than ones placed for it to find.
44
+ # They still count, but they are weaker evidence: the ledger is a floor on
45
+ # what exists in the app, never a complete oracle.
46
+ discovered_not_seeded: bool = False
47
+ #: The ref a fix for this defect landed on, once one has. Retires the entry
48
+ #: from the recall denominator without deleting it -- REVIEWER escalated a
49
+ #: correct fix because the schema had no way to say this, and CLAUDE.md
50
+ #: requires the ledger to change in the same commit as the defect. Deleting
51
+ #: the entry instead would lose the severity and domain expectations that
52
+ #: make a past score reproducible.
53
+ fixed_in: str | None = None
54
+
55
+ @property
56
+ def retired(self) -> bool:
57
+ """Fixed, so no longer expected to be present. Not scored for recall."""
58
+ return self.fixed_in is not None
59
+
60
+
61
+ @dataclass(frozen=True)
62
+ class PlantedNonDefect:
63
+ """Correct behaviour that looks wrong. Reporting one is a false positive."""
64
+
65
+ id: str
66
+ title: str
67
+ why_correct: str
68
+ endpoint: str | None = None
69
+ ui_route: str | None = None
70
+ paths: tuple[str, ...] = ()
71
+
72
+
73
+ @dataclass
74
+ class GoldenLedger:
75
+ defects: list[GoldenDefect]
76
+ not_defects: list[PlantedNonDefect]
77
+
78
+ @classmethod
79
+ def load(cls, path: Path | str) -> "GoldenLedger":
80
+ raw = yaml.safe_load(Path(path).read_text(encoding="utf-8"))
81
+ return cls(
82
+ defects=[_golden(d) for d in raw.get("defects", [])],
83
+ not_defects=[_planted(d) for d in raw.get("not_defects", [])],
84
+ )
85
+
86
+ def for_phase(self, phase: int) -> list[GoldenDefect]:
87
+ return [d for d in self.defects if d.phase <= phase]
88
+
89
+ def by_id(self, defect_id: str) -> GoldenDefect | None:
90
+ return next((d for d in self.defects if d.id == defect_id), None)
91
+
92
+
93
+ def _enum_or_die(kind: type[Domain] | type[Severity], value: Any, defect_id: str, field: str) -> str:
94
+ """A ledger value that must name an enum member, checked when it is read.
95
+
96
+ `domain` and `severity` arrived as free strings and were compared against
97
+ validated enums downstream, so the two failure modes were both silent and
98
+ both late. A `domain: ui` (not a `Domain`) scores 0.0 similarity against
99
+ every envelope forever: the defect is permanently `missed`, recall drops and
100
+ nothing says why. A `severity: high` (not a `Severity`) raises ValueError out
101
+ of `Match.severity_delta` — *after* a paid run has finished.
102
+
103
+ "A stale ledger silently corrupts every score" is the reason the ledger is a
104
+ human's responsibility at merge. This makes the failure loud and immediate
105
+ instead, which is the only part of that a program can help with.
106
+ """
107
+ try:
108
+ return kind(value).value
109
+ except ValueError:
110
+ allowed = ", ".join(sorted(m.value for m in kind))
111
+ raise ValueError(
112
+ f"{defect_id}: {field} '{value}' is not a {kind.__name__}. Use one of: {allowed}."
113
+ ) from None
114
+
115
+
116
+ def _golden(d: dict[str, Any]) -> GoldenDefect:
117
+ loc = d.get("location", {}) or {}
118
+ return GoldenDefect(
119
+ id=d["id"],
120
+ domain=_enum_or_die(Domain, d["domain"], d["id"], "domain"),
121
+ defect_class=d.get("class", "bug"),
122
+ severity=_enum_or_die(Severity, d["severity"], d["id"], "severity"),
123
+ title=d["title"],
124
+ detail=d.get("detail", ""),
125
+ endpoint=loc.get("endpoint"),
126
+ ui_route=loc.get("ui_route"),
127
+ paths=tuple(loc.get("paths", ())),
128
+ keywords=tuple(d.get("keywords", ())),
129
+ phase=int(d.get("phase", 1)),
130
+ security_relevant=bool(d.get("security_relevant", False)),
131
+ discovered_not_seeded=bool(d.get("discovered_not_seeded", False)),
132
+ fixed_in=(str(d["fixed_in"]) if d.get("fixed_in") else None),
133
+ )
134
+
135
+
136
+ def _planted(d: dict[str, Any]) -> PlantedNonDefect:
137
+ loc = d.get("location", {}) or {}
138
+ return PlantedNonDefect(
139
+ id=d["id"],
140
+ title=d["title"],
141
+ why_correct=d.get("why_correct", ""),
142
+ endpoint=loc.get("endpoint"),
143
+ ui_route=loc.get("ui_route"),
144
+ paths=tuple(loc.get("paths", ())),
145
+ )
146
+
147
+
148
+ # -- similarity -------------------------------------------------------------
149
+
150
+
151
+ def _norm_endpoint(endpoint: str | None) -> str | None:
152
+ """`GET /v1/orders/{order_id}` and `get /v1/orders/{id}` are the same endpoint."""
153
+ if not endpoint:
154
+ return None
155
+ e = re.sub(r"\{[^}]*\}", "{}", endpoint.strip().lower())
156
+ return re.sub(r"\s+", " ", e)
157
+
158
+
159
+ def _norm_path(path: str) -> str:
160
+ """Compare by tail, so `target-app/api/app/routes/orders.py:88` matches `api/app/routes/orders.py`.
161
+
162
+ Delegates to the envelope's own normalisation rather than restating it. The
163
+ two were separate implementations and drifted: the scorer handled `:104-112`
164
+ line ranges and the repo-root prefix, the fingerprint handled neither, so a
165
+ defect the scorer counted as one thing hashed as three.
166
+ """
167
+ return normalize_path(path)
168
+
169
+
170
+ def _path_overlap(a: Iterable[str], b: Iterable[str]) -> float:
171
+ """Fraction of the golden defect's files the report also names."""
172
+ golden = {_norm_path(p) for p in b}
173
+ if not golden:
174
+ return 0.0
175
+ reported = {_norm_path(p) for p in a}
176
+ hits = sum(
177
+ 1
178
+ for g in golden
179
+ if any(r == g or r.endswith("/" + g) or g.endswith("/" + r) for r in reported)
180
+ )
181
+ return hits / len(golden)
182
+
183
+
184
+ def _keyword_overlap(text: str, keywords: Iterable[str]) -> float:
185
+ terms = list(keywords)
186
+ if not terms:
187
+ return 0.0
188
+ blob = text.lower()
189
+ return sum(1 for t in terms if t.lower() in blob) / len(terms)
190
+
191
+
192
+ # Domains that describe the same surface. An agent choosing either one has
193
+ # classified the defect defensibly, so scoring must accept both.
194
+ #
195
+ # Both entries were learned from real runs, and both times the scorer was wrong
196
+ # rather than the agent: API filed a cross-tenant read under `security`, and
197
+ # BROWSER filed a missing label and a contrast failure under `ux`. Marking those
198
+ # as misses would have hidden a perfect discovery run behind a 50% score.
199
+ _EQUIVALENT_DOMAINS: dict[str, set[str]] = {
200
+ "ux": {"frontend"},
201
+ "frontend": {"ux"},
202
+ }
203
+
204
+
205
+ def _domains_compatible(reported: str, golden: GoldenDefect) -> bool:
206
+ """Whether a reported domain is an acceptable classification of this defect.
207
+
208
+ Domain stays a gate — naming the right file under a genuinely wrong domain
209
+ means the defect was misunderstood — but the gate accepts any defensible
210
+ reading, not only the one the ledger happened to write down.
211
+ """
212
+ if reported == golden.domain:
213
+ return True
214
+ if reported == "security" and golden.security_relevant:
215
+ return True
216
+ return golden.domain in _EQUIVALENT_DOMAINS.get(reported, set())
217
+
218
+
219
+ def similarity(env: DefectEnvelope, golden: GoldenDefect) -> float:
220
+ """0..1 confidence that `env` reports `golden`."""
221
+ if not _domains_compatible(env.domain.value, golden):
222
+ return 0.0
223
+
224
+ text = f"{env.title} {env.summary} {env.suggested_fix_area}"
225
+
226
+ endpoint_match = (
227
+ _norm_endpoint(env.location.endpoint) is not None
228
+ and _norm_endpoint(env.location.endpoint) == _norm_endpoint(golden.endpoint)
229
+ )
230
+ route_match = (
231
+ env.location.ui_route is not None
232
+ and golden.ui_route is not None
233
+ and env.location.ui_route.rstrip("/") == golden.ui_route.rstrip("/")
234
+ )
235
+
236
+ # A cross-cutting defect has no single endpoint or route to anchor on, and
237
+ # the files that best demonstrate it are a judgement call — a report of
238
+ # inconsistent error shapes may cite whichever two handlers differ. For
239
+ # those, the prose has to carry the identification.
240
+ anchored = bool(golden.endpoint or golden.ui_route)
241
+ weights = (0.45, 0.30, 0.35) if anchored else (0.0, 0.35, 0.65)
242
+ w_location, w_paths, w_keywords = weights
243
+
244
+ score = w_location if (endpoint_match or route_match) else 0.0
245
+ score += w_paths * _path_overlap(env.location.paths, golden.paths)
246
+ score += w_keywords * _keyword_overlap(text, golden.keywords)
247
+ return min(score, 1.0)
248
+
249
+
250
+ def resembles_planted(env: DefectEnvelope, planted: PlantedNonDefect) -> float:
251
+ """Whether a finding is a report of deliberately-correct behaviour.
252
+
253
+ This needs a real anchor — the same endpoint or the same route. Sharing a
254
+ file is not enough: `orders.py` holds six seeded defects as well as the
255
+ deliberately-correct legacy handler, so a path-only rule blamed agents for
256
+ reporting the legacy endpoint when they had reported something else entirely.
257
+ """
258
+ if planted.endpoint and _norm_endpoint(env.location.endpoint) == _norm_endpoint(planted.endpoint):
259
+ return 1.0
260
+ if planted.ui_route and env.location.ui_route == planted.ui_route:
261
+ return 1.0
262
+ return 0.0
263
+
264
+
265
+ # -- results ----------------------------------------------------------------
266
+
267
+
268
+ @dataclass
269
+ class Match:
270
+ golden_id: str
271
+ envelope_id: str
272
+ score: float
273
+ reported_severity: str
274
+ expected_severity: str
275
+
276
+ @property
277
+ def severity_delta(self) -> int:
278
+ """Ranks apart. 0 is agreement, positive means the report was too calm."""
279
+ return Severity(self.reported_severity).rank - Severity(self.expected_severity).rank
280
+
281
+
282
+ @dataclass
283
+ class Scorecard:
284
+ matches: list[Match] = field(default_factory=list)
285
+ missed: list[str] = field(default_factory=list)
286
+ false_positives: list[str] = field(default_factory=list)
287
+ duplicates: list[str] = field(default_factory=list)
288
+ regressions_on_planted: list[tuple[str, str]] = field(default_factory=list)
289
+ #: Findings that matched a defect already marked `fixed_in`. Neither a find
290
+ #: nor a false positive: the report is correct wherever the fix has not
291
+ #: landed, so scoring it either way would be a lie about the run.
292
+ retired_hits: list[tuple[str, str]] = field(default_factory=list)
293
+ total_golden: int = 0
294
+ total_envelopes: int = 0
295
+ cost_usd: float = 0.0
296
+
297
+ @property
298
+ def recall(self) -> float:
299
+ return len(self.matches) / self.total_golden if self.total_golden else 0.0
300
+
301
+ @property
302
+ def precision(self) -> float:
303
+ judged = len(self.matches) + len(self.false_positives)
304
+ return len(self.matches) / judged if judged else 0.0
305
+
306
+ @property
307
+ def false_positive_rate(self) -> float:
308
+ return len(self.false_positives) / self.total_envelopes if self.total_envelopes else 0.0
309
+
310
+ @property
311
+ def duplicate_rate(self) -> float:
312
+ return len(self.duplicates) / self.total_envelopes if self.total_envelopes else 0.0
313
+
314
+ @property
315
+ def severity_agreement(self) -> float:
316
+ """Share of matches scored within one rank of the expected severity."""
317
+ if not self.matches:
318
+ return 0.0
319
+ return sum(1 for m in self.matches if abs(m.severity_delta) <= 1) / len(self.matches)
320
+
321
+ @property
322
+ def cost_per_accepted(self) -> float | None:
323
+ return self.cost_usd / len(self.matches) if self.matches else None
324
+
325
+ def summary(self) -> dict[str, Any]:
326
+ return {
327
+ "found": len(self.matches),
328
+ "of": self.total_golden,
329
+ "recall": round(self.recall, 3),
330
+ "precision": round(self.precision, 3),
331
+ "false_positives": len(self.false_positives),
332
+ "false_positive_rate": round(self.false_positive_rate, 3),
333
+ "duplicates": len(self.duplicates),
334
+ "duplicate_rate": round(self.duplicate_rate, 3),
335
+ "severity_agreement": round(self.severity_agreement, 3),
336
+ "planted_misreported": len(self.regressions_on_planted),
337
+ "retired_hits": len(self.retired_hits),
338
+ "cost_usd": round(self.cost_usd, 4),
339
+ "cost_per_accepted": (
340
+ round(self.cost_per_accepted, 4) if self.cost_per_accepted is not None else None
341
+ ),
342
+ "missed": sorted(self.missed),
343
+ }
344
+
345
+
346
+ def score(
347
+ envelopes: list[DefectEnvelope],
348
+ ledger: GoldenLedger,
349
+ *,
350
+ phase: int = 1,
351
+ domains: set[str] | None = None,
352
+ cost_usd: float = 0.0,
353
+ threshold: float = MATCH_THRESHOLD,
354
+ ) -> Scorecard:
355
+ """Match findings to seeded defects, best pair first.
356
+
357
+ Assignment is one-to-one and greedy on the strongest pair remaining. A second
358
+ envelope for an already-matched defect is a duplicate, not a second find —
359
+ counting it as a find would reward exactly the ticket-spam this system is
360
+ built to avoid.
361
+ """
362
+ in_scope = [d for d in ledger.for_phase(phase) if domains is None or d.domain in domains]
363
+ # A defect marked `fixed_in` is still matched -- so a report of it is not
364
+ # written off as a false positive -- but it leaves the recall denominator.
365
+ # Counting a repaired defect as a miss on every future run is exactly the
366
+ # silent corruption CLAUDE.md warns about.
367
+ golden = [d for d in in_scope if not d.retired]
368
+ retired = {d.id for d in in_scope if d.retired}
369
+ matchable = in_scope
370
+ card = Scorecard(total_golden=len(golden), total_envelopes=len(envelopes), cost_usd=cost_usd)
371
+
372
+ pairs = sorted(
373
+ (
374
+ (similarity(env, g), env, g)
375
+ for env in envelopes
376
+ for g in matchable
377
+ if similarity(env, g) >= threshold
378
+ ),
379
+ key=lambda t: -t[0],
380
+ )
381
+
382
+ claimed_golden: set[str] = set()
383
+ claimed_env: set[str] = set()
384
+ contested: list[tuple[float, DefectEnvelope, GoldenDefect]] = []
385
+
386
+ # First pass: settle the unambiguous pairs, strongest first.
387
+ for sim, env, g in pairs:
388
+ if g.id in claimed_golden or env.id in claimed_env:
389
+ contested.append((sim, env, g))
390
+ continue
391
+ claimed_golden.add(g.id)
392
+ claimed_env.add(env.id)
393
+ if g.id in retired:
394
+ card.retired_hits.append((env.id, g.id))
395
+ else:
396
+ card.matches.append(
397
+ Match(
398
+ golden_id=g.id,
399
+ envelope_id=env.id,
400
+ score=round(sim, 3),
401
+ reported_severity=env.severity.value,
402
+ expected_severity=g.severity,
403
+ )
404
+ )
405
+
406
+ # Second pass: an envelope whose best match was taken gets its next-best
407
+ # before anything else. Branding it a duplicate here was a real bug — two
408
+ # defects in one file on one route (a missing empty state and an unhandled
409
+ # rejection, both on /orders in OrdersList.tsx) each match the other's
410
+ # golden entry, so whichever lost the first pass was written off entirely.
411
+ # Only an envelope with no unclaimed match left is genuinely a duplicate.
412
+ for sim, env, g in contested:
413
+ if env.id in claimed_env:
414
+ continue
415
+ if g.id not in claimed_golden:
416
+ claimed_golden.add(g.id)
417
+ claimed_env.add(env.id)
418
+ if g.id in retired:
419
+ card.retired_hits.append((env.id, g.id))
420
+ else:
421
+ card.matches.append(
422
+ Match(
423
+ golden_id=g.id,
424
+ envelope_id=env.id,
425
+ score=round(sim, 3),
426
+ reported_severity=env.severity.value,
427
+ expected_severity=g.severity,
428
+ )
429
+ )
430
+
431
+ for sim, env, g in contested:
432
+ if env.id not in claimed_env:
433
+ card.duplicates.append(env.id)
434
+ claimed_env.add(env.id)
435
+
436
+ card.missed = [g.id for g in golden if g.id not in claimed_golden]
437
+
438
+ for env in envelopes:
439
+ if env.id in claimed_env:
440
+ continue
441
+ card.false_positives.append(env.id)
442
+ for planted in ledger.not_defects:
443
+ if resembles_planted(env, planted) >= MATCH_THRESHOLD:
444
+ card.regressions_on_planted.append((env.id, planted.id))
445
+ break
446
+
447
+ card.matches.sort(key=lambda m: m.golden_id)
448
+ return card