qaas-python 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- qaas/adapters/__init__.py +19 -0
- qaas/adapters/tracker.py +1783 -0
- qaas/adapters/vcs.py +555 -0
- qaas/cli.py +1757 -0
- qaas/config.py +409 -0
- qaas/defaults/config/agents/api.yaml +18 -0
- qaas/defaults/config/agents/architect.yaml +21 -0
- qaas/defaults/config/agents/auditor.yaml +19 -0
- qaas/defaults/config/agents/browser.yaml +15 -0
- qaas/defaults/config/agents/dba.yaml +20 -0
- qaas/defaults/config/agents/fixer.yaml +55 -0
- qaas/defaults/config/agents/guide.yaml +23 -0
- qaas/defaults/config/agents/load.yaml +26 -0
- qaas/defaults/config/agents/mapper.yaml +19 -0
- qaas/defaults/config/agents/reporter.yaml +19 -0
- qaas/defaults/config/agents/reproducer.yaml +21 -0
- qaas/defaults/config/agents/reviewer.yaml +18 -0
- qaas/defaults/config/agents/socket.yaml +23 -0
- qaas/defaults/config/agents/triage.yaml +20 -0
- qaas/defaults/config/agents/verifier.yaml +20 -0
- qaas/defaults/config/system.yaml +64 -0
- qaas/discover.py +242 -0
- qaas/envelope.py +318 -0
- qaas/envfile.py +100 -0
- qaas/guardrails.py +589 -0
- qaas/mcp/__init__.py +0 -0
- qaas/mcp/context.py +78 -0
- qaas/mcp/contract_diff.py +1011 -0
- qaas/mcp/defect_memory.py +495 -0
- qaas/mcp/env_control.py +925 -0
- qaas/mcp/envelope_server.py +463 -0
- qaas/mcp/test_runner.py +842 -0
- qaas/mcp/tracker.py +420 -0
- qaas/mcp/vcs.py +501 -0
- qaas/paths.py +317 -0
- qaas/plugin/.claude-plugin/plugin.json +9 -0
- qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
- qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
- qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
- qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
- qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
- qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
- qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
- qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
- qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
- qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
- qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
- qaas/plugin/skills/flake-detection/SKILL.md +39 -0
- qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
- qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
- qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
- qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
- qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
- qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
- qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
- qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
- qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
- qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
- qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
- qaas/plugin/skills/routing-rules/SKILL.md +34 -0
- qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
- qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
- qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
- qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
- qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
- qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
- qaas/prompts/API.md +44 -0
- qaas/prompts/ARCHITECT.md +80 -0
- qaas/prompts/AUDITOR.md +62 -0
- qaas/prompts/BROWSER.md +46 -0
- qaas/prompts/DBA.md +59 -0
- qaas/prompts/FIXER.md +55 -0
- qaas/prompts/GUIDE.md +94 -0
- qaas/prompts/LOAD.md +109 -0
- qaas/prompts/MAPPER.md +46 -0
- qaas/prompts/REPORTER.md +61 -0
- qaas/prompts/REPRODUCER.md +43 -0
- qaas/prompts/REVIEWER.md +53 -0
- qaas/prompts/SOCKET.md +100 -0
- qaas/prompts/TRIAGE.md +45 -0
- qaas/prompts/VERIFIER.md +41 -0
- qaas/prompts/_shared.md +45 -0
- qaas/registry.py +496 -0
- qaas/router.py +581 -0
- qaas/runner.py +210 -0
- qaas/scorecard.py +448 -0
- qaas/sdk_compat.py +52 -0
- qaas/store.py +323 -0
- qaas/target.py +287 -0
- qaas/tasks.py +438 -0
- qaas/trace.py +342 -0
- qaas_python-0.0.1.dist-info/METADATA +429 -0
- qaas_python-0.0.1.dist-info/RECORD +96 -0
- qaas_python-0.0.1.dist-info/WHEEL +4 -0
- qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
- qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
qaas/runner.py
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
"""Running one agent: build its options, stream its turn, record what it cost.
|
|
2
|
+
|
|
3
|
+
Every invocation is its own `query()`. What comes back that matters is not the
|
|
4
|
+
agent's prose — that is a summary for the log — but what it wrote through its
|
|
5
|
+
tools, plus the cost and turn count the ledger needs.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import time
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from typing import Any, AsyncIterator, Callable
|
|
13
|
+
|
|
14
|
+
from claude_agent_sdk import (
|
|
15
|
+
AssistantMessage,
|
|
16
|
+
ClaudeAgentOptions,
|
|
17
|
+
ResultMessage,
|
|
18
|
+
SystemMessage,
|
|
19
|
+
TextBlock,
|
|
20
|
+
ToolUseBlock,
|
|
21
|
+
query,
|
|
22
|
+
)
|
|
23
|
+
|
|
24
|
+
from qaas.config import AgentSpec
|
|
25
|
+
from qaas.mcp.context import ToolContext
|
|
26
|
+
from qaas.registry import build_options
|
|
27
|
+
from qaas.store import AgentResult
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class RunOutcome:
|
|
32
|
+
"""What one agent invocation produced, beyond its side effects."""
|
|
33
|
+
|
|
34
|
+
result: AgentResult
|
|
35
|
+
final_text: str = ""
|
|
36
|
+
tool_calls: int = 0
|
|
37
|
+
|
|
38
|
+
@property
|
|
39
|
+
def ok(self) -> bool:
|
|
40
|
+
return self.result.subtype == "success" and self.result.error is None
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _check_skills_loaded(spec: AgentSpec, ctx: ToolContext, message: Any, emit) -> None:
|
|
44
|
+
"""Say something when the skills an agent declared did not load.
|
|
45
|
+
|
|
46
|
+
This exists because the failure has no symptom. Skills used to be found
|
|
47
|
+
through `setting_sources=["project"]`, resolved against the agent's cwd, so
|
|
48
|
+
a user whose repository had no `.claude/skills/` got none of them -- no
|
|
49
|
+
error, no warning, findings still produced, every procedure missing. It was
|
|
50
|
+
invisible for the life of the project and only surfaced when someone tried
|
|
51
|
+
to install the package.
|
|
52
|
+
|
|
53
|
+
The CLI's init message lists what it loaded. Comparing it against what was
|
|
54
|
+
asked for costs nothing and makes the next regression loud. A mismatch is
|
|
55
|
+
recorded and reported rather than raised: an agent with three of its four
|
|
56
|
+
skills is degraded, not broken, and killing the run would lose the work.
|
|
57
|
+
"""
|
|
58
|
+
declared = list(spec.skills)
|
|
59
|
+
if not declared:
|
|
60
|
+
return
|
|
61
|
+
data = getattr(message, "data", None) or {}
|
|
62
|
+
loaded = {str(n) for n in (data.get("slash_commands") or [])}
|
|
63
|
+
if not loaded:
|
|
64
|
+
return # nothing reported; do not cry wolf about a shape we do not know
|
|
65
|
+
missing = [
|
|
66
|
+
name for name in declared
|
|
67
|
+
if not any(c == name or c.endswith(f":{name}") for c in loaded)
|
|
68
|
+
]
|
|
69
|
+
if not missing:
|
|
70
|
+
return
|
|
71
|
+
ctx.store.log(
|
|
72
|
+
"skills_missing", agent=spec.name, declared=declared, missing=missing,
|
|
73
|
+
cwd=str(data.get("cwd") or ""),
|
|
74
|
+
)
|
|
75
|
+
emit("skills_missing", agent=spec.name, missing=missing)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
#: How much of the task goes inline in the ledger line. Enough to tell two
|
|
79
|
+
#: REPRODUCER invocations apart at a glance; the artifact holds the rest.
|
|
80
|
+
TASK_PREVIEW_CHARS = 300
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _record_task(ctx: ToolContext, agent: str, task: str) -> dict[str, Any]:
|
|
84
|
+
"""Persist the instruction an agent was actually given, and reference it.
|
|
85
|
+
|
|
86
|
+
`agent_started` recorded `task_chars=len(task)` -- the *length* of the
|
|
87
|
+
prompt. So the one thing needed to explain why an agent did what it did, or
|
|
88
|
+
to replay it, was the one thing the ledger threw away; REPRODUCER runs once per
|
|
89
|
+
finding and its five lines were distinguishable only by character count.
|
|
90
|
+
|
|
91
|
+
The task goes to the artifact store rather than inline because a task is
|
|
92
|
+
kilobytes and `qaas trace` has to stay readable. A preview stays on the line
|
|
93
|
+
so the common case needs no second lookup.
|
|
94
|
+
|
|
95
|
+
Never raises: an unwritable artifact store must not stop the agent from
|
|
96
|
+
running. Provenance degrades to the preview.
|
|
97
|
+
"""
|
|
98
|
+
detail: dict[str, Any] = {"task_preview": task[:TASK_PREVIEW_CHARS]}
|
|
99
|
+
try:
|
|
100
|
+
# Numbered off what is already on disk, not off a ToolContext counter:
|
|
101
|
+
# the router builds a fresh context per dispatch, so an in-memory
|
|
102
|
+
# counter would restart at 1 and each REPRODUCER invocation would overwrite
|
|
103
|
+
# the previous one's task. This is the bug `put_result` already had.
|
|
104
|
+
existing = len(list((ctx.store.dir / "artifacts").glob(f"task-{agent}-*.md")))
|
|
105
|
+
detail["task_uri"] = ctx.store.put_artifact(f"task-{agent}-{existing + 1:02d}.md", task)
|
|
106
|
+
except OSError:
|
|
107
|
+
pass
|
|
108
|
+
return detail
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
async def run_agent(
|
|
112
|
+
spec: AgentSpec,
|
|
113
|
+
ctx: ToolContext,
|
|
114
|
+
task: str,
|
|
115
|
+
*,
|
|
116
|
+
options: ClaudeAgentOptions | None = None,
|
|
117
|
+
max_budget_usd: float | None = None,
|
|
118
|
+
on_event: Callable[[str, dict[str, Any]], None] | None = None,
|
|
119
|
+
) -> RunOutcome:
|
|
120
|
+
"""Invoke one agent and record the outcome.
|
|
121
|
+
|
|
122
|
+
Failures are captured, not raised. One agent falling over should cost the run
|
|
123
|
+
that agent's findings, not the whole run — the router decides whether to
|
|
124
|
+
retry, skip, or escalate.
|
|
125
|
+
"""
|
|
126
|
+
# Building the options can fail on its own — a missing prompt file
|
|
127
|
+
# (FileNotFoundError), an MCP server the config names but nothing provides
|
|
128
|
+
# (UnknownServer), a declared server whose ${VAR} is unset
|
|
129
|
+
# (MissingServerEnv). Outside the try below, those propagated out of a
|
|
130
|
+
# function whose contract is "failures are captured, not raised": the
|
|
131
|
+
# router's `_gather` calls `asyncio.gather` without `return_exceptions`
|
|
132
|
+
# and `run()` catches only BudgetExceeded, so one such agent aborted the
|
|
133
|
+
# whole run with no `run_finished` line — while its sibling discovery
|
|
134
|
+
# agents, which gather does not cancel, kept going and kept spending with
|
|
135
|
+
# nothing recording their cost.
|
|
136
|
+
try:
|
|
137
|
+
options = options or build_options(spec, ctx)
|
|
138
|
+
except Exception as exc: # noqa: BLE001 — same contract as the query below
|
|
139
|
+
error = f"{type(exc).__name__}: {exc}"
|
|
140
|
+
ctx.store.log("agent_error", agent=spec.name, error=error)
|
|
141
|
+
result = AgentResult(agent=spec.name, subtype="failure", error=error)
|
|
142
|
+
ctx.store.put_result(result)
|
|
143
|
+
return RunOutcome(result=result)
|
|
144
|
+
|
|
145
|
+
if max_budget_usd is not None:
|
|
146
|
+
options.max_budget_usd = max_budget_usd
|
|
147
|
+
started = time.monotonic()
|
|
148
|
+
ctx.store.log(
|
|
149
|
+
"agent_started", agent=spec.name, model=spec.model, task_chars=len(task),
|
|
150
|
+
**_record_task(ctx, spec.name, task),
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
before = {e.id for e in ctx.store.envelopes()}
|
|
154
|
+
final_text = ""
|
|
155
|
+
tool_calls = 0
|
|
156
|
+
subtype = "success"
|
|
157
|
+
error: str | None = None
|
|
158
|
+
cost = 0.0
|
|
159
|
+
turns = 0
|
|
160
|
+
|
|
161
|
+
def emit(kind: str, **detail: Any) -> None:
|
|
162
|
+
if on_event:
|
|
163
|
+
on_event(kind, detail)
|
|
164
|
+
|
|
165
|
+
try:
|
|
166
|
+
async for message in query(prompt=task, options=options):
|
|
167
|
+
if isinstance(message, SystemMessage) and message.subtype == "init":
|
|
168
|
+
_check_skills_loaded(spec, ctx, message, emit)
|
|
169
|
+
elif isinstance(message, AssistantMessage):
|
|
170
|
+
for block in message.content:
|
|
171
|
+
if isinstance(block, TextBlock):
|
|
172
|
+
final_text = block.text
|
|
173
|
+
elif isinstance(block, ToolUseBlock):
|
|
174
|
+
tool_calls += 1
|
|
175
|
+
emit("tool", agent=spec.name, tool=block.name)
|
|
176
|
+
elif isinstance(message, ResultMessage):
|
|
177
|
+
subtype = message.subtype or "success"
|
|
178
|
+
cost = message.total_cost_usd or 0.0
|
|
179
|
+
turns = message.num_turns or 0
|
|
180
|
+
if message.is_error:
|
|
181
|
+
error = _error_text(message)
|
|
182
|
+
if isinstance(message.result, str):
|
|
183
|
+
final_text = message.result
|
|
184
|
+
except Exception as exc: # noqa: BLE001 — the router decides what a failure means
|
|
185
|
+
subtype = "failure"
|
|
186
|
+
error = f"{type(exc).__name__}: {exc}"
|
|
187
|
+
ctx.store.log("agent_error", agent=spec.name, error=error)
|
|
188
|
+
|
|
189
|
+
produced = [e.id for e in ctx.store.envelopes() if e.id not in before]
|
|
190
|
+
result = AgentResult(
|
|
191
|
+
agent=spec.name,
|
|
192
|
+
subtype=subtype,
|
|
193
|
+
cost_usd=cost,
|
|
194
|
+
num_turns=turns,
|
|
195
|
+
duration_s=round(time.monotonic() - started, 2),
|
|
196
|
+
envelope_ids=produced,
|
|
197
|
+
error=error,
|
|
198
|
+
)
|
|
199
|
+
ctx.store.put_result(result)
|
|
200
|
+
emit("finished", agent=spec.name, cost=cost, envelopes=len(produced), subtype=subtype)
|
|
201
|
+
return RunOutcome(result=result, final_text=final_text.strip(), tool_calls=tool_calls)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _error_text(message: ResultMessage) -> str:
|
|
205
|
+
"""A usable error string out of whichever field this SDK version populated."""
|
|
206
|
+
for attr in ("errors", "terminal_reason", "stop_reason", "api_error_status"):
|
|
207
|
+
value = getattr(message, attr, None)
|
|
208
|
+
if value:
|
|
209
|
+
return f"{attr}={value}"
|
|
210
|
+
return f"result subtype={message.subtype}"
|
qaas/scorecard.py
ADDED
|
@@ -0,0 +1,448 @@
|
|
|
1
|
+
"""Scoring a run against the golden ledger.
|
|
2
|
+
|
|
3
|
+
This is the module that keeps the project honest. Everything else in the system
|
|
4
|
+
can look like it works — agents run, envelopes appear, tickets get filed — while
|
|
5
|
+
the findings are noise. The scorecard is what says otherwise, in numbers.
|
|
6
|
+
|
|
7
|
+
Matching is deliberately deterministic. Using a model to judge whether a finding
|
|
8
|
+
matches a seeded defect would make the score depend on the same class of system
|
|
9
|
+
being measured, and a generous judge would flatter the result exactly when the
|
|
10
|
+
result least deserves it.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import re
|
|
16
|
+
from dataclasses import dataclass, field
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any, Iterable
|
|
19
|
+
|
|
20
|
+
import yaml
|
|
21
|
+
|
|
22
|
+
from qaas.envelope import DefectEnvelope, Domain, normalize_path, Severity
|
|
23
|
+
|
|
24
|
+
MATCH_THRESHOLD = 0.5
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass(frozen=True)
|
|
28
|
+
class GoldenDefect:
|
|
29
|
+
"""One seeded defect, as recorded in target-app/defects.yaml."""
|
|
30
|
+
|
|
31
|
+
id: str
|
|
32
|
+
domain: str
|
|
33
|
+
defect_class: str
|
|
34
|
+
severity: str
|
|
35
|
+
title: str
|
|
36
|
+
detail: str
|
|
37
|
+
endpoint: str | None = None
|
|
38
|
+
ui_route: str | None = None
|
|
39
|
+
paths: tuple[str, ...] = ()
|
|
40
|
+
keywords: tuple[str, ...] = ()
|
|
41
|
+
phase: int = 1
|
|
42
|
+
security_relevant: bool = False
|
|
43
|
+
# True for defects the system found rather than ones placed for it to find.
|
|
44
|
+
# They still count, but they are weaker evidence: the ledger is a floor on
|
|
45
|
+
# what exists in the app, never a complete oracle.
|
|
46
|
+
discovered_not_seeded: bool = False
|
|
47
|
+
#: The ref a fix for this defect landed on, once one has. Retires the entry
|
|
48
|
+
#: from the recall denominator without deleting it -- REVIEWER escalated a
|
|
49
|
+
#: correct fix because the schema had no way to say this, and CLAUDE.md
|
|
50
|
+
#: requires the ledger to change in the same commit as the defect. Deleting
|
|
51
|
+
#: the entry instead would lose the severity and domain expectations that
|
|
52
|
+
#: make a past score reproducible.
|
|
53
|
+
fixed_in: str | None = None
|
|
54
|
+
|
|
55
|
+
@property
|
|
56
|
+
def retired(self) -> bool:
|
|
57
|
+
"""Fixed, so no longer expected to be present. Not scored for recall."""
|
|
58
|
+
return self.fixed_in is not None
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
@dataclass(frozen=True)
|
|
62
|
+
class PlantedNonDefect:
|
|
63
|
+
"""Correct behaviour that looks wrong. Reporting one is a false positive."""
|
|
64
|
+
|
|
65
|
+
id: str
|
|
66
|
+
title: str
|
|
67
|
+
why_correct: str
|
|
68
|
+
endpoint: str | None = None
|
|
69
|
+
ui_route: str | None = None
|
|
70
|
+
paths: tuple[str, ...] = ()
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass
|
|
74
|
+
class GoldenLedger:
|
|
75
|
+
defects: list[GoldenDefect]
|
|
76
|
+
not_defects: list[PlantedNonDefect]
|
|
77
|
+
|
|
78
|
+
@classmethod
|
|
79
|
+
def load(cls, path: Path | str) -> "GoldenLedger":
|
|
80
|
+
raw = yaml.safe_load(Path(path).read_text(encoding="utf-8"))
|
|
81
|
+
return cls(
|
|
82
|
+
defects=[_golden(d) for d in raw.get("defects", [])],
|
|
83
|
+
not_defects=[_planted(d) for d in raw.get("not_defects", [])],
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
def for_phase(self, phase: int) -> list[GoldenDefect]:
|
|
87
|
+
return [d for d in self.defects if d.phase <= phase]
|
|
88
|
+
|
|
89
|
+
def by_id(self, defect_id: str) -> GoldenDefect | None:
|
|
90
|
+
return next((d for d in self.defects if d.id == defect_id), None)
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _enum_or_die(kind: type[Domain] | type[Severity], value: Any, defect_id: str, field: str) -> str:
|
|
94
|
+
"""A ledger value that must name an enum member, checked when it is read.
|
|
95
|
+
|
|
96
|
+
`domain` and `severity` arrived as free strings and were compared against
|
|
97
|
+
validated enums downstream, so the two failure modes were both silent and
|
|
98
|
+
both late. A `domain: ui` (not a `Domain`) scores 0.0 similarity against
|
|
99
|
+
every envelope forever: the defect is permanently `missed`, recall drops and
|
|
100
|
+
nothing says why. A `severity: high` (not a `Severity`) raises ValueError out
|
|
101
|
+
of `Match.severity_delta` — *after* a paid run has finished.
|
|
102
|
+
|
|
103
|
+
"A stale ledger silently corrupts every score" is the reason the ledger is a
|
|
104
|
+
human's responsibility at merge. This makes the failure loud and immediate
|
|
105
|
+
instead, which is the only part of that a program can help with.
|
|
106
|
+
"""
|
|
107
|
+
try:
|
|
108
|
+
return kind(value).value
|
|
109
|
+
except ValueError:
|
|
110
|
+
allowed = ", ".join(sorted(m.value for m in kind))
|
|
111
|
+
raise ValueError(
|
|
112
|
+
f"{defect_id}: {field} '{value}' is not a {kind.__name__}. Use one of: {allowed}."
|
|
113
|
+
) from None
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _golden(d: dict[str, Any]) -> GoldenDefect:
|
|
117
|
+
loc = d.get("location", {}) or {}
|
|
118
|
+
return GoldenDefect(
|
|
119
|
+
id=d["id"],
|
|
120
|
+
domain=_enum_or_die(Domain, d["domain"], d["id"], "domain"),
|
|
121
|
+
defect_class=d.get("class", "bug"),
|
|
122
|
+
severity=_enum_or_die(Severity, d["severity"], d["id"], "severity"),
|
|
123
|
+
title=d["title"],
|
|
124
|
+
detail=d.get("detail", ""),
|
|
125
|
+
endpoint=loc.get("endpoint"),
|
|
126
|
+
ui_route=loc.get("ui_route"),
|
|
127
|
+
paths=tuple(loc.get("paths", ())),
|
|
128
|
+
keywords=tuple(d.get("keywords", ())),
|
|
129
|
+
phase=int(d.get("phase", 1)),
|
|
130
|
+
security_relevant=bool(d.get("security_relevant", False)),
|
|
131
|
+
discovered_not_seeded=bool(d.get("discovered_not_seeded", False)),
|
|
132
|
+
fixed_in=(str(d["fixed_in"]) if d.get("fixed_in") else None),
|
|
133
|
+
)
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def _planted(d: dict[str, Any]) -> PlantedNonDefect:
|
|
137
|
+
loc = d.get("location", {}) or {}
|
|
138
|
+
return PlantedNonDefect(
|
|
139
|
+
id=d["id"],
|
|
140
|
+
title=d["title"],
|
|
141
|
+
why_correct=d.get("why_correct", ""),
|
|
142
|
+
endpoint=loc.get("endpoint"),
|
|
143
|
+
ui_route=loc.get("ui_route"),
|
|
144
|
+
paths=tuple(loc.get("paths", ())),
|
|
145
|
+
)
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
# -- similarity -------------------------------------------------------------
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _norm_endpoint(endpoint: str | None) -> str | None:
|
|
152
|
+
"""`GET /v1/orders/{order_id}` and `get /v1/orders/{id}` are the same endpoint."""
|
|
153
|
+
if not endpoint:
|
|
154
|
+
return None
|
|
155
|
+
e = re.sub(r"\{[^}]*\}", "{}", endpoint.strip().lower())
|
|
156
|
+
return re.sub(r"\s+", " ", e)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _norm_path(path: str) -> str:
|
|
160
|
+
"""Compare by tail, so `target-app/api/app/routes/orders.py:88` matches `api/app/routes/orders.py`.
|
|
161
|
+
|
|
162
|
+
Delegates to the envelope's own normalisation rather than restating it. The
|
|
163
|
+
two were separate implementations and drifted: the scorer handled `:104-112`
|
|
164
|
+
line ranges and the repo-root prefix, the fingerprint handled neither, so a
|
|
165
|
+
defect the scorer counted as one thing hashed as three.
|
|
166
|
+
"""
|
|
167
|
+
return normalize_path(path)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _path_overlap(a: Iterable[str], b: Iterable[str]) -> float:
|
|
171
|
+
"""Fraction of the golden defect's files the report also names."""
|
|
172
|
+
golden = {_norm_path(p) for p in b}
|
|
173
|
+
if not golden:
|
|
174
|
+
return 0.0
|
|
175
|
+
reported = {_norm_path(p) for p in a}
|
|
176
|
+
hits = sum(
|
|
177
|
+
1
|
|
178
|
+
for g in golden
|
|
179
|
+
if any(r == g or r.endswith("/" + g) or g.endswith("/" + r) for r in reported)
|
|
180
|
+
)
|
|
181
|
+
return hits / len(golden)
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def _keyword_overlap(text: str, keywords: Iterable[str]) -> float:
|
|
185
|
+
terms = list(keywords)
|
|
186
|
+
if not terms:
|
|
187
|
+
return 0.0
|
|
188
|
+
blob = text.lower()
|
|
189
|
+
return sum(1 for t in terms if t.lower() in blob) / len(terms)
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
# Domains that describe the same surface. An agent choosing either one has
|
|
193
|
+
# classified the defect defensibly, so scoring must accept both.
|
|
194
|
+
#
|
|
195
|
+
# Both entries were learned from real runs, and both times the scorer was wrong
|
|
196
|
+
# rather than the agent: API filed a cross-tenant read under `security`, and
|
|
197
|
+
# BROWSER filed a missing label and a contrast failure under `ux`. Marking those
|
|
198
|
+
# as misses would have hidden a perfect discovery run behind a 50% score.
|
|
199
|
+
_EQUIVALENT_DOMAINS: dict[str, set[str]] = {
|
|
200
|
+
"ux": {"frontend"},
|
|
201
|
+
"frontend": {"ux"},
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _domains_compatible(reported: str, golden: GoldenDefect) -> bool:
|
|
206
|
+
"""Whether a reported domain is an acceptable classification of this defect.
|
|
207
|
+
|
|
208
|
+
Domain stays a gate — naming the right file under a genuinely wrong domain
|
|
209
|
+
means the defect was misunderstood — but the gate accepts any defensible
|
|
210
|
+
reading, not only the one the ledger happened to write down.
|
|
211
|
+
"""
|
|
212
|
+
if reported == golden.domain:
|
|
213
|
+
return True
|
|
214
|
+
if reported == "security" and golden.security_relevant:
|
|
215
|
+
return True
|
|
216
|
+
return golden.domain in _EQUIVALENT_DOMAINS.get(reported, set())
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def similarity(env: DefectEnvelope, golden: GoldenDefect) -> float:
|
|
220
|
+
"""0..1 confidence that `env` reports `golden`."""
|
|
221
|
+
if not _domains_compatible(env.domain.value, golden):
|
|
222
|
+
return 0.0
|
|
223
|
+
|
|
224
|
+
text = f"{env.title} {env.summary} {env.suggested_fix_area}"
|
|
225
|
+
|
|
226
|
+
endpoint_match = (
|
|
227
|
+
_norm_endpoint(env.location.endpoint) is not None
|
|
228
|
+
and _norm_endpoint(env.location.endpoint) == _norm_endpoint(golden.endpoint)
|
|
229
|
+
)
|
|
230
|
+
route_match = (
|
|
231
|
+
env.location.ui_route is not None
|
|
232
|
+
and golden.ui_route is not None
|
|
233
|
+
and env.location.ui_route.rstrip("/") == golden.ui_route.rstrip("/")
|
|
234
|
+
)
|
|
235
|
+
|
|
236
|
+
# A cross-cutting defect has no single endpoint or route to anchor on, and
|
|
237
|
+
# the files that best demonstrate it are a judgement call — a report of
|
|
238
|
+
# inconsistent error shapes may cite whichever two handlers differ. For
|
|
239
|
+
# those, the prose has to carry the identification.
|
|
240
|
+
anchored = bool(golden.endpoint or golden.ui_route)
|
|
241
|
+
weights = (0.45, 0.30, 0.35) if anchored else (0.0, 0.35, 0.65)
|
|
242
|
+
w_location, w_paths, w_keywords = weights
|
|
243
|
+
|
|
244
|
+
score = w_location if (endpoint_match or route_match) else 0.0
|
|
245
|
+
score += w_paths * _path_overlap(env.location.paths, golden.paths)
|
|
246
|
+
score += w_keywords * _keyword_overlap(text, golden.keywords)
|
|
247
|
+
return min(score, 1.0)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
def resembles_planted(env: DefectEnvelope, planted: PlantedNonDefect) -> float:
|
|
251
|
+
"""Whether a finding is a report of deliberately-correct behaviour.
|
|
252
|
+
|
|
253
|
+
This needs a real anchor — the same endpoint or the same route. Sharing a
|
|
254
|
+
file is not enough: `orders.py` holds six seeded defects as well as the
|
|
255
|
+
deliberately-correct legacy handler, so a path-only rule blamed agents for
|
|
256
|
+
reporting the legacy endpoint when they had reported something else entirely.
|
|
257
|
+
"""
|
|
258
|
+
if planted.endpoint and _norm_endpoint(env.location.endpoint) == _norm_endpoint(planted.endpoint):
|
|
259
|
+
return 1.0
|
|
260
|
+
if planted.ui_route and env.location.ui_route == planted.ui_route:
|
|
261
|
+
return 1.0
|
|
262
|
+
return 0.0
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
# -- results ----------------------------------------------------------------
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
@dataclass
|
|
269
|
+
class Match:
|
|
270
|
+
golden_id: str
|
|
271
|
+
envelope_id: str
|
|
272
|
+
score: float
|
|
273
|
+
reported_severity: str
|
|
274
|
+
expected_severity: str
|
|
275
|
+
|
|
276
|
+
@property
|
|
277
|
+
def severity_delta(self) -> int:
|
|
278
|
+
"""Ranks apart. 0 is agreement, positive means the report was too calm."""
|
|
279
|
+
return Severity(self.reported_severity).rank - Severity(self.expected_severity).rank
|
|
280
|
+
|
|
281
|
+
|
|
282
|
+
@dataclass
|
|
283
|
+
class Scorecard:
|
|
284
|
+
matches: list[Match] = field(default_factory=list)
|
|
285
|
+
missed: list[str] = field(default_factory=list)
|
|
286
|
+
false_positives: list[str] = field(default_factory=list)
|
|
287
|
+
duplicates: list[str] = field(default_factory=list)
|
|
288
|
+
regressions_on_planted: list[tuple[str, str]] = field(default_factory=list)
|
|
289
|
+
#: Findings that matched a defect already marked `fixed_in`. Neither a find
|
|
290
|
+
#: nor a false positive: the report is correct wherever the fix has not
|
|
291
|
+
#: landed, so scoring it either way would be a lie about the run.
|
|
292
|
+
retired_hits: list[tuple[str, str]] = field(default_factory=list)
|
|
293
|
+
total_golden: int = 0
|
|
294
|
+
total_envelopes: int = 0
|
|
295
|
+
cost_usd: float = 0.0
|
|
296
|
+
|
|
297
|
+
@property
|
|
298
|
+
def recall(self) -> float:
|
|
299
|
+
return len(self.matches) / self.total_golden if self.total_golden else 0.0
|
|
300
|
+
|
|
301
|
+
@property
|
|
302
|
+
def precision(self) -> float:
|
|
303
|
+
judged = len(self.matches) + len(self.false_positives)
|
|
304
|
+
return len(self.matches) / judged if judged else 0.0
|
|
305
|
+
|
|
306
|
+
@property
|
|
307
|
+
def false_positive_rate(self) -> float:
|
|
308
|
+
return len(self.false_positives) / self.total_envelopes if self.total_envelopes else 0.0
|
|
309
|
+
|
|
310
|
+
@property
|
|
311
|
+
def duplicate_rate(self) -> float:
|
|
312
|
+
return len(self.duplicates) / self.total_envelopes if self.total_envelopes else 0.0
|
|
313
|
+
|
|
314
|
+
@property
|
|
315
|
+
def severity_agreement(self) -> float:
|
|
316
|
+
"""Share of matches scored within one rank of the expected severity."""
|
|
317
|
+
if not self.matches:
|
|
318
|
+
return 0.0
|
|
319
|
+
return sum(1 for m in self.matches if abs(m.severity_delta) <= 1) / len(self.matches)
|
|
320
|
+
|
|
321
|
+
@property
|
|
322
|
+
def cost_per_accepted(self) -> float | None:
|
|
323
|
+
return self.cost_usd / len(self.matches) if self.matches else None
|
|
324
|
+
|
|
325
|
+
def summary(self) -> dict[str, Any]:
|
|
326
|
+
return {
|
|
327
|
+
"found": len(self.matches),
|
|
328
|
+
"of": self.total_golden,
|
|
329
|
+
"recall": round(self.recall, 3),
|
|
330
|
+
"precision": round(self.precision, 3),
|
|
331
|
+
"false_positives": len(self.false_positives),
|
|
332
|
+
"false_positive_rate": round(self.false_positive_rate, 3),
|
|
333
|
+
"duplicates": len(self.duplicates),
|
|
334
|
+
"duplicate_rate": round(self.duplicate_rate, 3),
|
|
335
|
+
"severity_agreement": round(self.severity_agreement, 3),
|
|
336
|
+
"planted_misreported": len(self.regressions_on_planted),
|
|
337
|
+
"retired_hits": len(self.retired_hits),
|
|
338
|
+
"cost_usd": round(self.cost_usd, 4),
|
|
339
|
+
"cost_per_accepted": (
|
|
340
|
+
round(self.cost_per_accepted, 4) if self.cost_per_accepted is not None else None
|
|
341
|
+
),
|
|
342
|
+
"missed": sorted(self.missed),
|
|
343
|
+
}
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def score(
|
|
347
|
+
envelopes: list[DefectEnvelope],
|
|
348
|
+
ledger: GoldenLedger,
|
|
349
|
+
*,
|
|
350
|
+
phase: int = 1,
|
|
351
|
+
domains: set[str] | None = None,
|
|
352
|
+
cost_usd: float = 0.0,
|
|
353
|
+
threshold: float = MATCH_THRESHOLD,
|
|
354
|
+
) -> Scorecard:
|
|
355
|
+
"""Match findings to seeded defects, best pair first.
|
|
356
|
+
|
|
357
|
+
Assignment is one-to-one and greedy on the strongest pair remaining. A second
|
|
358
|
+
envelope for an already-matched defect is a duplicate, not a second find —
|
|
359
|
+
counting it as a find would reward exactly the ticket-spam this system is
|
|
360
|
+
built to avoid.
|
|
361
|
+
"""
|
|
362
|
+
in_scope = [d for d in ledger.for_phase(phase) if domains is None or d.domain in domains]
|
|
363
|
+
# A defect marked `fixed_in` is still matched -- so a report of it is not
|
|
364
|
+
# written off as a false positive -- but it leaves the recall denominator.
|
|
365
|
+
# Counting a repaired defect as a miss on every future run is exactly the
|
|
366
|
+
# silent corruption CLAUDE.md warns about.
|
|
367
|
+
golden = [d for d in in_scope if not d.retired]
|
|
368
|
+
retired = {d.id for d in in_scope if d.retired}
|
|
369
|
+
matchable = in_scope
|
|
370
|
+
card = Scorecard(total_golden=len(golden), total_envelopes=len(envelopes), cost_usd=cost_usd)
|
|
371
|
+
|
|
372
|
+
pairs = sorted(
|
|
373
|
+
(
|
|
374
|
+
(similarity(env, g), env, g)
|
|
375
|
+
for env in envelopes
|
|
376
|
+
for g in matchable
|
|
377
|
+
if similarity(env, g) >= threshold
|
|
378
|
+
),
|
|
379
|
+
key=lambda t: -t[0],
|
|
380
|
+
)
|
|
381
|
+
|
|
382
|
+
claimed_golden: set[str] = set()
|
|
383
|
+
claimed_env: set[str] = set()
|
|
384
|
+
contested: list[tuple[float, DefectEnvelope, GoldenDefect]] = []
|
|
385
|
+
|
|
386
|
+
# First pass: settle the unambiguous pairs, strongest first.
|
|
387
|
+
for sim, env, g in pairs:
|
|
388
|
+
if g.id in claimed_golden or env.id in claimed_env:
|
|
389
|
+
contested.append((sim, env, g))
|
|
390
|
+
continue
|
|
391
|
+
claimed_golden.add(g.id)
|
|
392
|
+
claimed_env.add(env.id)
|
|
393
|
+
if g.id in retired:
|
|
394
|
+
card.retired_hits.append((env.id, g.id))
|
|
395
|
+
else:
|
|
396
|
+
card.matches.append(
|
|
397
|
+
Match(
|
|
398
|
+
golden_id=g.id,
|
|
399
|
+
envelope_id=env.id,
|
|
400
|
+
score=round(sim, 3),
|
|
401
|
+
reported_severity=env.severity.value,
|
|
402
|
+
expected_severity=g.severity,
|
|
403
|
+
)
|
|
404
|
+
)
|
|
405
|
+
|
|
406
|
+
# Second pass: an envelope whose best match was taken gets its next-best
|
|
407
|
+
# before anything else. Branding it a duplicate here was a real bug — two
|
|
408
|
+
# defects in one file on one route (a missing empty state and an unhandled
|
|
409
|
+
# rejection, both on /orders in OrdersList.tsx) each match the other's
|
|
410
|
+
# golden entry, so whichever lost the first pass was written off entirely.
|
|
411
|
+
# Only an envelope with no unclaimed match left is genuinely a duplicate.
|
|
412
|
+
for sim, env, g in contested:
|
|
413
|
+
if env.id in claimed_env:
|
|
414
|
+
continue
|
|
415
|
+
if g.id not in claimed_golden:
|
|
416
|
+
claimed_golden.add(g.id)
|
|
417
|
+
claimed_env.add(env.id)
|
|
418
|
+
if g.id in retired:
|
|
419
|
+
card.retired_hits.append((env.id, g.id))
|
|
420
|
+
else:
|
|
421
|
+
card.matches.append(
|
|
422
|
+
Match(
|
|
423
|
+
golden_id=g.id,
|
|
424
|
+
envelope_id=env.id,
|
|
425
|
+
score=round(sim, 3),
|
|
426
|
+
reported_severity=env.severity.value,
|
|
427
|
+
expected_severity=g.severity,
|
|
428
|
+
)
|
|
429
|
+
)
|
|
430
|
+
|
|
431
|
+
for sim, env, g in contested:
|
|
432
|
+
if env.id not in claimed_env:
|
|
433
|
+
card.duplicates.append(env.id)
|
|
434
|
+
claimed_env.add(env.id)
|
|
435
|
+
|
|
436
|
+
card.missed = [g.id for g in golden if g.id not in claimed_golden]
|
|
437
|
+
|
|
438
|
+
for env in envelopes:
|
|
439
|
+
if env.id in claimed_env:
|
|
440
|
+
continue
|
|
441
|
+
card.false_positives.append(env.id)
|
|
442
|
+
for planted in ledger.not_defects:
|
|
443
|
+
if resembles_planted(env, planted) >= MATCH_THRESHOLD:
|
|
444
|
+
card.regressions_on_planted.append((env.id, planted.id))
|
|
445
|
+
break
|
|
446
|
+
|
|
447
|
+
card.matches.sort(key=lambda m: m.golden_id)
|
|
448
|
+
return card
|