qaas-python 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (96) hide show
  1. qaas/adapters/__init__.py +19 -0
  2. qaas/adapters/tracker.py +1783 -0
  3. qaas/adapters/vcs.py +555 -0
  4. qaas/cli.py +1757 -0
  5. qaas/config.py +409 -0
  6. qaas/defaults/config/agents/api.yaml +18 -0
  7. qaas/defaults/config/agents/architect.yaml +21 -0
  8. qaas/defaults/config/agents/auditor.yaml +19 -0
  9. qaas/defaults/config/agents/browser.yaml +15 -0
  10. qaas/defaults/config/agents/dba.yaml +20 -0
  11. qaas/defaults/config/agents/fixer.yaml +55 -0
  12. qaas/defaults/config/agents/guide.yaml +23 -0
  13. qaas/defaults/config/agents/load.yaml +26 -0
  14. qaas/defaults/config/agents/mapper.yaml +19 -0
  15. qaas/defaults/config/agents/reporter.yaml +19 -0
  16. qaas/defaults/config/agents/reproducer.yaml +21 -0
  17. qaas/defaults/config/agents/reviewer.yaml +18 -0
  18. qaas/defaults/config/agents/socket.yaml +23 -0
  19. qaas/defaults/config/agents/triage.yaml +20 -0
  20. qaas/defaults/config/agents/verifier.yaml +20 -0
  21. qaas/defaults/config/system.yaml +64 -0
  22. qaas/discover.py +242 -0
  23. qaas/envelope.py +318 -0
  24. qaas/envfile.py +100 -0
  25. qaas/guardrails.py +589 -0
  26. qaas/mcp/__init__.py +0 -0
  27. qaas/mcp/context.py +78 -0
  28. qaas/mcp/contract_diff.py +1011 -0
  29. qaas/mcp/defect_memory.py +495 -0
  30. qaas/mcp/env_control.py +925 -0
  31. qaas/mcp/envelope_server.py +463 -0
  32. qaas/mcp/test_runner.py +842 -0
  33. qaas/mcp/tracker.py +420 -0
  34. qaas/mcp/vcs.py +501 -0
  35. qaas/paths.py +317 -0
  36. qaas/plugin/.claude-plugin/plugin.json +9 -0
  37. qaas/plugin/skills/a11y-audit/SKILL.md +34 -0
  38. qaas/plugin/skills/adversarial-review/SKILL.md +120 -0
  39. qaas/plugin/skills/api-surface-extraction/SKILL.md +38 -0
  40. qaas/plugin/skills/authz-matrix-check/SKILL.md +46 -0
  41. qaas/plugin/skills/console-error-triage/SKILL.md +39 -0
  42. qaas/plugin/skills/contract-test-generation/SKILL.md +36 -0
  43. qaas/plugin/skills/dedupe-strategy/SKILL.md +39 -0
  44. qaas/plugin/skills/environment-pinning/SKILL.md +35 -0
  45. qaas/plugin/skills/error-taxonomy/SKILL.md +42 -0
  46. qaas/plugin/skills/exploratory-ui-walk/SKILL.md +46 -0
  47. qaas/plugin/skills/failing-test-authoring/SKILL.md +47 -0
  48. qaas/plugin/skills/flake-detection/SKILL.md +39 -0
  49. qaas/plugin/skills/form-state-probe/SKILL.md +36 -0
  50. qaas/plugin/skills/minimal-diff-discipline/SKILL.md +70 -0
  51. qaas/plugin/skills/openapi-diff/SKILL.md +45 -0
  52. qaas/plugin/skills/ownership-resolution/SKILL.md +31 -0
  53. qaas/plugin/skills/product-task-graph/SKILL.md +35 -0
  54. qaas/plugin/skills/regression-risk-scoring/SKILL.md +59 -0
  55. qaas/plugin/skills/regression-suite-selection/SKILL.md +36 -0
  56. qaas/plugin/skills/repo-cartography/SKILL.md +38 -0
  57. qaas/plugin/skills/repro-minimisation/SKILL.md +41 -0
  58. qaas/plugin/skills/rollback-plan-authoring/SKILL.md +81 -0
  59. qaas/plugin/skills/root-cause-vs-symptom/SKILL.md +67 -0
  60. qaas/plugin/skills/routing-rules/SKILL.md +34 -0
  61. qaas/plugin/skills/severity-rubric/SKILL.md +42 -0
  62. qaas/plugin/skills/test-first-fix/SKILL.md +66 -0
  63. qaas/plugin/skills/test-quality-audit/SKILL.md +58 -0
  64. qaas/plugin/skills/ticket-writer/SKILL.md +40 -0
  65. qaas/plugin/skills/verdict-reporting/SKILL.md +35 -0
  66. qaas/plugin/skills/verification-protocol/SKILL.md +39 -0
  67. qaas/prompts/API.md +44 -0
  68. qaas/prompts/ARCHITECT.md +80 -0
  69. qaas/prompts/AUDITOR.md +62 -0
  70. qaas/prompts/BROWSER.md +46 -0
  71. qaas/prompts/DBA.md +59 -0
  72. qaas/prompts/FIXER.md +55 -0
  73. qaas/prompts/GUIDE.md +94 -0
  74. qaas/prompts/LOAD.md +109 -0
  75. qaas/prompts/MAPPER.md +46 -0
  76. qaas/prompts/REPORTER.md +61 -0
  77. qaas/prompts/REPRODUCER.md +43 -0
  78. qaas/prompts/REVIEWER.md +53 -0
  79. qaas/prompts/SOCKET.md +100 -0
  80. qaas/prompts/TRIAGE.md +45 -0
  81. qaas/prompts/VERIFIER.md +41 -0
  82. qaas/prompts/_shared.md +45 -0
  83. qaas/registry.py +496 -0
  84. qaas/router.py +581 -0
  85. qaas/runner.py +210 -0
  86. qaas/scorecard.py +448 -0
  87. qaas/sdk_compat.py +52 -0
  88. qaas/store.py +323 -0
  89. qaas/target.py +287 -0
  90. qaas/tasks.py +438 -0
  91. qaas/trace.py +342 -0
  92. qaas_python-0.0.1.dist-info/METADATA +429 -0
  93. qaas_python-0.0.1.dist-info/RECORD +96 -0
  94. qaas_python-0.0.1.dist-info/WHEEL +4 -0
  95. qaas_python-0.0.1.dist-info/entry_points.txt +2 -0
  96. qaas_python-0.0.1.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,1011 @@
1
+ """The `contract_diff` MCP server — what changed for the people calling you.
2
+
3
+ A text diff of two OpenAPI documents tells an agent that lines moved. It does
4
+ not tell it that `currency` vanished from the Invoice response and every
5
+ consumer reading that field now gets a KeyError. This server answers the second
6
+ question: it walks both documents structurally, resolves `$ref`s, and reports
7
+ changes as consumer-visible facts with a breaking/non-breaking verdict attached.
8
+
9
+ Two things follow from API's brief (§4.5):
10
+
11
+ * **The declared contract is the reference.** `spec_a` defaults to
12
+ the target's declared spec and `spec_b` to the running app's `/openapi.json`,
13
+ so "drift" here means the implementation disagrees with the published spec —
14
+ which is the defect, not the other way round.
15
+ * **A finding ships with a failing test.** `generate_contract_test` emits a
16
+ standalone pytest module that asserts the spec's promises against a live
17
+ server. That file is the evidence an envelope cites; it must fail on the
18
+ violating implementation and pass on a conforming one, or it is worthless.
19
+
20
+ Classification follows one fixed rule set, applied identically by `diff_openapi`
21
+ and `classify_breaking` so the two can never disagree: losing a guarantee is
22
+ breaking (a removed endpoint or field, a dropped required-ness, a narrowed type,
23
+ a newly required request input, a changed status code); gaining an optional one
24
+ is not.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import json
30
+ import os
31
+ import re
32
+ import urllib.error
33
+ import urllib.parse
34
+ import urllib.request
35
+ from pathlib import Path
36
+ from typing import Any, Iterable
37
+
38
+ import yaml
39
+ from claude_agent_sdk import create_sdk_mcp_server, tool
40
+
41
+ from qaas.mcp.context import ToolContext, err, ok
42
+
43
+ DEFAULT_SPEC_FILE = "openapi.yaml"
44
+ DEFAULT_LIVE_SPEC_URL = "http://localhost:8000/openapi.json"
45
+ FETCH_TIMEOUT_S = 10
46
+ GENERATED_DIR = "generated"
47
+
48
+ HTTP_METHODS = ("get", "put", "post", "delete", "patch", "options", "head", "trace")
49
+
50
+ #: Source extensions worth grepping for call sites. Everything else is noise.
51
+ CONSUMER_SUFFIXES = frozenset(
52
+ {".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".vue", ".svelte",
53
+ ".py", ".go", ".rb", ".java", ".kt", ".rs", ".php", ".cs", ".swift"}
54
+ )
55
+ SKIP_DIRS = frozenset({".git", ".venv", "node_modules", "__pycache__", "dist", "build", ".qaas", ".pytest_cache", ".mypy_cache"})
56
+ MAX_CONSUMER_HITS = 200
57
+
58
+ #: kind -> (verdict, why). The single source of truth for breaking-ness: every
59
+ #: change `diff_openapi` emits carries a kind from this table, and
60
+ #: `classify_breaking` is a lookup into it. One table, one answer.
61
+ RULES: dict[str, tuple[str, str]] = {
62
+ "endpoint_removed": ("breaking", "Consumers calling this endpoint now get a 404."),
63
+ "endpoint_added": ("non_breaking", "New surface; no existing caller is affected."),
64
+ "status_code_removed": ("breaking", "A documented outcome disappeared; consumers branching on it are wrong."),
65
+ "status_code_added": ("non_breaking", "An additional documented outcome; existing handling still applies."),
66
+ "response_field_removed": ("breaking", "Consumers reading this field get nothing back."),
67
+ "response_field_renamed": ("breaking", "A rename is a removal and an addition; every reader of the old name breaks."),
68
+ "response_field_added": ("non_breaking", "An added optional response field is ignored by existing consumers."),
69
+ "response_required_dropped": ("breaking", "The field was guaranteed present and no longer is."),
70
+ "response_required_added": ("non_breaking", "A field that was optional is now always present; that only helps."),
71
+ "response_type_narrowed": ("breaking", "The value set shrank; consumers may receive nothing they can use."),
72
+ "response_type_widened": ("non_breaking", "The declared value set grew; previously valid values still arrive."),
73
+ "response_enum_narrowed": ("breaking", "Documented values were withdrawn."),
74
+ "response_enum_widened": ("non_breaking", "Additional documented values. Consumers with exhaustive switches should still be told."),
75
+ "request_field_removed": ("breaking", "Requests that carried this field may now be rejected."),
76
+ "request_field_added_required": ("breaking", "Existing requests omit it and will now fail validation."),
77
+ "request_field_added_optional": ("non_breaking", "Existing requests remain valid."),
78
+ "request_required_added": ("breaking", "A previously optional input is now mandatory."),
79
+ "request_required_dropped": ("non_breaking", "Fewer inputs are mandatory; existing requests still validate."),
80
+ "request_type_narrowed": ("breaking", "Values that used to validate no longer do."),
81
+ "request_type_widened": ("non_breaking", "More values validate than before."),
82
+ "parameter_removed": ("breaking", "Callers passing this parameter silently lose the behaviour it controlled."),
83
+ "parameter_added_required": ("breaking", "Existing callers omit it and will now fail."),
84
+ "parameter_added_optional": ("non_breaking", "Existing callers are unaffected."),
85
+ "parameter_required_added": ("breaking", "A previously optional parameter is now mandatory."),
86
+ "parameter_type_narrowed": ("breaking", "Values callers already send may now be rejected."),
87
+ "parameter_type_widened": ("non_breaking", "More values are accepted than before."),
88
+ "security_added": ("breaking", "An endpoint that accepted anonymous calls now requires credentials."),
89
+ "security_removed": ("non_breaking", "Compatibility is unaffected, but dropping auth is a security finding in its own right."),
90
+ }
91
+
92
+
93
+ # --------------------------------------------------------------------------
94
+ # spec loading
95
+ # --------------------------------------------------------------------------
96
+
97
+
98
+ def _is_url(value: str) -> bool:
99
+ return value.startswith(("http://", "https://"))
100
+
101
+
102
+ def _fetch_spec(url: str) -> tuple[dict[str, Any] | None, str | None]:
103
+ try:
104
+ with urllib.request.urlopen(url, timeout=FETCH_TIMEOUT_S) as resp: # noqa: S310 - http(s) only, checked by caller
105
+ body = resp.read().decode(errors="replace")
106
+ except urllib.error.HTTPError as exc:
107
+ return None, f"HTTP {exc.code} from {url}"
108
+ except (urllib.error.URLError, OSError, TimeoutError, ValueError) as exc:
109
+ return None, f"could not reach {url} ({exc})"
110
+ try:
111
+ doc = json.loads(body)
112
+ except json.JSONDecodeError:
113
+ try:
114
+ doc = yaml.safe_load(body)
115
+ except yaml.YAMLError as exc:
116
+ return None, f"{url} returned something that is neither JSON nor YAML ({exc})"
117
+ return (doc, None) if isinstance(doc, dict) else (None, f"{url} did not return an object")
118
+
119
+
120
+ def _login_path(ctx: ToolContext) -> str:
121
+ """The login path this target declares, defaulting to a conventional one.
122
+
123
+ `env_control` was repaired for this exact class of bug and records why in
124
+ its own comment: `profile.auth` has existed since the schema was written and
125
+ nothing read it, so impersonation was welded to demo accounts and a literal
126
+ password. These were the last such literals in `src/` — a generated contract
127
+ test hard-coded `admin@northwind.test` / `password123` and POSTed them at
128
+ `/v1/auth/login`.
129
+
130
+ Against any target but the bundled demo, that made every generated test fail
131
+ at its `token` fixture, and the envelope then cited a "failing contract test"
132
+ that was really a login failure. Fabricated evidence is what `qaas score`'s
133
+ precision metric exists to catch, and it would not have caught this.
134
+ """
135
+ profile = getattr(ctx.config, "profile", None)
136
+ auth = getattr(profile, "auth", None) if profile else None
137
+ declared = getattr(auth, "login_endpoint", None) if auth else None
138
+ if not declared:
139
+ return "/v1/auth/login"
140
+ path = declared.split(None, 1)[-1].strip() if " " in declared else declared.strip()
141
+ return path if path.startswith("/") else f"/{path}"
142
+
143
+
144
+ def _spec_origin() -> str:
145
+ """The one origin a spec may be fetched from: the app under test."""
146
+ base = os.environ.get("QAAS_TARGET_BASE_URL") or DEFAULT_LIVE_SPEC_URL
147
+ parts = urllib.parse.urlsplit(base)
148
+ return f"{parts.scheme}://{parts.netloc}"
149
+
150
+
151
+ def _read_spec(ref: str, target_root: Path) -> tuple[dict[str, Any] | None, str | None]:
152
+ """Load a spec from the target's URL or a path inside the checkout.
153
+
154
+ Both halves were open. A `spec` argument is agent-supplied, and it went
155
+ either to `urlopen` for any http(s) URL the agent named, or to `read_text`
156
+ for any path — absolute or `../`-relative — whose contents then came back in
157
+ the tool result. `find_consumers` had the containment check these two did
158
+ not, which is the usual shape of this bug: the rule exists one function over.
159
+
160
+ The URL half is narrowed to the target app's own origin rather than dropped,
161
+ because comparing the committed spec against the running one is what the
162
+ tool is for. Fetching anything else is network research, which `guardrails`
163
+ already refuses outright (`WebFetch`/`WebSearch`, "findings come from the
164
+ code and the running app, not the web") — this makes the two agree.
165
+ """
166
+ if _is_url(ref):
167
+ allowed = _spec_origin()
168
+ parts = urllib.parse.urlsplit(ref)
169
+ if f"{parts.scheme}://{parts.netloc}" != allowed:
170
+ return None, (
171
+ f"refusing to fetch {ref}: a spec may only be read from the application "
172
+ f"under test ({allowed}) or from a file inside the checkout. Findings "
173
+ "come from the code and the running app, not the web."
174
+ )
175
+ return _fetch_spec(ref)
176
+ root = target_root.resolve()
177
+ candidate = Path(ref)
178
+ path = (candidate if candidate.is_absolute() else root / candidate).resolve()
179
+ if path != root and not path.is_relative_to(root):
180
+ return None, (
181
+ f"spec '{ref}' resolves to {path}, outside the repository ({root}). "
182
+ "Specs are read from inside the checkout."
183
+ )
184
+ if not path.exists():
185
+ return None, f"no such spec file: {path}"
186
+ try:
187
+ doc = yaml.safe_load(path.read_text(encoding="utf-8"))
188
+ except (OSError, yaml.YAMLError) as exc:
189
+ return None, f"could not parse {path}: {exc}"
190
+ return (doc, None) if isinstance(doc, dict) else (None, f"{path} does not contain an OpenAPI object")
191
+
192
+
193
+ # --------------------------------------------------------------------------
194
+ # schema walking
195
+ # --------------------------------------------------------------------------
196
+
197
+
198
+ def _resolve(node: Any, doc: dict[str, Any], seen: frozenset[str] = frozenset()) -> tuple[Any, frozenset[str]]:
199
+ """Follow local `$ref`s. Cycles stop at the second visit rather than recurse."""
200
+ guard = 0
201
+ while isinstance(node, dict) and "$ref" in node and guard < 20:
202
+ ref = str(node["$ref"])
203
+ if not ref.startswith("#/") or ref in seen:
204
+ return {}, seen
205
+ seen = seen | {ref}
206
+ target: Any = doc
207
+ for part in ref[2:].split("/"):
208
+ part = part.replace("~1", "/").replace("~0", "~")
209
+ if not isinstance(target, dict) or part not in target:
210
+ return {}, seen
211
+ target = target[part]
212
+ node = target
213
+ guard += 1
214
+ return node, seen
215
+
216
+
217
+ def _merged(schema: Any, doc: dict[str, Any], seen: frozenset[str]) -> tuple[dict[str, Any], frozenset[str]]:
218
+ """Resolve a schema and flatten a single level of allOf into it."""
219
+ schema, seen = _resolve(schema, doc, seen)
220
+ if not isinstance(schema, dict):
221
+ return {}, seen
222
+ if "allOf" not in schema:
223
+ return schema, seen
224
+ merged: dict[str, Any] = {k: v for k, v in schema.items() if k != "allOf"}
225
+ props: dict[str, Any] = dict(merged.get("properties") or {})
226
+ required: list[str] = list(merged.get("required") or [])
227
+ for part in schema["allOf"]:
228
+ sub, seen = _merged(part, doc, seen)
229
+ props.update(sub.get("properties") or {})
230
+ required.extend(sub.get("required") or [])
231
+ for key, value in sub.items():
232
+ if key not in ("properties", "required"):
233
+ merged.setdefault(key, value)
234
+ merged["properties"] = props
235
+ merged["required"] = sorted(set(required))
236
+ return merged, seen
237
+
238
+
239
+ def _types(schema: dict[str, Any]) -> frozenset[str]:
240
+ """The JSON types a schema admits, unioned across type lists and anyOf/oneOf."""
241
+ out: set[str] = set()
242
+ raw = schema.get("type")
243
+ if isinstance(raw, str):
244
+ out.add(raw)
245
+ elif isinstance(raw, list):
246
+ out.update(str(t) for t in raw)
247
+ for key in ("anyOf", "oneOf"):
248
+ for part in schema.get(key) or []:
249
+ if isinstance(part, dict):
250
+ out |= _types(part)
251
+ if not out and "properties" in schema:
252
+ out.add("object")
253
+ if not out and "items" in schema:
254
+ out.add("array")
255
+ return frozenset(out)
256
+
257
+
258
+ def _enum(schema: dict[str, Any]) -> frozenset[str] | None:
259
+ values = schema.get("enum")
260
+ if isinstance(values, list):
261
+ return frozenset(json.dumps(v, sort_keys=True) for v in values)
262
+ for key in ("anyOf", "oneOf"):
263
+ collected: set[str] = set()
264
+ for part in schema.get(key) or []:
265
+ sub = _enum(part) if isinstance(part, dict) else None
266
+ if sub:
267
+ collected |= set(sub)
268
+ if collected:
269
+ return frozenset(collected)
270
+ return None
271
+
272
+
273
+ def flatten_schema(schema: Any, doc: dict[str, Any], prefix: str = "", *, depth: int = 0,
274
+ seen: frozenset[str] = frozenset()) -> dict[str, dict[str, Any]]:
275
+ """Field path -> {types, enum, required}. Arrays flatten as `items[].field`.
276
+
277
+ Dotted paths are what a consumer actually reads (`items[].currency`), so a
278
+ diff expressed over them lands in the same vocabulary as the bug report.
279
+ """
280
+ out: dict[str, dict[str, Any]] = {}
281
+ if depth > 12:
282
+ return out
283
+ node, seen = _merged(schema, doc, seen)
284
+ if not isinstance(node, dict):
285
+ return out
286
+
287
+ if "items" in node:
288
+ out.update(flatten_schema(node["items"], doc, f"{prefix}[]", depth=depth + 1, seen=seen))
289
+
290
+ required = set(node.get("required") or [])
291
+ for name, sub in (node.get("properties") or {}).items():
292
+ path = f"{prefix}.{name}" if prefix else str(name)
293
+ resolved, sub_seen = _merged(sub, doc, seen)
294
+ out[path] = {
295
+ "types": _types(resolved),
296
+ "enum": _enum(resolved),
297
+ "required": name in required,
298
+ }
299
+ out.update(flatten_schema(resolved, doc, path, depth=depth + 1, seen=sub_seen))
300
+ return out
301
+
302
+
303
+ def _json_schema(container: Any, doc: dict[str, Any]) -> Any:
304
+ """The JSON body schema out of a responses/requestBody entry, if there is one."""
305
+ node, _ = _resolve(container, doc)
306
+ if not isinstance(node, dict):
307
+ return None
308
+ content = node.get("content")
309
+ if not isinstance(content, dict):
310
+ return None
311
+ for media, spec in content.items():
312
+ if "json" in str(media):
313
+ return (spec or {}).get("schema")
314
+ first = next(iter(content.values()), None)
315
+ return (first or {}).get("schema") if isinstance(first, dict) else None
316
+
317
+
318
+ def _operations(doc: dict[str, Any]) -> dict[tuple[str, str], dict[str, Any]]:
319
+ """(path, METHOD) -> operation, with path-level parameters folded in."""
320
+ ops: dict[tuple[str, str], dict[str, Any]] = {}
321
+ for path, item in (doc.get("paths") or {}).items():
322
+ if not isinstance(item, dict):
323
+ continue
324
+ shared = item.get("parameters") or []
325
+ for method in HTTP_METHODS:
326
+ op = item.get(method)
327
+ if not isinstance(op, dict):
328
+ continue
329
+ merged = dict(op)
330
+ merged["parameters"] = list(shared) + list(op.get("parameters") or [])
331
+ ops[(str(path), method.upper())] = merged
332
+ return ops
333
+
334
+
335
+ def _requires_auth(op: dict[str, Any], doc: dict[str, Any]) -> bool:
336
+ security = op.get("security", doc.get("security", []))
337
+ return bool(security) and any(bool(entry) for entry in security)
338
+
339
+
340
+ def _parameters(op: dict[str, Any], doc: dict[str, Any]) -> dict[tuple[str, str], dict[str, Any]]:
341
+ out: dict[tuple[str, str], dict[str, Any]] = {}
342
+ for raw in op.get("parameters") or []:
343
+ param, _ = _resolve(raw, doc)
344
+ if not isinstance(param, dict) or "name" not in param:
345
+ continue
346
+ schema, _ = _merged(param.get("schema") or {}, doc, frozenset())
347
+ out[(str(param.get("in", "query")), str(param["name"]))] = {
348
+ "required": bool(param.get("required", False)),
349
+ "types": _types(schema),
350
+ "enum": _enum(schema),
351
+ }
352
+ return out
353
+
354
+
355
+ # --------------------------------------------------------------------------
356
+ # the diff itself
357
+ # --------------------------------------------------------------------------
358
+
359
+
360
+ def _change(kind: str, path: str, method: str, detail: str) -> dict[str, Any]:
361
+ verdict, _ = RULES.get(kind, ("unknown", ""))
362
+ return {"kind": kind, "path": path, "method": method, "detail": detail, "breaking": verdict == "breaking"}
363
+
364
+
365
+ def _kind(prefix: str, facet: str, direction: str) -> str:
366
+ """`response_enum_widened` if that rule exists, else the `_type_` equivalent.
367
+
368
+ Only responses get their own enum rules; for requests and parameters an enum
369
+ change is just a change to the admissible value space, which is what the
370
+ type rules already say.
371
+ """
372
+ candidate = f"{prefix}_{facet}_{direction}"
373
+ return candidate if candidate in RULES else f"{prefix}_type_{direction}"
374
+
375
+
376
+ def _compare_value_space(a: dict[str, Any], b: dict[str, Any], prefix: str) -> str | None:
377
+ """The widened/narrowed kind for a field's value space, or None if unchanged.
378
+
379
+ Direction is decided by subset relation, not by name: only a strict shrink of
380
+ the admissible values can break a consumer that already works. An unrelated
381
+ change (string -> integer, one enum swapped for another) counts as narrowing,
382
+ because at least one value the consumer handled is now impossible.
383
+ """
384
+ enum_a, enum_b = a.get("enum"), b.get("enum")
385
+ if enum_a != enum_b:
386
+ if enum_a is None: # was unconstrained, now restricted
387
+ return _kind(prefix, "enum", "narrowed")
388
+ if enum_b is None: # was restricted, now open
389
+ return _kind(prefix, "enum", "widened")
390
+ return _kind(prefix, "enum", "widened" if enum_a < enum_b else "narrowed")
391
+
392
+ types_a, types_b = a.get("types") or frozenset(), b.get("types") or frozenset()
393
+ if types_a == types_b:
394
+ return None
395
+ # An empty set is not "no types" — it is a schema with no `type` at all,
396
+ # which admits *anything*. Treating it as the empty set made a widening to an
397
+ # untyped schema fall through to narrowed, and the report then said BREAKING
398
+ # over a detail that read, in as many words, `string -> any`. Untyped is the
399
+ # top of this lattice, so a move toward it widens and a move away narrows.
400
+ if not types_b:
401
+ return f"{prefix}_type_widened"
402
+ if not types_a:
403
+ return f"{prefix}_type_narrowed"
404
+ if types_a < types_b:
405
+ return f"{prefix}_type_widened"
406
+ return f"{prefix}_type_narrowed"
407
+
408
+
409
+ def _detect_renames(removed: list[str], added: list[str], fields_a: dict[str, dict[str, Any]],
410
+ fields_b: dict[str, dict[str, Any]]) -> list[tuple[str, str]]:
411
+ """Pair a removal with an addition when the sibling and shape both match.
412
+
413
+ Heuristic, deliberately conservative: a pair is only a rename when exactly
414
+ one candidate matches, so an object that lost two fields and gained two is
415
+ reported as four changes rather than two invented renames.
416
+ """
417
+ pairs: list[tuple[str, str]] = []
418
+ taken: set[str] = set()
419
+ for old in removed:
420
+ parent = old.rsplit(".", 1)[0] if "." in old else ""
421
+ shape = (fields_a[old]["types"], fields_a[old]["enum"], fields_a[old]["required"])
422
+ candidates = [
423
+ new for new in added
424
+ if new not in taken
425
+ and (new.rsplit(".", 1)[0] if "." in new else "") == parent
426
+ and (fields_b[new]["types"], fields_b[new]["enum"], fields_b[new]["required"]) == shape
427
+ ]
428
+ if len(candidates) == 1:
429
+ taken.add(candidates[0])
430
+ pairs.append((old, candidates[0]))
431
+ return pairs
432
+
433
+
434
+ def diff_specs(spec_a: dict[str, Any], spec_b: dict[str, Any]) -> list[dict[str, Any]]:
435
+ """Structural diff of two OpenAPI documents, A being the reference."""
436
+ ops_a, ops_b = _operations(spec_a), _operations(spec_b)
437
+ changes: list[dict[str, Any]] = []
438
+
439
+ for key in sorted(set(ops_a) - set(ops_b)):
440
+ changes.append(_change("endpoint_removed", key[0], key[1], f"{key[1]} {key[0]} is declared in A but absent from B."))
441
+ for key in sorted(set(ops_b) - set(ops_a)):
442
+ changes.append(_change("endpoint_added", key[0], key[1], f"{key[1]} {key[0]} exists in B but is undeclared in A."))
443
+
444
+ for key in sorted(set(ops_a) & set(ops_b)):
445
+ path, method = key
446
+ op_a, op_b = ops_a[key], ops_b[key]
447
+ changes.extend(_diff_security(op_a, op_b, spec_a, spec_b, path, method))
448
+ changes.extend(_diff_parameters(op_a, op_b, spec_a, spec_b, path, method))
449
+ changes.extend(_diff_request(op_a, op_b, spec_a, spec_b, path, method))
450
+ changes.extend(_diff_responses(op_a, op_b, spec_a, spec_b, path, method))
451
+
452
+ return changes
453
+
454
+
455
+ def _diff_security(op_a: dict[str, Any], op_b: dict[str, Any], doc_a: dict[str, Any], doc_b: dict[str, Any],
456
+ path: str, method: str) -> list[dict[str, Any]]:
457
+ auth_a, auth_b = _requires_auth(op_a, doc_a), _requires_auth(op_b, doc_b)
458
+ if auth_a == auth_b:
459
+ return []
460
+ kind = "security_added" if auth_b else "security_removed"
461
+ verb = "now requires" if auth_b else "no longer requires"
462
+ return [_change(kind, path, method, f"{method} {path} {verb} authentication.")]
463
+
464
+
465
+ def _diff_parameters(op_a: dict[str, Any], op_b: dict[str, Any], doc_a: dict[str, Any], doc_b: dict[str, Any],
466
+ path: str, method: str) -> list[dict[str, Any]]:
467
+ params_a, params_b = _parameters(op_a, doc_a), _parameters(op_b, doc_b)
468
+ changes: list[dict[str, Any]] = []
469
+ for key in sorted(set(params_a) - set(params_b)):
470
+ changes.append(_change("parameter_removed", path, method, f"{key[0]} parameter '{key[1]}' was removed."))
471
+ for key in sorted(set(params_b) - set(params_a)):
472
+ kind = "parameter_added_required" if params_b[key]["required"] else "parameter_added_optional"
473
+ changes.append(_change(kind, path, method, f"{key[0]} parameter '{key[1]}' was added."))
474
+ for key in sorted(set(params_a) & set(params_b)):
475
+ a, b = params_a[key], params_b[key]
476
+ if not a["required"] and b["required"]:
477
+ changes.append(_change("parameter_required_added", path, method, f"{key[0]} parameter '{key[1]}' became required."))
478
+ kind = _compare_value_space(a, b, "parameter")
479
+ if kind:
480
+ changes.append(_change(kind, path, method,
481
+ f"{key[0]} parameter '{key[1]}': {_describe(a)} -> {_describe(b)}."))
482
+ return changes
483
+
484
+
485
+ def _diff_request(op_a: dict[str, Any], op_b: dict[str, Any], doc_a: dict[str, Any], doc_b: dict[str, Any],
486
+ path: str, method: str) -> list[dict[str, Any]]:
487
+ schema_a = _json_schema(op_a.get("requestBody") or {}, doc_a)
488
+ schema_b = _json_schema(op_b.get("requestBody") or {}, doc_b)
489
+ if schema_a is None and schema_b is None:
490
+ return []
491
+ fields_a = flatten_schema(schema_a or {}, doc_a)
492
+ fields_b = flatten_schema(schema_b or {}, doc_b)
493
+ changes: list[dict[str, Any]] = []
494
+ for name in sorted(set(fields_a) - set(fields_b)):
495
+ changes.append(_change("request_field_removed", path, method, f"request field '{name}' was removed."))
496
+ for name in sorted(set(fields_b) - set(fields_a)):
497
+ kind = "request_field_added_required" if fields_b[name]["required"] else "request_field_added_optional"
498
+ changes.append(_change(kind, path, method, f"request field '{name}' was added."))
499
+ for name in sorted(set(fields_a) & set(fields_b)):
500
+ a, b = fields_a[name], fields_b[name]
501
+ if not a["required"] and b["required"]:
502
+ changes.append(_change("request_required_added", path, method, f"request field '{name}' became required."))
503
+ elif a["required"] and not b["required"]:
504
+ changes.append(_change("request_required_dropped", path, method, f"request field '{name}' is no longer required."))
505
+ kind = _compare_value_space(a, b, "request")
506
+ if kind:
507
+ changes.append(_change(kind, path, method, f"request field '{name}': {_describe(a)} -> {_describe(b)}."))
508
+ return changes
509
+
510
+
511
+ def _diff_responses(op_a: dict[str, Any], op_b: dict[str, Any], doc_a: dict[str, Any], doc_b: dict[str, Any],
512
+ path: str, method: str) -> list[dict[str, Any]]:
513
+ responses_a = {str(k): v for k, v in (op_a.get("responses") or {}).items()}
514
+ responses_b = {str(k): v for k, v in (op_b.get("responses") or {}).items()}
515
+ changes: list[dict[str, Any]] = []
516
+
517
+ for code in sorted(set(responses_a) - set(responses_b)):
518
+ changes.append(_change("status_code_removed", path, method, f"documented status {code} is gone."))
519
+ for code in sorted(set(responses_b) - set(responses_a)):
520
+ changes.append(_change("status_code_added", path, method, f"status {code} is documented in B only."))
521
+
522
+ for code in sorted(set(responses_a) & set(responses_b)):
523
+ schema_a = _json_schema(responses_a[code], doc_a)
524
+ schema_b = _json_schema(responses_b[code], doc_b)
525
+ if schema_a is None and schema_b is None:
526
+ continue
527
+ fields_a = flatten_schema(schema_a or {}, doc_a)
528
+ fields_b = flatten_schema(schema_b or {}, doc_b)
529
+ removed = sorted(set(fields_a) - set(fields_b))
530
+ added = sorted(set(fields_b) - set(fields_a))
531
+ renamed = _detect_renames(removed, added, fields_a, fields_b)
532
+ renamed_old = {old for old, _ in renamed}
533
+ renamed_new = {new for _, new in renamed}
534
+
535
+ for old, new in renamed:
536
+ changes.append(_change("response_field_renamed", path, method,
537
+ f"{code} response field '{old}' appears to have been renamed to '{new}'."))
538
+ for name in removed:
539
+ if name in renamed_old:
540
+ continue
541
+ qualifier = "required " if fields_a[name]["required"] else ""
542
+ changes.append(_change("response_field_removed", path, method,
543
+ f"{code} response is missing the {qualifier}field '{name}' the reference declares."))
544
+ for name in added:
545
+ if name in renamed_new:
546
+ continue
547
+ qualifier = "required" if fields_b[name]["required"] else "optional"
548
+ changes.append(_change("response_field_added", path, method,
549
+ f"{code} response gained the {qualifier} field '{name}'."))
550
+ for name in sorted(set(fields_a) & set(fields_b)):
551
+ a, b = fields_a[name], fields_b[name]
552
+ if a["required"] and not b["required"]:
553
+ changes.append(_change("response_required_dropped", path, method,
554
+ f"{code} response field '{name}' is no longer guaranteed present."))
555
+ elif not a["required"] and b["required"]:
556
+ changes.append(_change("response_required_added", path, method,
557
+ f"{code} response field '{name}' is now always present."))
558
+ kind = _compare_value_space(a, b, "response")
559
+ if kind:
560
+ changes.append(_change(kind, path, method,
561
+ f"{code} response field '{name}': {_describe(a)} -> {_describe(b)}."))
562
+ return changes
563
+
564
+
565
+ def _describe(field: dict[str, Any]) -> str:
566
+ types = "/".join(sorted(field.get("types") or [])) or "any"
567
+ enum = field.get("enum")
568
+ if enum:
569
+ # Enum members are stored as canonical JSON so they can be set-compared;
570
+ # decode them again for a message a human reads.
571
+ values = sorted(str(json.loads(v)) for v in enum)
572
+ return f"{types} enum[{', '.join(values)}]"
573
+ return types
574
+
575
+
576
+ # --------------------------------------------------------------------------
577
+ # consumers and generated tests
578
+ # --------------------------------------------------------------------------
579
+
580
+
581
+ def _split_endpoint(raw: str) -> tuple[str | None, str]:
582
+ """'GET /v1/orders' -> ('GET', '/v1/orders'); a bare path -> (None, path)."""
583
+ parts = raw.strip().split()
584
+ if len(parts) == 2 and parts[0].upper() in {m.upper() for m in HTTP_METHODS}:
585
+ return parts[0].upper(), parts[1]
586
+ return None, parts[-1] if parts else raw.strip()
587
+
588
+
589
+ def _search_terms(path: str) -> list[str]:
590
+ """Literal needles for a templated path: the whole thing and its stable prefix."""
591
+ terms = {path}
592
+ head = path.split("{", 1)[0].rstrip("/")
593
+ if head and head != path:
594
+ terms.add(head)
595
+ return sorted(terms, key=len, reverse=True)
596
+
597
+
598
+ def _walk_sources(root: Path) -> Iterable[Path]:
599
+ for dirpath, dirnames, filenames in os.walk(root):
600
+ dirnames[:] = [d for d in dirnames if d not in SKIP_DIRS and not d.startswith(".")]
601
+ for name in filenames:
602
+ if Path(name).suffix in CONSUMER_SUFFIXES:
603
+ yield Path(dirpath) / name
604
+
605
+
606
+ def _identifier(method: str, path: str) -> str:
607
+ slug = re.sub(r"[^a-z0-9]+", "_", path.lower()).strip("_")
608
+ return f"{method.lower()}_{slug}" or "endpoint"
609
+
610
+
611
+ def _required_fields(schema: Any, doc: dict[str, Any]) -> tuple[list[str], list[str]]:
612
+ """(top-level required fields, required fields of `items[]`) for a response schema."""
613
+ fields = flatten_schema(schema or {}, doc)
614
+ top = sorted(n for n, f in fields.items() if f["required"] and "." not in n and "[]" not in n)
615
+ item = sorted(
616
+ n.split("items[].", 1)[1]
617
+ for n, f in fields.items()
618
+ if f["required"] and n.startswith("items[].") and "." not in n.split("items[].", 1)[1]
619
+ )
620
+ return top, item
621
+
622
+
623
+ TEST_TEMPLATE = '''"""Contract test for {method} {path} — generated by API from {spec_name}.
624
+
625
+ {why}
626
+
627
+ Runs against a live server: set QAAS_TARGET_BASE_URL (default {default_base}).
628
+ Supply QAAS_TARGET_TOKEN to skip the login round-trip. Standard library only, so
629
+ it runs anywhere pytest does — including in a fix branch's CI.
630
+ """
631
+
632
+ import json
633
+ import os
634
+ import urllib.error
635
+ import urllib.request
636
+
637
+ import pytest
638
+
639
+ BASE_URL = os.environ.get("QAAS_TARGET_BASE_URL", "{default_base}").rstrip("/")
640
+ PATH = {path!r}
641
+ METHOD = {method!r}
642
+ EXPECTED_STATUS = {expected_status}
643
+ REQUIRED_TOP_LEVEL_FIELDS = {required_top!r}
644
+ REQUIRED_ITEM_FIELDS = {required_item!r}
645
+ AUTH_REQUIRED = {auth_required!r}
646
+ LOGIN_PATH = {login_path!r}
647
+ LOGIN_EMAIL = os.environ.get("QAAS_TARGET_USER", "")
648
+ LOGIN_PASSWORD = os.environ.get("QAAS_TARGET_PASSWORD", "")
649
+
650
+
651
+ def _call(path, method="GET", token=None, body=None):
652
+ data = json.dumps(body).encode() if body is not None else None
653
+ headers = {{"Accept": "application/json"}}
654
+ if data is not None:
655
+ headers["Content-Type"] = "application/json"
656
+ if token:
657
+ headers["Authorization"] = "Bearer " + token
658
+ request = urllib.request.Request(BASE_URL + path, data=data, headers=headers, method=method)
659
+ try:
660
+ with urllib.request.urlopen(request, timeout=15) as resp:
661
+ return resp.status, resp.read().decode(errors="replace")
662
+ except urllib.error.HTTPError as exc:
663
+ return exc.code, exc.read().decode(errors="replace")
664
+ except (urllib.error.URLError, OSError) as exc:
665
+ pytest.fail("target app unreachable at " + BASE_URL + ": " + str(exc))
666
+
667
+
668
+ @pytest.fixture(scope="module")
669
+ def token():
670
+ if not AUTH_REQUIRED:
671
+ return None
672
+ preset = os.environ.get("QAAS_TARGET_TOKEN")
673
+ if preset:
674
+ return preset
675
+ if not (LOGIN_EMAIL and LOGIN_PASSWORD):
676
+ pytest.skip(
677
+ "This endpoint needs auth. Set QAAS_TARGET_TOKEN, or QAAS_TARGET_USER and "
678
+ "QAAS_TARGET_PASSWORD for " + LOGIN_PATH + "."
679
+ )
680
+ status, body = _call(LOGIN_PATH, "POST", body={{"email": LOGIN_EMAIL, "password": LOGIN_PASSWORD}})
681
+ assert status == 200, "could not log in to fetch a token: HTTP %s %s" % (status, body[:300])
682
+ return json.loads(body)["access_token"]
683
+
684
+
685
+ @pytest.fixture(scope="module")
686
+ def response(token):
687
+ status, body = _call(PATH, METHOD, token=token)
688
+ return status, body
689
+
690
+
691
+ def test_status_code_matches_the_spec(response):
692
+ status, body = response
693
+ assert status == EXPECTED_STATUS, (
694
+ "%s %s: spec declares %s, server returned %s. Body: %s"
695
+ % (METHOD, PATH, EXPECTED_STATUS, status, body[:300])
696
+ )
697
+
698
+
699
+ def test_response_is_json(response):
700
+ _, body = response
701
+ try:
702
+ json.loads(body)
703
+ except json.JSONDecodeError:
704
+ pytest.fail("%s %s did not return JSON: %s" % (METHOD, PATH, body[:300]))
705
+
706
+
707
+ def test_required_response_fields_are_present(response):
708
+ if not REQUIRED_TOP_LEVEL_FIELDS:
709
+ pytest.skip("the spec declares no required top-level response fields")
710
+ _, body = response
711
+ payload = json.loads(body)
712
+ missing = [name for name in REQUIRED_TOP_LEVEL_FIELDS if name not in payload]
713
+ assert not missing, (
714
+ "%s %s response is missing spec-required field(s): %s"
715
+ % (METHOD, PATH, ", ".join(missing))
716
+ )
717
+
718
+
719
+ def test_collection_items_carry_required_fields(response):
720
+ if not REQUIRED_ITEM_FIELDS:
721
+ pytest.skip("this response is not a collection with a declared item schema")
722
+ _, body = response
723
+ payload = json.loads(body)
724
+ items = payload.get("items") if isinstance(payload, dict) else None
725
+ if not items:
726
+ pytest.skip("no items returned; seed the fixture to exercise this assertion")
727
+ missing = sorted({{name for item in items for name in REQUIRED_ITEM_FIELDS if name not in item}})
728
+ assert not missing, (
729
+ "%s %s items are missing spec-required field(s): %s"
730
+ % (METHOD, PATH, ", ".join(missing))
731
+ )
732
+
733
+
734
+ def test_endpoint_requires_authentication():
735
+ if not AUTH_REQUIRED:
736
+ pytest.skip("the spec marks this endpoint as public")
737
+ status, body = _call(PATH, METHOD)
738
+ assert status == 401, (
739
+ "%s %s is declared as authenticated but answered an anonymous call with %s"
740
+ % (METHOD, PATH, status)
741
+ )
742
+ '''
743
+
744
+
745
+ # --------------------------------------------------------------------------
746
+ # tools
747
+ # --------------------------------------------------------------------------
748
+
749
+
750
+ def build_tools(ctx: ToolContext) -> list:
751
+ """The contract_diff tools, bound to one agent's run context.
752
+
753
+ Split from `build` so tests can call the handlers directly without standing
754
+ up an MCP transport.
755
+ """
756
+
757
+ # `layout.spec` has existed since the schema was written and nothing read
758
+ # it, so every target was assumed to keep an `openapi.yaml` at its root.
759
+ _profile = getattr(ctx.config, "profile", None)
760
+ _declared_spec = getattr(getattr(_profile, "layout", None), "spec", None) if _profile else None
761
+ default_spec = ctx.target_root / (_declared_spec or DEFAULT_SPEC_FILE)
762
+ _spec_label = str(_declared_spec or DEFAULT_SPEC_FILE)
763
+
764
+ @tool(
765
+ "diff_openapi",
766
+ "Compare two OpenAPI documents semantically. Defaults to the declared contract "
767
+ f"({_spec_label}) against the running app's /openapi.json, so a reported change "
768
+ "means the implementation disagrees with the spec. Pass file paths when the app is down.",
769
+ {
770
+ "type": "object",
771
+ "properties": {
772
+ "spec_a": {"type": "string", "description": "Reference spec: a path or an http(s) URL. Default: the declared openapi.yaml."},
773
+ "spec_b": {"type": "string", "description": "Spec under test. Default: the running app's /openapi.json."},
774
+ "only_breaking": {"type": "boolean", "description": "Return only changes classified breaking."},
775
+ },
776
+ },
777
+ )
778
+ async def diff_openapi(args: dict[str, Any]) -> dict[str, Any]:
779
+ ref_a = str(args.get("spec_a") or default_spec)
780
+ live_default = os.environ.get("QAAS_TARGET_BASE_URL")
781
+ live_default = f"{live_default.rstrip('/')}/openapi.json" if live_default else DEFAULT_LIVE_SPEC_URL
782
+ ref_b = str(args.get("spec_b") or live_default)
783
+
784
+ doc_a, problem = _read_spec(ref_a, ctx.target_root)
785
+ if problem:
786
+ return err(f"Could not load spec_a: {problem}")
787
+ doc_b, problem = _read_spec(ref_b, ctx.target_root)
788
+ if problem:
789
+ hint = (
790
+ " The target app does not appear to be running. Start it with env_control.spin_up, "
791
+ "or pass spec_b as a file path to compare two documents on disk."
792
+ if _is_url(ref_b) else ""
793
+ )
794
+ return err(f"Could not load spec_b: {problem}.{hint}")
795
+
796
+ assert doc_a is not None and doc_b is not None
797
+ try:
798
+ changes = diff_specs(doc_a, doc_b)
799
+ except RecursionError:
800
+ return err("The specs contain a recursion this differ cannot follow; compare a subset of paths.")
801
+
802
+ if args.get("only_breaking"):
803
+ changes = [c for c in changes if c["breaking"]]
804
+ breaking = [c for c in changes if c["breaking"]]
805
+
806
+ if not changes:
807
+ return ok(f"No semantic differences between {ref_a} and {ref_b}.", changes=[], breaking_count=0)
808
+
809
+ lines = [f"{'BREAKING' if c['breaking'] else 'compatible'} {c['method']} {c['path']} [{c['kind']}] {c['detail']}"
810
+ for c in changes[:120]]
811
+ more = f"\n(+{len(changes) - 120} more)" if len(changes) > 120 else ""
812
+ return ok(
813
+ f"{len(changes)} change(s), {len(breaking)} breaking, comparing {ref_a} (reference) "
814
+ f"against {ref_b}:\n" + "\n".join(lines) + more,
815
+ changes=changes, breaking_count=len(breaking), spec_a=ref_a, spec_b=ref_b,
816
+ )
817
+
818
+ @tool(
819
+ "classify_breaking",
820
+ "Classify one change as breaking, non_breaking or unknown, with the consumer-impact reason. "
821
+ "Pass a change object from diff_openapi.",
822
+ {
823
+ "type": "object",
824
+ "required": ["change"],
825
+ "properties": {
826
+ "change": {
827
+ "type": "object",
828
+ "description": "A change from diff_openapi: {kind, path, method, detail}.",
829
+ "properties": {
830
+ "kind": {"type": "string"},
831
+ "path": {"type": "string"},
832
+ "method": {"type": "string"},
833
+ "detail": {"type": "string"},
834
+ },
835
+ }
836
+ },
837
+ },
838
+ )
839
+ async def classify_breaking(args: dict[str, Any]) -> dict[str, Any]:
840
+ change = args.get("change")
841
+ if not isinstance(change, dict):
842
+ change = {k: v for k, v in args.items() if k != "change"}
843
+ kind = str(change.get("kind") or "").strip()
844
+ if not kind:
845
+ return err(
846
+ "The change has no 'kind'. Pass a change object exactly as diff_openapi returned it; "
847
+ f"recognised kinds are: {', '.join(sorted(RULES))}."
848
+ )
849
+ verdict, reason = RULES.get(kind, (
850
+ "unknown",
851
+ f"'{kind}' is not a kind this server emits, so its consumer impact cannot be decided here. "
852
+ "Judge it by hand, or re-run diff_openapi and classify one of its changes.",
853
+ ))
854
+ where = f"{change.get('method', '')} {change.get('path', '')}".strip()
855
+ return ok(
856
+ f"{kind}{f' on {where}' if where else ''}: {verdict}. {reason}",
857
+ verdict=verdict, reason=reason, kind=kind, breaking=verdict == "breaking",
858
+ )
859
+
860
+ @tool(
861
+ "find_consumers",
862
+ "Find likely call sites of an endpoint in this repository, across its declared frontend roots. "
863
+ "Heuristic: it matches the literal path string and the stable prefix before any {param}, "
864
+ "so a client that assembles its URL from fragments will be missed and an unrelated string "
865
+ "that happens to contain the path will be reported.",
866
+ {
867
+ "type": "object",
868
+ "required": ["endpoint"],
869
+ "properties": {
870
+ "endpoint": {"type": "string", "description": "'GET /v1/orders' or just '/v1/orders'."},
871
+ "root": {"type": "string", "description": "Subdirectory to search, repo-relative. Default: the whole repo."},
872
+ },
873
+ },
874
+ )
875
+ async def find_consumers(args: dict[str, Any]) -> dict[str, Any]:
876
+ _, path = _split_endpoint(str(args["endpoint"]))
877
+ if not path.startswith("/"):
878
+ return err(f"'{args['endpoint']}' does not name a path. Use 'GET /v1/orders' or '/v1/orders'.")
879
+
880
+ root = ctx.target_root
881
+ if args.get("root"):
882
+ candidate = (ctx.target_root / str(args["root"])).resolve()
883
+ if not candidate.is_relative_to(ctx.target_root.resolve()):
884
+ return err("root must stay inside the repository.")
885
+ if not candidate.exists():
886
+ return err(f"No such directory: {candidate}.")
887
+ root = candidate
888
+
889
+ terms = _search_terms(path)
890
+ hits: list[dict[str, Any]] = []
891
+ for file in _walk_sources(root):
892
+ try:
893
+ text = file.read_text(errors="replace")
894
+ except OSError:
895
+ continue
896
+ if not any(term in text for term in terms):
897
+ continue
898
+ for number, line in enumerate(text.splitlines(), start=1):
899
+ if any(term in line for term in terms):
900
+ hits.append({
901
+ "file": str(file.relative_to(ctx.target_root)) if file.is_relative_to(ctx.target_root) else str(file),
902
+ "line": number,
903
+ "text": line.strip()[:200],
904
+ })
905
+ if len(hits) >= MAX_CONSUMER_HITS:
906
+ break
907
+ if len(hits) >= MAX_CONSUMER_HITS:
908
+ break
909
+
910
+ if not hits:
911
+ return ok(
912
+ f"No call sites matched {path} under {root}. That is weak evidence: a client that builds "
913
+ "the URL from fragments would not match. Check by hand before claiming no consumers exist.",
914
+ consumers=[], searched=str(root), terms=terms,
915
+ )
916
+ listing = "\n".join(f"{h['file']}:{h['line']} {h['text']}" for h in hits)
917
+ return ok(f"{len(hits)} possible call site(s) for {path} (heuristic):\n{listing}",
918
+ consumers=hits, searched=str(root), terms=terms)
919
+
920
+ @tool(
921
+ "generate_contract_test",
922
+ "Emit a runnable pytest module asserting the spec's contract for one endpoint: status code, "
923
+ "required response fields, and the auth requirement. Written under .qaas/generated/ and stored "
924
+ "as an artifact so the envelope can cite it. Run it before filing — a contract test that does "
925
+ "not fail against the current implementation is not evidence.",
926
+ {
927
+ "type": "object",
928
+ "required": ["endpoint", "method"],
929
+ "properties": {
930
+ "endpoint": {"type": "string", "description": "Spec path, e.g. '/v1/invoices'."},
931
+ "method": {"type": "string", "description": "HTTP method, e.g. 'GET'."},
932
+ "expectation": {"type": "string", "description": "Why you are generating it — what you expect to break. Recorded in the test's docstring."},
933
+ "spec": {"type": "string", "description": f"Spec to read the contract from. Default: {_spec_label}."},
934
+ "expected_status": {"type": "integer", "description": "Override the success status. Default: the first 2xx in the spec."},
935
+ },
936
+ },
937
+ )
938
+ async def generate_contract_test(args: dict[str, Any]) -> dict[str, Any]:
939
+ ref = str(args.get("spec") or default_spec)
940
+ doc, problem = _read_spec(ref, ctx.target_root)
941
+ if problem:
942
+ return err(f"Could not load the spec: {problem}")
943
+ assert doc is not None
944
+
945
+ _, path = _split_endpoint(str(args["endpoint"]))
946
+ method = str(args["method"]).upper()
947
+ if "{" in path:
948
+ return err(
949
+ f"{path} has path parameters this generator cannot fill in — it would have to invent "
950
+ "an id, and a test that 404s proves nothing. Generate a test for a collection endpoint, "
951
+ "or write the parameterised case by hand."
952
+ )
953
+ ops = _operations(doc)
954
+ op = ops.get((path, method))
955
+ if op is None:
956
+ known = ", ".join(f"{m} {p}" for p, m in sorted(ops)[:25])
957
+ return err(f"{method} {path} is not in {ref}. It declares: {known}")
958
+
959
+ responses = {str(k): v for k, v in (op.get("responses") or {}).items()}
960
+ override = args.get("expected_status")
961
+ if override is not None:
962
+ expected = int(override)
963
+ if str(expected) not in responses:
964
+ return err(f"{method} {path} does not declare status {expected}; it declares {', '.join(sorted(responses))}.")
965
+ else:
966
+ success = sorted(c for c in responses if c.isdigit() and 200 <= int(c) < 300)
967
+ if not success:
968
+ return err(f"{method} {path} declares no 2xx response, so there is no success contract to assert.")
969
+ expected = int(success[0])
970
+
971
+ required_top, required_item = _required_fields(_json_schema(responses[str(expected)], doc), doc)
972
+ source = TEST_TEMPLATE.format(
973
+ method=method,
974
+ path=path,
975
+ spec_name=Path(ref).name if not _is_url(ref) else ref,
976
+ # The expectation is free text from an agent and lands inside a
977
+ # docstring, so a stray triple quote would produce a file that will
978
+ # not import. Neutralise it rather than reject the call.
979
+ why=str(args.get("expectation") or "Asserts the published contract for this endpoint.").strip().replace('"""', "'''").replace("\\", "/"),
980
+ default_base=os.environ.get("QAAS_TARGET_BASE_URL", "http://localhost:8000").rstrip("/"),
981
+ login_path=_login_path(ctx),
982
+ expected_status=expected,
983
+ required_top=required_top,
984
+ required_item=required_item,
985
+ auth_required=_requires_auth(op, doc),
986
+ )
987
+
988
+ out_dir = ctx.store.root / GENERATED_DIR
989
+ out_dir.mkdir(parents=True, exist_ok=True)
990
+ filename = f"test_contract_{_identifier(method, path)}.py"
991
+ target = out_dir / filename
992
+ target.write_text(source, encoding="utf-8")
993
+ uri = ctx.store.put_artifact(filename, source)
994
+ ctx.store.log("contract_test", agent=ctx.agent.name, endpoint=f"{method} {path}", path=str(target))
995
+
996
+ return ok(
997
+ f"Wrote {target}. It asserts HTTP {expected}"
998
+ + (f", required response fields {', '.join(required_top)}" if required_top else "")
999
+ + (f", required item fields {', '.join(required_item)}" if required_item else "")
1000
+ + (", and that anonymous calls get 401" if _requires_auth(op, doc) else "")
1001
+ + f". Run it with QAAS_TARGET_BASE_URL set, then cite {uri} as evidence.\n\n{source}",
1002
+ path=str(target), uri=uri, source=source, expected_status=expected,
1003
+ required_fields=required_top, required_item_fields=required_item,
1004
+ )
1005
+
1006
+ return [diff_openapi, classify_breaking, find_consumers, generate_contract_test]
1007
+
1008
+
1009
+ def build(ctx: ToolContext):
1010
+ """Construct the contract_diff MCP server bound to one agent's run context."""
1011
+ return create_sdk_mcp_server(name="contract_diff", version="1.0.0", tools=build_tools(ctx))