fusiontest 0.2.0__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fusiontest-0.2.0 → fusiontest-0.2.2}/PKG-INFO +1 -1
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/action_model.py +19 -87
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/desktop/playwright_adapter.py +25 -0
- fusiontest-0.2.2/fusiontest/discovery/goal_generator.py +287 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/PKG-INFO +1 -1
- {fusiontest-0.2.0 → fusiontest-0.2.2}/pyproject.toml +1 -1
- fusiontest-0.2.0/fusiontest/discovery/goal_generator.py +0 -696
- {fusiontest-0.2.0 → fusiontest-0.2.2}/README.md +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/__init__.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/cli.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/__init__.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/goal_verifier.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/replay.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/runner.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/screen_parser.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/secrets.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/tokens.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/desktop/__init__.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/desktop/macos_adapter.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/desktop/windows_adapter.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/discovery/__init__.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/guardrails/__init__.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/guardrails/engine.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/mobile/__init__.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/mobile/android_adapter.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/mobile/ios_adapter.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/recording/__init__.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/recording/recorder.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/reporting/__init__.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/reporting/reporter.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/training/__init__.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/training/data_collector.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/training/dataset_builder.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/training/trainer.py +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/SOURCES.txt +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/dependency_links.txt +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/entry_points.txt +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/not-zip-safe +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/requires.txt +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/top_level.txt +0 -0
- {fusiontest-0.2.0 → fusiontest-0.2.2}/setup.cfg +0 -0
|
@@ -73,6 +73,7 @@ def _error_status_code(err: Exception) -> int | None:
|
|
|
73
73
|
_BACKEND_SPECIFIC_STATUS_CODES = {401, 403, 404}
|
|
74
74
|
_BACKEND_SPECIFIC_TOKENS = (
|
|
75
75
|
"model_not_found", "does not exist", "do not have access",
|
|
76
|
+
"this model only supports", "is not supported for this model", # wrong kind of model
|
|
76
77
|
"invalid api key", "invalid_api_key", "is not set",
|
|
77
78
|
"api key not valid", "api_key_invalid", # Google returns these as HTTP 400
|
|
78
79
|
)
|
|
@@ -185,52 +186,6 @@ class Action:
|
|
|
185
186
|
|
|
186
187
|
ActionBackend = Literal["mpnet", "gpt4o", "claude", "gemini", "ollama", "groq"]
|
|
187
188
|
|
|
188
|
-
# ── Gemini model auto-detection ───────────────────────────────────────────────
|
|
189
|
-
|
|
190
|
-
_gemini_model_cache: str | None = None
|
|
191
|
-
_GEMINI_PREFERRED = ["flash", "pro"] # tier preference, newest date wins
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
def _resolve_gemini_model(client) -> str:
|
|
195
|
-
"""Return the best available Gemini model for this API key (cached).
|
|
196
|
-
|
|
197
|
-
Preference: newest flash > older flash > newest pro > older pro.
|
|
198
|
-
Sort is DESCENDING by name so '2.5' beats '2.0' beats '1.5'.
|
|
199
|
-
|
|
200
|
-
Accepts a ``google.genai.Client`` instance from the new unified SDK
|
|
201
|
-
(``pip install google-genai``). The deprecated ``google-generativeai``
|
|
202
|
-
package is no longer supported.
|
|
203
|
-
"""
|
|
204
|
-
global _gemini_model_cache # noqa: PLW0603
|
|
205
|
-
if _gemini_model_cache is not None:
|
|
206
|
-
return _gemini_model_cache
|
|
207
|
-
|
|
208
|
-
try:
|
|
209
|
-
all_models: list[str] = []
|
|
210
|
-
for m in client.models.list():
|
|
211
|
-
name = (getattr(m, "name", "") or "").replace("models/", "")
|
|
212
|
-
if not name or "gemini" not in name.lower():
|
|
213
|
-
continue
|
|
214
|
-
if not any(t in name.lower() for t in ("flash", "pro")):
|
|
215
|
-
continue
|
|
216
|
-
all_models.append(name)
|
|
217
|
-
|
|
218
|
-
# Newest flash first, then newest pro — reverse=True so '2.5' > '2.0' > '1.5'
|
|
219
|
-
flash = sorted([m for m in all_models if "flash" in m.lower()], reverse=True)
|
|
220
|
-
pro = sorted([m for m in all_models if "flash" not in m.lower()], reverse=True)
|
|
221
|
-
ordered = flash + pro
|
|
222
|
-
|
|
223
|
-
if ordered:
|
|
224
|
-
_gemini_model_cache = ordered[0]
|
|
225
|
-
logger.info("Auto-selected Gemini model: %s", _gemini_model_cache)
|
|
226
|
-
return _gemini_model_cache
|
|
227
|
-
except Exception as exc:
|
|
228
|
-
logger.warning("Could not list Gemini models: %s — falling back to gemini-2.5-flash", exc)
|
|
229
|
-
|
|
230
|
-
_gemini_model_cache = "gemini-2.5-flash"
|
|
231
|
-
return _gemini_model_cache
|
|
232
|
-
|
|
233
|
-
|
|
234
189
|
# ── Fallback-chain construction ───────────────────────────────────────────────
|
|
235
190
|
|
|
236
191
|
# Model "" means "resolve the default for this backend at call time"
|
|
@@ -243,6 +198,11 @@ _DEFAULT_CHAIN_MODEL = ""
|
|
|
243
198
|
GROQ_DEFAULT_MODEL = "openai/gpt-oss-120b"
|
|
244
199
|
GROQ_FAST_MODEL = "openai/gpt-oss-20b" # separate rate-limit bucket
|
|
245
200
|
|
|
201
|
+
# Pinned Gemini model; GEMINI_MODEL overrides it. Not auto-selected from the
|
|
202
|
+
# model list — names like *-tts, *-image or Interactions-only models sort
|
|
203
|
+
# above the general-purpose one and reject normal requests.
|
|
204
|
+
GEMINI_DEFAULT_MODEL = "gemini-2.5-flash"
|
|
205
|
+
|
|
246
206
|
|
|
247
207
|
# Output-token cap for LLM calls. Reasoning models (Groq gpt-oss, Gemini 2.5)
|
|
248
208
|
# spend part of this on hidden reasoning before any answer, so a tight cap
|
|
@@ -264,7 +224,7 @@ def default_model_for_backend(backend: str) -> str:
|
|
|
264
224
|
if backend == "claude":
|
|
265
225
|
return "claude-sonnet-4-6"
|
|
266
226
|
if backend == "gemini":
|
|
267
|
-
return
|
|
227
|
+
return os.getenv("GEMINI_MODEL", GEMINI_DEFAULT_MODEL)
|
|
268
228
|
if backend == "ollama":
|
|
269
229
|
return os.getenv("OLLAMA_MODEL", "qwen2.5:7b")
|
|
270
230
|
if backend == "groq":
|
|
@@ -334,7 +294,7 @@ def _build_default_fallback_chain(primary: str) -> list[tuple[str, str]]:
|
|
|
334
294
|
seen_backends = {primary}
|
|
335
295
|
|
|
336
296
|
if "gemini" not in seen_backends and os.getenv("GEMINI_API_KEY"):
|
|
337
|
-
chain.append(("gemini",
|
|
297
|
+
chain.append(("gemini", default_model_for_backend("gemini")))
|
|
338
298
|
seen_backends.add("gemini")
|
|
339
299
|
|
|
340
300
|
if "groq" not in seen_backends and os.getenv("GROQ_API_KEY"):
|
|
@@ -360,6 +320,7 @@ class LLMRequest:
|
|
|
360
320
|
reminder: str = "" # appended to the prompt for backends without a JSON mode
|
|
361
321
|
json_object: bool = False # OpenAI-compatible response_format={"type": "json_object"}
|
|
362
322
|
empty: str = "" # returned when the model produces no text
|
|
323
|
+
max_output_tokens: int = LLM_MAX_OUTPUT_TOKENS # includes hidden reasoning/thinking
|
|
363
324
|
|
|
364
325
|
|
|
365
326
|
class LLMCaller:
|
|
@@ -381,6 +342,8 @@ class LLMCaller:
|
|
|
381
342
|
self.temperature = temperature
|
|
382
343
|
self.max_retries_per_backend = max_retries_per_backend
|
|
383
344
|
self.backoff_seconds = backoff_seconds
|
|
345
|
+
# "backend:model" of the entry that answered the last call.
|
|
346
|
+
self.last_backend: str = ""
|
|
384
347
|
# Per-(backend, model) client cache. Lazily populated.
|
|
385
348
|
self._clients: dict[tuple[str, str], tuple[Any, str]] = {}
|
|
386
349
|
|
|
@@ -416,11 +379,8 @@ class LLMCaller:
|
|
|
416
379
|
def _get_client(self, backend: str, model: str) -> tuple[Any, str]:
|
|
417
380
|
"""
|
|
418
381
|
Lazily create (and cache) the client for a given backend+model.
|
|
419
|
-
Returns (client, resolved_model)
|
|
420
|
-
|
|
421
|
-
(client, resolved_model) together so subsequent calls keep using the
|
|
422
|
-
resolved name instead of the empty sentinel that `default_model_for_backend`
|
|
423
|
-
returns for Gemini.
|
|
382
|
+
Returns (client, resolved_model); an empty `model` means the backend's
|
|
383
|
+
default (see default_model_for_backend).
|
|
424
384
|
"""
|
|
425
385
|
cache_key = (backend, model)
|
|
426
386
|
if cache_key in self._clients:
|
|
@@ -493,12 +453,6 @@ class LLMCaller:
|
|
|
493
453
|
"export GEMINI_API_KEY=AIza..."
|
|
494
454
|
)
|
|
495
455
|
client = genai.Client(api_key=api_key)
|
|
496
|
-
if not resolved_model:
|
|
497
|
-
resolved_model = _resolve_gemini_model(client)
|
|
498
|
-
logger.info(
|
|
499
|
-
"Gemini client created; resolved model=%r (cache_key=%r)",
|
|
500
|
-
resolved_model, cache_key,
|
|
501
|
-
)
|
|
502
456
|
|
|
503
457
|
else:
|
|
504
458
|
raise RuntimeError(f"Unknown backend: {backend}")
|
|
@@ -542,7 +496,9 @@ class LLMCaller:
|
|
|
542
496
|
|
|
543
497
|
for attempt in range(self.max_retries_per_backend + 1):
|
|
544
498
|
try:
|
|
545
|
-
|
|
499
|
+
answer = self._call_llm_single(prompt, backend, model, request)
|
|
500
|
+
self.last_backend = f"{backend}:{model or default_model_for_backend(backend)}"
|
|
501
|
+
return answer
|
|
546
502
|
except Exception as err: # noqa: BLE001
|
|
547
503
|
last_error = err
|
|
548
504
|
transient = _is_transient_error(err)
|
|
@@ -613,30 +569,6 @@ class LLMCaller:
|
|
|
613
569
|
"""
|
|
614
570
|
client, resolved_model = self._get_client(backend, model)
|
|
615
571
|
|
|
616
|
-
# Defensive guard: the Gemini SDK raises a cryptic
|
|
617
|
-
# ``ValueError: model is required.`` when ``model=""`` reaches
|
|
618
|
-
# ``generate_content``. This has happened in prod when a cache path
|
|
619
|
-
# or future refactor lets the sentinel `_DEFAULT_CHAIN_MODEL` leak
|
|
620
|
-
# through. Force a re-resolve and raise a loud, actionable error if
|
|
621
|
-
# that still fails — it's better to fail the attempt with context
|
|
622
|
-
# than to hand an empty string to the SDK.
|
|
623
|
-
if backend == "gemini" and not resolved_model:
|
|
624
|
-
logger.error(
|
|
625
|
-
"Gemini resolved_model is empty after _get_client(backend=%r, model=%r). "
|
|
626
|
-
"Attempting emergency re-resolve.",
|
|
627
|
-
backend, model,
|
|
628
|
-
)
|
|
629
|
-
resolved_model = _resolve_gemini_model(client)
|
|
630
|
-
if not resolved_model:
|
|
631
|
-
raise RuntimeError(
|
|
632
|
-
"Could not resolve a Gemini model name. "
|
|
633
|
-
"This should never happen — _resolve_gemini_model falls back to "
|
|
634
|
-
"'gemini-2.5-flash' on listing failures. Check for a corrupted "
|
|
635
|
-
"_gemini_model_cache."
|
|
636
|
-
)
|
|
637
|
-
# Self-heal the cache so the next call doesn't hit this path.
|
|
638
|
-
self._clients[(backend, model)] = (client, resolved_model)
|
|
639
|
-
|
|
640
572
|
if backend in ("gpt4o", "ollama", "groq"):
|
|
641
573
|
resp = client.chat.completions.create(
|
|
642
574
|
model=resolved_model,
|
|
@@ -645,7 +577,7 @@ class LLMCaller:
|
|
|
645
577
|
{"role": "user", "content": prompt},
|
|
646
578
|
],
|
|
647
579
|
temperature=self.temperature,
|
|
648
|
-
max_tokens=
|
|
580
|
+
max_tokens=request.max_output_tokens,
|
|
649
581
|
**({"response_format": {"type": "json_object"}} if request.json_object else {}),
|
|
650
582
|
**openai_compat_extra_args(resolved_model),
|
|
651
583
|
)
|
|
@@ -658,7 +590,7 @@ class LLMCaller:
|
|
|
658
590
|
if backend == "claude":
|
|
659
591
|
resp = client.messages.create(
|
|
660
592
|
model=resolved_model,
|
|
661
|
-
max_tokens=
|
|
593
|
+
max_tokens=request.max_output_tokens,
|
|
662
594
|
system=request.system,
|
|
663
595
|
messages=[{"role": "user", "content": prompt + request.reminder}],
|
|
664
596
|
)
|
|
@@ -676,7 +608,7 @@ class LLMCaller:
|
|
|
676
608
|
config=genai_types.GenerateContentConfig(
|
|
677
609
|
system_instruction=request.system,
|
|
678
610
|
temperature=self.temperature,
|
|
679
|
-
max_output_tokens=
|
|
611
|
+
max_output_tokens=request.max_output_tokens,
|
|
680
612
|
),
|
|
681
613
|
)
|
|
682
614
|
meta = getattr(resp, "usage_metadata", None)
|
|
@@ -140,6 +140,31 @@ class PlaywrightAdapter:
|
|
|
140
140
|
|
|
141
141
|
# ── Screen reading ─────────────────────────────────────────────────────────
|
|
142
142
|
|
|
143
|
+
def site_links(self, limit: int = 10) -> list[str]:
|
|
144
|
+
"""Same-origin page URLs linked from the header/navigation (fallback: any link)."""
|
|
145
|
+
return self._get_loop().run_until_complete(self._async_site_links(limit))
|
|
146
|
+
|
|
147
|
+
async def _async_site_links(self, limit: int) -> list[str]:
|
|
148
|
+
from urllib.parse import urldefrag, urljoin, urlsplit # noqa: PLC0415
|
|
149
|
+
|
|
150
|
+
hrefs: list[str] = await self._page.locator("header a[href], nav a[href]").evaluate_all(
|
|
151
|
+
"els => els.map(e => e.getAttribute('href'))"
|
|
152
|
+
)
|
|
153
|
+
if not hrefs:
|
|
154
|
+
hrefs = await self._page.locator("a[href]").evaluate_all(
|
|
155
|
+
"els => els.map(e => e.getAttribute('href'))"
|
|
156
|
+
)
|
|
157
|
+
base = self._page.url
|
|
158
|
+
origin = urlsplit(base).netloc
|
|
159
|
+
links: list[str] = []
|
|
160
|
+
for href in hrefs:
|
|
161
|
+
if not href or href.startswith(("mailto:", "tel:", "javascript:")):
|
|
162
|
+
continue
|
|
163
|
+
url = urldefrag(urljoin(base, href)).url.rstrip("/")
|
|
164
|
+
if urlsplit(url).netloc == origin and url != base.rstrip("/") and url not in links:
|
|
165
|
+
links.append(url)
|
|
166
|
+
return links[:limit]
|
|
167
|
+
|
|
143
168
|
def current_url(self) -> str:
|
|
144
169
|
"""URL of the current page ("" before start). Used for secret site binding."""
|
|
145
170
|
return self._page.url if self._page else ""
|
|
@@ -0,0 +1,287 @@
|
|
|
1
|
+
"""
|
|
2
|
+
fusiontest/discovery/goal_generator.py
|
|
3
|
+
|
|
4
|
+
Autonomous goal discovery: propose test goals for a site from what it
|
|
5
|
+
actually contains.
|
|
6
|
+
|
|
7
|
+
1. ``snapshot_site`` opens the site in a headless browser and captures the
|
|
8
|
+
start page plus up to a few pages linked from its navigation, using the
|
|
9
|
+
same screen representation the runner uses (headings, text, form fields,
|
|
10
|
+
buttons).
|
|
11
|
+
2. ``GoalGenerator.generate`` asks an LLM (through the shared fallback chain)
|
|
12
|
+
for goals, each quoting *evidence* from the captured pages.
|
|
13
|
+
3. Goals whose evidence isn't on those pages are dropped — the model may not
|
|
14
|
+
invent features (login, search, consent checkboxes…) the site doesn't have.
|
|
15
|
+
|
|
16
|
+
There is no template fallback: if no LLM can answer, generation fails with the
|
|
17
|
+
underlying error rather than presenting generic goals as "discovered".
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import json
|
|
23
|
+
import logging
|
|
24
|
+
import os
|
|
25
|
+
import re
|
|
26
|
+
from dataclasses import dataclass, field
|
|
27
|
+
from typing import Literal
|
|
28
|
+
from urllib.parse import urlsplit
|
|
29
|
+
|
|
30
|
+
from fusiontest.core.action_model import (
|
|
31
|
+
LLMCaller,
|
|
32
|
+
LLMRequest,
|
|
33
|
+
LLMUnavailableError,
|
|
34
|
+
_is_ollama_reachable,
|
|
35
|
+
)
|
|
36
|
+
from fusiontest.core.replay import select_evidence
|
|
37
|
+
|
|
38
|
+
logger = logging.getLogger(__name__)
|
|
39
|
+
|
|
40
|
+
GoalCategory = Literal["critical_path", "happy_path", "edge_cases"]
|
|
41
|
+
_CATEGORIES = ("critical_path", "happy_path", "edge_cases")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class GoalGenerationError(RuntimeError):
|
|
45
|
+
"""Goals couldn't be generated (site unreachable, no LLM, unusable answer)."""
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
@dataclass
|
|
49
|
+
class DiscoveredGoal:
|
|
50
|
+
id: str
|
|
51
|
+
name: str # short 5-8 word title for the YAML output
|
|
52
|
+
text: str
|
|
53
|
+
category: GoalCategory
|
|
54
|
+
confidence: float # 0.0 – 1.0
|
|
55
|
+
selected: bool = True # user can deselect before running
|
|
56
|
+
edited: bool = False # has the user overridden the text?
|
|
57
|
+
evidence: str = "" # quote from the site content the goal relies on
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
@dataclass
|
|
61
|
+
class GoalGeneratorConfig:
|
|
62
|
+
max_goals: int = 20
|
|
63
|
+
max_pages: int = 8 # start page + pages linked from navigation
|
|
64
|
+
max_chars_per_page: int = 4000
|
|
65
|
+
temperature: float = 0.3
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass
|
|
69
|
+
class SiteSnapshot:
|
|
70
|
+
url: str
|
|
71
|
+
pages: list[tuple[str, str]] = field(default_factory=list) # (url, screen text)
|
|
72
|
+
|
|
73
|
+
@property
|
|
74
|
+
def content(self) -> str:
|
|
75
|
+
return "\n\n".join(f"=== PAGE {urlsplit(u).path or '/'} ===\n{text}" for u, text in self.pages)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
# ── Site snapshot ─────────────────────────────────────────────────────────────
|
|
79
|
+
|
|
80
|
+
def snapshot_site(url: str, config: GoalGeneratorConfig | None = None) -> SiteSnapshot:
|
|
81
|
+
"""Capture the start page and pages linked from its navigation."""
|
|
82
|
+
from fusiontest.desktop.playwright_adapter import PlaywrightAdapter # noqa: PLC0415
|
|
83
|
+
|
|
84
|
+
cfg = config or GoalGeneratorConfig()
|
|
85
|
+
snapshot = SiteSnapshot(url=url)
|
|
86
|
+
adapter = PlaywrightAdapter(headless=True)
|
|
87
|
+
try:
|
|
88
|
+
adapter.start(url)
|
|
89
|
+
snapshot.pages.append((url, _page_text(adapter, cfg)))
|
|
90
|
+
for link in adapter.site_links(limit=cfg.max_pages - 1):
|
|
91
|
+
try:
|
|
92
|
+
adapter.navigate(link)
|
|
93
|
+
snapshot.pages.append((link, _page_text(adapter, cfg)))
|
|
94
|
+
except Exception as exc: # noqa: BLE001 — one bad page shouldn't sink discovery
|
|
95
|
+
logger.warning("Discovery: skipping %s (%s)", link, exc)
|
|
96
|
+
except Exception as exc: # noqa: BLE001
|
|
97
|
+
raise GoalGenerationError(f"Could not open {url}: {exc}") from exc
|
|
98
|
+
finally:
|
|
99
|
+
try:
|
|
100
|
+
adapter.stop()
|
|
101
|
+
except Exception: # noqa: BLE001
|
|
102
|
+
pass
|
|
103
|
+
logger.info("Discovery snapshot of %s: %d page(s), %d chars",
|
|
104
|
+
url, len(snapshot.pages), len(snapshot.content))
|
|
105
|
+
return snapshot
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _page_text(adapter, cfg: GoalGeneratorConfig) -> str:
|
|
109
|
+
_, screen = adapter.get_screen()
|
|
110
|
+
return screen[: cfg.max_chars_per_page]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# ── Prompt ────────────────────────────────────────────────────────────────────
|
|
114
|
+
|
|
115
|
+
_SYSTEM_PROMPT = """\
|
|
116
|
+
You are a senior QA engineer writing automated UI test goals for a specific website.
|
|
117
|
+
You are given the content of the site's pages as an accessibility snapshot:
|
|
118
|
+
interactive elements (links, buttons, fields with their labels) and page text.
|
|
119
|
+
|
|
120
|
+
Rules:
|
|
121
|
+
- Only write goals for features that appear in the provided content. Do NOT
|
|
122
|
+
assume features that aren't shown — no login, sign-up, search, checkout,
|
|
123
|
+
account, or consent-checkbox goals unless those controls appear in the content.
|
|
124
|
+
- Each goal is one complete, self-contained action sentence a tester can follow,
|
|
125
|
+
e.g. "Navigate to the Pricing page and verify all three plan tiers are displayed".
|
|
126
|
+
- Refer to things by their visible text (link, button, heading or field label).
|
|
127
|
+
No CSS selectors, IDs, or code.
|
|
128
|
+
- Do not submit real forms, send messages, or make purchases: this may be a
|
|
129
|
+
live production site. For forms, verify the fields and client-side validation
|
|
130
|
+
(e.g. required-field errors) instead of submitting successfully.
|
|
131
|
+
- Cover navigation, key content, forms and validation — as far as the site has them.
|
|
132
|
+
- evidence: a short EXACT quote (3-80 characters) copied from the provided
|
|
133
|
+
content that the goal depends on — a heading, link, button or field label.
|
|
134
|
+
|
|
135
|
+
Return ONLY a JSON object: {"goals": [ ... ]}. Each goal has exactly these keys:
|
|
136
|
+
name (short 5-8 word title)
|
|
137
|
+
text (the full goal sentence)
|
|
138
|
+
category ("critical_path" | "happy_path" | "edge_cases")
|
|
139
|
+
confidence (number 0.0-1.0: how clearly the content supports this goal)
|
|
140
|
+
evidence (exact quote from the content)
|
|
141
|
+
"""
|
|
142
|
+
|
|
143
|
+
_REQUEST = LLMRequest(
|
|
144
|
+
system=_SYSTEM_PROMPT,
|
|
145
|
+
reminder='\n\nIMPORTANT: respond with ONLY the JSON object {"goals": [...]}. No other text.',
|
|
146
|
+
json_object=True,
|
|
147
|
+
empty="{}",
|
|
148
|
+
# 20 goals of JSON plus the model's hidden reasoning/thinking.
|
|
149
|
+
max_output_tokens=8192,
|
|
150
|
+
)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _build_user_message(snapshot: SiteSnapshot, max_goals: int) -> str:
|
|
154
|
+
return (
|
|
155
|
+
f"Site: {snapshot.url}\n\n"
|
|
156
|
+
f"SITE CONTENT ({len(snapshot.pages)} page(s)):\n{snapshot.content}\n\n"
|
|
157
|
+
f"Write up to {max_goals} test goals for this site, grounded in the content above, "
|
|
158
|
+
f"across critical_path (the site's most important flows), happy_path (navigation "
|
|
159
|
+
f"and features) and edge_cases (validation and error states)."
|
|
160
|
+
)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
# ── Parsing ───────────────────────────────────────────────────────────────────
|
|
164
|
+
|
|
165
|
+
def _derive_name(text: str) -> str:
|
|
166
|
+
"""Fallback: condense a goal sentence into a short 5-8 word title."""
|
|
167
|
+
cleaned = re.sub(
|
|
168
|
+
r"^(verify\s+(that\s+)?|navigate\s+to\s+(the\s+)?|check\s+(that\s+)?|"
|
|
169
|
+
r"attempt\s+to\s+|click\s+|submit\s+|open\s+(the\s+)?|ensure\s+(that\s+)?)",
|
|
170
|
+
"",
|
|
171
|
+
text,
|
|
172
|
+
flags=re.I,
|
|
173
|
+
).strip()
|
|
174
|
+
words = cleaned.split()
|
|
175
|
+
name = " ".join(words[:7])
|
|
176
|
+
if len(words) > 7:
|
|
177
|
+
name += "…"
|
|
178
|
+
return name.capitalize()
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def _parse_goals(raw: str, max_goals: int, content: str) -> list[DiscoveredGoal]:
|
|
182
|
+
"""Parse the LLM's JSON and keep only goals whose evidence is in the content."""
|
|
183
|
+
raw = re.sub(r"^```[a-z]*\n?|\n?```$", "", raw.strip(), flags=re.M)
|
|
184
|
+
try:
|
|
185
|
+
data = json.loads(raw)
|
|
186
|
+
except json.JSONDecodeError as exc:
|
|
187
|
+
raise GoalGenerationError(f"The model's answer wasn't valid JSON: {exc}") from exc
|
|
188
|
+
items = data if isinstance(data, list) else data.get("goals", [])
|
|
189
|
+
|
|
190
|
+
goals: list[DiscoveredGoal] = []
|
|
191
|
+
dropped: list[str] = []
|
|
192
|
+
for item in items:
|
|
193
|
+
if not isinstance(item, dict):
|
|
194
|
+
continue
|
|
195
|
+
text = str(item.get("text", "")).strip()
|
|
196
|
+
if not text:
|
|
197
|
+
continue
|
|
198
|
+
evidence = select_evidence([str(item.get("evidence", ""))], content, limit=1)
|
|
199
|
+
if not evidence:
|
|
200
|
+
dropped.append(text)
|
|
201
|
+
continue
|
|
202
|
+
category = item.get("category", "happy_path")
|
|
203
|
+
if category not in _CATEGORIES:
|
|
204
|
+
category = "happy_path"
|
|
205
|
+
try:
|
|
206
|
+
confidence = max(0.0, min(1.0, float(item.get("confidence", 0.8))))
|
|
207
|
+
except (TypeError, ValueError):
|
|
208
|
+
confidence = 0.8
|
|
209
|
+
goals.append(DiscoveredGoal(
|
|
210
|
+
id=f"goal-{len(goals):03d}",
|
|
211
|
+
name=str(item.get("name", "")).strip() or _derive_name(text),
|
|
212
|
+
text=text,
|
|
213
|
+
category=category, # type: ignore[arg-type]
|
|
214
|
+
confidence=confidence,
|
|
215
|
+
selected=confidence >= 0.75,
|
|
216
|
+
evidence=evidence[0],
|
|
217
|
+
))
|
|
218
|
+
if len(goals) >= max_goals:
|
|
219
|
+
break
|
|
220
|
+
|
|
221
|
+
if dropped:
|
|
222
|
+
logger.info("Dropped %d goal(s) not grounded in the site content: %s",
|
|
223
|
+
len(dropped), "; ".join(d[:60] for d in dropped))
|
|
224
|
+
return goals
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
# ── Backend choice ────────────────────────────────────────────────────────────
|
|
228
|
+
|
|
229
|
+
_BACKEND_KEYS = {"groq": "GROQ_API_KEY", "claude": "ANTHROPIC_API_KEY",
|
|
230
|
+
"gemini": "GEMINI_API_KEY", "gpt4o": "OPENAI_API_KEY"}
|
|
231
|
+
_BACKEND_ALIASES = {"openai": "gpt4o"}
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _primary_backend(preferred: str | None) -> str:
|
|
235
|
+
"""The user's backend if usable, else the first one with credentials."""
|
|
236
|
+
choice = (preferred or "").strip().lower()
|
|
237
|
+
choice = _BACKEND_ALIASES.get(choice, choice)
|
|
238
|
+
if choice in _BACKEND_KEYS and os.getenv(_BACKEND_KEYS[choice]):
|
|
239
|
+
return choice
|
|
240
|
+
if choice == "ollama" and _is_ollama_reachable():
|
|
241
|
+
return "ollama"
|
|
242
|
+
for backend, key in _BACKEND_KEYS.items():
|
|
243
|
+
if os.getenv(key):
|
|
244
|
+
return backend
|
|
245
|
+
if _is_ollama_reachable():
|
|
246
|
+
return "ollama"
|
|
247
|
+
raise GoalGenerationError(
|
|
248
|
+
"No LLM configured: set GROQ_API_KEY, ANTHROPIC_API_KEY, GEMINI_API_KEY or "
|
|
249
|
+
"OPENAI_API_KEY (or run Ollama)."
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
# ── Public interface ──────────────────────────────────────────────────────────
|
|
254
|
+
|
|
255
|
+
class GoalGenerator:
|
|
256
|
+
"""Generates prioritised, site-specific test goals for a given URL."""
|
|
257
|
+
|
|
258
|
+
def __init__(self, config: GoalGeneratorConfig | None = None) -> None:
|
|
259
|
+
self.config = config or GoalGeneratorConfig()
|
|
260
|
+
# Set after generate(): "backend:model" that produced the goals.
|
|
261
|
+
self.last_backend_used: str = ""
|
|
262
|
+
|
|
263
|
+
def generate(
|
|
264
|
+
self,
|
|
265
|
+
url: str,
|
|
266
|
+
preferred_backend: str | None = None,
|
|
267
|
+
snapshot: SiteSnapshot | None = None,
|
|
268
|
+
) -> list[DiscoveredGoal]:
|
|
269
|
+
"""Goals grounded in the site's content. Raises GoalGenerationError."""
|
|
270
|
+
snapshot = snapshot or snapshot_site(url, self.config)
|
|
271
|
+
if not snapshot.content.strip():
|
|
272
|
+
raise GoalGenerationError(f"No content could be read from {url}")
|
|
273
|
+
|
|
274
|
+
caller = LLMCaller(_primary_backend(preferred_backend), temperature=self.config.temperature)
|
|
275
|
+
try:
|
|
276
|
+
raw, _ = caller._call_llm(_build_user_message(snapshot, self.config.max_goals), _REQUEST)
|
|
277
|
+
except LLMUnavailableError as exc:
|
|
278
|
+
raise GoalGenerationError(str(exc)) from exc
|
|
279
|
+
self.last_backend_used = caller.last_backend
|
|
280
|
+
|
|
281
|
+
goals = _parse_goals(raw, self.config.max_goals, snapshot.content)
|
|
282
|
+
if not goals:
|
|
283
|
+
raise GoalGenerationError(
|
|
284
|
+
"The model didn't propose any goals grounded in the site's content."
|
|
285
|
+
)
|
|
286
|
+
logger.info("Generated %d goal(s) for %s via %s", len(goals), url, self.last_backend_used)
|
|
287
|
+
return goals
|