fusiontest 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (41) hide show
  1. {fusiontest-0.2.0 → fusiontest-0.2.2}/PKG-INFO +1 -1
  2. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/action_model.py +19 -87
  3. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/desktop/playwright_adapter.py +25 -0
  4. fusiontest-0.2.2/fusiontest/discovery/goal_generator.py +287 -0
  5. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/PKG-INFO +1 -1
  6. {fusiontest-0.2.0 → fusiontest-0.2.2}/pyproject.toml +1 -1
  7. fusiontest-0.2.0/fusiontest/discovery/goal_generator.py +0 -696
  8. {fusiontest-0.2.0 → fusiontest-0.2.2}/README.md +0 -0
  9. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/__init__.py +0 -0
  10. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/cli.py +0 -0
  11. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/__init__.py +0 -0
  12. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/goal_verifier.py +0 -0
  13. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/replay.py +0 -0
  14. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/runner.py +0 -0
  15. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/screen_parser.py +0 -0
  16. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/secrets.py +0 -0
  17. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/core/tokens.py +0 -0
  18. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/desktop/__init__.py +0 -0
  19. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/desktop/macos_adapter.py +0 -0
  20. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/desktop/windows_adapter.py +0 -0
  21. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/discovery/__init__.py +0 -0
  22. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/guardrails/__init__.py +0 -0
  23. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/guardrails/engine.py +0 -0
  24. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/mobile/__init__.py +0 -0
  25. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/mobile/android_adapter.py +0 -0
  26. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/mobile/ios_adapter.py +0 -0
  27. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/recording/__init__.py +0 -0
  28. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/recording/recorder.py +0 -0
  29. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/reporting/__init__.py +0 -0
  30. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/reporting/reporter.py +0 -0
  31. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/training/__init__.py +0 -0
  32. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/training/data_collector.py +0 -0
  33. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/training/dataset_builder.py +0 -0
  34. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest/training/trainer.py +0 -0
  35. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/SOURCES.txt +0 -0
  36. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/dependency_links.txt +0 -0
  37. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/entry_points.txt +0 -0
  38. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/not-zip-safe +0 -0
  39. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/requires.txt +0 -0
  40. {fusiontest-0.2.0 → fusiontest-0.2.2}/fusiontest.egg-info/top_level.txt +0 -0
  41. {fusiontest-0.2.0 → fusiontest-0.2.2}/setup.cfg +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: fusiontest
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: AI-powered UI testing for mobile and desktop — by FusionLeap.io
5
5
  Author-email: FusionLeap <hello@fusionleap.io>
6
6
  License: MIT
@@ -73,6 +73,7 @@ def _error_status_code(err: Exception) -> int | None:
73
73
  _BACKEND_SPECIFIC_STATUS_CODES = {401, 403, 404}
74
74
  _BACKEND_SPECIFIC_TOKENS = (
75
75
  "model_not_found", "does not exist", "do not have access",
76
+ "this model only supports", "is not supported for this model", # wrong kind of model
76
77
  "invalid api key", "invalid_api_key", "is not set",
77
78
  "api key not valid", "api_key_invalid", # Google returns these as HTTP 400
78
79
  )
@@ -185,52 +186,6 @@ class Action:
185
186
 
186
187
  ActionBackend = Literal["mpnet", "gpt4o", "claude", "gemini", "ollama", "groq"]
187
188
 
188
- # ── Gemini model auto-detection ───────────────────────────────────────────────
189
-
190
- _gemini_model_cache: str | None = None
191
- _GEMINI_PREFERRED = ["flash", "pro"] # tier preference, newest date wins
192
-
193
-
194
- def _resolve_gemini_model(client) -> str:
195
- """Return the best available Gemini model for this API key (cached).
196
-
197
- Preference: newest flash > older flash > newest pro > older pro.
198
- Sort is DESCENDING by name so '2.5' beats '2.0' beats '1.5'.
199
-
200
- Accepts a ``google.genai.Client`` instance from the new unified SDK
201
- (``pip install google-genai``). The deprecated ``google-generativeai``
202
- package is no longer supported.
203
- """
204
- global _gemini_model_cache # noqa: PLW0603
205
- if _gemini_model_cache is not None:
206
- return _gemini_model_cache
207
-
208
- try:
209
- all_models: list[str] = []
210
- for m in client.models.list():
211
- name = (getattr(m, "name", "") or "").replace("models/", "")
212
- if not name or "gemini" not in name.lower():
213
- continue
214
- if not any(t in name.lower() for t in ("flash", "pro")):
215
- continue
216
- all_models.append(name)
217
-
218
- # Newest flash first, then newest pro — reverse=True so '2.5' > '2.0' > '1.5'
219
- flash = sorted([m for m in all_models if "flash" in m.lower()], reverse=True)
220
- pro = sorted([m for m in all_models if "flash" not in m.lower()], reverse=True)
221
- ordered = flash + pro
222
-
223
- if ordered:
224
- _gemini_model_cache = ordered[0]
225
- logger.info("Auto-selected Gemini model: %s", _gemini_model_cache)
226
- return _gemini_model_cache
227
- except Exception as exc:
228
- logger.warning("Could not list Gemini models: %s — falling back to gemini-2.5-flash", exc)
229
-
230
- _gemini_model_cache = "gemini-2.5-flash"
231
- return _gemini_model_cache
232
-
233
-
234
189
  # ── Fallback-chain construction ───────────────────────────────────────────────
235
190
 
236
191
  # Model "" means "resolve the default for this backend at call time"
@@ -243,6 +198,11 @@ _DEFAULT_CHAIN_MODEL = ""
243
198
  GROQ_DEFAULT_MODEL = "openai/gpt-oss-120b"
244
199
  GROQ_FAST_MODEL = "openai/gpt-oss-20b" # separate rate-limit bucket
245
200
 
201
+ # Pinned Gemini model; GEMINI_MODEL overrides it. Not auto-selected from the
202
+ # model list — names like *-tts, *-image or Interactions-only models sort
203
+ # above the general-purpose one and reject normal requests.
204
+ GEMINI_DEFAULT_MODEL = "gemini-2.5-flash"
205
+
246
206
 
247
207
  # Output-token cap for LLM calls. Reasoning models (Groq gpt-oss, Gemini 2.5)
248
208
  # spend part of this on hidden reasoning before any answer, so a tight cap
@@ -264,7 +224,7 @@ def default_model_for_backend(backend: str) -> str:
264
224
  if backend == "claude":
265
225
  return "claude-sonnet-4-6"
266
226
  if backend == "gemini":
267
- return _DEFAULT_CHAIN_MODEL # resolved lazily
227
+ return os.getenv("GEMINI_MODEL", GEMINI_DEFAULT_MODEL)
268
228
  if backend == "ollama":
269
229
  return os.getenv("OLLAMA_MODEL", "qwen2.5:7b")
270
230
  if backend == "groq":
@@ -334,7 +294,7 @@ def _build_default_fallback_chain(primary: str) -> list[tuple[str, str]]:
334
294
  seen_backends = {primary}
335
295
 
336
296
  if "gemini" not in seen_backends and os.getenv("GEMINI_API_KEY"):
337
- chain.append(("gemini", _DEFAULT_CHAIN_MODEL))
297
+ chain.append(("gemini", default_model_for_backend("gemini")))
338
298
  seen_backends.add("gemini")
339
299
 
340
300
  if "groq" not in seen_backends and os.getenv("GROQ_API_KEY"):
@@ -360,6 +320,7 @@ class LLMRequest:
360
320
  reminder: str = "" # appended to the prompt for backends without a JSON mode
361
321
  json_object: bool = False # OpenAI-compatible response_format={"type": "json_object"}
362
322
  empty: str = "" # returned when the model produces no text
323
+ max_output_tokens: int = LLM_MAX_OUTPUT_TOKENS # includes hidden reasoning/thinking
363
324
 
364
325
 
365
326
  class LLMCaller:
@@ -381,6 +342,8 @@ class LLMCaller:
381
342
  self.temperature = temperature
382
343
  self.max_retries_per_backend = max_retries_per_backend
383
344
  self.backoff_seconds = backoff_seconds
345
+ # "backend:model" of the entry that answered the last call.
346
+ self.last_backend: str = ""
384
347
  # Per-(backend, model) client cache. Lazily populated.
385
348
  self._clients: dict[tuple[str, str], tuple[Any, str]] = {}
386
349
 
@@ -416,11 +379,8 @@ class LLMCaller:
416
379
  def _get_client(self, backend: str, model: str) -> tuple[Any, str]:
417
380
  """
418
381
  Lazily create (and cache) the client for a given backend+model.
419
- Returns (client, resolved_model). The resolved_model may differ from
420
- the requested `model` when Gemini auto-resolution is used — we cache
421
- (client, resolved_model) together so subsequent calls keep using the
422
- resolved name instead of the empty sentinel that `default_model_for_backend`
423
- returns for Gemini.
382
+ Returns (client, resolved_model); an empty `model` means the backend's
383
+ default (see default_model_for_backend).
424
384
  """
425
385
  cache_key = (backend, model)
426
386
  if cache_key in self._clients:
@@ -493,12 +453,6 @@ class LLMCaller:
493
453
  "export GEMINI_API_KEY=AIza..."
494
454
  )
495
455
  client = genai.Client(api_key=api_key)
496
- if not resolved_model:
497
- resolved_model = _resolve_gemini_model(client)
498
- logger.info(
499
- "Gemini client created; resolved model=%r (cache_key=%r)",
500
- resolved_model, cache_key,
501
- )
502
456
 
503
457
  else:
504
458
  raise RuntimeError(f"Unknown backend: {backend}")
@@ -542,7 +496,9 @@ class LLMCaller:
542
496
 
543
497
  for attempt in range(self.max_retries_per_backend + 1):
544
498
  try:
545
- return self._call_llm_single(prompt, backend, model, request)
499
+ answer = self._call_llm_single(prompt, backend, model, request)
500
+ self.last_backend = f"{backend}:{model or default_model_for_backend(backend)}"
501
+ return answer
546
502
  except Exception as err: # noqa: BLE001
547
503
  last_error = err
548
504
  transient = _is_transient_error(err)
@@ -613,30 +569,6 @@ class LLMCaller:
613
569
  """
614
570
  client, resolved_model = self._get_client(backend, model)
615
571
 
616
- # Defensive guard: the Gemini SDK raises a cryptic
617
- # ``ValueError: model is required.`` when ``model=""`` reaches
618
- # ``generate_content``. This has happened in prod when a cache path
619
- # or future refactor lets the sentinel `_DEFAULT_CHAIN_MODEL` leak
620
- # through. Force a re-resolve and raise a loud, actionable error if
621
- # that still fails — it's better to fail the attempt with context
622
- # than to hand an empty string to the SDK.
623
- if backend == "gemini" and not resolved_model:
624
- logger.error(
625
- "Gemini resolved_model is empty after _get_client(backend=%r, model=%r). "
626
- "Attempting emergency re-resolve.",
627
- backend, model,
628
- )
629
- resolved_model = _resolve_gemini_model(client)
630
- if not resolved_model:
631
- raise RuntimeError(
632
- "Could not resolve a Gemini model name. "
633
- "This should never happen — _resolve_gemini_model falls back to "
634
- "'gemini-2.5-flash' on listing failures. Check for a corrupted "
635
- "_gemini_model_cache."
636
- )
637
- # Self-heal the cache so the next call doesn't hit this path.
638
- self._clients[(backend, model)] = (client, resolved_model)
639
-
640
572
  if backend in ("gpt4o", "ollama", "groq"):
641
573
  resp = client.chat.completions.create(
642
574
  model=resolved_model,
@@ -645,7 +577,7 @@ class LLMCaller:
645
577
  {"role": "user", "content": prompt},
646
578
  ],
647
579
  temperature=self.temperature,
648
- max_tokens=LLM_MAX_OUTPUT_TOKENS,
580
+ max_tokens=request.max_output_tokens,
649
581
  **({"response_format": {"type": "json_object"}} if request.json_object else {}),
650
582
  **openai_compat_extra_args(resolved_model),
651
583
  )
@@ -658,7 +590,7 @@ class LLMCaller:
658
590
  if backend == "claude":
659
591
  resp = client.messages.create(
660
592
  model=resolved_model,
661
- max_tokens=LLM_MAX_OUTPUT_TOKENS,
593
+ max_tokens=request.max_output_tokens,
662
594
  system=request.system,
663
595
  messages=[{"role": "user", "content": prompt + request.reminder}],
664
596
  )
@@ -676,7 +608,7 @@ class LLMCaller:
676
608
  config=genai_types.GenerateContentConfig(
677
609
  system_instruction=request.system,
678
610
  temperature=self.temperature,
679
- max_output_tokens=LLM_MAX_OUTPUT_TOKENS,
611
+ max_output_tokens=request.max_output_tokens,
680
612
  ),
681
613
  )
682
614
  meta = getattr(resp, "usage_metadata", None)
@@ -140,6 +140,31 @@ class PlaywrightAdapter:
140
140
 
141
141
  # ── Screen reading ─────────────────────────────────────────────────────────
142
142
 
143
+ def site_links(self, limit: int = 10) -> list[str]:
144
+ """Same-origin page URLs linked from the header/navigation (fallback: any link)."""
145
+ return self._get_loop().run_until_complete(self._async_site_links(limit))
146
+
147
+ async def _async_site_links(self, limit: int) -> list[str]:
148
+ from urllib.parse import urldefrag, urljoin, urlsplit # noqa: PLC0415
149
+
150
+ hrefs: list[str] = await self._page.locator("header a[href], nav a[href]").evaluate_all(
151
+ "els => els.map(e => e.getAttribute('href'))"
152
+ )
153
+ if not hrefs:
154
+ hrefs = await self._page.locator("a[href]").evaluate_all(
155
+ "els => els.map(e => e.getAttribute('href'))"
156
+ )
157
+ base = self._page.url
158
+ origin = urlsplit(base).netloc
159
+ links: list[str] = []
160
+ for href in hrefs:
161
+ if not href or href.startswith(("mailto:", "tel:", "javascript:")):
162
+ continue
163
+ url = urldefrag(urljoin(base, href)).url.rstrip("/")
164
+ if urlsplit(url).netloc == origin and url != base.rstrip("/") and url not in links:
165
+ links.append(url)
166
+ return links[:limit]
167
+
143
168
  def current_url(self) -> str:
144
169
  """URL of the current page ("" before start). Used for secret site binding."""
145
170
  return self._page.url if self._page else ""
@@ -0,0 +1,287 @@
1
+ """
2
+ fusiontest/discovery/goal_generator.py
3
+
4
+ Autonomous goal discovery: propose test goals for a site from what it
5
+ actually contains.
6
+
7
+ 1. ``snapshot_site`` opens the site in a headless browser and captures the
8
+ start page plus up to a few pages linked from its navigation, using the
9
+ same screen representation the runner uses (headings, text, form fields,
10
+ buttons).
11
+ 2. ``GoalGenerator.generate`` asks an LLM (through the shared fallback chain)
12
+ for goals, each quoting *evidence* from the captured pages.
13
+ 3. Goals whose evidence isn't on those pages are dropped — the model may not
14
+ invent features (login, search, consent checkboxes…) the site doesn't have.
15
+
16
+ There is no template fallback: if no LLM can answer, generation fails with the
17
+ underlying error rather than presenting generic goals as "discovered".
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ import json
23
+ import logging
24
+ import os
25
+ import re
26
+ from dataclasses import dataclass, field
27
+ from typing import Literal
28
+ from urllib.parse import urlsplit
29
+
30
+ from fusiontest.core.action_model import (
31
+ LLMCaller,
32
+ LLMRequest,
33
+ LLMUnavailableError,
34
+ _is_ollama_reachable,
35
+ )
36
+ from fusiontest.core.replay import select_evidence
37
+
38
+ logger = logging.getLogger(__name__)
39
+
40
+ GoalCategory = Literal["critical_path", "happy_path", "edge_cases"]
41
+ _CATEGORIES = ("critical_path", "happy_path", "edge_cases")
42
+
43
+
44
+ class GoalGenerationError(RuntimeError):
45
+ """Goals couldn't be generated (site unreachable, no LLM, unusable answer)."""
46
+
47
+
48
+ @dataclass
49
+ class DiscoveredGoal:
50
+ id: str
51
+ name: str # short 5-8 word title for the YAML output
52
+ text: str
53
+ category: GoalCategory
54
+ confidence: float # 0.0 – 1.0
55
+ selected: bool = True # user can deselect before running
56
+ edited: bool = False # has the user overridden the text?
57
+ evidence: str = "" # quote from the site content the goal relies on
58
+
59
+
60
+ @dataclass
61
+ class GoalGeneratorConfig:
62
+ max_goals: int = 20
63
+ max_pages: int = 8 # start page + pages linked from navigation
64
+ max_chars_per_page: int = 4000
65
+ temperature: float = 0.3
66
+
67
+
68
+ @dataclass
69
+ class SiteSnapshot:
70
+ url: str
71
+ pages: list[tuple[str, str]] = field(default_factory=list) # (url, screen text)
72
+
73
+ @property
74
+ def content(self) -> str:
75
+ return "\n\n".join(f"=== PAGE {urlsplit(u).path or '/'} ===\n{text}" for u, text in self.pages)
76
+
77
+
78
+ # ── Site snapshot ─────────────────────────────────────────────────────────────
79
+
80
+ def snapshot_site(url: str, config: GoalGeneratorConfig | None = None) -> SiteSnapshot:
81
+ """Capture the start page and pages linked from its navigation."""
82
+ from fusiontest.desktop.playwright_adapter import PlaywrightAdapter # noqa: PLC0415
83
+
84
+ cfg = config or GoalGeneratorConfig()
85
+ snapshot = SiteSnapshot(url=url)
86
+ adapter = PlaywrightAdapter(headless=True)
87
+ try:
88
+ adapter.start(url)
89
+ snapshot.pages.append((url, _page_text(adapter, cfg)))
90
+ for link in adapter.site_links(limit=cfg.max_pages - 1):
91
+ try:
92
+ adapter.navigate(link)
93
+ snapshot.pages.append((link, _page_text(adapter, cfg)))
94
+ except Exception as exc: # noqa: BLE001 — one bad page shouldn't sink discovery
95
+ logger.warning("Discovery: skipping %s (%s)", link, exc)
96
+ except Exception as exc: # noqa: BLE001
97
+ raise GoalGenerationError(f"Could not open {url}: {exc}") from exc
98
+ finally:
99
+ try:
100
+ adapter.stop()
101
+ except Exception: # noqa: BLE001
102
+ pass
103
+ logger.info("Discovery snapshot of %s: %d page(s), %d chars",
104
+ url, len(snapshot.pages), len(snapshot.content))
105
+ return snapshot
106
+
107
+
108
+ def _page_text(adapter, cfg: GoalGeneratorConfig) -> str:
109
+ _, screen = adapter.get_screen()
110
+ return screen[: cfg.max_chars_per_page]
111
+
112
+
113
+ # ── Prompt ────────────────────────────────────────────────────────────────────
114
+
115
+ _SYSTEM_PROMPT = """\
116
+ You are a senior QA engineer writing automated UI test goals for a specific website.
117
+ You are given the content of the site's pages as an accessibility snapshot:
118
+ interactive elements (links, buttons, fields with their labels) and page text.
119
+
120
+ Rules:
121
+ - Only write goals for features that appear in the provided content. Do NOT
122
+ assume features that aren't shown — no login, sign-up, search, checkout,
123
+ account, or consent-checkbox goals unless those controls appear in the content.
124
+ - Each goal is one complete, self-contained action sentence a tester can follow,
125
+ e.g. "Navigate to the Pricing page and verify all three plan tiers are displayed".
126
+ - Refer to things by their visible text (link, button, heading or field label).
127
+ No CSS selectors, IDs, or code.
128
+ - Do not submit real forms, send messages, or make purchases: this may be a
129
+ live production site. For forms, verify the fields and client-side validation
130
+ (e.g. required-field errors) instead of submitting successfully.
131
+ - Cover navigation, key content, forms and validation — as far as the site has them.
132
+ - evidence: a short EXACT quote (3-80 characters) copied from the provided
133
+ content that the goal depends on — a heading, link, button or field label.
134
+
135
+ Return ONLY a JSON object: {"goals": [ ... ]}. Each goal has exactly these keys:
136
+ name (short 5-8 word title)
137
+ text (the full goal sentence)
138
+ category ("critical_path" | "happy_path" | "edge_cases")
139
+ confidence (number 0.0-1.0: how clearly the content supports this goal)
140
+ evidence (exact quote from the content)
141
+ """
142
+
143
+ _REQUEST = LLMRequest(
144
+ system=_SYSTEM_PROMPT,
145
+ reminder='\n\nIMPORTANT: respond with ONLY the JSON object {"goals": [...]}. No other text.',
146
+ json_object=True,
147
+ empty="{}",
148
+ # 20 goals of JSON plus the model's hidden reasoning/thinking.
149
+ max_output_tokens=8192,
150
+ )
151
+
152
+
153
+ def _build_user_message(snapshot: SiteSnapshot, max_goals: int) -> str:
154
+ return (
155
+ f"Site: {snapshot.url}\n\n"
156
+ f"SITE CONTENT ({len(snapshot.pages)} page(s)):\n{snapshot.content}\n\n"
157
+ f"Write up to {max_goals} test goals for this site, grounded in the content above, "
158
+ f"across critical_path (the site's most important flows), happy_path (navigation "
159
+ f"and features) and edge_cases (validation and error states)."
160
+ )
161
+
162
+
163
+ # ── Parsing ───────────────────────────────────────────────────────────────────
164
+
165
+ def _derive_name(text: str) -> str:
166
+ """Fallback: condense a goal sentence into a short 5-8 word title."""
167
+ cleaned = re.sub(
168
+ r"^(verify\s+(that\s+)?|navigate\s+to\s+(the\s+)?|check\s+(that\s+)?|"
169
+ r"attempt\s+to\s+|click\s+|submit\s+|open\s+(the\s+)?|ensure\s+(that\s+)?)",
170
+ "",
171
+ text,
172
+ flags=re.I,
173
+ ).strip()
174
+ words = cleaned.split()
175
+ name = " ".join(words[:7])
176
+ if len(words) > 7:
177
+ name += "…"
178
+ return name.capitalize()
179
+
180
+
181
+ def _parse_goals(raw: str, max_goals: int, content: str) -> list[DiscoveredGoal]:
182
+ """Parse the LLM's JSON and keep only goals whose evidence is in the content."""
183
+ raw = re.sub(r"^```[a-z]*\n?|\n?```$", "", raw.strip(), flags=re.M)
184
+ try:
185
+ data = json.loads(raw)
186
+ except json.JSONDecodeError as exc:
187
+ raise GoalGenerationError(f"The model's answer wasn't valid JSON: {exc}") from exc
188
+ items = data if isinstance(data, list) else data.get("goals", [])
189
+
190
+ goals: list[DiscoveredGoal] = []
191
+ dropped: list[str] = []
192
+ for item in items:
193
+ if not isinstance(item, dict):
194
+ continue
195
+ text = str(item.get("text", "")).strip()
196
+ if not text:
197
+ continue
198
+ evidence = select_evidence([str(item.get("evidence", ""))], content, limit=1)
199
+ if not evidence:
200
+ dropped.append(text)
201
+ continue
202
+ category = item.get("category", "happy_path")
203
+ if category not in _CATEGORIES:
204
+ category = "happy_path"
205
+ try:
206
+ confidence = max(0.0, min(1.0, float(item.get("confidence", 0.8))))
207
+ except (TypeError, ValueError):
208
+ confidence = 0.8
209
+ goals.append(DiscoveredGoal(
210
+ id=f"goal-{len(goals):03d}",
211
+ name=str(item.get("name", "")).strip() or _derive_name(text),
212
+ text=text,
213
+ category=category, # type: ignore[arg-type]
214
+ confidence=confidence,
215
+ selected=confidence >= 0.75,
216
+ evidence=evidence[0],
217
+ ))
218
+ if len(goals) >= max_goals:
219
+ break
220
+
221
+ if dropped:
222
+ logger.info("Dropped %d goal(s) not grounded in the site content: %s",
223
+ len(dropped), "; ".join(d[:60] for d in dropped))
224
+ return goals
225
+
226
+
227
+ # ── Backend choice ────────────────────────────────────────────────────────────
228
+
229
+ _BACKEND_KEYS = {"groq": "GROQ_API_KEY", "claude": "ANTHROPIC_API_KEY",
230
+ "gemini": "GEMINI_API_KEY", "gpt4o": "OPENAI_API_KEY"}
231
+ _BACKEND_ALIASES = {"openai": "gpt4o"}
232
+
233
+
234
+ def _primary_backend(preferred: str | None) -> str:
235
+ """The user's backend if usable, else the first one with credentials."""
236
+ choice = (preferred or "").strip().lower()
237
+ choice = _BACKEND_ALIASES.get(choice, choice)
238
+ if choice in _BACKEND_KEYS and os.getenv(_BACKEND_KEYS[choice]):
239
+ return choice
240
+ if choice == "ollama" and _is_ollama_reachable():
241
+ return "ollama"
242
+ for backend, key in _BACKEND_KEYS.items():
243
+ if os.getenv(key):
244
+ return backend
245
+ if _is_ollama_reachable():
246
+ return "ollama"
247
+ raise GoalGenerationError(
248
+ "No LLM configured: set GROQ_API_KEY, ANTHROPIC_API_KEY, GEMINI_API_KEY or "
249
+ "OPENAI_API_KEY (or run Ollama)."
250
+ )
251
+
252
+
253
+ # ── Public interface ──────────────────────────────────────────────────────────
254
+
255
+ class GoalGenerator:
256
+ """Generates prioritised, site-specific test goals for a given URL."""
257
+
258
+ def __init__(self, config: GoalGeneratorConfig | None = None) -> None:
259
+ self.config = config or GoalGeneratorConfig()
260
+ # Set after generate(): "backend:model" that produced the goals.
261
+ self.last_backend_used: str = ""
262
+
263
+ def generate(
264
+ self,
265
+ url: str,
266
+ preferred_backend: str | None = None,
267
+ snapshot: SiteSnapshot | None = None,
268
+ ) -> list[DiscoveredGoal]:
269
+ """Goals grounded in the site's content. Raises GoalGenerationError."""
270
+ snapshot = snapshot or snapshot_site(url, self.config)
271
+ if not snapshot.content.strip():
272
+ raise GoalGenerationError(f"No content could be read from {url}")
273
+
274
+ caller = LLMCaller(_primary_backend(preferred_backend), temperature=self.config.temperature)
275
+ try:
276
+ raw, _ = caller._call_llm(_build_user_message(snapshot, self.config.max_goals), _REQUEST)
277
+ except LLMUnavailableError as exc:
278
+ raise GoalGenerationError(str(exc)) from exc
279
+ self.last_backend_used = caller.last_backend
280
+
281
+ goals = _parse_goals(raw, self.config.max_goals, snapshot.content)
282
+ if not goals:
283
+ raise GoalGenerationError(
284
+ "The model didn't propose any goals grounded in the site's content."
285
+ )
286
+ logger.info("Generated %d goal(s) for %s via %s", len(goals), url, self.last_backend_used)
287
+ return goals
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: fusiontest
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: AI-powered UI testing for mobile and desktop — by FusionLeap.io
5
5
  Author-email: FusionLeap <hello@fusionleap.io>
6
6
  License: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "fusiontest"
7
- version = "0.2.0"
7
+ version = "0.2.2"
8
8
  description = "AI-powered UI testing for mobile and desktop — by FusionLeap.io"
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.10"