fusiontest 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {fusiontest-0.2.2 → fusiontest-0.2.3}/PKG-INFO +1 -1
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/core/action_model.py +56 -1
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/core/runner.py +14 -3
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/core/tokens.py +6 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/discovery/goal_generator.py +1 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest.egg-info/PKG-INFO +1 -1
- {fusiontest-0.2.2 → fusiontest-0.2.3}/pyproject.toml +1 -1
- {fusiontest-0.2.2 → fusiontest-0.2.3}/README.md +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/__init__.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/cli.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/core/__init__.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/core/goal_verifier.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/core/replay.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/core/screen_parser.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/core/secrets.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/desktop/__init__.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/desktop/macos_adapter.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/desktop/playwright_adapter.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/desktop/windows_adapter.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/discovery/__init__.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/guardrails/__init__.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/guardrails/engine.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/mobile/__init__.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/mobile/android_adapter.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/mobile/ios_adapter.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/recording/__init__.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/recording/recorder.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/reporting/__init__.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/reporting/reporter.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/training/__init__.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/training/data_collector.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/training/dataset_builder.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest/training/trainer.py +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest.egg-info/SOURCES.txt +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest.egg-info/dependency_links.txt +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest.egg-info/entry_points.txt +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest.egg-info/not-zip-safe +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest.egg-info/requires.txt +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/fusiontest.egg-info/top_level.txt +0 -0
- {fusiontest-0.2.2 → fusiontest-0.2.3}/setup.cfg +0 -0
|
@@ -210,6 +210,52 @@ GEMINI_DEFAULT_MODEL = "gemini-2.5-flash"
|
|
|
210
210
|
LLM_MAX_OUTPUT_TOKENS = 1024
|
|
211
211
|
|
|
212
212
|
|
|
213
|
+
# Page text the action model sees per step (ADR-007 cost work). It needs every
|
|
214
|
+
# clickable element but only the gist of the page — the verifier, which judges
|
|
215
|
+
# content, gets the full page. Measured on fusionleap.io, page text was ~65% of
|
|
216
|
+
# each ~2.6k-token step prompt.
|
|
217
|
+
_ACTION_CONTENT_CHARS = 1500
|
|
218
|
+
_ACTION_LABEL_CHARS = 80
|
|
219
|
+
_ACTION_ELEMENT_CHARS = 120
|
|
220
|
+
_CONTENT_LINE_RE = re.compile(r'^\s*(\w+): "(.*)"$')
|
|
221
|
+
_ELEMENT_LINE_RE = re.compile(r'^\s*\[\d+\] ')
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def compact_screen_for_action(screen_text: str) -> str:
|
|
225
|
+
"""Trim a screen's page-content section for the per-step action prompt.
|
|
226
|
+
|
|
227
|
+
Every interactive element is kept (the model can only act on what it sees),
|
|
228
|
+
but very long labels — e.g. list items holding whole paragraphs — are cut:
|
|
229
|
+
the model acts by index. In the page content, headings are kept whole,
|
|
230
|
+
other text is cut to 80 characters, and the section stops at ~1,500 characters.
|
|
231
|
+
"""
|
|
232
|
+
lines: list[str] = []
|
|
233
|
+
in_content = False
|
|
234
|
+
used = 0
|
|
235
|
+
for line in screen_text.splitlines():
|
|
236
|
+
if line.startswith("-- "):
|
|
237
|
+
in_content = line.startswith("-- Page content")
|
|
238
|
+
lines.append(line)
|
|
239
|
+
continue
|
|
240
|
+
if not in_content:
|
|
241
|
+
if len(line) > _ACTION_ELEMENT_CHARS and _ELEMENT_LINE_RE.match(line):
|
|
242
|
+
line = line[:_ACTION_ELEMENT_CHARS] + '…"'
|
|
243
|
+
lines.append(line)
|
|
244
|
+
continue
|
|
245
|
+
match = _CONTENT_LINE_RE.match(line)
|
|
246
|
+
if match:
|
|
247
|
+
kind, label = match.groups()
|
|
248
|
+
if kind != "heading" and len(label) > _ACTION_LABEL_CHARS:
|
|
249
|
+
label = label[:_ACTION_LABEL_CHARS] + "…"
|
|
250
|
+
line = f' {kind}: "{label}"'
|
|
251
|
+
if used + len(line) > _ACTION_CONTENT_CHARS:
|
|
252
|
+
lines.append(" … (more page text omitted for brevity)")
|
|
253
|
+
break
|
|
254
|
+
used += len(line)
|
|
255
|
+
lines.append(line)
|
|
256
|
+
return "\n".join(lines)
|
|
257
|
+
|
|
258
|
+
|
|
213
259
|
def openai_compat_extra_args(model: str) -> dict:
|
|
214
260
|
"""Extra chat.completions args for OpenAI-compatible models (Groq/OpenAI/Ollama)."""
|
|
215
261
|
if model.startswith("openai/gpt-oss"):
|
|
@@ -321,6 +367,10 @@ class LLMRequest:
|
|
|
321
367
|
json_object: bool = False # OpenAI-compatible response_format={"type": "json_object"}
|
|
322
368
|
empty: str = "" # returned when the model produces no text
|
|
323
369
|
max_output_tokens: int = LLM_MAX_OUTPUT_TOKENS # includes hidden reasoning/thinking
|
|
370
|
+
# Let reasoning models think before answering. Off for per-step action and
|
|
371
|
+
# verifier calls (short answers; thinking tokens are billed as output and
|
|
372
|
+
# add latency); on for open-ended generation like goal discovery.
|
|
373
|
+
thinking: bool = False
|
|
324
374
|
|
|
325
375
|
|
|
326
376
|
class LLMCaller:
|
|
@@ -609,6 +659,11 @@ class LLMCaller:
|
|
|
609
659
|
system_instruction=request.system,
|
|
610
660
|
temperature=self.temperature,
|
|
611
661
|
max_output_tokens=request.max_output_tokens,
|
|
662
|
+
# Flash models can skip thinking entirely; Pro models can't.
|
|
663
|
+
thinking_config=(
|
|
664
|
+
genai_types.ThinkingConfig(thinking_budget=0)
|
|
665
|
+
if not request.thinking and "flash" in resolved_model else None
|
|
666
|
+
),
|
|
612
667
|
),
|
|
613
668
|
)
|
|
614
669
|
meta = getattr(resp, "usage_metadata", None)
|
|
@@ -820,7 +875,7 @@ class ActionModel(LLMCaller):
|
|
|
820
875
|
pruned_invalid = invalid_actions[-self.max_invalid_actions_in_prompt:]
|
|
821
876
|
sections += ["", "ACTIONS THAT FAILED (do NOT repeat):", *[f" - {a}" for a in pruned_invalid]]
|
|
822
877
|
|
|
823
|
-
trimmed_screen = screen_text
|
|
878
|
+
trimmed_screen = compact_screen_for_action(screen_text)
|
|
824
879
|
if self.max_screen_chars and len(trimmed_screen) > self.max_screen_chars:
|
|
825
880
|
# Truncate at the last newline before the limit to avoid cutting an
|
|
826
881
|
# element in half. Interactive elements are listed first, so only
|
|
@@ -60,7 +60,8 @@ class StepResult:
|
|
|
60
60
|
screenshots: list[bytes] = field(default_factory=list)
|
|
61
61
|
error: str = ""
|
|
62
62
|
duration_seconds: float = 0.0
|
|
63
|
-
token_usage: TokenUsage = field(default_factory=TokenUsage)
|
|
63
|
+
token_usage: TokenUsage = field(default_factory=TokenUsage) # all LLM calls
|
|
64
|
+
verifier_tokens: TokenUsage = field(default_factory=TokenUsage) # the verifier's share
|
|
64
65
|
# The goal could not be evaluated (LLM capacity / infrastructure). Not a
|
|
65
66
|
# test result: excluded from stability and reported as an error (ADR-007).
|
|
66
67
|
infra_error: bool = False
|
|
@@ -80,6 +81,10 @@ class StepResult:
|
|
|
80
81
|
"error": self.error,
|
|
81
82
|
"duration_seconds": self.duration_seconds,
|
|
82
83
|
"token_usage": self.token_usage.to_dict(),
|
|
84
|
+
"token_breakdown": {
|
|
85
|
+
"action": (self.token_usage - self.verifier_tokens).to_dict(),
|
|
86
|
+
"verifier": self.verifier_tokens.to_dict(),
|
|
87
|
+
},
|
|
83
88
|
"mode": self.mode,
|
|
84
89
|
}
|
|
85
90
|
|
|
@@ -506,7 +511,7 @@ class FusionTestRunner:
|
|
|
506
511
|
action_str = _redact(f"done: {action.reasoning}")
|
|
507
512
|
step.actions.append(action_str)
|
|
508
513
|
verified = self.verifier.is_complete(goal, screen_text, history)
|
|
509
|
-
|
|
514
|
+
self._add_verifier_tokens(step)
|
|
510
515
|
step.success = verified
|
|
511
516
|
if verified:
|
|
512
517
|
self._last_recording = self._make_recording(recorded, screen_text)
|
|
@@ -607,7 +612,7 @@ class FusionTestRunner:
|
|
|
607
612
|
f" Goal verified complete at max_steps "
|
|
608
613
|
f"({self.config.max_steps_per_goal}) — last-chance check passed"
|
|
609
614
|
)
|
|
610
|
-
|
|
615
|
+
self._add_verifier_tokens(step)
|
|
611
616
|
|
|
612
617
|
if not step.success and self.config.screenshot_on_failure:
|
|
613
618
|
try:
|
|
@@ -744,6 +749,12 @@ class FusionTestRunner:
|
|
|
744
749
|
)
|
|
745
750
|
return replace(action, value=resolver.resolve(action.value)), None
|
|
746
751
|
|
|
752
|
+
def _add_verifier_tokens(self, step: StepResult) -> None:
|
|
753
|
+
usage = getattr(self.verifier, "last_token_usage", None)
|
|
754
|
+
if isinstance(usage, TokenUsage):
|
|
755
|
+
step.token_usage += usage
|
|
756
|
+
step.verifier_tokens += usage
|
|
757
|
+
|
|
747
758
|
def _write_step_log(
|
|
748
759
|
self, goal: str, screen_text: str, action: str, outcome: str
|
|
749
760
|
) -> None:
|
|
@@ -25,6 +25,12 @@ class TokenUsage:
|
|
|
25
25
|
output_tokens=self.output_tokens + other.output_tokens,
|
|
26
26
|
)
|
|
27
27
|
|
|
28
|
+
def __sub__(self, other: TokenUsage) -> TokenUsage:
|
|
29
|
+
return TokenUsage(
|
|
30
|
+
input_tokens=self.input_tokens - other.input_tokens,
|
|
31
|
+
output_tokens=self.output_tokens - other.output_tokens,
|
|
32
|
+
)
|
|
33
|
+
|
|
28
34
|
def __iadd__(self, other: TokenUsage) -> TokenUsage:
|
|
29
35
|
self.input_tokens += other.input_tokens
|
|
30
36
|
self.output_tokens += other.output_tokens
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|