prettyplay 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. prettyplay/.usages/lifecycle.md +50 -0
  2. prettyplay/.usages/steps.md +56 -0
  3. prettyplay/CODEMANIFEST +182 -0
  4. prettyplay/__init__.py +13 -0
  5. prettyplay/cache/.usages/addressing.md +31 -0
  6. prettyplay/cache/.usages/budgets.md +21 -0
  7. prettyplay/cache/.usages/storage.md +32 -0
  8. prettyplay/cache/CODEMANIFEST +152 -0
  9. prettyplay/cache/__init__.py +8 -0
  10. prettyplay/cache/budgets.py +73 -0
  11. prettyplay/cache/models.py +60 -0
  12. prettyplay/cache/store.py +260 -0
  13. prettyplay/cache/text.py +35 -0
  14. prettyplay/config/.usages/configuration.md +87 -0
  15. prettyplay/config/CODEMANIFEST +132 -0
  16. prettyplay/config/__init__.py +6 -0
  17. prettyplay/config/loader.py +192 -0
  18. prettyplay/config/models.py +92 -0
  19. prettyplay/driver/.usages/facade.md +66 -0
  20. prettyplay/driver/CODEMANIFEST +162 -0
  21. prettyplay/driver/__init__.py +6 -0
  22. prettyplay/driver/page.py +350 -0
  23. prettyplay/driver/session.py +288 -0
  24. prettyplay/engine/.usages/generation.md +45 -0
  25. prettyplay/engine/.usages/healing.md +31 -0
  26. prettyplay/engine/CODEMANIFEST +215 -0
  27. prettyplay/engine/__init__.py +8 -0
  28. prettyplay/engine/classification.py +69 -0
  29. prettyplay/engine/execution.py +25 -0
  30. prettyplay/engine/generator.py +318 -0
  31. prettyplay/engine/healer.py +116 -0
  32. prettyplay/engine/text.py +19 -0
  33. prettyplay/executor.py +109 -0
  34. prettyplay/failures/.usages/taxonomy.md +40 -0
  35. prettyplay/failures/CODEMANIFEST +117 -0
  36. prettyplay/failures/__init__.py +17 -0
  37. prettyplay/failures/errors.py +147 -0
  38. prettyplay/llm/.usages/classification.md +25 -0
  39. prettyplay/llm/.usages/providers.md +33 -0
  40. prettyplay/llm/CODEMANIFEST +125 -0
  41. prettyplay/llm/__init__.py +8 -0
  42. prettyplay/llm/_request.py +216 -0
  43. prettyplay/llm/anthropic_provider.py +213 -0
  44. prettyplay/llm/models.py +22 -0
  45. prettyplay/llm/openai_provider.py +187 -0
  46. prettyplay/llm/provider.py +115 -0
  47. prettyplay/reporting/.usages/hooks.md +41 -0
  48. prettyplay/reporting/CODEMANIFEST +79 -0
  49. prettyplay/reporting/__init__.py +6 -0
  50. prettyplay/reporting/hooks.py +37 -0
  51. prettyplay/reporting/reporter.py +70 -0
  52. prettyplay/runtime.py +107 -0
  53. prettyplay/scenario.py +291 -0
  54. prettyplay-0.0.0.dist-info/METADATA +236 -0
  55. prettyplay-0.0.0.dist-info/RECORD +58 -0
  56. prettyplay-0.0.0.dist-info/WHEEL +5 -0
  57. prettyplay-0.0.0.dist-info/licenses/LICENSE +28 -0
  58. prettyplay-0.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,31 @@
1
+ # Step healing
2
+
3
+ Domain: healing a failed cached step. Audience: library internals and engineers reasoning about healed runs.
4
+
5
+ ## Heal
6
+
7
+ ```python
8
+ healed = healer.heal(
9
+ step=failed_step, error="element not found: button «Sign in»", previous_steps=["open the login page"], page=page
10
+ )
11
+ ```
12
+
13
+ The classification verdict decides the path:
14
+
15
+ | Category | Path |
16
+ |---|---|
17
+ | rot | regenerate from the current page within the healing budget (default 2), execute, save back to the cache, report loudly |
18
+ | product_defect | raise ProductDefectError carrying the verdict — category, explanation and recommendation all reach the exception message, the on_step_verdict hook and the log |
19
+ | incurable | raise IncurableStepError carrying the verdict; the reason names the incurability cause |
20
+
21
+ ## Rules
22
+
23
+ - Anti-masking: healing never turns a product defect into a green test
24
+ - The healed code replaces the cached code only after a successful execution
25
+ - Generation and healing attempts live in one per-test registry — owned by the runtime of the test — with separate per-step limits (default 3 and 2)
26
+ - A regeneration budget exhaustion after rot raises IncurableStepError carrying the verdict of the original rot classification — no extra LLM request
27
+ - Provider unavailability during the classification raises LlmUnavailableError — an explicit infrastructure failure
28
+
29
+ ## Verdicts
30
+
31
+ Every terminal failure carries its verdict in full: the exception message starts with the primary reason and appends the verdict render; the same three fields reach on_step_verdict and the structured log record.
@@ -0,0 +1,215 @@
1
+ Imports:
2
+ - Types:
3
+ - Config
4
+ From: prettyplay/config
5
+ - Types:
6
+ - StepReporter
7
+ From: prettyplay/reporting
8
+ - Types:
9
+ - ProductDefectError
10
+ - IncurableStepError
11
+ - LlmUnavailableError
12
+ - FailureVerdict
13
+ From: prettyplay/failures
14
+ - Types:
15
+ - PageFacade
16
+ Usages:
17
+ - facade
18
+ From: prettyplay/driver
19
+ - Types:
20
+ - StepCache
21
+ - StepIdentity
22
+ - CachedStep
23
+ - RunBudgets
24
+ From: prettyplay/cache
25
+ - Types:
26
+ - LlmProvider
27
+ - FailureClassification
28
+ Usages:
29
+ - classification
30
+ From: prettyplay/llm
31
+
32
+ Usages:
33
+ conventions: .goga/usages/conventions.md
34
+ system_prompt: |
35
+ You generate executable Python code for one step of a web UI test.
36
+
37
+ Input you receive:
38
+ - STEP: the step sentence in a natural language
39
+ - PREVIOUS STEPS: the sentences of the previous steps of the test, in order
40
+ - PAGE SNAPSHOT: the accessibility snapshot of the current page
41
+ - SCREENSHOT: an image of the page, when attached
42
+ - PAGE API: the exact surface listing of the page facade — call nothing outside it
43
+ - USER INSTRUCTIONS: the project's code style guidance, when configured
44
+ - CODE: the existing step code that failed (regeneration requests only)
45
+ - ERROR: the failure description of the existing code (regeneration requests only)
46
+
47
+ Output exactly one Python code block with one function of the fixed form:
48
+
49
+ def step(page) -> None:
50
+ ...
51
+
52
+ Rules:
53
+ - The function receives exactly one argument: the page facade. Never import anything, never use other libraries
54
+ - Work only through the page API: the request carries the exact surface listing of the page facade — call nothing outside it
55
+ - For an assertion sentence end with an expectation call; for an action sentence perform the actions
56
+ - Locating by role and accessible name is preferred; by visible text next; by label for form fields
57
+ - Attribute, CSS and XPath locating exist for elements without accessible names — the accessibility-first priority stands unless USER INSTRUCTIONS say otherwise
58
+ - Scroll abilities exist for scenario scrolling: bring an element into view, scroll by an amount, to the page end or start, inside a scrollable container
59
+ - No fixed delays, no sleeps, no explicit waits — the facade waits itself
60
+ - The step must complete exactly what STEP says — nothing more, nothing less
61
+ - Output only the code block, no explanations
62
+ classification_prompt: |
63
+ You classify a failure of a web UI test step.
64
+
65
+ Input you receive:
66
+ - STEP: the step sentence
67
+ - CODE: the step code that failed
68
+ - ERROR: the failure description
69
+ - PAGE SNAPSHOT: the accessibility snapshot of the current page
70
+ - SCREENSHOT: an image of the page, when attached
71
+
72
+ Answer with exactly one line of the form:
73
+ category | explanation | recommendation
74
+
75
+ where category is one of:
76
+ - rot — the UI changed (selectors, texts, structure) and the step can be regenerated for the same intent
77
+ - product_defect — the step works as written but the expected behavior of the application is genuinely broken
78
+ - incurable — the step sentence no longer matches reality, the intent is ambiguous, or regeneration cannot help
79
+
80
+ explanation: one short sentence why. recommendation: one short sentence what the engineer should do.
81
+ Output only that single line — no code, no extra text.
82
+
83
+ Annotations: |
84
+ Use `conventions` for code writing rules and testing.
85
+ Use `facade` from Imports for the page API the generated code works through.
86
+ Use `classification` from Imports for the healing decision categories.
87
+ Use `system_prompt` as the system prompt of every code generation request.
88
+ Use `classification_prompt` as the system prompt of every failure classification request.
89
+ Use `facade` from Imports as the single source of the page API surface for generation requests.
90
+
91
+ The fixed form of step code: one function receiving exactly one argument — the `PageFacade` of the test; the function body works only through `PageFacade` and the LocatorFacade element API.
92
+ Every attempt — generation or healing — consumes the shared per-step budget of the test; exhaustion is the incurable failure, never an infinite loop.
93
+ Healing never masks a product defect: a classified product_defect fails the test loudly; a healed step is reported loudly and written back to the cache.
94
+ Every terminal failure carries a verdict — a `FailureVerdict` of category, explanation, recommendation — fully present in the exception message, the verdict hook event and the log; an unavailable LLM skips the verdict quietly with a WARNING, the failure itself is never delayed or distorted — the quiet skip applies to verdicts enriching an already-decided failure; the classification driving the healing decision surfaces as the infrastructure failure.
95
+ A failed check of a candidate stops the generation retries: a check that executed and did not hold is classified, not regenerated.
96
+ The page API surface listing sent to the provider mirrors `facade` from Imports exactly — the listing and the practice change together.
97
+ The user instructions of the project settings — the generation_prompt field of `Config` — reach generation and regeneration requests only; classification requests never carry them; the instructions take no part in the step address: a cached step never regenerates because the instructions changed.
98
+
99
+ ---
100
+
101
+ "StepGenerator(config: Config, provider: LlmProvider, cache: StepCache, budgets: RunBudgets, reporter: StepReporter)":
102
+ location: generator.py
103
+ annotations: |
104
+ The generation engine: produce working step code for an unknown step, executing candidates against the live page.
105
+
106
+ `config`: project settings — the screenshot flag and the generation instructions.
107
+ `provider`: the LLM port.
108
+ `cache`: the step cache — a generated step is stored on success.
109
+ `budgets`: the per-test attempt registry.
110
+ `reporter`: the visibility point.
111
+ methods:
112
+ "generate(identity: StepIdentity, step_text: str, previous_steps: list[str], page: PageFacade) -> step: CachedStep": |
113
+ Generate and store a new step.
114
+
115
+ Algorithm:
116
+ 1. Ask the budgets registry try_generation for the step identity; a refused attempt is the incurable failure
117
+ 2. Collect the request inputs: the page accessibility snapshot, the step sentence, `previous_steps`, and the page API surface from `facade`; add the page screenshot when the project settings enable screenshots
118
+ 3. Request step code from the provider port generate_step_code passing `system_prompt` as the system prompt, the page API surface listing taken from `facade`, and the user instructions — the effective config generation_prompt — when non-empty
119
+ 4. Execute the candidate with `run_step_code` against `page`
120
+ 5. On success: build `CachedStep`, save it to the cache, return it
121
+ 6. On a candidate failure that is an AssertionError — a check that executed and did not hold: stop the retries immediately and classify via `classify_step_failure`; a product_defect verdict raises `ProductDefectError` carrying the verdict, any other verdict raises `IncurableStepError` carrying it — the reason names the failed candidate check; provider unavailability at this classification is skipped quietly with a WARNING — `IncurableStepError` is raised without a verdict, the reason naming the failed candidate check (the failed check is the primary signal, the verdict is enrichment)
122
+ 7. On any other candidate failure: repeat from step 1 with the fresh failure description and the fresh snapshot, while attempts remain
123
+ 8. On budget exhaustion: classify the last candidate via `classify_step_failure` and raise `IncurableStepError` carrying the verdict — the reason names the exhausted pool and the last candidate failure; provider unavailability at this classification is skipped quietly with a WARNING, the failure raises without a verdict
124
+ 9. Report on_generation_started for every attempt
125
+
126
+ Requirements:
127
+ - Every generation request carries the exact page API surface taken from `facade` from Imports: the model always sees the precise list of calls it may use
128
+ - Provider unavailability of a generation request surfaces as `LlmUnavailableError` immediately — no retry on it
129
+ - Provider unavailability of a failed-check classification is skipped quietly with a WARNING: the failure raises without a verdict — the failed check itself is the primary signal
130
+ - Exactly one failed check stops the retries: the attempt budget is never spent on a legitimately failing assertion
131
+ - A verdict requested on this path fully reaches the raised error
132
+ "regenerate(identity: StepIdentity, step_text: str, previous_steps: list[str], page: PageFacade, existing_code: str, error: str) -> step: CachedStep": |
133
+ Regenerate a failed step for healing.
134
+
135
+ Algorithm:
136
+ 1. The same loop as the generate method with three additions: every provider request carries `existing_code` and `error` and the user instructions — the effective config generation_prompt — when non-empty; attempts consume the healing budget via try_healing; a budget exhaustion raises `IncurableStepError` without classification — the healer attaches the verdict of its own classification, no extra LLM request is made (the failed-check classification of the generate loop applies on both pools)
137
+
138
+ "run_step_code(code: str, page: PageFacade)":
139
+ location: execution.py
140
+ annotations: |
141
+ Execute step code of the fixed form against the page facade of the test.
142
+
143
+ `code`: the step code text.
144
+ `page`: the page facade of the current test.
145
+
146
+ Algorithm:
147
+ 1. Compile and load `code` as a module in an isolated namespace
148
+ 2. Resolve the step function of the fixed form — the single callable receiving the page facade
149
+ 3. Call it with `page`
150
+
151
+ Requirements:
152
+ - A failure inside the step code propagates to the caller as-is: the engine classifies it, this routine never swallows or retries
153
+ - Executing step code loads no LLM provider and touches no network beyond the page itself
154
+
155
+ Constraints:
156
+ - Execute only step code produced by generation or loaded from the cache — never arbitrary file content
157
+
158
+ "classify_step_failure(config: Config, provider: LlmProvider, step_text: str, code: str, error: str, page: PageFacade) -> classification: FailureClassification":
159
+ location: classification.py
160
+ annotations: |
161
+ Classify a step failure: collect the page state and ask the provider — the single classification call for both engines.
162
+
163
+ `config`: project settings — the screenshot flag.
164
+ `provider`: the LLM port.
165
+ `step_text`: the sentence of the failed step.
166
+ `code`: the step code that failed.
167
+ `error`: the human-readable failure description.
168
+ `page`: the page facade of the current test.
169
+ `classification`: the `FailureClassification` verdict.
170
+
171
+ Algorithm:
172
+ 1. Collect the classification inputs: the step sentence, the failed code, the `error` text, the fresh page snapshot — plus the screenshot when enabled
173
+ 2. Ask the provider port classify_failure passing `classification_prompt` as the system prompt
174
+ 3. Return the verdict
175
+
176
+ Constraints:
177
+ - Provider unavailability propagates to the caller: this routine never swallows it — the calling path decides whether it is a terminal infrastructure failure or a quiet verdict skip
178
+
179
+ "StepHealer(config: Config, provider: LlmProvider, generator: StepGenerator, cache: StepCache, budgets: RunBudgets, reporter: StepReporter)":
180
+ location: healer.py
181
+ annotations: |
182
+ The healing engine: classify a failed cached step, regenerate rot, never mask a defect.
183
+
184
+ `config`: project settings — the screenshot flag.
185
+ `provider`: the LLM port for classification.
186
+ `generator`: the regeneration engine.
187
+ `cache`: the step cache for the healed write-back.
188
+ `budgets`: the per-test attempt registry.
189
+ `reporter`: the visibility point.
190
+ methods:
191
+ "heal(step: CachedStep, error: str, previous_steps: list[str], page: PageFacade) -> step: CachedStep": |
192
+ Classify and heal a failed cached step.
193
+
194
+ `previous_steps`: the sentences of the previous steps of the test, in execution order — scenario context for regeneration.
195
+
196
+ Algorithm:
197
+ 1. Classify the failure via `classify_step_failure` — the verdict is a `FailureClassification`
198
+ 2. Report on_healing_started with the category
199
+ 3. product_defect: raise `ProductDefectError` carrying the verdict built from the classification — the message states what was expected against what was observed; the recommendation reaches the error through the verdict
200
+ 4. incurable: raise `IncurableStepError` carrying the verdict; the reason names the classification explanation of incurability
201
+ 5. rot: regenerate via the generator regenerate — its loop executes the candidate and stores the healed step on success; report on_healed with the explanation of what was rot and what changed, return the healed step
202
+ 6. A regeneration budget exhaustion inside step 5 surfaces as `IncurableStepError` carrying the verdict of the step 1 classification — the reason names the exhausted pool; no extra LLM request is made
203
+ 7. Provider unavailability of the classification surfaces as `LlmUnavailableError` — an explicit infrastructure failure
204
+
205
+ Requirements:
206
+ - Anti-masking: healing may only turn a rot-failed step green; a classified product defect always fails the test
207
+ - The healed code replaces the cached code only after a successful execution
208
+ - Every verdict produced on the paths of this method fully reaches the raised error
209
+
210
+ ---
211
+
212
+ Author: Goga
213
+ CreatedAt: 07/09/26
214
+ Description: |
215
+ The agent engine of prettyplay: step code generation with execution in the loop and failed-check classification, the fixed-form execution routine, the shared classification call, and healing with anti-masking and verdicts on terminal failures.
@@ -0,0 +1,8 @@
1
+ """Facade of the prettyplay.engine cell: generation, execution and healing of step code."""
2
+
3
+ from .classification import classify_step_failure
4
+ from .execution import run_step_code
5
+ from .generator import StepGenerator
6
+ from .healer import StepHealer
7
+
8
+ __all__ = ["StepGenerator", "StepHealer", "classify_step_failure", "run_step_code"]
@@ -0,0 +1,69 @@
1
+ """The shared classification call: collect the page state and ask the provider."""
2
+
3
+ from ..config import Config
4
+ from ..driver import PageFacade
5
+ from ..llm import FailureClassification, LlmProvider
6
+
7
+ #: System prompt of every classification request; used by both engines, applied verbatim.
8
+ CLASSIFICATION_PROMPT = """You classify a failure of a web UI test step.
9
+
10
+ Input you receive:
11
+ - STEP: the step sentence
12
+ - CODE: the step code that failed
13
+ - ERROR: the failure description
14
+ - PAGE SNAPSHOT: the accessibility snapshot of the current page
15
+ - SCREENSHOT: an image of the page, when attached
16
+
17
+ Answer with exactly one line of the form:
18
+ category | explanation | recommendation
19
+
20
+ where category is one of:
21
+ - rot — the UI changed (selectors, texts, structure) and the step can be regenerated for the same intent
22
+ - product_defect — the step works as written but the expected behavior of the application is genuinely broken
23
+ - incurable — the step sentence no longer matches reality, the intent is ambiguous, or regeneration cannot help
24
+
25
+ explanation: one short sentence why. recommendation: one short sentence what the engineer should do.
26
+ Output only that single line — no code, no extra text."""
27
+
28
+
29
+ def classify_step_failure( # noqa: PLR0913, PLR0917 — the parameter list is fixed by the engine contract
30
+ config: Config,
31
+ provider: LlmProvider,
32
+ step_text: str,
33
+ code: str,
34
+ error: str,
35
+ page: PageFacade,
36
+ ) -> FailureClassification:
37
+ """Classify a step failure: collect the page state and ask the provider.
38
+
39
+ The single classification call for both engines — the generator failed-check
40
+ stop and exhaustion, the healer verdict. Provider unavailability propagates
41
+ to the caller: this routine never swallows it — the calling path decides
42
+ whether it is a terminal infrastructure failure or a quiet verdict skip.
43
+
44
+ Args:
45
+ config: project settings; ``send_screenshots`` attaches page images.
46
+ provider: the LLM port implementation classifying the failure.
47
+ step_text: the sentence of the failed step.
48
+ code: the step code that failed.
49
+ error: the human-readable failure description.
50
+ page: the page facade of the current test.
51
+
52
+ Returns:
53
+ The classification verdict.
54
+
55
+ Raises:
56
+ LlmUnavailableError: the provider service failed; the calling path
57
+ decides the handling.
58
+ """
59
+ snapshot = page.aria_snapshot()
60
+ screenshot = page.screenshot() if config.send_screenshots else None
61
+
62
+ return provider.classify_failure(
63
+ prompt=CLASSIFICATION_PROMPT,
64
+ step_text=step_text,
65
+ code=code,
66
+ error=error,
67
+ snapshot=snapshot,
68
+ screenshot=screenshot,
69
+ )
@@ -0,0 +1,25 @@
1
+ """Execution of step code of the fixed form against the page facade of the test."""
2
+
3
+ from ..driver import PageFacade
4
+
5
+
6
+ def run_step_code(code: str, page: PageFacade) -> None:
7
+ """Execute step code of the fixed form against the page facade of the test.
8
+
9
+ The code is compiled and loaded in an isolated namespace; the step
10
+ function of the fixed form — the single callable named ``step`` — is
11
+ resolved and called with the page facade. A failure inside the step code
12
+ propagates to the caller as-is: the engine classifies it, this routine
13
+ never swallows or retries. No module is registered in ``sys.modules`` and
14
+ no provider or network beyond the page itself is touched.
15
+
16
+ Args:
17
+ code: the step code text produced by generation or loaded from the cache.
18
+ page: the page facade of the current test.
19
+ """
20
+ namespace: dict[str, object] = {}
21
+ # the engine contract: exec of the fixed-form step code in an isolated namespace
22
+ exec(compile(code, "<prettyplay-step>", "exec"), namespace)
23
+
24
+ fn = namespace["step"]
25
+ fn(page)
@@ -0,0 +1,318 @@
1
+ """Generation of working step code: LLM candidates executed against the live page in a loop."""
2
+
3
+ import logging
4
+ from datetime import date
5
+
6
+ from ..cache import CachedStep, RunBudgets, StepCache, StepIdentity
7
+ from ..config import Config
8
+ from ..driver import PageFacade
9
+ from ..failures import FailureVerdict, IncurableStepError, LlmUnavailableError, ProductDefectError
10
+ from ..llm import FailureClassification, LlmProvider
11
+ from ..reporting import StepReporter
12
+ from .classification import classify_step_failure
13
+ from .execution import run_step_code
14
+ from .text import first_line_short
15
+
16
+ logger = logging.getLogger("prettyplay")
17
+
18
+ #: System prompt of every generation request; applied verbatim by the provider.
19
+ SYSTEM_PROMPT = """You generate executable Python code for one step of a web UI test.
20
+
21
+ Input you receive:
22
+ - STEP: the step sentence in a natural language
23
+ - PREVIOUS STEPS: the sentences of the previous steps of the test, in order
24
+ - PAGE SNAPSHOT: the accessibility snapshot of the current page
25
+ - SCREENSHOT: an image of the page, when attached
26
+ - PAGE API: the exact surface listing of the page facade — call nothing outside it
27
+ - USER INSTRUCTIONS: the project's code style guidance, when configured
28
+ - CODE: the existing step code that failed (regeneration requests only)
29
+ - ERROR: the failure description of the existing code (regeneration requests only)
30
+
31
+ Output exactly one Python code block with one function of the fixed form:
32
+
33
+ def step(page) -> None:
34
+ ...
35
+
36
+ Rules:
37
+ - The function receives exactly one argument: the page facade. Never import anything, never use other libraries
38
+ - Work only through the page API: the request carries the exact surface listing of the page facade — call nothing outside it
39
+ - For an assertion sentence end with an expectation call; for an action sentence perform the actions
40
+ - Locating by role and accessible name is preferred; by visible text next; by label for form fields
41
+ - Attribute, CSS and XPath locating exist for elements without accessible names — the accessibility-first priority stands unless USER INSTRUCTIONS say otherwise
42
+ - Scroll abilities exist for scenario scrolling: bring an element into view, scroll by an amount, to the page end or start, inside a scrollable container
43
+ - No fixed delays, no sleeps, no explicit waits — the facade waits itself
44
+ - The step must complete exactly what STEP says — nothing more, nothing less
45
+ - Output only the code block, no explanations"""
46
+
47
+ #: Frozen surface listing of the driver facade — the only calls step code may make.
48
+ #: Mirrors ``prettyplay/driver/.usages/facade.md`` verbatim; the driver facade is a
49
+ #: backward-compatibility contract, so this constant changes only together with it.
50
+ #: ``close()`` stays out: it is a runtime method of PrettyTest, not of step code.
51
+ PAGE_API_SURFACE = """page.open(url) — navigate and wait for load
52
+ page.find_by_role(role, name) — element by aria role and accessible name
53
+ page.find_by_label(label) — element by associated label
54
+ page.find_by_text(text) — element by visible text
55
+ page.find_by_attribute(name, value) — element by attribute value — data-* attributes
56
+ page.find_by_css(selector) — element by CSS selector
57
+ page.find_by_xpath(xpath) — element by XPath expression
58
+ page.aria_snapshot() — accessibility-tree page state
59
+ page.screenshot() — full-page PNG bytes
60
+ page.url — current URL
61
+ page.scroll_to_element(element) — bring an element into the viewport (works inside scrollable ancestors)
62
+ page.scroll_down(pixels) — scroll the page down by an amount
63
+ page.scroll_up(pixels) — scroll the page up by an amount
64
+ page.scroll_to_bottom() — scroll to the end of the page
65
+ page.scroll_to_top() — scroll to the start of the page
66
+ page.scroll_into_view(element, container) — bring an element into view inside a specific scrollable container
67
+ page.scroll_container_down(container, pixels) — scroll a scrollable container down by an amount
68
+ page.scroll_container_up(container, pixels) — scroll a scrollable container up by an amount
69
+ element.click() — click with auto-wait
70
+ element.fill(value) — set input text
71
+ element.select_option(value) — choose an option
72
+ element.expect_visible() — assert visible
73
+ element.expect_text(text) — assert text
74
+ element.expect_enabled() — assert enabled"""
75
+
76
+
77
+ class StepGenerator:
78
+ """Generates working step code by executing LLM candidates against the live page.
79
+
80
+ Each attempt is one provider request: the generator snapshots the page,
81
+ asks for code of the fixed form, and immediately executes the candidate.
82
+ A failing candidate is retried as a regeneration request carrying the code
83
+ and its error, until one candidate works or the attempt budget of the step
84
+ runs out. Only a proven candidate is cached — failures are never stored.
85
+
86
+ Attributes:
87
+ _config: project settings; the screenshot flag feeds the requests.
88
+ _provider: the LLM port implementation doing the requests.
89
+ _cache: the store where working steps are saved.
90
+ _budgets: the per-test attempt registry of the engine.
91
+ _reporter: the visibility point for generation and cache events.
92
+ """
93
+
94
+ def __init__(
95
+ self,
96
+ config: Config,
97
+ provider: LlmProvider,
98
+ cache: StepCache,
99
+ budgets: RunBudgets,
100
+ reporter: StepReporter,
101
+ ) -> None:
102
+ """Keep the collaborators of the generation loop.
103
+
104
+ Args:
105
+ config: project settings; ``send_screenshots`` attaches page images.
106
+ provider: the LLM port implementation.
107
+ cache: the store of working steps.
108
+ budgets: the per-test attempt registry.
109
+ reporter: the visibility point for engine events.
110
+ """
111
+ self._config = config
112
+ self._provider = provider
113
+ self._cache = cache
114
+ self._budgets = budgets
115
+ self._reporter = reporter
116
+
117
+ def generate(
118
+ self,
119
+ identity: StepIdentity,
120
+ step_text: str,
121
+ previous_steps: list[str],
122
+ page: PageFacade,
123
+ ) -> CachedStep:
124
+ """Generate step code until a candidate works, then cache it.
125
+
126
+ Args:
127
+ identity: the address of the step.
128
+ step_text: the sentence of the step.
129
+ previous_steps: the sentences of the previous steps of the test.
130
+ page: the live page facade the candidates run against.
131
+
132
+ Returns:
133
+ The cached step holding the proven code.
134
+
135
+ Raises:
136
+ IncurableStepError: the generation attempt budget is exhausted.
137
+ LlmUnavailableError: the provider service failed; no retry.
138
+ """
139
+ return self._loop(identity, step_text, previous_steps, page, "generation", None, None)
140
+
141
+ def regenerate( # noqa: PLR0913, PLR0917 — the signature is fixed by the engine contract
142
+ self,
143
+ identity: StepIdentity,
144
+ step_text: str,
145
+ previous_steps: list[str],
146
+ page: PageFacade,
147
+ existing_code: str,
148
+ error: str,
149
+ ) -> CachedStep:
150
+ """Regenerate step code starting from the failed candidate.
151
+
152
+ The loop is the generation loop; the differences are the healing budget
153
+ pool and the failed code and error carried by the first request.
154
+
155
+ Args:
156
+ identity: the address of the step.
157
+ step_text: the sentence of the step.
158
+ previous_steps: the sentences of the previous steps of the test.
159
+ page: the live page facade the candidates run against.
160
+ existing_code: the cached step code that failed.
161
+ error: the failure description of the existing code.
162
+
163
+ Returns:
164
+ The cached step holding the proven regenerated code.
165
+
166
+ Raises:
167
+ IncurableStepError: the healing attempt budget is exhausted.
168
+ LlmUnavailableError: the provider service failed; no retry.
169
+ """
170
+ return self._loop(identity, step_text, previous_steps, page, "healing", existing_code, error)
171
+
172
+ def _loop( # noqa: PLR0913, PLR0917 — the shared attempt loop with its fixed inputs
173
+ self,
174
+ identity: StepIdentity,
175
+ step_text: str,
176
+ previous_steps: list[str],
177
+ page: PageFacade,
178
+ pool: str,
179
+ existing_code: str | None,
180
+ error: str | None,
181
+ ) -> CachedStep:
182
+ """Run the shared attempt loop until a candidate works or the budget runs out.
183
+
184
+ A failed check — a candidate ``AssertionError`` — stops the retries at
185
+ once and is classified: the attempt budget is never spent on a
186
+ legitimately failing assertion. Any other candidate failure is retried
187
+ as a regeneration request carrying the code and its error. When the
188
+ budget runs out, the generation pool classifies the last candidate;
189
+ the healing pool raises without classification — the healer attaches
190
+ the verdict it already holds, so no extra LLM request is made.
191
+
192
+ Args:
193
+ identity: the address of the step.
194
+ step_text: the sentence of the step.
195
+ previous_steps: the sentences of the previous steps of the test.
196
+ page: the live page facade the candidates run against.
197
+ pool: the budget pool name — "generation" or "healing".
198
+ existing_code: the failed code of the first request, if any.
199
+ error: the failure description of the first request, if any.
200
+
201
+ Returns:
202
+ The cached step holding the proven code.
203
+
204
+ Raises:
205
+ ProductDefectError: a failed candidate check classified as a
206
+ genuine product defect.
207
+ IncurableStepError: the attempt budget of the step is exhausted,
208
+ or a failure classified as incurable.
209
+ LlmUnavailableError: the provider service failed; no retry.
210
+ """
211
+ attempt = 0
212
+ code = None
213
+ spend = self._budgets.try_generation if pool == "generation" else self._budgets.try_healing
214
+
215
+ while True:
216
+ if not spend(identity):
217
+ # candidate cause is part of the reason contract: «the specific incurability cause»
218
+ reason = f"{pool} attempt budget exhausted"
219
+
220
+ if error:
221
+ reason = f"{reason}; last failure: {error}"
222
+ if pool == "healing":
223
+ raise IncurableStepError(
224
+ step_text,
225
+ reason,
226
+ None, # healer attaches the verdict — no second LLM request
227
+ )
228
+ if error is None:
229
+ raise IncurableStepError(
230
+ step_text,
231
+ reason,
232
+ None, # nothing to classify: no candidates existed
233
+ )
234
+
235
+ raise IncurableStepError(step_text, reason, self._classify(step_text, code, error, page))
236
+
237
+ attempt += 1
238
+ self._reporter.emit("on_generation_started", {"step_text": step_text, "attempt": attempt})
239
+
240
+ snapshot = page.aria_snapshot()
241
+ screenshot = page.screenshot() if self._config.send_screenshots else None
242
+
243
+ code = self._provider.generate_step_code(
244
+ prompt=SYSTEM_PROMPT,
245
+ user_instructions=self._config.generation_prompt,
246
+ step_text=step_text,
247
+ previous_steps=previous_steps,
248
+ snapshot=snapshot,
249
+ screenshot=screenshot,
250
+ page_api=PAGE_API_SURFACE,
251
+ existing_code=existing_code,
252
+ error=error,
253
+ )
254
+
255
+ try:
256
+ run_step_code(code, page)
257
+ except AssertionError as check_failure:
258
+ # failed check: attempts not spent — classify and stop
259
+ reason = f"candidate check failed: {first_line_short(check_failure)}"
260
+ verdict = self._classify(step_text, code, reason, page)
261
+ if verdict is not None and verdict.category == "product_defect":
262
+ raise ProductDefectError(step_text, first_line_short(check_failure), verdict) from None
263
+ raise IncurableStepError(step_text, reason, verdict) from None
264
+ except Exception as candidate_error: # other candidate failures heal via retry
265
+ existing_code = code
266
+ error = first_line_short(candidate_error)
267
+ else:
268
+ break
269
+
270
+ step = CachedStep(
271
+ identity=identity,
272
+ code=code,
273
+ created_at=date.today().isoformat(), # noqa: DTZ011 — calendar date of step creation
274
+ )
275
+ self._cache.save(step)
276
+
277
+ return step
278
+
279
+ def _classify(self, step_text: str, code: str | None, error: str, page: PageFacade) -> FailureVerdict | None:
280
+ """Classify a failure through the shared routine with the quiet skip.
281
+
282
+ Every classification inside the loop enriches an already-decided
283
+ failure, so an unavailable LLM yields no verdict — the failure never
284
+ waits for it and never turns into an infrastructure error.
285
+
286
+ Args:
287
+ step_text: the sentence of the failed step.
288
+ code: the code of the last candidate.
289
+ error: the failure description of the candidate.
290
+ page: the live page facade of the test.
291
+
292
+ Returns:
293
+ The verdict of the classification, or ``None`` when the LLM was
294
+ unavailable — the quiet skip logs a WARNING.
295
+ """
296
+ try:
297
+ classification = classify_step_failure(self._config, self._provider, step_text, code, error, page)
298
+ except LlmUnavailableError:
299
+ logger.warning("verdict skipped: llm unavailable")
300
+ return None
301
+
302
+ return _verdict(classification)
303
+
304
+
305
+ def _verdict(classification: FailureClassification) -> FailureVerdict:
306
+ """Build the verdict value object carried by the terminal errors.
307
+
308
+ Args:
309
+ classification: the classification verdict of the provider.
310
+
311
+ Returns:
312
+ The frozen verdict of the terminal failure.
313
+ """
314
+ return FailureVerdict(
315
+ category=classification.category,
316
+ explanation=classification.explanation,
317
+ recommendation=classification.recommendation,
318
+ )