prettyplay 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- prettyplay/.usages/lifecycle.md +50 -0
- prettyplay/.usages/steps.md +56 -0
- prettyplay/CODEMANIFEST +182 -0
- prettyplay/__init__.py +13 -0
- prettyplay/cache/.usages/addressing.md +31 -0
- prettyplay/cache/.usages/budgets.md +21 -0
- prettyplay/cache/.usages/storage.md +32 -0
- prettyplay/cache/CODEMANIFEST +152 -0
- prettyplay/cache/__init__.py +8 -0
- prettyplay/cache/budgets.py +73 -0
- prettyplay/cache/models.py +60 -0
- prettyplay/cache/store.py +260 -0
- prettyplay/cache/text.py +35 -0
- prettyplay/config/.usages/configuration.md +87 -0
- prettyplay/config/CODEMANIFEST +132 -0
- prettyplay/config/__init__.py +6 -0
- prettyplay/config/loader.py +192 -0
- prettyplay/config/models.py +92 -0
- prettyplay/driver/.usages/facade.md +66 -0
- prettyplay/driver/CODEMANIFEST +162 -0
- prettyplay/driver/__init__.py +6 -0
- prettyplay/driver/page.py +350 -0
- prettyplay/driver/session.py +288 -0
- prettyplay/engine/.usages/generation.md +45 -0
- prettyplay/engine/.usages/healing.md +31 -0
- prettyplay/engine/CODEMANIFEST +215 -0
- prettyplay/engine/__init__.py +8 -0
- prettyplay/engine/classification.py +69 -0
- prettyplay/engine/execution.py +25 -0
- prettyplay/engine/generator.py +318 -0
- prettyplay/engine/healer.py +116 -0
- prettyplay/engine/text.py +19 -0
- prettyplay/executor.py +109 -0
- prettyplay/failures/.usages/taxonomy.md +40 -0
- prettyplay/failures/CODEMANIFEST +117 -0
- prettyplay/failures/__init__.py +17 -0
- prettyplay/failures/errors.py +147 -0
- prettyplay/llm/.usages/classification.md +25 -0
- prettyplay/llm/.usages/providers.md +33 -0
- prettyplay/llm/CODEMANIFEST +125 -0
- prettyplay/llm/__init__.py +8 -0
- prettyplay/llm/_request.py +216 -0
- prettyplay/llm/anthropic_provider.py +213 -0
- prettyplay/llm/models.py +22 -0
- prettyplay/llm/openai_provider.py +187 -0
- prettyplay/llm/provider.py +115 -0
- prettyplay/reporting/.usages/hooks.md +41 -0
- prettyplay/reporting/CODEMANIFEST +79 -0
- prettyplay/reporting/__init__.py +6 -0
- prettyplay/reporting/hooks.py +37 -0
- prettyplay/reporting/reporter.py +70 -0
- prettyplay/runtime.py +107 -0
- prettyplay/scenario.py +291 -0
- prettyplay-0.0.0.dist-info/METADATA +236 -0
- prettyplay-0.0.0.dist-info/RECORD +58 -0
- prettyplay-0.0.0.dist-info/WHEEL +5 -0
- prettyplay-0.0.0.dist-info/licenses/LICENSE +28 -0
- prettyplay-0.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
# Step healing
|
|
2
|
+
|
|
3
|
+
Domain: healing a failed cached step. Audience: library internals and engineers reasoning about healed runs.
|
|
4
|
+
|
|
5
|
+
## Heal
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
healed = healer.heal(
|
|
9
|
+
step=failed_step, error="element not found: button «Sign in»", previous_steps=["open the login page"], page=page
|
|
10
|
+
)
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
The classification verdict decides the path:
|
|
14
|
+
|
|
15
|
+
| Category | Path |
|
|
16
|
+
|---|---|
|
|
17
|
+
| rot | regenerate from the current page within the healing budget (default 2), execute, save back to the cache, report loudly |
|
|
18
|
+
| product_defect | raise ProductDefectError carrying the verdict — category, explanation and recommendation all reach the exception message, the on_step_verdict hook and the log |
|
|
19
|
+
| incurable | raise IncurableStepError carrying the verdict; the reason names the incurability cause |
|
|
20
|
+
|
|
21
|
+
## Rules
|
|
22
|
+
|
|
23
|
+
- Anti-masking: healing never turns a product defect into a green test
|
|
24
|
+
- The healed code replaces the cached code only after a successful execution
|
|
25
|
+
- Generation and healing attempts live in one per-test registry — owned by the runtime of the test — with separate per-step limits (default 3 and 2)
|
|
26
|
+
- A regeneration budget exhaustion after rot raises IncurableStepError carrying the verdict of the original rot classification — no extra LLM request
|
|
27
|
+
- Provider unavailability during the classification raises LlmUnavailableError — an explicit infrastructure failure
|
|
28
|
+
|
|
29
|
+
## Verdicts
|
|
30
|
+
|
|
31
|
+
Every terminal failure carries its verdict in full: the exception message starts with the primary reason and appends the verdict render; the same three fields reach on_step_verdict and the structured log record.
|
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
Imports:
|
|
2
|
+
- Types:
|
|
3
|
+
- Config
|
|
4
|
+
From: prettyplay/config
|
|
5
|
+
- Types:
|
|
6
|
+
- StepReporter
|
|
7
|
+
From: prettyplay/reporting
|
|
8
|
+
- Types:
|
|
9
|
+
- ProductDefectError
|
|
10
|
+
- IncurableStepError
|
|
11
|
+
- LlmUnavailableError
|
|
12
|
+
- FailureVerdict
|
|
13
|
+
From: prettyplay/failures
|
|
14
|
+
- Types:
|
|
15
|
+
- PageFacade
|
|
16
|
+
Usages:
|
|
17
|
+
- facade
|
|
18
|
+
From: prettyplay/driver
|
|
19
|
+
- Types:
|
|
20
|
+
- StepCache
|
|
21
|
+
- StepIdentity
|
|
22
|
+
- CachedStep
|
|
23
|
+
- RunBudgets
|
|
24
|
+
From: prettyplay/cache
|
|
25
|
+
- Types:
|
|
26
|
+
- LlmProvider
|
|
27
|
+
- FailureClassification
|
|
28
|
+
Usages:
|
|
29
|
+
- classification
|
|
30
|
+
From: prettyplay/llm
|
|
31
|
+
|
|
32
|
+
Usages:
|
|
33
|
+
conventions: .goga/usages/conventions.md
|
|
34
|
+
system_prompt: |
|
|
35
|
+
You generate executable Python code for one step of a web UI test.
|
|
36
|
+
|
|
37
|
+
Input you receive:
|
|
38
|
+
- STEP: the step sentence in a natural language
|
|
39
|
+
- PREVIOUS STEPS: the sentences of the previous steps of the test, in order
|
|
40
|
+
- PAGE SNAPSHOT: the accessibility snapshot of the current page
|
|
41
|
+
- SCREENSHOT: an image of the page, when attached
|
|
42
|
+
- PAGE API: the exact surface listing of the page facade — call nothing outside it
|
|
43
|
+
- USER INSTRUCTIONS: the project's code style guidance, when configured
|
|
44
|
+
- CODE: the existing step code that failed (regeneration requests only)
|
|
45
|
+
- ERROR: the failure description of the existing code (regeneration requests only)
|
|
46
|
+
|
|
47
|
+
Output exactly one Python code block with one function of the fixed form:
|
|
48
|
+
|
|
49
|
+
def step(page) -> None:
|
|
50
|
+
...
|
|
51
|
+
|
|
52
|
+
Rules:
|
|
53
|
+
- The function receives exactly one argument: the page facade. Never import anything, never use other libraries
|
|
54
|
+
- Work only through the page API: the request carries the exact surface listing of the page facade — call nothing outside it
|
|
55
|
+
- For an assertion sentence end with an expectation call; for an action sentence perform the actions
|
|
56
|
+
- Locating by role and accessible name is preferred; by visible text next; by label for form fields
|
|
57
|
+
- Attribute, CSS and XPath locating exist for elements without accessible names — the accessibility-first priority stands unless USER INSTRUCTIONS say otherwise
|
|
58
|
+
- Scroll abilities exist for scenario scrolling: bring an element into view, scroll by an amount, to the page end or start, inside a scrollable container
|
|
59
|
+
- No fixed delays, no sleeps, no explicit waits — the facade waits itself
|
|
60
|
+
- The step must complete exactly what STEP says — nothing more, nothing less
|
|
61
|
+
- Output only the code block, no explanations
|
|
62
|
+
classification_prompt: |
|
|
63
|
+
You classify a failure of a web UI test step.
|
|
64
|
+
|
|
65
|
+
Input you receive:
|
|
66
|
+
- STEP: the step sentence
|
|
67
|
+
- CODE: the step code that failed
|
|
68
|
+
- ERROR: the failure description
|
|
69
|
+
- PAGE SNAPSHOT: the accessibility snapshot of the current page
|
|
70
|
+
- SCREENSHOT: an image of the page, when attached
|
|
71
|
+
|
|
72
|
+
Answer with exactly one line of the form:
|
|
73
|
+
category | explanation | recommendation
|
|
74
|
+
|
|
75
|
+
where category is one of:
|
|
76
|
+
- rot — the UI changed (selectors, texts, structure) and the step can be regenerated for the same intent
|
|
77
|
+
- product_defect — the step works as written but the expected behavior of the application is genuinely broken
|
|
78
|
+
- incurable — the step sentence no longer matches reality, the intent is ambiguous, or regeneration cannot help
|
|
79
|
+
|
|
80
|
+
explanation: one short sentence why. recommendation: one short sentence what the engineer should do.
|
|
81
|
+
Output only that single line — no code, no extra text.
|
|
82
|
+
|
|
83
|
+
Annotations: |
|
|
84
|
+
Use `conventions` for code writing rules and testing.
|
|
85
|
+
Use `facade` from Imports for the page API the generated code works through.
|
|
86
|
+
Use `classification` from Imports for the healing decision categories.
|
|
87
|
+
Use `system_prompt` as the system prompt of every code generation request.
|
|
88
|
+
Use `classification_prompt` as the system prompt of every failure classification request.
|
|
89
|
+
Use `facade` from Imports as the single source of the page API surface for generation requests.
|
|
90
|
+
|
|
91
|
+
The fixed form of step code: one function receiving exactly one argument — the `PageFacade` of the test; the function body works only through `PageFacade` and the LocatorFacade element API.
|
|
92
|
+
Every attempt — generation or healing — consumes the shared per-step budget of the test; exhaustion is the incurable failure, never an infinite loop.
|
|
93
|
+
Healing never masks a product defect: a classified product_defect fails the test loudly; a healed step is reported loudly and written back to the cache.
|
|
94
|
+
Every terminal failure carries a verdict — a `FailureVerdict` of category, explanation, recommendation — fully present in the exception message, the verdict hook event and the log; an unavailable LLM skips the verdict quietly with a WARNING, the failure itself is never delayed or distorted — the quiet skip applies to verdicts enriching an already-decided failure; the classification driving the healing decision surfaces as the infrastructure failure.
|
|
95
|
+
A failed check of a candidate stops the generation retries: a check that executed and did not hold is classified, not regenerated.
|
|
96
|
+
The page API surface listing sent to the provider mirrors `facade` from Imports exactly — the listing and the practice change together.
|
|
97
|
+
The user instructions of the project settings — the generation_prompt field of `Config` — reach generation and regeneration requests only; classification requests never carry them; the instructions take no part in the step address: a cached step never regenerates because the instructions changed.
|
|
98
|
+
|
|
99
|
+
---
|
|
100
|
+
|
|
101
|
+
"StepGenerator(config: Config, provider: LlmProvider, cache: StepCache, budgets: RunBudgets, reporter: StepReporter)":
|
|
102
|
+
location: generator.py
|
|
103
|
+
annotations: |
|
|
104
|
+
The generation engine: produce working step code for an unknown step, executing candidates against the live page.
|
|
105
|
+
|
|
106
|
+
`config`: project settings — the screenshot flag and the generation instructions.
|
|
107
|
+
`provider`: the LLM port.
|
|
108
|
+
`cache`: the step cache — a generated step is stored on success.
|
|
109
|
+
`budgets`: the per-test attempt registry.
|
|
110
|
+
`reporter`: the visibility point.
|
|
111
|
+
methods:
|
|
112
|
+
"generate(identity: StepIdentity, step_text: str, previous_steps: list[str], page: PageFacade) -> step: CachedStep": |
|
|
113
|
+
Generate and store a new step.
|
|
114
|
+
|
|
115
|
+
Algorithm:
|
|
116
|
+
1. Ask the budgets registry try_generation for the step identity; a refused attempt is the incurable failure
|
|
117
|
+
2. Collect the request inputs: the page accessibility snapshot, the step sentence, `previous_steps`, and the page API surface from `facade`; add the page screenshot when the project settings enable screenshots
|
|
118
|
+
3. Request step code from the provider port generate_step_code passing `system_prompt` as the system prompt, the page API surface listing taken from `facade`, and the user instructions — the effective config generation_prompt — when non-empty
|
|
119
|
+
4. Execute the candidate with `run_step_code` against `page`
|
|
120
|
+
5. On success: build `CachedStep`, save it to the cache, return it
|
|
121
|
+
6. On a candidate failure that is an AssertionError — a check that executed and did not hold: stop the retries immediately and classify via `classify_step_failure`; a product_defect verdict raises `ProductDefectError` carrying the verdict, any other verdict raises `IncurableStepError` carrying it — the reason names the failed candidate check; provider unavailability at this classification is skipped quietly with a WARNING — `IncurableStepError` is raised without a verdict, the reason naming the failed candidate check (the failed check is the primary signal, the verdict is enrichment)
|
|
122
|
+
7. On any other candidate failure: repeat from step 1 with the fresh failure description and the fresh snapshot, while attempts remain
|
|
123
|
+
8. On budget exhaustion: classify the last candidate via `classify_step_failure` and raise `IncurableStepError` carrying the verdict — the reason names the exhausted pool and the last candidate failure; provider unavailability at this classification is skipped quietly with a WARNING, the failure raises without a verdict
|
|
124
|
+
9. Report on_generation_started for every attempt
|
|
125
|
+
|
|
126
|
+
Requirements:
|
|
127
|
+
- Every generation request carries the exact page API surface taken from `facade` from Imports: the model always sees the precise list of calls it may use
|
|
128
|
+
- Provider unavailability of a generation request surfaces as `LlmUnavailableError` immediately — no retry on it
|
|
129
|
+
- Provider unavailability of a failed-check classification is skipped quietly with a WARNING: the failure raises without a verdict — the failed check itself is the primary signal
|
|
130
|
+
- Exactly one failed check stops the retries: the attempt budget is never spent on a legitimately failing assertion
|
|
131
|
+
- A verdict requested on this path fully reaches the raised error
|
|
132
|
+
"regenerate(identity: StepIdentity, step_text: str, previous_steps: list[str], page: PageFacade, existing_code: str, error: str) -> step: CachedStep": |
|
|
133
|
+
Regenerate a failed step for healing.
|
|
134
|
+
|
|
135
|
+
Algorithm:
|
|
136
|
+
1. The same loop as the generate method with three additions: every provider request carries `existing_code` and `error` and the user instructions — the effective config generation_prompt — when non-empty; attempts consume the healing budget via try_healing; a budget exhaustion raises `IncurableStepError` without classification — the healer attaches the verdict of its own classification, no extra LLM request is made (the failed-check classification of the generate loop applies on both pools)
|
|
137
|
+
|
|
138
|
+
"run_step_code(code: str, page: PageFacade)":
|
|
139
|
+
location: execution.py
|
|
140
|
+
annotations: |
|
|
141
|
+
Execute step code of the fixed form against the page facade of the test.
|
|
142
|
+
|
|
143
|
+
`code`: the step code text.
|
|
144
|
+
`page`: the page facade of the current test.
|
|
145
|
+
|
|
146
|
+
Algorithm:
|
|
147
|
+
1. Compile and load `code` as a module in an isolated namespace
|
|
148
|
+
2. Resolve the step function of the fixed form — the single callable receiving the page facade
|
|
149
|
+
3. Call it with `page`
|
|
150
|
+
|
|
151
|
+
Requirements:
|
|
152
|
+
- A failure inside the step code propagates to the caller as-is: the engine classifies it, this routine never swallows or retries
|
|
153
|
+
- Executing step code loads no LLM provider and touches no network beyond the page itself
|
|
154
|
+
|
|
155
|
+
Constraints:
|
|
156
|
+
- Execute only step code produced by generation or loaded from the cache — never arbitrary file content
|
|
157
|
+
|
|
158
|
+
"classify_step_failure(config: Config, provider: LlmProvider, step_text: str, code: str, error: str, page: PageFacade) -> classification: FailureClassification":
|
|
159
|
+
location: classification.py
|
|
160
|
+
annotations: |
|
|
161
|
+
Classify a step failure: collect the page state and ask the provider — the single classification call for both engines.
|
|
162
|
+
|
|
163
|
+
`config`: project settings — the screenshot flag.
|
|
164
|
+
`provider`: the LLM port.
|
|
165
|
+
`step_text`: the sentence of the failed step.
|
|
166
|
+
`code`: the step code that failed.
|
|
167
|
+
`error`: the human-readable failure description.
|
|
168
|
+
`page`: the page facade of the current test.
|
|
169
|
+
`classification`: the `FailureClassification` verdict.
|
|
170
|
+
|
|
171
|
+
Algorithm:
|
|
172
|
+
1. Collect the classification inputs: the step sentence, the failed code, the `error` text, the fresh page snapshot — plus the screenshot when enabled
|
|
173
|
+
2. Ask the provider port classify_failure passing `classification_prompt` as the system prompt
|
|
174
|
+
3. Return the verdict
|
|
175
|
+
|
|
176
|
+
Constraints:
|
|
177
|
+
- Provider unavailability propagates to the caller: this routine never swallows it — the calling path decides whether it is a terminal infrastructure failure or a quiet verdict skip
|
|
178
|
+
|
|
179
|
+
"StepHealer(config: Config, provider: LlmProvider, generator: StepGenerator, cache: StepCache, budgets: RunBudgets, reporter: StepReporter)":
|
|
180
|
+
location: healer.py
|
|
181
|
+
annotations: |
|
|
182
|
+
The healing engine: classify a failed cached step, regenerate rot, never mask a defect.
|
|
183
|
+
|
|
184
|
+
`config`: project settings — the screenshot flag.
|
|
185
|
+
`provider`: the LLM port for classification.
|
|
186
|
+
`generator`: the regeneration engine.
|
|
187
|
+
`cache`: the step cache for the healed write-back.
|
|
188
|
+
`budgets`: the per-test attempt registry.
|
|
189
|
+
`reporter`: the visibility point.
|
|
190
|
+
methods:
|
|
191
|
+
"heal(step: CachedStep, error: str, previous_steps: list[str], page: PageFacade) -> step: CachedStep": |
|
|
192
|
+
Classify and heal a failed cached step.
|
|
193
|
+
|
|
194
|
+
`previous_steps`: the sentences of the previous steps of the test, in execution order — scenario context for regeneration.
|
|
195
|
+
|
|
196
|
+
Algorithm:
|
|
197
|
+
1. Classify the failure via `classify_step_failure` — the verdict is a `FailureClassification`
|
|
198
|
+
2. Report on_healing_started with the category
|
|
199
|
+
3. product_defect: raise `ProductDefectError` carrying the verdict built from the classification — the message states what was expected against what was observed; the recommendation reaches the error through the verdict
|
|
200
|
+
4. incurable: raise `IncurableStepError` carrying the verdict; the reason names the classification explanation of incurability
|
|
201
|
+
5. rot: regenerate via the generator regenerate — its loop executes the candidate and stores the healed step on success; report on_healed with the explanation of what was rot and what changed, return the healed step
|
|
202
|
+
6. A regeneration budget exhaustion inside step 5 surfaces as `IncurableStepError` carrying the verdict of the step 1 classification — the reason names the exhausted pool; no extra LLM request is made
|
|
203
|
+
7. Provider unavailability of the classification surfaces as `LlmUnavailableError` — an explicit infrastructure failure
|
|
204
|
+
|
|
205
|
+
Requirements:
|
|
206
|
+
- Anti-masking: healing may only turn a rot-failed step green; a classified product defect always fails the test
|
|
207
|
+
- The healed code replaces the cached code only after a successful execution
|
|
208
|
+
- Every verdict produced on the paths of this method fully reaches the raised error
|
|
209
|
+
|
|
210
|
+
---
|
|
211
|
+
|
|
212
|
+
Author: Goga
|
|
213
|
+
CreatedAt: 07/09/26
|
|
214
|
+
Description: |
|
|
215
|
+
The agent engine of prettyplay: step code generation with execution in the loop and failed-check classification, the fixed-form execution routine, the shared classification call, and healing with anti-masking and verdicts on terminal failures.
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""Facade of the prettyplay.engine cell: generation, execution and healing of step code."""
|
|
2
|
+
|
|
3
|
+
from .classification import classify_step_failure
|
|
4
|
+
from .execution import run_step_code
|
|
5
|
+
from .generator import StepGenerator
|
|
6
|
+
from .healer import StepHealer
|
|
7
|
+
|
|
8
|
+
__all__ = ["StepGenerator", "StepHealer", "classify_step_failure", "run_step_code"]
|
|
@@ -0,0 +1,69 @@
|
|
|
1
|
+
"""The shared classification call: collect the page state and ask the provider."""
|
|
2
|
+
|
|
3
|
+
from ..config import Config
|
|
4
|
+
from ..driver import PageFacade
|
|
5
|
+
from ..llm import FailureClassification, LlmProvider
|
|
6
|
+
|
|
7
|
+
#: System prompt of every classification request; used by both engines, applied verbatim.
|
|
8
|
+
CLASSIFICATION_PROMPT = """You classify a failure of a web UI test step.
|
|
9
|
+
|
|
10
|
+
Input you receive:
|
|
11
|
+
- STEP: the step sentence
|
|
12
|
+
- CODE: the step code that failed
|
|
13
|
+
- ERROR: the failure description
|
|
14
|
+
- PAGE SNAPSHOT: the accessibility snapshot of the current page
|
|
15
|
+
- SCREENSHOT: an image of the page, when attached
|
|
16
|
+
|
|
17
|
+
Answer with exactly one line of the form:
|
|
18
|
+
category | explanation | recommendation
|
|
19
|
+
|
|
20
|
+
where category is one of:
|
|
21
|
+
- rot — the UI changed (selectors, texts, structure) and the step can be regenerated for the same intent
|
|
22
|
+
- product_defect — the step works as written but the expected behavior of the application is genuinely broken
|
|
23
|
+
- incurable — the step sentence no longer matches reality, the intent is ambiguous, or regeneration cannot help
|
|
24
|
+
|
|
25
|
+
explanation: one short sentence why. recommendation: one short sentence what the engineer should do.
|
|
26
|
+
Output only that single line — no code, no extra text."""
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def classify_step_failure( # noqa: PLR0913, PLR0917 — the parameter list is fixed by the engine contract
|
|
30
|
+
config: Config,
|
|
31
|
+
provider: LlmProvider,
|
|
32
|
+
step_text: str,
|
|
33
|
+
code: str,
|
|
34
|
+
error: str,
|
|
35
|
+
page: PageFacade,
|
|
36
|
+
) -> FailureClassification:
|
|
37
|
+
"""Classify a step failure: collect the page state and ask the provider.
|
|
38
|
+
|
|
39
|
+
The single classification call for both engines — the generator failed-check
|
|
40
|
+
stop and exhaustion, the healer verdict. Provider unavailability propagates
|
|
41
|
+
to the caller: this routine never swallows it — the calling path decides
|
|
42
|
+
whether it is a terminal infrastructure failure or a quiet verdict skip.
|
|
43
|
+
|
|
44
|
+
Args:
|
|
45
|
+
config: project settings; ``send_screenshots`` attaches page images.
|
|
46
|
+
provider: the LLM port implementation classifying the failure.
|
|
47
|
+
step_text: the sentence of the failed step.
|
|
48
|
+
code: the step code that failed.
|
|
49
|
+
error: the human-readable failure description.
|
|
50
|
+
page: the page facade of the current test.
|
|
51
|
+
|
|
52
|
+
Returns:
|
|
53
|
+
The classification verdict.
|
|
54
|
+
|
|
55
|
+
Raises:
|
|
56
|
+
LlmUnavailableError: the provider service failed; the calling path
|
|
57
|
+
decides the handling.
|
|
58
|
+
"""
|
|
59
|
+
snapshot = page.aria_snapshot()
|
|
60
|
+
screenshot = page.screenshot() if config.send_screenshots else None
|
|
61
|
+
|
|
62
|
+
return provider.classify_failure(
|
|
63
|
+
prompt=CLASSIFICATION_PROMPT,
|
|
64
|
+
step_text=step_text,
|
|
65
|
+
code=code,
|
|
66
|
+
error=error,
|
|
67
|
+
snapshot=snapshot,
|
|
68
|
+
screenshot=screenshot,
|
|
69
|
+
)
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
"""Execution of step code of the fixed form against the page facade of the test."""
|
|
2
|
+
|
|
3
|
+
from ..driver import PageFacade
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def run_step_code(code: str, page: PageFacade) -> None:
|
|
7
|
+
"""Execute step code of the fixed form against the page facade of the test.
|
|
8
|
+
|
|
9
|
+
The code is compiled and loaded in an isolated namespace; the step
|
|
10
|
+
function of the fixed form — the single callable named ``step`` — is
|
|
11
|
+
resolved and called with the page facade. A failure inside the step code
|
|
12
|
+
propagates to the caller as-is: the engine classifies it, this routine
|
|
13
|
+
never swallows or retries. No module is registered in ``sys.modules`` and
|
|
14
|
+
no provider or network beyond the page itself is touched.
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
code: the step code text produced by generation or loaded from the cache.
|
|
18
|
+
page: the page facade of the current test.
|
|
19
|
+
"""
|
|
20
|
+
namespace: dict[str, object] = {}
|
|
21
|
+
# the engine contract: exec of the fixed-form step code in an isolated namespace
|
|
22
|
+
exec(compile(code, "<prettyplay-step>", "exec"), namespace)
|
|
23
|
+
|
|
24
|
+
fn = namespace["step"]
|
|
25
|
+
fn(page)
|
|
@@ -0,0 +1,318 @@
|
|
|
1
|
+
"""Generation of working step code: LLM candidates executed against the live page in a loop."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
from datetime import date
|
|
5
|
+
|
|
6
|
+
from ..cache import CachedStep, RunBudgets, StepCache, StepIdentity
|
|
7
|
+
from ..config import Config
|
|
8
|
+
from ..driver import PageFacade
|
|
9
|
+
from ..failures import FailureVerdict, IncurableStepError, LlmUnavailableError, ProductDefectError
|
|
10
|
+
from ..llm import FailureClassification, LlmProvider
|
|
11
|
+
from ..reporting import StepReporter
|
|
12
|
+
from .classification import classify_step_failure
|
|
13
|
+
from .execution import run_step_code
|
|
14
|
+
from .text import first_line_short
|
|
15
|
+
|
|
16
|
+
logger = logging.getLogger("prettyplay")
|
|
17
|
+
|
|
18
|
+
#: System prompt of every generation request; applied verbatim by the provider.
|
|
19
|
+
SYSTEM_PROMPT = """You generate executable Python code for one step of a web UI test.
|
|
20
|
+
|
|
21
|
+
Input you receive:
|
|
22
|
+
- STEP: the step sentence in a natural language
|
|
23
|
+
- PREVIOUS STEPS: the sentences of the previous steps of the test, in order
|
|
24
|
+
- PAGE SNAPSHOT: the accessibility snapshot of the current page
|
|
25
|
+
- SCREENSHOT: an image of the page, when attached
|
|
26
|
+
- PAGE API: the exact surface listing of the page facade — call nothing outside it
|
|
27
|
+
- USER INSTRUCTIONS: the project's code style guidance, when configured
|
|
28
|
+
- CODE: the existing step code that failed (regeneration requests only)
|
|
29
|
+
- ERROR: the failure description of the existing code (regeneration requests only)
|
|
30
|
+
|
|
31
|
+
Output exactly one Python code block with one function of the fixed form:
|
|
32
|
+
|
|
33
|
+
def step(page) -> None:
|
|
34
|
+
...
|
|
35
|
+
|
|
36
|
+
Rules:
|
|
37
|
+
- The function receives exactly one argument: the page facade. Never import anything, never use other libraries
|
|
38
|
+
- Work only through the page API: the request carries the exact surface listing of the page facade — call nothing outside it
|
|
39
|
+
- For an assertion sentence end with an expectation call; for an action sentence perform the actions
|
|
40
|
+
- Locating by role and accessible name is preferred; by visible text next; by label for form fields
|
|
41
|
+
- Attribute, CSS and XPath locating exist for elements without accessible names — the accessibility-first priority stands unless USER INSTRUCTIONS say otherwise
|
|
42
|
+
- Scroll abilities exist for scenario scrolling: bring an element into view, scroll by an amount, to the page end or start, inside a scrollable container
|
|
43
|
+
- No fixed delays, no sleeps, no explicit waits — the facade waits itself
|
|
44
|
+
- The step must complete exactly what STEP says — nothing more, nothing less
|
|
45
|
+
- Output only the code block, no explanations"""
|
|
46
|
+
|
|
47
|
+
#: Frozen surface listing of the driver facade — the only calls step code may make.
|
|
48
|
+
#: Mirrors ``prettyplay/driver/.usages/facade.md`` verbatim; the driver facade is a
|
|
49
|
+
#: backward-compatibility contract, so this constant changes only together with it.
|
|
50
|
+
#: ``close()`` stays out: it is a runtime method of PrettyTest, not of step code.
|
|
51
|
+
PAGE_API_SURFACE = """page.open(url) — navigate and wait for load
|
|
52
|
+
page.find_by_role(role, name) — element by aria role and accessible name
|
|
53
|
+
page.find_by_label(label) — element by associated label
|
|
54
|
+
page.find_by_text(text) — element by visible text
|
|
55
|
+
page.find_by_attribute(name, value) — element by attribute value — data-* attributes
|
|
56
|
+
page.find_by_css(selector) — element by CSS selector
|
|
57
|
+
page.find_by_xpath(xpath) — element by XPath expression
|
|
58
|
+
page.aria_snapshot() — accessibility-tree page state
|
|
59
|
+
page.screenshot() — full-page PNG bytes
|
|
60
|
+
page.url — current URL
|
|
61
|
+
page.scroll_to_element(element) — bring an element into the viewport (works inside scrollable ancestors)
|
|
62
|
+
page.scroll_down(pixels) — scroll the page down by an amount
|
|
63
|
+
page.scroll_up(pixels) — scroll the page up by an amount
|
|
64
|
+
page.scroll_to_bottom() — scroll to the end of the page
|
|
65
|
+
page.scroll_to_top() — scroll to the start of the page
|
|
66
|
+
page.scroll_into_view(element, container) — bring an element into view inside a specific scrollable container
|
|
67
|
+
page.scroll_container_down(container, pixels) — scroll a scrollable container down by an amount
|
|
68
|
+
page.scroll_container_up(container, pixels) — scroll a scrollable container up by an amount
|
|
69
|
+
element.click() — click with auto-wait
|
|
70
|
+
element.fill(value) — set input text
|
|
71
|
+
element.select_option(value) — choose an option
|
|
72
|
+
element.expect_visible() — assert visible
|
|
73
|
+
element.expect_text(text) — assert text
|
|
74
|
+
element.expect_enabled() — assert enabled"""
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
class StepGenerator:
|
|
78
|
+
"""Generates working step code by executing LLM candidates against the live page.
|
|
79
|
+
|
|
80
|
+
Each attempt is one provider request: the generator snapshots the page,
|
|
81
|
+
asks for code of the fixed form, and immediately executes the candidate.
|
|
82
|
+
A failing candidate is retried as a regeneration request carrying the code
|
|
83
|
+
and its error, until one candidate works or the attempt budget of the step
|
|
84
|
+
runs out. Only a proven candidate is cached — failures are never stored.
|
|
85
|
+
|
|
86
|
+
Attributes:
|
|
87
|
+
_config: project settings; the screenshot flag feeds the requests.
|
|
88
|
+
_provider: the LLM port implementation doing the requests.
|
|
89
|
+
_cache: the store where working steps are saved.
|
|
90
|
+
_budgets: the per-test attempt registry of the engine.
|
|
91
|
+
_reporter: the visibility point for generation and cache events.
|
|
92
|
+
"""
|
|
93
|
+
|
|
94
|
+
def __init__(
|
|
95
|
+
self,
|
|
96
|
+
config: Config,
|
|
97
|
+
provider: LlmProvider,
|
|
98
|
+
cache: StepCache,
|
|
99
|
+
budgets: RunBudgets,
|
|
100
|
+
reporter: StepReporter,
|
|
101
|
+
) -> None:
|
|
102
|
+
"""Keep the collaborators of the generation loop.
|
|
103
|
+
|
|
104
|
+
Args:
|
|
105
|
+
config: project settings; ``send_screenshots`` attaches page images.
|
|
106
|
+
provider: the LLM port implementation.
|
|
107
|
+
cache: the store of working steps.
|
|
108
|
+
budgets: the per-test attempt registry.
|
|
109
|
+
reporter: the visibility point for engine events.
|
|
110
|
+
"""
|
|
111
|
+
self._config = config
|
|
112
|
+
self._provider = provider
|
|
113
|
+
self._cache = cache
|
|
114
|
+
self._budgets = budgets
|
|
115
|
+
self._reporter = reporter
|
|
116
|
+
|
|
117
|
+
def generate(
|
|
118
|
+
self,
|
|
119
|
+
identity: StepIdentity,
|
|
120
|
+
step_text: str,
|
|
121
|
+
previous_steps: list[str],
|
|
122
|
+
page: PageFacade,
|
|
123
|
+
) -> CachedStep:
|
|
124
|
+
"""Generate step code until a candidate works, then cache it.
|
|
125
|
+
|
|
126
|
+
Args:
|
|
127
|
+
identity: the address of the step.
|
|
128
|
+
step_text: the sentence of the step.
|
|
129
|
+
previous_steps: the sentences of the previous steps of the test.
|
|
130
|
+
page: the live page facade the candidates run against.
|
|
131
|
+
|
|
132
|
+
Returns:
|
|
133
|
+
The cached step holding the proven code.
|
|
134
|
+
|
|
135
|
+
Raises:
|
|
136
|
+
IncurableStepError: the generation attempt budget is exhausted.
|
|
137
|
+
LlmUnavailableError: the provider service failed; no retry.
|
|
138
|
+
"""
|
|
139
|
+
return self._loop(identity, step_text, previous_steps, page, "generation", None, None)
|
|
140
|
+
|
|
141
|
+
def regenerate( # noqa: PLR0913, PLR0917 — the signature is fixed by the engine contract
|
|
142
|
+
self,
|
|
143
|
+
identity: StepIdentity,
|
|
144
|
+
step_text: str,
|
|
145
|
+
previous_steps: list[str],
|
|
146
|
+
page: PageFacade,
|
|
147
|
+
existing_code: str,
|
|
148
|
+
error: str,
|
|
149
|
+
) -> CachedStep:
|
|
150
|
+
"""Regenerate step code starting from the failed candidate.
|
|
151
|
+
|
|
152
|
+
The loop is the generation loop; the differences are the healing budget
|
|
153
|
+
pool and the failed code and error carried by the first request.
|
|
154
|
+
|
|
155
|
+
Args:
|
|
156
|
+
identity: the address of the step.
|
|
157
|
+
step_text: the sentence of the step.
|
|
158
|
+
previous_steps: the sentences of the previous steps of the test.
|
|
159
|
+
page: the live page facade the candidates run against.
|
|
160
|
+
existing_code: the cached step code that failed.
|
|
161
|
+
error: the failure description of the existing code.
|
|
162
|
+
|
|
163
|
+
Returns:
|
|
164
|
+
The cached step holding the proven regenerated code.
|
|
165
|
+
|
|
166
|
+
Raises:
|
|
167
|
+
IncurableStepError: the healing attempt budget is exhausted.
|
|
168
|
+
LlmUnavailableError: the provider service failed; no retry.
|
|
169
|
+
"""
|
|
170
|
+
return self._loop(identity, step_text, previous_steps, page, "healing", existing_code, error)
|
|
171
|
+
|
|
172
|
+
def _loop( # noqa: PLR0913, PLR0917 — the shared attempt loop with its fixed inputs
|
|
173
|
+
self,
|
|
174
|
+
identity: StepIdentity,
|
|
175
|
+
step_text: str,
|
|
176
|
+
previous_steps: list[str],
|
|
177
|
+
page: PageFacade,
|
|
178
|
+
pool: str,
|
|
179
|
+
existing_code: str | None,
|
|
180
|
+
error: str | None,
|
|
181
|
+
) -> CachedStep:
|
|
182
|
+
"""Run the shared attempt loop until a candidate works or the budget runs out.
|
|
183
|
+
|
|
184
|
+
A failed check — a candidate ``AssertionError`` — stops the retries at
|
|
185
|
+
once and is classified: the attempt budget is never spent on a
|
|
186
|
+
legitimately failing assertion. Any other candidate failure is retried
|
|
187
|
+
as a regeneration request carrying the code and its error. When the
|
|
188
|
+
budget runs out, the generation pool classifies the last candidate;
|
|
189
|
+
the healing pool raises without classification — the healer attaches
|
|
190
|
+
the verdict it already holds, so no extra LLM request is made.
|
|
191
|
+
|
|
192
|
+
Args:
|
|
193
|
+
identity: the address of the step.
|
|
194
|
+
step_text: the sentence of the step.
|
|
195
|
+
previous_steps: the sentences of the previous steps of the test.
|
|
196
|
+
page: the live page facade the candidates run against.
|
|
197
|
+
pool: the budget pool name — "generation" or "healing".
|
|
198
|
+
existing_code: the failed code of the first request, if any.
|
|
199
|
+
error: the failure description of the first request, if any.
|
|
200
|
+
|
|
201
|
+
Returns:
|
|
202
|
+
The cached step holding the proven code.
|
|
203
|
+
|
|
204
|
+
Raises:
|
|
205
|
+
ProductDefectError: a failed candidate check classified as a
|
|
206
|
+
genuine product defect.
|
|
207
|
+
IncurableStepError: the attempt budget of the step is exhausted,
|
|
208
|
+
or a failure classified as incurable.
|
|
209
|
+
LlmUnavailableError: the provider service failed; no retry.
|
|
210
|
+
"""
|
|
211
|
+
attempt = 0
|
|
212
|
+
code = None
|
|
213
|
+
spend = self._budgets.try_generation if pool == "generation" else self._budgets.try_healing
|
|
214
|
+
|
|
215
|
+
while True:
|
|
216
|
+
if not spend(identity):
|
|
217
|
+
# candidate cause is part of the reason contract: «the specific incurability cause»
|
|
218
|
+
reason = f"{pool} attempt budget exhausted"
|
|
219
|
+
|
|
220
|
+
if error:
|
|
221
|
+
reason = f"{reason}; last failure: {error}"
|
|
222
|
+
if pool == "healing":
|
|
223
|
+
raise IncurableStepError(
|
|
224
|
+
step_text,
|
|
225
|
+
reason,
|
|
226
|
+
None, # healer attaches the verdict — no second LLM request
|
|
227
|
+
)
|
|
228
|
+
if error is None:
|
|
229
|
+
raise IncurableStepError(
|
|
230
|
+
step_text,
|
|
231
|
+
reason,
|
|
232
|
+
None, # nothing to classify: no candidates existed
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
raise IncurableStepError(step_text, reason, self._classify(step_text, code, error, page))
|
|
236
|
+
|
|
237
|
+
attempt += 1
|
|
238
|
+
self._reporter.emit("on_generation_started", {"step_text": step_text, "attempt": attempt})
|
|
239
|
+
|
|
240
|
+
snapshot = page.aria_snapshot()
|
|
241
|
+
screenshot = page.screenshot() if self._config.send_screenshots else None
|
|
242
|
+
|
|
243
|
+
code = self._provider.generate_step_code(
|
|
244
|
+
prompt=SYSTEM_PROMPT,
|
|
245
|
+
user_instructions=self._config.generation_prompt,
|
|
246
|
+
step_text=step_text,
|
|
247
|
+
previous_steps=previous_steps,
|
|
248
|
+
snapshot=snapshot,
|
|
249
|
+
screenshot=screenshot,
|
|
250
|
+
page_api=PAGE_API_SURFACE,
|
|
251
|
+
existing_code=existing_code,
|
|
252
|
+
error=error,
|
|
253
|
+
)
|
|
254
|
+
|
|
255
|
+
try:
|
|
256
|
+
run_step_code(code, page)
|
|
257
|
+
except AssertionError as check_failure:
|
|
258
|
+
# failed check: attempts not spent — classify and stop
|
|
259
|
+
reason = f"candidate check failed: {first_line_short(check_failure)}"
|
|
260
|
+
verdict = self._classify(step_text, code, reason, page)
|
|
261
|
+
if verdict is not None and verdict.category == "product_defect":
|
|
262
|
+
raise ProductDefectError(step_text, first_line_short(check_failure), verdict) from None
|
|
263
|
+
raise IncurableStepError(step_text, reason, verdict) from None
|
|
264
|
+
except Exception as candidate_error: # other candidate failures heal via retry
|
|
265
|
+
existing_code = code
|
|
266
|
+
error = first_line_short(candidate_error)
|
|
267
|
+
else:
|
|
268
|
+
break
|
|
269
|
+
|
|
270
|
+
step = CachedStep(
|
|
271
|
+
identity=identity,
|
|
272
|
+
code=code,
|
|
273
|
+
created_at=date.today().isoformat(), # noqa: DTZ011 — calendar date of step creation
|
|
274
|
+
)
|
|
275
|
+
self._cache.save(step)
|
|
276
|
+
|
|
277
|
+
return step
|
|
278
|
+
|
|
279
|
+
def _classify(self, step_text: str, code: str | None, error: str, page: PageFacade) -> FailureVerdict | None:
|
|
280
|
+
"""Classify a failure through the shared routine with the quiet skip.
|
|
281
|
+
|
|
282
|
+
Every classification inside the loop enriches an already-decided
|
|
283
|
+
failure, so an unavailable LLM yields no verdict — the failure never
|
|
284
|
+
waits for it and never turns into an infrastructure error.
|
|
285
|
+
|
|
286
|
+
Args:
|
|
287
|
+
step_text: the sentence of the failed step.
|
|
288
|
+
code: the code of the last candidate.
|
|
289
|
+
error: the failure description of the candidate.
|
|
290
|
+
page: the live page facade of the test.
|
|
291
|
+
|
|
292
|
+
Returns:
|
|
293
|
+
The verdict of the classification, or ``None`` when the LLM was
|
|
294
|
+
unavailable — the quiet skip logs a WARNING.
|
|
295
|
+
"""
|
|
296
|
+
try:
|
|
297
|
+
classification = classify_step_failure(self._config, self._provider, step_text, code, error, page)
|
|
298
|
+
except LlmUnavailableError:
|
|
299
|
+
logger.warning("verdict skipped: llm unavailable")
|
|
300
|
+
return None
|
|
301
|
+
|
|
302
|
+
return _verdict(classification)
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def _verdict(classification: FailureClassification) -> FailureVerdict:
|
|
306
|
+
"""Build the verdict value object carried by the terminal errors.
|
|
307
|
+
|
|
308
|
+
Args:
|
|
309
|
+
classification: the classification verdict of the provider.
|
|
310
|
+
|
|
311
|
+
Returns:
|
|
312
|
+
The frozen verdict of the terminal failure.
|
|
313
|
+
"""
|
|
314
|
+
return FailureVerdict(
|
|
315
|
+
category=classification.category,
|
|
316
|
+
explanation=classification.explanation,
|
|
317
|
+
recommendation=classification.recommendation,
|
|
318
|
+
)
|