prettyplay 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of prettyplay might be problematic. Click here for more details.
- prettyplay/.usages/lifecycle.md +50 -0
- prettyplay/.usages/steps.md +56 -0
- prettyplay/CODEMANIFEST +182 -0
- prettyplay/__init__.py +13 -0
- prettyplay/cache/.usages/addressing.md +31 -0
- prettyplay/cache/.usages/budgets.md +21 -0
- prettyplay/cache/.usages/storage.md +32 -0
- prettyplay/cache/CODEMANIFEST +152 -0
- prettyplay/cache/__init__.py +8 -0
- prettyplay/cache/budgets.py +73 -0
- prettyplay/cache/models.py +60 -0
- prettyplay/cache/store.py +260 -0
- prettyplay/cache/text.py +35 -0
- prettyplay/config/.usages/configuration.md +87 -0
- prettyplay/config/CODEMANIFEST +132 -0
- prettyplay/config/__init__.py +6 -0
- prettyplay/config/loader.py +192 -0
- prettyplay/config/models.py +92 -0
- prettyplay/driver/.usages/facade.md +66 -0
- prettyplay/driver/CODEMANIFEST +162 -0
- prettyplay/driver/__init__.py +6 -0
- prettyplay/driver/page.py +350 -0
- prettyplay/driver/session.py +288 -0
- prettyplay/engine/.usages/generation.md +45 -0
- prettyplay/engine/.usages/healing.md +31 -0
- prettyplay/engine/CODEMANIFEST +215 -0
- prettyplay/engine/__init__.py +8 -0
- prettyplay/engine/classification.py +69 -0
- prettyplay/engine/execution.py +25 -0
- prettyplay/engine/generator.py +318 -0
- prettyplay/engine/healer.py +116 -0
- prettyplay/engine/text.py +19 -0
- prettyplay/executor.py +109 -0
- prettyplay/failures/.usages/taxonomy.md +40 -0
- prettyplay/failures/CODEMANIFEST +117 -0
- prettyplay/failures/__init__.py +17 -0
- prettyplay/failures/errors.py +147 -0
- prettyplay/llm/.usages/classification.md +25 -0
- prettyplay/llm/.usages/providers.md +33 -0
- prettyplay/llm/CODEMANIFEST +125 -0
- prettyplay/llm/__init__.py +8 -0
- prettyplay/llm/_request.py +216 -0
- prettyplay/llm/anthropic_provider.py +213 -0
- prettyplay/llm/models.py +22 -0
- prettyplay/llm/openai_provider.py +187 -0
- prettyplay/llm/provider.py +115 -0
- prettyplay/reporting/.usages/hooks.md +41 -0
- prettyplay/reporting/CODEMANIFEST +79 -0
- prettyplay/reporting/__init__.py +6 -0
- prettyplay/reporting/hooks.py +37 -0
- prettyplay/reporting/reporter.py +70 -0
- prettyplay/runtime.py +107 -0
- prettyplay/scenario.py +291 -0
- prettyplay-0.0.0.dist-info/METADATA +236 -0
- prettyplay-0.0.0.dist-info/RECORD +58 -0
- prettyplay-0.0.0.dist-info/WHEEL +5 -0
- prettyplay-0.0.0.dist-info/licenses/LICENSE +28 -0
- prettyplay-0.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
"""Healing of failed cached steps: classification verdict decides the branch."""
|
|
2
|
+
|
|
3
|
+
from ..cache import CachedStep, RunBudgets, StepCache
|
|
4
|
+
from ..config import Config
|
|
5
|
+
from ..driver import PageFacade
|
|
6
|
+
from ..failures import FailureVerdict, IncurableStepError, ProductDefectError
|
|
7
|
+
from ..llm import LlmProvider
|
|
8
|
+
from ..reporting import StepReporter
|
|
9
|
+
from .classification import classify_step_failure
|
|
10
|
+
from .generator import StepGenerator
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class StepHealer:
|
|
14
|
+
"""Heals a failed cached step according to the classification verdict.
|
|
15
|
+
|
|
16
|
+
The healer never heals blindly. It first asks the provider to classify
|
|
17
|
+
the failure of the cached code, then follows the verdict: a product
|
|
18
|
+
defect propagates loudly as the signal the test suite exists for
|
|
19
|
+
(the cache stays untouched — nothing to regenerate), an incurable step
|
|
20
|
+
propagates with the verdict fields, and rot delegates to the generator,
|
|
21
|
+
which regenerates the code and rewrites the cache only after the healed
|
|
22
|
+
candidate has actually worked on the page.
|
|
23
|
+
|
|
24
|
+
Attributes:
|
|
25
|
+
_config: project settings; the screenshot flag feeds the requests.
|
|
26
|
+
_provider: the LLM port implementation classifying the failure.
|
|
27
|
+
_generator: the regeneration loop of the rot branch.
|
|
28
|
+
_reporter: the visibility point for healing events.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
def __init__( # noqa: PLR0913, PLR0917 — the signature is fixed by the engine contract
|
|
32
|
+
self,
|
|
33
|
+
config: Config,
|
|
34
|
+
provider: LlmProvider,
|
|
35
|
+
generator: StepGenerator,
|
|
36
|
+
cache: StepCache, # noqa: ARG002 — written by the generator only; kept for contract symmetry
|
|
37
|
+
budgets: RunBudgets, # noqa: ARG002 — spent by the generator only; kept for contract symmetry
|
|
38
|
+
reporter: StepReporter,
|
|
39
|
+
) -> None:
|
|
40
|
+
"""Keep the collaborators of the healing branch.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
config: project settings; ``send_screenshots`` attaches page images.
|
|
44
|
+
provider: the LLM port implementation classifying the failure.
|
|
45
|
+
generator: the regeneration loop handling the rot verdict.
|
|
46
|
+
cache: the store of the failed step; written by the generator
|
|
47
|
+
only — accepted for contract symmetry, never read here.
|
|
48
|
+
budgets: the per-test attempt registry; spent by the generator —
|
|
49
|
+
accepted for contract symmetry, never read here.
|
|
50
|
+
reporter: the visibility point for engine events.
|
|
51
|
+
"""
|
|
52
|
+
self._config = config
|
|
53
|
+
self._provider = provider
|
|
54
|
+
self._generator = generator
|
|
55
|
+
self._reporter = reporter
|
|
56
|
+
|
|
57
|
+
def heal(
|
|
58
|
+
self,
|
|
59
|
+
step: CachedStep,
|
|
60
|
+
error: str,
|
|
61
|
+
previous_steps: list[str],
|
|
62
|
+
page: PageFacade,
|
|
63
|
+
) -> CachedStep:
|
|
64
|
+
"""Heal a failed cached step according to the classification verdict.
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
step: the cached step whose code failed.
|
|
68
|
+
error: the failure description of the cached code.
|
|
69
|
+
previous_steps: the sentences of the previous steps of the test.
|
|
70
|
+
page: the live page facade the healed code runs against.
|
|
71
|
+
|
|
72
|
+
Returns:
|
|
73
|
+
The healed step with proven code, already cached by the generator.
|
|
74
|
+
|
|
75
|
+
Raises:
|
|
76
|
+
ProductDefectError: the expectation of the step is genuinely
|
|
77
|
+
broken in the product; the cache stays untouched.
|
|
78
|
+
IncurableStepError: the verdict says regeneration cannot help,
|
|
79
|
+
or the regeneration attempt budget is exhausted — the
|
|
80
|
+
exhaustion reuses the verdict of this classification, no
|
|
81
|
+
second LLM request is made.
|
|
82
|
+
LlmUnavailableError: the provider service failed; no retry.
|
|
83
|
+
"""
|
|
84
|
+
step_text = step.identity.normalized_text
|
|
85
|
+
classification = classify_step_failure(self._config, self._provider, step_text, step.code, error, page)
|
|
86
|
+
|
|
87
|
+
verdict = FailureVerdict(
|
|
88
|
+
category=classification.category,
|
|
89
|
+
explanation=classification.explanation,
|
|
90
|
+
recommendation=classification.recommendation,
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
self._reporter.emit("on_healing_started", {"step_text": step_text, "category": classification.category})
|
|
94
|
+
|
|
95
|
+
if classification.category == "product_defect":
|
|
96
|
+
raise ProductDefectError(step_text, classification.explanation, verdict)
|
|
97
|
+
if classification.category == "incurable":
|
|
98
|
+
raise IncurableStepError(step_text, classification.explanation, verdict)
|
|
99
|
+
|
|
100
|
+
try:
|
|
101
|
+
healed = self._generator.regenerate(
|
|
102
|
+
identity=step.identity,
|
|
103
|
+
step_text=step_text,
|
|
104
|
+
previous_steps=previous_steps,
|
|
105
|
+
page=page,
|
|
106
|
+
existing_code=step.code,
|
|
107
|
+
error=error,
|
|
108
|
+
)
|
|
109
|
+
except IncurableStepError as incurable:
|
|
110
|
+
if incurable.verdict is None:
|
|
111
|
+
# regeneration exhausted: verdict of this classification, no second LLM request
|
|
112
|
+
raise IncurableStepError(step_text, incurable.reason, verdict) from incurable
|
|
113
|
+
raise # a fresh failed-check verdict is never overwritten
|
|
114
|
+
|
|
115
|
+
self._reporter.emit("on_healed", {"step_text": step_text, "explanation": classification.explanation})
|
|
116
|
+
return healed
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
"""Shared short-failure text of the engine: one truncation policy for reports and requests."""
|
|
2
|
+
|
|
3
|
+
#: Upper bound of the short failure description carried by reports and healing requests.
|
|
4
|
+
SHORT_ERROR_LENGTH = 200
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def first_line_short(exc: Exception) -> str:
|
|
8
|
+
"""Return the first line of the exception text, cut to 200 characters.
|
|
9
|
+
|
|
10
|
+
A message-less exception (a bare ``assert`` in step code) yields an
|
|
11
|
+
empty description instead of crashing the reporting and healing paths.
|
|
12
|
+
|
|
13
|
+
Args:
|
|
14
|
+
exc: the exception raised by the failed step or candidate branch.
|
|
15
|
+
|
|
16
|
+
Returns:
|
|
17
|
+
The short failure description carried by reports and requests.
|
|
18
|
+
"""
|
|
19
|
+
return str(exc).partition("\n")[0][:SHORT_ERROR_LENGTH]
|
prettyplay/executor.py
ADDED
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""The step cycle: cache hit — execute; cache miss — generate; cached failure — heal."""
|
|
2
|
+
|
|
3
|
+
from .cache import RunBudgets, StepCache, StepIdentity, normalize_step_text
|
|
4
|
+
from .driver import PageFacade
|
|
5
|
+
from .engine import StepGenerator, StepHealer, run_step_code
|
|
6
|
+
from .engine.text import first_line_short
|
|
7
|
+
from .failures import IncurableStepError, ProductDefectError
|
|
8
|
+
from .reporting import StepReporter
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class StepExecutor:
|
|
12
|
+
"""The owner of the step cycle of one test.
|
|
13
|
+
|
|
14
|
+
One step goes through the full cycle: a cache hit executes the cached
|
|
15
|
+
code with no LLM involvement whatsoever; a miss delegates to the
|
|
16
|
+
generator, which stores the step after the candidate has actually
|
|
17
|
+
worked; a failed cached step delegates to the healer, whose verdict
|
|
18
|
+
decides between a loud product defect, an incurable step, and rot
|
|
19
|
+
regeneration. The scenario context — the sentences of the previous
|
|
20
|
+
steps of this test — feeds every generation and healing request.
|
|
21
|
+
|
|
22
|
+
Attributes:
|
|
23
|
+
cache_key: the context key of the owning test object.
|
|
24
|
+
_cache: the step cache of the test.
|
|
25
|
+
_generator: the generation engine of the cycle.
|
|
26
|
+
_healer: the healing engine of the cycle.
|
|
27
|
+
_reporter: the visibility point of the test.
|
|
28
|
+
_scenario: the sentences of the previous steps of this test.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
def __init__( # noqa: PLR0913, PLR0917 — the signature is fixed by the root cell contract
|
|
32
|
+
self,
|
|
33
|
+
cache_key: str,
|
|
34
|
+
cache: StepCache,
|
|
35
|
+
generator: StepGenerator,
|
|
36
|
+
healer: StepHealer,
|
|
37
|
+
budgets: RunBudgets, # noqa: ARG002 — spent by the engine; kept for contract symmetry
|
|
38
|
+
reporter: StepReporter,
|
|
39
|
+
) -> None:
|
|
40
|
+
"""Keep the collaborators of the step cycle and reset the scenario context.
|
|
41
|
+
|
|
42
|
+
Args:
|
|
43
|
+
cache_key: the context key of the owning test object.
|
|
44
|
+
cache: the step cache of the test.
|
|
45
|
+
generator: the generation engine of the cycle.
|
|
46
|
+
healer: the healing engine of the cycle.
|
|
47
|
+
budgets: the per-test attempt registry; attempts are spent
|
|
48
|
+
by the engine, not by the executor.
|
|
49
|
+
reporter: the visibility point of the test.
|
|
50
|
+
"""
|
|
51
|
+
self.cache_key = cache_key
|
|
52
|
+
self._cache = cache
|
|
53
|
+
self._generator = generator
|
|
54
|
+
self._healer = healer
|
|
55
|
+
self._reporter = reporter
|
|
56
|
+
self._scenario: list[str] = [] # test scenario context
|
|
57
|
+
|
|
58
|
+
def execute(self, step_text: str, step_type: str, page: PageFacade) -> None:
|
|
59
|
+
"""Run one step through the full cycle.
|
|
60
|
+
|
|
61
|
+
Args:
|
|
62
|
+
step_text: the sentence of the step as written by the engineer.
|
|
63
|
+
step_type: the kind of the step sentence ({action, assertion}).
|
|
64
|
+
page: the live page facade of the current test.
|
|
65
|
+
|
|
66
|
+
Raises:
|
|
67
|
+
ProductDefectError: the healed step verdict says the expectation
|
|
68
|
+
of the step is genuinely broken in the product.
|
|
69
|
+
IncurableStepError: the step never generated successfully, or the
|
|
70
|
+
verdict says regeneration cannot help.
|
|
71
|
+
LlmUnavailableError: the provider service failed; no retry.
|
|
72
|
+
"""
|
|
73
|
+
try:
|
|
74
|
+
self._reporter.emit("on_step_started", {"step_text": step_text, "step_type": step_type})
|
|
75
|
+
|
|
76
|
+
identity = StepIdentity(
|
|
77
|
+
cache_key=self.cache_key,
|
|
78
|
+
step_type=step_type,
|
|
79
|
+
normalized_text=normalize_step_text(step_text),
|
|
80
|
+
)
|
|
81
|
+
cached = self._cache.load(identity)
|
|
82
|
+
|
|
83
|
+
if cached is not None:
|
|
84
|
+
try:
|
|
85
|
+
run_step_code(cached.code, page)
|
|
86
|
+
except Exception as error: # cached code failed — context goes to healing
|
|
87
|
+
self._healer.heal(cached, first_line_short(error), self._scenario, page) # healed = re-executed
|
|
88
|
+
else:
|
|
89
|
+
self._generator.generate(identity, step_text, self._scenario, page)
|
|
90
|
+
|
|
91
|
+
self._scenario.append(step_text)
|
|
92
|
+
self._reporter.emit("on_step_passed", {"step_text": step_text, "step_type": step_type})
|
|
93
|
+
except Exception as error:
|
|
94
|
+
self._reporter.emit(
|
|
95
|
+
"on_step_failed",
|
|
96
|
+
{"step_text": step_text, "step_type": step_type, "error": first_line_short(error)},
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
if isinstance(error, (ProductDefectError, IncurableStepError)) and error.verdict is not None:
|
|
100
|
+
self._reporter.emit(
|
|
101
|
+
"on_step_verdict",
|
|
102
|
+
{
|
|
103
|
+
"step_text": step_text,
|
|
104
|
+
"category": error.verdict.category,
|
|
105
|
+
"explanation": error.verdict.explanation,
|
|
106
|
+
"recommendation": error.verdict.recommendation,
|
|
107
|
+
},
|
|
108
|
+
)
|
|
109
|
+
raise
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
# Failure taxonomy
|
|
2
|
+
|
|
3
|
+
Domain: failure kinds of prettyplay. Audience: integrators wiring library failures into runner and CI reporting.
|
|
4
|
+
|
|
5
|
+
Every step failure is one of three distinct kinds; all derive from `PrettyplayError`, so one except clause catches any prettyplay failure. A fourth kind — the configuration error — joins the base from the config cell.
|
|
6
|
+
|
|
7
|
+
| Exception | Meaning | When it happens | Recommended reaction |
|
|
8
|
+
|---|---|---|---|
|
|
9
|
+
| ProductDefectError | real product regression | an assertion expectation legitimately failed | treat as a bug: file it, fix the product — this failure is the value of the suite |
|
|
10
|
+
| IncurableStepError | the step cannot be (re)generated | attempt budget exhausted, step text no longer matches reality, ambiguity | follow `recommendation`: reword the step or refresh the cache |
|
|
11
|
+
| LlmUnavailableError | LLM infrastructure down | generation or healing ran while the provider was unavailable | restore provider access or keys; cached steps are unaffected |
|
|
12
|
+
| ConfigurationError | settings are invalid | the first library use loaded an invalid [tool.prettyplay] section | fix the named setting — the message lists the allowed values |
|
|
13
|
+
|
|
14
|
+
## Verdicts on terminal failures
|
|
15
|
+
|
|
16
|
+
ProductDefectError and IncurableStepError carry an optional verdict: category, explanation, recommendation. It is fully present in the exception message, the on_step_verdict hook event and the log. When the LLM is unavailable the verdict is skipped quietly (WARNING in the log) — the failure itself is never delayed or distorted.
|
|
17
|
+
|
|
18
|
+
```python
|
|
19
|
+
import pytest
|
|
20
|
+
|
|
21
|
+
from prettyplay.failures import IncurableStepError
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def test_reports_only_library_failures():
|
|
25
|
+
with pytest.raises(IncurableStepError) as info:
|
|
26
|
+
...
|
|
27
|
+
assert info.value.recommendation
|
|
28
|
+
# info.value.verdict may be None when the LLM was unavailable
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## Assertion semantics
|
|
32
|
+
|
|
33
|
+
ProductDefectError is also an AssertionError: unittest reports the failed check as a failure (not an error), pytest shows it as an ordinary assertion failure, the traceback is folded to the library boundary. Catch it with `except PrettyplayError` or `except AssertionError` — both work.
|
|
34
|
+
|
|
35
|
+
## Rules
|
|
36
|
+
|
|
37
|
+
- Every failure message is actionable: what happened, on which step, what to do next
|
|
38
|
+
- A healed run never turns a ProductDefectError into a green test
|
|
39
|
+
- LlmUnavailableError never occurs on the cached path — a cached suite runs without any LLM
|
|
40
|
+
- The verdict explanation never replaces the primary failure — it is appended
|
|
@@ -0,0 +1,117 @@
|
|
|
1
|
+
Usages:
|
|
2
|
+
conventions: .goga/usages/conventions.md
|
|
3
|
+
|
|
4
|
+
Annotations: |
|
|
5
|
+
Use `conventions` for code writing rules and testing.
|
|
6
|
+
|
|
7
|
+
The failure taxonomy of the library: three distinct, user-distinguishable step failure kinds plus the shared verdict type; every failure carries an actionable message.
|
|
8
|
+
All failure types derive from the single library base `PrettyplayError`; ProductDefectError additionally derives from AssertionError — the failed check is a failure, never an error, in any runner.
|
|
9
|
+
The rendered message of a terminal failure starts with its primary reason; the verdict render is appended, never interleaved.
|
|
10
|
+
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
"PrettyplayError(message: str)":
|
|
14
|
+
location: errors.py
|
|
15
|
+
annotations: |
|
|
16
|
+
The common base of every library failure. Exists for one consumer need: catch any prettyplay failure with a single except clause at the test-suite boundary.
|
|
17
|
+
|
|
18
|
+
`message`: the failure description.
|
|
19
|
+
|
|
20
|
+
"FailureVerdict(category: str, explanation: str, recommendation: str)":
|
|
21
|
+
location: errors.py
|
|
22
|
+
annotations: |
|
|
23
|
+
The verdict of a terminal step failure: what the LLM saw on the page at the moment of the failure and what the engineer should do next.
|
|
24
|
+
|
|
25
|
+
`category`: the classification label: rot, product_defect or incurable.
|
|
26
|
+
`explanation`: what happened on the page — one short sentence.
|
|
27
|
+
`recommendation`: the recommended engineer action — one short sentence.
|
|
28
|
+
|
|
29
|
+
Requirements:
|
|
30
|
+
- Built by the engines from the failure classification; the failure types never request it themselves
|
|
31
|
+
properties:
|
|
32
|
+
"category -> str": |
|
|
33
|
+
The classification label: rot, product_defect or incurable.
|
|
34
|
+
"explanation -> str": |
|
|
35
|
+
What happened on the page at the moment of the failure.
|
|
36
|
+
"recommendation -> str": |
|
|
37
|
+
The recommended engineer action.
|
|
38
|
+
methods:
|
|
39
|
+
"render() -> text: str": |
|
|
40
|
+
Render the verdict as stable labelled lines — the single render used by the exception message tail, the hook event payload and the log record.
|
|
41
|
+
|
|
42
|
+
`text`: the rendered verdict.
|
|
43
|
+
|
|
44
|
+
Algorithm:
|
|
45
|
+
1. Build one line per non-empty field: category:, explanation:, recommendation:
|
|
46
|
+
2. Join the lines
|
|
47
|
+
|
|
48
|
+
Requirements:
|
|
49
|
+
- Labels are stable lowercase words — integrators parse them
|
|
50
|
+
|
|
51
|
+
"PrettyplayError::ProductDefectError(step_text: str, message: str, verdict: FailureVerdict | None)":
|
|
52
|
+
location: errors.py
|
|
53
|
+
annotations: |
|
|
54
|
+
A real functional product defect: the expectation of an assertion step legitimately did not hold against the current application state. This is the signal the test suite exists for.
|
|
55
|
+
|
|
56
|
+
`step_text`: the sentence of the failed step.
|
|
57
|
+
`message`: what exactly was expected and what was observed — the primary reason.
|
|
58
|
+
`verdict`: the optional `FailureVerdict`; absent when the LLM was unavailable — the failure never waits for it.
|
|
59
|
+
|
|
60
|
+
Requirements:
|
|
61
|
+
- Derives from `PrettyplayError` and AssertionError: catchable as any library failure and as an assertion failure in the same except clauses
|
|
62
|
+
- The rendered message starts with `message`; the `verdict` render is appended when present
|
|
63
|
+
- The traceback a runner sees starts at the library boundary — internal library frames are folded away
|
|
64
|
+
- Propagates to the test runner as a failing test: no retry, no healing
|
|
65
|
+
|
|
66
|
+
Constraints:
|
|
67
|
+
- The library facade stays framework-agnostic: the runner alignment is done by the type itself, never by a runner plugin or integration
|
|
68
|
+
properties:
|
|
69
|
+
"step_text -> str": |
|
|
70
|
+
The sentence of the failed step.
|
|
71
|
+
"message -> str": |
|
|
72
|
+
What exactly was expected and what was observed.
|
|
73
|
+
"verdict -> FailureVerdict | None": |
|
|
74
|
+
The optional failure verdict; None — the explicit absence when the LLM was unavailable.
|
|
75
|
+
|
|
76
|
+
"PrettyplayError::IncurableStepError(step_text: str, reason: str, verdict: FailureVerdict | None)":
|
|
77
|
+
location: errors.py
|
|
78
|
+
annotations: |
|
|
79
|
+
An incurable step: regeneration cannot produce working code — the attempt budget is exhausted, the step text no longer matches the application reality, or the intent is ambiguous.
|
|
80
|
+
|
|
81
|
+
`step_text`: the sentence of the failed step.
|
|
82
|
+
`reason`: the specific incurability cause — the primary reason of the rendered message.
|
|
83
|
+
`verdict`: the optional `FailureVerdict` — reused from a classification that already happened or requested at budget exhaustion; absent when the LLM was unavailable.
|
|
84
|
+
|
|
85
|
+
Requirements:
|
|
86
|
+
- The recommendation is carried by `verdict`; the recommendation property derives from it, falling back to the built-in path guidance when the verdict is absent
|
|
87
|
+
- The rendered message starts with `reason`, then appends the `verdict` render when present; the fallback recommendation keeps the message actionable without a verdict
|
|
88
|
+
- An execution failure, not a failed check: derives from `PrettyplayError` only, never from AssertionError
|
|
89
|
+
properties:
|
|
90
|
+
"step_text -> str": |
|
|
91
|
+
The sentence of the failed step.
|
|
92
|
+
"reason -> str": |
|
|
93
|
+
The specific incurability cause.
|
|
94
|
+
"recommendation -> str": |
|
|
95
|
+
The recommended engineer action: verdict recommendation when present, the built-in path guidance otherwise.
|
|
96
|
+
"verdict -> FailureVerdict | None": |
|
|
97
|
+
The optional failure verdict; None — the explicit absence when the LLM was unavailable.
|
|
98
|
+
|
|
99
|
+
"PrettyplayError::LlmUnavailableError(message: str)":
|
|
100
|
+
location: errors.py
|
|
101
|
+
annotations: |
|
|
102
|
+
LLM infrastructure failure: the provider service is unreachable, times out, rate-limits or rejects authentication. Blocks only code generation and healing; cached steps keep running.
|
|
103
|
+
|
|
104
|
+
`message`: the failure description naming the provider.
|
|
105
|
+
|
|
106
|
+
Requirements:
|
|
107
|
+
- Raised only on generation or healing paths; never on a cached step execution path
|
|
108
|
+
properties:
|
|
109
|
+
"message -> str": |
|
|
110
|
+
The failure description naming the provider.
|
|
111
|
+
|
|
112
|
+
---
|
|
113
|
+
|
|
114
|
+
Author: Goga
|
|
115
|
+
CreatedAt: 07/09/26
|
|
116
|
+
Description: |
|
|
117
|
+
The failure taxonomy of prettyplay: product defect, incurable step, LLM infrastructure — mutations of one library base — and the shared verdict of a terminal failure.
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""Facade of the prettyplay.failures cell: the failure taxonomy of the library."""
|
|
2
|
+
|
|
3
|
+
from .errors import (
|
|
4
|
+
FailureVerdict,
|
|
5
|
+
IncurableStepError,
|
|
6
|
+
LlmUnavailableError,
|
|
7
|
+
PrettyplayError,
|
|
8
|
+
ProductDefectError,
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"FailureVerdict",
|
|
13
|
+
"IncurableStepError",
|
|
14
|
+
"LlmUnavailableError",
|
|
15
|
+
"PrettyplayError",
|
|
16
|
+
"ProductDefectError",
|
|
17
|
+
]
|
|
@@ -0,0 +1,147 @@
|
|
|
1
|
+
"""The failure taxonomy of prettyplay: three distinct kinds, one library base.
|
|
2
|
+
|
|
3
|
+
Every failure the library raises derives from :class:`PrettyplayError`, so a
|
|
4
|
+
test suite catches any prettyplay failure with a single ``except`` clause at
|
|
5
|
+
its boundary. Each kind carries actionable fields instead of an opaque string.
|
|
6
|
+
The rendered message of a terminal failure starts with its primary reason;
|
|
7
|
+
the verdict render is appended, never interleaved.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
class PrettyplayError(Exception):
|
|
14
|
+
"""The common base of every library failure.
|
|
15
|
+
|
|
16
|
+
Args:
|
|
17
|
+
message: the failure description.
|
|
18
|
+
"""
|
|
19
|
+
|
|
20
|
+
def __init__(self, message: str) -> None:
|
|
21
|
+
self.message = message
|
|
22
|
+
|
|
23
|
+
super().__init__(message)
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True)
|
|
27
|
+
class FailureVerdict:
|
|
28
|
+
"""The verdict of a terminal step failure, carried by the errors that stop the run.
|
|
29
|
+
|
|
30
|
+
Built by the engines from a :class:`~prettyplay.llm.FailureClassification`;
|
|
31
|
+
the failure types never request it themselves. A plain frozen value object,
|
|
32
|
+
not a pydantic model — the source classification already validated the data,
|
|
33
|
+
and failure paths stay cheap.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
category: the classification label: rot, product_defect or incurable.
|
|
37
|
+
explanation: what happened on the page — one short sentence.
|
|
38
|
+
recommendation: the recommended engineer action — one short sentence.
|
|
39
|
+
"""
|
|
40
|
+
|
|
41
|
+
category: str
|
|
42
|
+
explanation: str
|
|
43
|
+
recommendation: str
|
|
44
|
+
|
|
45
|
+
def render(self) -> str:
|
|
46
|
+
"""Render the verdict as stable labelled lines.
|
|
47
|
+
|
|
48
|
+
The single render used by the exception message tail, the hook event
|
|
49
|
+
payload content and the log record; the labels are stable lowercase
|
|
50
|
+
words integrators parse. An empty field yields no line.
|
|
51
|
+
|
|
52
|
+
Returns:
|
|
53
|
+
The rendered verdict — one line per non-empty field, joined with newlines.
|
|
54
|
+
"""
|
|
55
|
+
fields = (
|
|
56
|
+
("category", self.category),
|
|
57
|
+
("explanation", self.explanation),
|
|
58
|
+
("recommendation", self.recommendation),
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
return "\n".join(f"{label}: {value}" for label, value in fields if value)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class ProductDefectError(PrettyplayError, AssertionError):
|
|
65
|
+
"""A real functional product defect: the expectation of an assertion step did not hold.
|
|
66
|
+
|
|
67
|
+
The signal the test suite exists for: no retry, no healing — propagates to
|
|
68
|
+
the test runner as a failing test. Derives from both :class:`PrettyplayError`
|
|
69
|
+
and ``AssertionError``, so any runner counts it as a failure, never an error.
|
|
70
|
+
|
|
71
|
+
Args:
|
|
72
|
+
step_text: the sentence of the failed step.
|
|
73
|
+
message: what exactly was expected and what was observed — the primary reason.
|
|
74
|
+
verdict: the optional :class:`FailureVerdict`; ``None`` — the explicit
|
|
75
|
+
absence when the LLM was unavailable, the failure never waits for it.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
def __init__(self, step_text: str, message: str, verdict: FailureVerdict | None = None) -> None:
|
|
79
|
+
self.step_text = step_text
|
|
80
|
+
self.message = message
|
|
81
|
+
self.verdict = verdict
|
|
82
|
+
|
|
83
|
+
PrettyplayError.__init__(self, message)
|
|
84
|
+
|
|
85
|
+
def __str__(self) -> str:
|
|
86
|
+
if self.verdict is None:
|
|
87
|
+
return self.message
|
|
88
|
+
|
|
89
|
+
return self.message + "\n" + self.verdict.render()
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
class IncurableStepError(PrettyplayError):
|
|
93
|
+
"""An incurable step: regeneration cannot produce working code.
|
|
94
|
+
|
|
95
|
+
Raised when the attempt budget is exhausted, the step text no longer
|
|
96
|
+
matches the application reality, or the intent is ambiguous. An execution
|
|
97
|
+
failure, not a failed check — derives from :class:`PrettyplayError` only,
|
|
98
|
+
never from ``AssertionError``.
|
|
99
|
+
|
|
100
|
+
Args:
|
|
101
|
+
step_text: the sentence of the failed step.
|
|
102
|
+
reason: the specific incurability cause — the primary reason.
|
|
103
|
+
verdict: the optional :class:`FailureVerdict` — reused from a
|
|
104
|
+
classification that already happened or requested at budget
|
|
105
|
+
exhaustion; ``None`` when the LLM was unavailable.
|
|
106
|
+
"""
|
|
107
|
+
|
|
108
|
+
def __init__(self, step_text: str, reason: str, verdict: FailureVerdict | None = None) -> None:
|
|
109
|
+
self.step_text = step_text
|
|
110
|
+
self.reason = reason
|
|
111
|
+
self.verdict = verdict
|
|
112
|
+
|
|
113
|
+
super().__init__(reason)
|
|
114
|
+
|
|
115
|
+
@property
|
|
116
|
+
def recommendation(self) -> str:
|
|
117
|
+
"""The recommended engineer action.
|
|
118
|
+
|
|
119
|
+
Returns:
|
|
120
|
+
The verdict recommendation when present, the built-in path guidance otherwise.
|
|
121
|
+
"""
|
|
122
|
+
if self.verdict is not None:
|
|
123
|
+
return self.verdict.recommendation
|
|
124
|
+
|
|
125
|
+
return "reword the step or refresh the cache"
|
|
126
|
+
|
|
127
|
+
def __str__(self) -> str:
|
|
128
|
+
if self.verdict is not None:
|
|
129
|
+
return self.reason + "\n" + self.verdict.render()
|
|
130
|
+
|
|
131
|
+
return self.reason + "\nrecommendation: reword the step or refresh the cache"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
class LlmUnavailableError(PrettyplayError):
|
|
135
|
+
"""LLM infrastructure failure: the provider service is unreachable or rejects the request.
|
|
136
|
+
|
|
137
|
+
Blocks only code generation and healing; cached steps keep running. No
|
|
138
|
+
retries.
|
|
139
|
+
|
|
140
|
+
Args:
|
|
141
|
+
message: the failure description naming the provider.
|
|
142
|
+
"""
|
|
143
|
+
|
|
144
|
+
def __init__(self, message: str) -> None:
|
|
145
|
+
self.message = message
|
|
146
|
+
|
|
147
|
+
super().__init__(message)
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# Failure classification
|
|
2
|
+
|
|
3
|
+
Domain: classifying a failed cached step before healing. Audience: engineers reasoning about healing decisions.
|
|
4
|
+
|
|
5
|
+
## Categories
|
|
6
|
+
|
|
7
|
+
| Category | Meaning | Consequence |
|
|
8
|
+
|---|---|---|
|
|
9
|
+
| rot | the UI changed: selectors, texts, structure | the step is regenerated from the current page and retried |
|
|
10
|
+
| product_defect | the expectation legitimately failed | the test fails loudly — never healed green |
|
|
11
|
+
| incurable | regeneration cannot help: budget exhausted, text no longer matches reality, ambiguity | the incurable failure carries step, reason, recommendation |
|
|
12
|
+
|
|
13
|
+
## Call
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
classification = provider.classify_failure(
|
|
17
|
+
prompt=system_prompt, # the system prompt text comes from the calling engine
|
|
18
|
+
step_text="click the «Sign in» button",
|
|
19
|
+
code=step_code,
|
|
20
|
+
error="element not found: button «Sign in»",
|
|
21
|
+
snapshot=snapshot_text,
|
|
22
|
+
screenshot=None,
|
|
23
|
+
)
|
|
24
|
+
print(classification.category, classification.explanation, classification.recommendation)
|
|
25
|
+
```
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Providers
|
|
2
|
+
|
|
3
|
+
Domain: LLM provider selection and parity. Audience: integrators choosing a provider and setting models.
|
|
4
|
+
|
|
5
|
+
## Select a provider
|
|
6
|
+
|
|
7
|
+
```python
|
|
8
|
+
from prettyplay.config import Config
|
|
9
|
+
from prettyplay.llm import create_provider
|
|
10
|
+
|
|
11
|
+
config = Config(provider="anthropic", model="claude-sonnet-4-5")
|
|
12
|
+
provider = create_provider(config)
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
The provider is a project setting: openai or anthropic; env override PRETTYPLAY_PROVIDER. API keys come only from environment variables: OPENAI_API_KEY for openai, ANTHROPIC_API_KEY for anthropic.
|
|
16
|
+
|
|
17
|
+
## Models
|
|
18
|
+
|
|
19
|
+
| Setting | Purpose | Fallback |
|
|
20
|
+
|---|---|---|
|
|
21
|
+
| model | the main model for both operations | — |
|
|
22
|
+
| generation_model | code generation only | model |
|
|
23
|
+
| classification_model | failure classification only | model |
|
|
24
|
+
|
|
25
|
+
base_url overrides the provider endpoint when set.
|
|
26
|
+
|
|
27
|
+
## Parity
|
|
28
|
+
|
|
29
|
+
Both providers expose the same two operations — generate_step_code and classify_failure — with identical inputs, identical output shapes and the identical failure taxonomy: a provider service failure raises LlmUnavailableError; cached step code never depends on the provider. One request per attempt; attempt budgets belong to the calling engine.
|
|
30
|
+
|
|
31
|
+
## Answer shape
|
|
32
|
+
|
|
33
|
+
generate_step_code returns step code of the fixed form. Models often answer with a fenced python block (```python … ```); the provider unwraps the first fenced block before returning, so the engine receives clean code either way — an answer with no closed fence is returned verbatim and, if unparsable, keeps failing downstream in execution.
|