prettyplay 0.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.


This version of prettyplay might be problematic. Click here for more details.

Files changed (58) hide show
  1. prettyplay/.usages/lifecycle.md +50 -0
  2. prettyplay/.usages/steps.md +56 -0
  3. prettyplay/CODEMANIFEST +182 -0
  4. prettyplay/__init__.py +13 -0
  5. prettyplay/cache/.usages/addressing.md +31 -0
  6. prettyplay/cache/.usages/budgets.md +21 -0
  7. prettyplay/cache/.usages/storage.md +32 -0
  8. prettyplay/cache/CODEMANIFEST +152 -0
  9. prettyplay/cache/__init__.py +8 -0
  10. prettyplay/cache/budgets.py +73 -0
  11. prettyplay/cache/models.py +60 -0
  12. prettyplay/cache/store.py +260 -0
  13. prettyplay/cache/text.py +35 -0
  14. prettyplay/config/.usages/configuration.md +87 -0
  15. prettyplay/config/CODEMANIFEST +132 -0
  16. prettyplay/config/__init__.py +6 -0
  17. prettyplay/config/loader.py +192 -0
  18. prettyplay/config/models.py +92 -0
  19. prettyplay/driver/.usages/facade.md +66 -0
  20. prettyplay/driver/CODEMANIFEST +162 -0
  21. prettyplay/driver/__init__.py +6 -0
  22. prettyplay/driver/page.py +350 -0
  23. prettyplay/driver/session.py +288 -0
  24. prettyplay/engine/.usages/generation.md +45 -0
  25. prettyplay/engine/.usages/healing.md +31 -0
  26. prettyplay/engine/CODEMANIFEST +215 -0
  27. prettyplay/engine/__init__.py +8 -0
  28. prettyplay/engine/classification.py +69 -0
  29. prettyplay/engine/execution.py +25 -0
  30. prettyplay/engine/generator.py +318 -0
  31. prettyplay/engine/healer.py +116 -0
  32. prettyplay/engine/text.py +19 -0
  33. prettyplay/executor.py +109 -0
  34. prettyplay/failures/.usages/taxonomy.md +40 -0
  35. prettyplay/failures/CODEMANIFEST +117 -0
  36. prettyplay/failures/__init__.py +17 -0
  37. prettyplay/failures/errors.py +147 -0
  38. prettyplay/llm/.usages/classification.md +25 -0
  39. prettyplay/llm/.usages/providers.md +33 -0
  40. prettyplay/llm/CODEMANIFEST +125 -0
  41. prettyplay/llm/__init__.py +8 -0
  42. prettyplay/llm/_request.py +216 -0
  43. prettyplay/llm/anthropic_provider.py +213 -0
  44. prettyplay/llm/models.py +22 -0
  45. prettyplay/llm/openai_provider.py +187 -0
  46. prettyplay/llm/provider.py +115 -0
  47. prettyplay/reporting/.usages/hooks.md +41 -0
  48. prettyplay/reporting/CODEMANIFEST +79 -0
  49. prettyplay/reporting/__init__.py +6 -0
  50. prettyplay/reporting/hooks.py +37 -0
  51. prettyplay/reporting/reporter.py +70 -0
  52. prettyplay/runtime.py +107 -0
  53. prettyplay/scenario.py +291 -0
  54. prettyplay-0.0.0.dist-info/METADATA +236 -0
  55. prettyplay-0.0.0.dist-info/RECORD +58 -0
  56. prettyplay-0.0.0.dist-info/WHEEL +5 -0
  57. prettyplay-0.0.0.dist-info/licenses/LICENSE +28 -0
  58. prettyplay-0.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,116 @@
1
+ """Healing of failed cached steps: classification verdict decides the branch."""
2
+
3
+ from ..cache import CachedStep, RunBudgets, StepCache
4
+ from ..config import Config
5
+ from ..driver import PageFacade
6
+ from ..failures import FailureVerdict, IncurableStepError, ProductDefectError
7
+ from ..llm import LlmProvider
8
+ from ..reporting import StepReporter
9
+ from .classification import classify_step_failure
10
+ from .generator import StepGenerator
11
+
12
+
13
+ class StepHealer:
14
+ """Heals a failed cached step according to the classification verdict.
15
+
16
+ The healer never heals blindly. It first asks the provider to classify
17
+ the failure of the cached code, then follows the verdict: a product
18
+ defect propagates loudly as the signal the test suite exists for
19
+ (the cache stays untouched — nothing to regenerate), an incurable step
20
+ propagates with the verdict fields, and rot delegates to the generator,
21
+ which regenerates the code and rewrites the cache only after the healed
22
+ candidate has actually worked on the page.
23
+
24
+ Attributes:
25
+ _config: project settings; the screenshot flag feeds the requests.
26
+ _provider: the LLM port implementation classifying the failure.
27
+ _generator: the regeneration loop of the rot branch.
28
+ _reporter: the visibility point for healing events.
29
+ """
30
+
31
+ def __init__( # noqa: PLR0913, PLR0917 — the signature is fixed by the engine contract
32
+ self,
33
+ config: Config,
34
+ provider: LlmProvider,
35
+ generator: StepGenerator,
36
+ cache: StepCache, # noqa: ARG002 — written by the generator only; kept for contract symmetry
37
+ budgets: RunBudgets, # noqa: ARG002 — spent by the generator only; kept for contract symmetry
38
+ reporter: StepReporter,
39
+ ) -> None:
40
+ """Keep the collaborators of the healing branch.
41
+
42
+ Args:
43
+ config: project settings; ``send_screenshots`` attaches page images.
44
+ provider: the LLM port implementation classifying the failure.
45
+ generator: the regeneration loop handling the rot verdict.
46
+ cache: the store of the failed step; written by the generator
47
+ only — accepted for contract symmetry, never read here.
48
+ budgets: the per-test attempt registry; spent by the generator —
49
+ accepted for contract symmetry, never read here.
50
+ reporter: the visibility point for engine events.
51
+ """
52
+ self._config = config
53
+ self._provider = provider
54
+ self._generator = generator
55
+ self._reporter = reporter
56
+
57
+ def heal(
58
+ self,
59
+ step: CachedStep,
60
+ error: str,
61
+ previous_steps: list[str],
62
+ page: PageFacade,
63
+ ) -> CachedStep:
64
+ """Heal a failed cached step according to the classification verdict.
65
+
66
+ Args:
67
+ step: the cached step whose code failed.
68
+ error: the failure description of the cached code.
69
+ previous_steps: the sentences of the previous steps of the test.
70
+ page: the live page facade the healed code runs against.
71
+
72
+ Returns:
73
+ The healed step with proven code, already cached by the generator.
74
+
75
+ Raises:
76
+ ProductDefectError: the expectation of the step is genuinely
77
+ broken in the product; the cache stays untouched.
78
+ IncurableStepError: the verdict says regeneration cannot help,
79
+ or the regeneration attempt budget is exhausted — the
80
+ exhaustion reuses the verdict of this classification, no
81
+ second LLM request is made.
82
+ LlmUnavailableError: the provider service failed; no retry.
83
+ """
84
+ step_text = step.identity.normalized_text
85
+ classification = classify_step_failure(self._config, self._provider, step_text, step.code, error, page)
86
+
87
+ verdict = FailureVerdict(
88
+ category=classification.category,
89
+ explanation=classification.explanation,
90
+ recommendation=classification.recommendation,
91
+ )
92
+
93
+ self._reporter.emit("on_healing_started", {"step_text": step_text, "category": classification.category})
94
+
95
+ if classification.category == "product_defect":
96
+ raise ProductDefectError(step_text, classification.explanation, verdict)
97
+ if classification.category == "incurable":
98
+ raise IncurableStepError(step_text, classification.explanation, verdict)
99
+
100
+ try:
101
+ healed = self._generator.regenerate(
102
+ identity=step.identity,
103
+ step_text=step_text,
104
+ previous_steps=previous_steps,
105
+ page=page,
106
+ existing_code=step.code,
107
+ error=error,
108
+ )
109
+ except IncurableStepError as incurable:
110
+ if incurable.verdict is None:
111
+ # regeneration exhausted: verdict of this classification, no second LLM request
112
+ raise IncurableStepError(step_text, incurable.reason, verdict) from incurable
113
+ raise # a fresh failed-check verdict is never overwritten
114
+
115
+ self._reporter.emit("on_healed", {"step_text": step_text, "explanation": classification.explanation})
116
+ return healed
@@ -0,0 +1,19 @@
1
+ """Shared short-failure text of the engine: one truncation policy for reports and requests."""
2
+
3
+ #: Upper bound of the short failure description carried by reports and healing requests.
4
+ SHORT_ERROR_LENGTH = 200
5
+
6
+
7
+ def first_line_short(exc: Exception) -> str:
8
+ """Return the first line of the exception text, cut to 200 characters.
9
+
10
+ A message-less exception (a bare ``assert`` in step code) yields an
11
+ empty description instead of crashing the reporting and healing paths.
12
+
13
+ Args:
14
+ exc: the exception raised by the failed step or candidate branch.
15
+
16
+ Returns:
17
+ The short failure description carried by reports and requests.
18
+ """
19
+ return str(exc).partition("\n")[0][:SHORT_ERROR_LENGTH]
prettyplay/executor.py ADDED
@@ -0,0 +1,109 @@
1
+ """The step cycle: cache hit — execute; cache miss — generate; cached failure — heal."""
2
+
3
+ from .cache import RunBudgets, StepCache, StepIdentity, normalize_step_text
4
+ from .driver import PageFacade
5
+ from .engine import StepGenerator, StepHealer, run_step_code
6
+ from .engine.text import first_line_short
7
+ from .failures import IncurableStepError, ProductDefectError
8
+ from .reporting import StepReporter
9
+
10
+
11
+ class StepExecutor:
12
+ """The owner of the step cycle of one test.
13
+
14
+ One step goes through the full cycle: a cache hit executes the cached
15
+ code with no LLM involvement whatsoever; a miss delegates to the
16
+ generator, which stores the step after the candidate has actually
17
+ worked; a failed cached step delegates to the healer, whose verdict
18
+ decides between a loud product defect, an incurable step, and rot
19
+ regeneration. The scenario context — the sentences of the previous
20
+ steps of this test — feeds every generation and healing request.
21
+
22
+ Attributes:
23
+ cache_key: the context key of the owning test object.
24
+ _cache: the step cache of the test.
25
+ _generator: the generation engine of the cycle.
26
+ _healer: the healing engine of the cycle.
27
+ _reporter: the visibility point of the test.
28
+ _scenario: the sentences of the previous steps of this test.
29
+ """
30
+
31
+ def __init__( # noqa: PLR0913, PLR0917 — the signature is fixed by the root cell contract
32
+ self,
33
+ cache_key: str,
34
+ cache: StepCache,
35
+ generator: StepGenerator,
36
+ healer: StepHealer,
37
+ budgets: RunBudgets, # noqa: ARG002 — spent by the engine; kept for contract symmetry
38
+ reporter: StepReporter,
39
+ ) -> None:
40
+ """Keep the collaborators of the step cycle and reset the scenario context.
41
+
42
+ Args:
43
+ cache_key: the context key of the owning test object.
44
+ cache: the step cache of the test.
45
+ generator: the generation engine of the cycle.
46
+ healer: the healing engine of the cycle.
47
+ budgets: the per-test attempt registry; attempts are spent
48
+ by the engine, not by the executor.
49
+ reporter: the visibility point of the test.
50
+ """
51
+ self.cache_key = cache_key
52
+ self._cache = cache
53
+ self._generator = generator
54
+ self._healer = healer
55
+ self._reporter = reporter
56
+ self._scenario: list[str] = [] # test scenario context
57
+
58
+ def execute(self, step_text: str, step_type: str, page: PageFacade) -> None:
59
+ """Run one step through the full cycle.
60
+
61
+ Args:
62
+ step_text: the sentence of the step as written by the engineer.
63
+ step_type: the kind of the step sentence ({action, assertion}).
64
+ page: the live page facade of the current test.
65
+
66
+ Raises:
67
+ ProductDefectError: the healed step verdict says the expectation
68
+ of the step is genuinely broken in the product.
69
+ IncurableStepError: the step never generated successfully, or the
70
+ verdict says regeneration cannot help.
71
+ LlmUnavailableError: the provider service failed; no retry.
72
+ """
73
+ try:
74
+ self._reporter.emit("on_step_started", {"step_text": step_text, "step_type": step_type})
75
+
76
+ identity = StepIdentity(
77
+ cache_key=self.cache_key,
78
+ step_type=step_type,
79
+ normalized_text=normalize_step_text(step_text),
80
+ )
81
+ cached = self._cache.load(identity)
82
+
83
+ if cached is not None:
84
+ try:
85
+ run_step_code(cached.code, page)
86
+ except Exception as error: # cached code failed — context goes to healing
87
+ self._healer.heal(cached, first_line_short(error), self._scenario, page) # healed = re-executed
88
+ else:
89
+ self._generator.generate(identity, step_text, self._scenario, page)
90
+
91
+ self._scenario.append(step_text)
92
+ self._reporter.emit("on_step_passed", {"step_text": step_text, "step_type": step_type})
93
+ except Exception as error:
94
+ self._reporter.emit(
95
+ "on_step_failed",
96
+ {"step_text": step_text, "step_type": step_type, "error": first_line_short(error)},
97
+ )
98
+
99
+ if isinstance(error, (ProductDefectError, IncurableStepError)) and error.verdict is not None:
100
+ self._reporter.emit(
101
+ "on_step_verdict",
102
+ {
103
+ "step_text": step_text,
104
+ "category": error.verdict.category,
105
+ "explanation": error.verdict.explanation,
106
+ "recommendation": error.verdict.recommendation,
107
+ },
108
+ )
109
+ raise
@@ -0,0 +1,40 @@
1
+ # Failure taxonomy
2
+
3
+ Domain: failure kinds of prettyplay. Audience: integrators wiring library failures into runner and CI reporting.
4
+
5
+ Every step failure is one of three distinct kinds; all derive from `PrettyplayError`, so one except clause catches any prettyplay failure. A fourth kind — the configuration error — joins the base from the config cell.
6
+
7
+ | Exception | Meaning | When it happens | Recommended reaction |
8
+ |---|---|---|---|
9
+ | ProductDefectError | real product regression | an assertion expectation legitimately failed | treat as a bug: file it, fix the product — this failure is the value of the suite |
10
+ | IncurableStepError | the step cannot be (re)generated | attempt budget exhausted, step text no longer matches reality, ambiguity | follow `recommendation`: reword the step or refresh the cache |
11
+ | LlmUnavailableError | LLM infrastructure down | generation or healing ran while the provider was unavailable | restore provider access or keys; cached steps are unaffected |
12
+ | ConfigurationError | settings are invalid | the first library use loaded an invalid [tool.prettyplay] section | fix the named setting — the message lists the allowed values |
13
+
14
+ ## Verdicts on terminal failures
15
+
16
+ ProductDefectError and IncurableStepError carry an optional verdict: category, explanation, recommendation. It is fully present in the exception message, the on_step_verdict hook event and the log. When the LLM is unavailable the verdict is skipped quietly (WARNING in the log) — the failure itself is never delayed or distorted.
17
+
18
+ ```python
19
+ import pytest
20
+
21
+ from prettyplay.failures import IncurableStepError
22
+
23
+
24
+ def test_reports_only_library_failures():
25
+ with pytest.raises(IncurableStepError) as info:
26
+ ...
27
+ assert info.value.recommendation
28
+ # info.value.verdict may be None when the LLM was unavailable
29
+ ```
30
+
31
+ ## Assertion semantics
32
+
33
+ ProductDefectError is also an AssertionError: unittest reports the failed check as a failure (not an error), pytest shows it as an ordinary assertion failure, the traceback is folded to the library boundary. Catch it with `except PrettyplayError` or `except AssertionError` — both work.
34
+
35
+ ## Rules
36
+
37
+ - Every failure message is actionable: what happened, on which step, what to do next
38
+ - A healed run never turns a ProductDefectError into a green test
39
+ - LlmUnavailableError never occurs on the cached path — a cached suite runs without any LLM
40
+ - The verdict explanation never replaces the primary failure — it is appended
@@ -0,0 +1,117 @@
1
+ Usages:
2
+ conventions: .goga/usages/conventions.md
3
+
4
+ Annotations: |
5
+ Use `conventions` for code writing rules and testing.
6
+
7
+ The failure taxonomy of the library: three distinct, user-distinguishable step failure kinds plus the shared verdict type; every failure carries an actionable message.
8
+ All failure types derive from the single library base `PrettyplayError`; ProductDefectError additionally derives from AssertionError — the failed check is a failure, never an error, in any runner.
9
+ The rendered message of a terminal failure starts with its primary reason; the verdict render is appended, never interleaved.
10
+
11
+ ---
12
+
13
+ "PrettyplayError(message: str)":
14
+ location: errors.py
15
+ annotations: |
16
+ The common base of every library failure. Exists for one consumer need: catch any prettyplay failure with a single except clause at the test-suite boundary.
17
+
18
+ `message`: the failure description.
19
+
20
+ "FailureVerdict(category: str, explanation: str, recommendation: str)":
21
+ location: errors.py
22
+ annotations: |
23
+ The verdict of a terminal step failure: what the LLM saw on the page at the moment of the failure and what the engineer should do next.
24
+
25
+ `category`: the classification label: rot, product_defect or incurable.
26
+ `explanation`: what happened on the page — one short sentence.
27
+ `recommendation`: the recommended engineer action — one short sentence.
28
+
29
+ Requirements:
30
+ - Built by the engines from the failure classification; the failure types never request it themselves
31
+ properties:
32
+ "category -> str": |
33
+ The classification label: rot, product_defect or incurable.
34
+ "explanation -> str": |
35
+ What happened on the page at the moment of the failure.
36
+ "recommendation -> str": |
37
+ The recommended engineer action.
38
+ methods:
39
+ "render() -> text: str": |
40
+ Render the verdict as stable labelled lines — the single render used by the exception message tail, the hook event payload and the log record.
41
+
42
+ `text`: the rendered verdict.
43
+
44
+ Algorithm:
45
+ 1. Build one line per non-empty field: category:, explanation:, recommendation:
46
+ 2. Join the lines
47
+
48
+ Requirements:
49
+ - Labels are stable lowercase words — integrators parse them
50
+
51
+ "PrettyplayError::ProductDefectError(step_text: str, message: str, verdict: FailureVerdict | None)":
52
+ location: errors.py
53
+ annotations: |
54
+ A real functional product defect: the expectation of an assertion step legitimately did not hold against the current application state. This is the signal the test suite exists for.
55
+
56
+ `step_text`: the sentence of the failed step.
57
+ `message`: what exactly was expected and what was observed — the primary reason.
58
+ `verdict`: the optional `FailureVerdict`; absent when the LLM was unavailable — the failure never waits for it.
59
+
60
+ Requirements:
61
+ - Derives from `PrettyplayError` and AssertionError: catchable as any library failure and as an assertion failure in the same except clauses
62
+ - The rendered message starts with `message`; the `verdict` render is appended when present
63
+ - The traceback a runner sees starts at the library boundary — internal library frames are folded away
64
+ - Propagates to the test runner as a failing test: no retry, no healing
65
+
66
+ Constraints:
67
+ - The library facade stays framework-agnostic: the runner alignment is done by the type itself, never by a runner plugin or integration
68
+ properties:
69
+ "step_text -> str": |
70
+ The sentence of the failed step.
71
+ "message -> str": |
72
+ What exactly was expected and what was observed.
73
+ "verdict -> FailureVerdict | None": |
74
+ The optional failure verdict; None — the explicit absence when the LLM was unavailable.
75
+
76
+ "PrettyplayError::IncurableStepError(step_text: str, reason: str, verdict: FailureVerdict | None)":
77
+ location: errors.py
78
+ annotations: |
79
+ An incurable step: regeneration cannot produce working code — the attempt budget is exhausted, the step text no longer matches the application reality, or the intent is ambiguous.
80
+
81
+ `step_text`: the sentence of the failed step.
82
+ `reason`: the specific incurability cause — the primary reason of the rendered message.
83
+ `verdict`: the optional `FailureVerdict` — reused from a classification that already happened or requested at budget exhaustion; absent when the LLM was unavailable.
84
+
85
+ Requirements:
86
+ - The recommendation is carried by `verdict`; the recommendation property derives from it, falling back to the built-in path guidance when the verdict is absent
87
+ - The rendered message starts with `reason`, then appends the `verdict` render when present; the fallback recommendation keeps the message actionable without a verdict
88
+ - An execution failure, not a failed check: derives from `PrettyplayError` only, never from AssertionError
89
+ properties:
90
+ "step_text -> str": |
91
+ The sentence of the failed step.
92
+ "reason -> str": |
93
+ The specific incurability cause.
94
+ "recommendation -> str": |
95
+ The recommended engineer action: verdict recommendation when present, the built-in path guidance otherwise.
96
+ "verdict -> FailureVerdict | None": |
97
+ The optional failure verdict; None — the explicit absence when the LLM was unavailable.
98
+
99
+ "PrettyplayError::LlmUnavailableError(message: str)":
100
+ location: errors.py
101
+ annotations: |
102
+ LLM infrastructure failure: the provider service is unreachable, times out, rate-limits or rejects authentication. Blocks only code generation and healing; cached steps keep running.
103
+
104
+ `message`: the failure description naming the provider.
105
+
106
+ Requirements:
107
+ - Raised only on generation or healing paths; never on a cached step execution path
108
+ properties:
109
+ "message -> str": |
110
+ The failure description naming the provider.
111
+
112
+ ---
113
+
114
+ Author: Goga
115
+ CreatedAt: 07/09/26
116
+ Description: |
117
+ The failure taxonomy of prettyplay: product defect, incurable step, LLM infrastructure — mutations of one library base — and the shared verdict of a terminal failure.
@@ -0,0 +1,17 @@
1
+ """Facade of the prettyplay.failures cell: the failure taxonomy of the library."""
2
+
3
+ from .errors import (
4
+ FailureVerdict,
5
+ IncurableStepError,
6
+ LlmUnavailableError,
7
+ PrettyplayError,
8
+ ProductDefectError,
9
+ )
10
+
11
+ __all__ = [
12
+ "FailureVerdict",
13
+ "IncurableStepError",
14
+ "LlmUnavailableError",
15
+ "PrettyplayError",
16
+ "ProductDefectError",
17
+ ]
@@ -0,0 +1,147 @@
1
+ """The failure taxonomy of prettyplay: three distinct kinds, one library base.
2
+
3
+ Every failure the library raises derives from :class:`PrettyplayError`, so a
4
+ test suite catches any prettyplay failure with a single ``except`` clause at
5
+ its boundary. Each kind carries actionable fields instead of an opaque string.
6
+ The rendered message of a terminal failure starts with its primary reason;
7
+ the verdict render is appended, never interleaved.
8
+ """
9
+
10
+ from dataclasses import dataclass
11
+
12
+
13
+ class PrettyplayError(Exception):
14
+ """The common base of every library failure.
15
+
16
+ Args:
17
+ message: the failure description.
18
+ """
19
+
20
+ def __init__(self, message: str) -> None:
21
+ self.message = message
22
+
23
+ super().__init__(message)
24
+
25
+
26
+ @dataclass(frozen=True)
27
+ class FailureVerdict:
28
+ """The verdict of a terminal step failure, carried by the errors that stop the run.
29
+
30
+ Built by the engines from a :class:`~prettyplay.llm.FailureClassification`;
31
+ the failure types never request it themselves. A plain frozen value object,
32
+ not a pydantic model — the source classification already validated the data,
33
+ and failure paths stay cheap.
34
+
35
+ Args:
36
+ category: the classification label: rot, product_defect or incurable.
37
+ explanation: what happened on the page — one short sentence.
38
+ recommendation: the recommended engineer action — one short sentence.
39
+ """
40
+
41
+ category: str
42
+ explanation: str
43
+ recommendation: str
44
+
45
+ def render(self) -> str:
46
+ """Render the verdict as stable labelled lines.
47
+
48
+ The single render used by the exception message tail, the hook event
49
+ payload content and the log record; the labels are stable lowercase
50
+ words integrators parse. An empty field yields no line.
51
+
52
+ Returns:
53
+ The rendered verdict — one line per non-empty field, joined with newlines.
54
+ """
55
+ fields = (
56
+ ("category", self.category),
57
+ ("explanation", self.explanation),
58
+ ("recommendation", self.recommendation),
59
+ )
60
+
61
+ return "\n".join(f"{label}: {value}" for label, value in fields if value)
62
+
63
+
64
+ class ProductDefectError(PrettyplayError, AssertionError):
65
+ """A real functional product defect: the expectation of an assertion step did not hold.
66
+
67
+ The signal the test suite exists for: no retry, no healing — propagates to
68
+ the test runner as a failing test. Derives from both :class:`PrettyplayError`
69
+ and ``AssertionError``, so any runner counts it as a failure, never an error.
70
+
71
+ Args:
72
+ step_text: the sentence of the failed step.
73
+ message: what exactly was expected and what was observed — the primary reason.
74
+ verdict: the optional :class:`FailureVerdict`; ``None`` — the explicit
75
+ absence when the LLM was unavailable, the failure never waits for it.
76
+ """
77
+
78
+ def __init__(self, step_text: str, message: str, verdict: FailureVerdict | None = None) -> None:
79
+ self.step_text = step_text
80
+ self.message = message
81
+ self.verdict = verdict
82
+
83
+ PrettyplayError.__init__(self, message)
84
+
85
+ def __str__(self) -> str:
86
+ if self.verdict is None:
87
+ return self.message
88
+
89
+ return self.message + "\n" + self.verdict.render()
90
+
91
+
92
+ class IncurableStepError(PrettyplayError):
93
+ """An incurable step: regeneration cannot produce working code.
94
+
95
+ Raised when the attempt budget is exhausted, the step text no longer
96
+ matches the application reality, or the intent is ambiguous. An execution
97
+ failure, not a failed check — derives from :class:`PrettyplayError` only,
98
+ never from ``AssertionError``.
99
+
100
+ Args:
101
+ step_text: the sentence of the failed step.
102
+ reason: the specific incurability cause — the primary reason.
103
+ verdict: the optional :class:`FailureVerdict` — reused from a
104
+ classification that already happened or requested at budget
105
+ exhaustion; ``None`` when the LLM was unavailable.
106
+ """
107
+
108
+ def __init__(self, step_text: str, reason: str, verdict: FailureVerdict | None = None) -> None:
109
+ self.step_text = step_text
110
+ self.reason = reason
111
+ self.verdict = verdict
112
+
113
+ super().__init__(reason)
114
+
115
+ @property
116
+ def recommendation(self) -> str:
117
+ """The recommended engineer action.
118
+
119
+ Returns:
120
+ The verdict recommendation when present, the built-in path guidance otherwise.
121
+ """
122
+ if self.verdict is not None:
123
+ return self.verdict.recommendation
124
+
125
+ return "reword the step or refresh the cache"
126
+
127
+ def __str__(self) -> str:
128
+ if self.verdict is not None:
129
+ return self.reason + "\n" + self.verdict.render()
130
+
131
+ return self.reason + "\nrecommendation: reword the step or refresh the cache"
132
+
133
+
134
+ class LlmUnavailableError(PrettyplayError):
135
+ """LLM infrastructure failure: the provider service is unreachable or rejects the request.
136
+
137
+ Blocks only code generation and healing; cached steps keep running. No
138
+ retries.
139
+
140
+ Args:
141
+ message: the failure description naming the provider.
142
+ """
143
+
144
+ def __init__(self, message: str) -> None:
145
+ self.message = message
146
+
147
+ super().__init__(message)
@@ -0,0 +1,25 @@
1
+ # Failure classification
2
+
3
+ Domain: classifying a failed cached step before healing. Audience: engineers reasoning about healing decisions.
4
+
5
+ ## Categories
6
+
7
+ | Category | Meaning | Consequence |
8
+ |---|---|---|
9
+ | rot | the UI changed: selectors, texts, structure | the step is regenerated from the current page and retried |
10
+ | product_defect | the expectation legitimately failed | the test fails loudly — never healed green |
11
+ | incurable | regeneration cannot help: budget exhausted, text no longer matches reality, ambiguity | the incurable failure carries step, reason, recommendation |
12
+
13
+ ## Call
14
+
15
+ ```python
16
+ classification = provider.classify_failure(
17
+ prompt=system_prompt, # the system prompt text comes from the calling engine
18
+ step_text="click the «Sign in» button",
19
+ code=step_code,
20
+ error="element not found: button «Sign in»",
21
+ snapshot=snapshot_text,
22
+ screenshot=None,
23
+ )
24
+ print(classification.category, classification.explanation, classification.recommendation)
25
+ ```
@@ -0,0 +1,33 @@
1
+ # Providers
2
+
3
+ Domain: LLM provider selection and parity. Audience: integrators choosing a provider and setting models.
4
+
5
+ ## Select a provider
6
+
7
+ ```python
8
+ from prettyplay.config import Config
9
+ from prettyplay.llm import create_provider
10
+
11
+ config = Config(provider="anthropic", model="claude-sonnet-4-5")
12
+ provider = create_provider(config)
13
+ ```
14
+
15
+ The provider is a project setting: openai or anthropic; env override PRETTYPLAY_PROVIDER. API keys come only from environment variables: OPENAI_API_KEY for openai, ANTHROPIC_API_KEY for anthropic.
16
+
17
+ ## Models
18
+
19
+ | Setting | Purpose | Fallback |
20
+ |---|---|---|
21
+ | model | the main model for both operations | — |
22
+ | generation_model | code generation only | model |
23
+ | classification_model | failure classification only | model |
24
+
25
+ base_url overrides the provider endpoint when set.
26
+
27
+ ## Parity
28
+
29
+ Both providers expose the same two operations — generate_step_code and classify_failure — with identical inputs, identical output shapes and the identical failure taxonomy: a provider service failure raises LlmUnavailableError; cached step code never depends on the provider. One request per attempt; attempt budgets belong to the calling engine.
30
+
31
+ ## Answer shape
32
+
33
+ generate_step_code returns step code of the fixed form. Models often answer with a fenced python block (```python … ```); the provider unwraps the first fenced block before returning, so the engine receives clean code either way — an answer with no closed fence is returned verbatim and, if unparsable, keeps failing downstream in execution.