prettyplay 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Potentially problematic release.
This version of prettyplay might be problematic. Click here for more details.
- prettyplay/.usages/lifecycle.md +50 -0
- prettyplay/.usages/steps.md +56 -0
- prettyplay/CODEMANIFEST +182 -0
- prettyplay/__init__.py +13 -0
- prettyplay/cache/.usages/addressing.md +31 -0
- prettyplay/cache/.usages/budgets.md +21 -0
- prettyplay/cache/.usages/storage.md +32 -0
- prettyplay/cache/CODEMANIFEST +152 -0
- prettyplay/cache/__init__.py +8 -0
- prettyplay/cache/budgets.py +73 -0
- prettyplay/cache/models.py +60 -0
- prettyplay/cache/store.py +260 -0
- prettyplay/cache/text.py +35 -0
- prettyplay/config/.usages/configuration.md +87 -0
- prettyplay/config/CODEMANIFEST +132 -0
- prettyplay/config/__init__.py +6 -0
- prettyplay/config/loader.py +192 -0
- prettyplay/config/models.py +92 -0
- prettyplay/driver/.usages/facade.md +66 -0
- prettyplay/driver/CODEMANIFEST +162 -0
- prettyplay/driver/__init__.py +6 -0
- prettyplay/driver/page.py +350 -0
- prettyplay/driver/session.py +288 -0
- prettyplay/engine/.usages/generation.md +45 -0
- prettyplay/engine/.usages/healing.md +31 -0
- prettyplay/engine/CODEMANIFEST +215 -0
- prettyplay/engine/__init__.py +8 -0
- prettyplay/engine/classification.py +69 -0
- prettyplay/engine/execution.py +25 -0
- prettyplay/engine/generator.py +318 -0
- prettyplay/engine/healer.py +116 -0
- prettyplay/engine/text.py +19 -0
- prettyplay/executor.py +109 -0
- prettyplay/failures/.usages/taxonomy.md +40 -0
- prettyplay/failures/CODEMANIFEST +117 -0
- prettyplay/failures/__init__.py +17 -0
- prettyplay/failures/errors.py +147 -0
- prettyplay/llm/.usages/classification.md +25 -0
- prettyplay/llm/.usages/providers.md +33 -0
- prettyplay/llm/CODEMANIFEST +125 -0
- prettyplay/llm/__init__.py +8 -0
- prettyplay/llm/_request.py +216 -0
- prettyplay/llm/anthropic_provider.py +213 -0
- prettyplay/llm/models.py +22 -0
- prettyplay/llm/openai_provider.py +187 -0
- prettyplay/llm/provider.py +115 -0
- prettyplay/reporting/.usages/hooks.md +41 -0
- prettyplay/reporting/CODEMANIFEST +79 -0
- prettyplay/reporting/__init__.py +6 -0
- prettyplay/reporting/hooks.py +37 -0
- prettyplay/reporting/reporter.py +70 -0
- prettyplay/runtime.py +107 -0
- prettyplay/scenario.py +291 -0
- prettyplay-0.0.0.dist-info/METADATA +236 -0
- prettyplay-0.0.0.dist-info/RECORD +58 -0
- prettyplay-0.0.0.dist-info/WHEEL +5 -0
- prettyplay-0.0.0.dist-info/licenses/LICENSE +28 -0
- prettyplay-0.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
"""The openai SDK implementation of the LlmProvider port."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
|
|
5
|
+
from openai import OpenAI, OpenAIError
|
|
6
|
+
|
|
7
|
+
from ..config import Config
|
|
8
|
+
from ..failures import LlmUnavailableError
|
|
9
|
+
from ._request import (
|
|
10
|
+
build_classification_fields,
|
|
11
|
+
build_fields_text,
|
|
12
|
+
extract_code_block,
|
|
13
|
+
openai_user_content,
|
|
14
|
+
parse_classification_line,
|
|
15
|
+
require_completion_text,
|
|
16
|
+
unparsable_classification,
|
|
17
|
+
)
|
|
18
|
+
from .models import FailureClassification
|
|
19
|
+
from .provider import LlmProvider
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _first_choice_text(response: object) -> str | None:
|
|
23
|
+
"""Extract the message text of the first choice of an openai response.
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
response: the SDK response of a ``chat.completions.create`` call.
|
|
27
|
+
|
|
28
|
+
Returns:
|
|
29
|
+
The text of the first choice message, or ``None`` when the response
|
|
30
|
+
carries no choice at all (empty or missing ``choices``/``message``).
|
|
31
|
+
"""
|
|
32
|
+
choices = getattr(response, "choices", None) or []
|
|
33
|
+
message = getattr(choices[0], "message", None) if choices else None
|
|
34
|
+
|
|
35
|
+
return getattr(message, "content", None)
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
class OpenAiProvider(LlmProvider):
|
|
39
|
+
"""The LlmProvider implementation served by the openai SDK.
|
|
40
|
+
|
|
41
|
+
Full parity with :class:`AnthropicProvider`: the same operations, the
|
|
42
|
+
same inputs, the same output shapes. The constructor reads no
|
|
43
|
+
environment and constructs no client; the SDK client is created lazily
|
|
44
|
+
on the first request, so the library starts without LLM credentials.
|
|
45
|
+
Every service failure maps to
|
|
46
|
+
:class:`~prettyplay.failures.LlmUnavailableError` naming the provider;
|
|
47
|
+
the API key is read from the environment only and never logged.
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
def __init__(self, config: Config) -> None:
|
|
51
|
+
"""Keep the config; the SDK client stays unconstructed until the first request.
|
|
52
|
+
|
|
53
|
+
Args:
|
|
54
|
+
config: project settings; the effective models and the optional
|
|
55
|
+
base_url endpoint override come from it.
|
|
56
|
+
"""
|
|
57
|
+
self._config = config
|
|
58
|
+
self._client: OpenAI | None = None
|
|
59
|
+
|
|
60
|
+
def _get_client(self) -> OpenAI:
|
|
61
|
+
"""Construct the SDK client on the first request.
|
|
62
|
+
|
|
63
|
+
Returns:
|
|
64
|
+
The lazily constructed openai SDK client.
|
|
65
|
+
|
|
66
|
+
Raises:
|
|
67
|
+
LlmUnavailableError: the OPENAI_API_KEY environment variable is
|
|
68
|
+
missing or empty — generation and healing are blocked.
|
|
69
|
+
"""
|
|
70
|
+
if self._client is None:
|
|
71
|
+
api_key = os.environ.get("OPENAI_API_KEY")
|
|
72
|
+
|
|
73
|
+
if not api_key:
|
|
74
|
+
raise LlmUnavailableError("llm unavailable: openai: OPENAI_API_KEY is not set")
|
|
75
|
+
|
|
76
|
+
self._client = OpenAI(api_key=api_key, base_url=self._config.base_url or None)
|
|
77
|
+
return self._client
|
|
78
|
+
|
|
79
|
+
def generate_step_code( # noqa: PLR0913, PLR0917 — the signature is fixed by the port contract
|
|
80
|
+
self,
|
|
81
|
+
prompt: str,
|
|
82
|
+
user_instructions: str,
|
|
83
|
+
step_text: str,
|
|
84
|
+
previous_steps: list[str],
|
|
85
|
+
snapshot: str,
|
|
86
|
+
screenshot: bytes | None,
|
|
87
|
+
page_api: str,
|
|
88
|
+
existing_code: str | None,
|
|
89
|
+
error: str | None,
|
|
90
|
+
) -> str:
|
|
91
|
+
"""Generate step code of the fixed form working only through the driver facade.
|
|
92
|
+
|
|
93
|
+
Args:
|
|
94
|
+
prompt: the system prompt text supplied by the calling engine;
|
|
95
|
+
applied verbatim as the system message.
|
|
96
|
+
user_instructions: the project's code style instructions from the
|
|
97
|
+
generation_prompt setting; empty — the request carries no
|
|
98
|
+
instructions block, non-empty — rendered verbatim as a
|
|
99
|
+
separate USER INSTRUCTIONS block of the user content,
|
|
100
|
+
identically to the anthropic implementation.
|
|
101
|
+
step_text: the sentence of the step to generate.
|
|
102
|
+
previous_steps: the sentences of the previous steps of the test,
|
|
103
|
+
in execution order — scenario context.
|
|
104
|
+
snapshot: the accessibility snapshot of the current page.
|
|
105
|
+
screenshot: an optional PNG image of the page; passed only when
|
|
106
|
+
the project enables screenshots.
|
|
107
|
+
page_api: the exact page facade surface listing — the calls the
|
|
108
|
+
model may use.
|
|
109
|
+
existing_code: the existing step code that failed; non-empty only
|
|
110
|
+
on regeneration requests.
|
|
111
|
+
error: the failure description of the existing code; non-empty
|
|
112
|
+
only on regeneration requests.
|
|
113
|
+
|
|
114
|
+
Returns:
|
|
115
|
+
The generated step code of the fixed form.
|
|
116
|
+
|
|
117
|
+
Raises:
|
|
118
|
+
LlmUnavailableError: the SDK client is unavailable or the
|
|
119
|
+
service request failed.
|
|
120
|
+
"""
|
|
121
|
+
text = build_fields_text(user_instructions, step_text, previous_steps, snapshot, page_api, existing_code, error)
|
|
122
|
+
messages = [
|
|
123
|
+
{"role": "system", "content": prompt},
|
|
124
|
+
{"role": "user", "content": openai_user_content(text, screenshot)},
|
|
125
|
+
]
|
|
126
|
+
|
|
127
|
+
try:
|
|
128
|
+
response = self._get_client().chat.completions.create(
|
|
129
|
+
model=self._config.effective_generation_model,
|
|
130
|
+
messages=messages,
|
|
131
|
+
)
|
|
132
|
+
except OpenAIError as sdk_error:
|
|
133
|
+
raise LlmUnavailableError("llm unavailable: openai request failed") from sdk_error
|
|
134
|
+
|
|
135
|
+
return extract_code_block(require_completion_text(_first_choice_text(response), "openai"))
|
|
136
|
+
|
|
137
|
+
def classify_failure( # noqa: PLR0913, PLR0917 — the signature is fixed by the port contract
|
|
138
|
+
self,
|
|
139
|
+
prompt: str,
|
|
140
|
+
step_text: str,
|
|
141
|
+
code: str,
|
|
142
|
+
error: str,
|
|
143
|
+
snapshot: str,
|
|
144
|
+
screenshot: bytes | None,
|
|
145
|
+
) -> FailureClassification:
|
|
146
|
+
"""Classify a failed cached step.
|
|
147
|
+
|
|
148
|
+
Args:
|
|
149
|
+
prompt: the system prompt text supplied by the calling engine;
|
|
150
|
+
applied verbatim as the system message.
|
|
151
|
+
step_text: the sentence of the failed step.
|
|
152
|
+
code: the existing step code that failed.
|
|
153
|
+
error: the human-readable description of the failure.
|
|
154
|
+
snapshot: the accessibility snapshot of the current page.
|
|
155
|
+
screenshot: an optional PNG image of the page; passed only when
|
|
156
|
+
the project enables screenshots.
|
|
157
|
+
|
|
158
|
+
Returns:
|
|
159
|
+
The classification verdict; an unparsable or unknown answer maps
|
|
160
|
+
protectively to the incurable category.
|
|
161
|
+
|
|
162
|
+
Raises:
|
|
163
|
+
LlmUnavailableError: the SDK client is unavailable or the
|
|
164
|
+
service request failed.
|
|
165
|
+
"""
|
|
166
|
+
text = build_classification_fields(step_text, code, error, snapshot)
|
|
167
|
+
messages = [
|
|
168
|
+
{"role": "system", "content": prompt},
|
|
169
|
+
{"role": "user", "content": openai_user_content(text, screenshot)},
|
|
170
|
+
]
|
|
171
|
+
|
|
172
|
+
try:
|
|
173
|
+
response = self._get_client().chat.completions.create(
|
|
174
|
+
model=self._config.effective_classification_model,
|
|
175
|
+
messages=messages,
|
|
176
|
+
)
|
|
177
|
+
except OpenAIError as sdk_error:
|
|
178
|
+
raise LlmUnavailableError("llm unavailable: openai request failed") from sdk_error
|
|
179
|
+
|
|
180
|
+
answer = require_completion_text(_first_choice_text(response), "openai")
|
|
181
|
+
parsed = parse_classification_line(answer)
|
|
182
|
+
|
|
183
|
+
if parsed is None:
|
|
184
|
+
return FailureClassification(**unparsable_classification())
|
|
185
|
+
|
|
186
|
+
category, explanation, recommendation = parsed
|
|
187
|
+
return FailureClassification(category=category, explanation=explanation, recommendation=recommendation)
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""The unified LLM port of the library and the provider factory."""
|
|
2
|
+
|
|
3
|
+
from ..config import Config
|
|
4
|
+
from .models import FailureClassification
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
class LlmProvider:
|
|
8
|
+
"""The single LLM port: step code generation and failure classification.
|
|
9
|
+
|
|
10
|
+
One contract, two interchangeable SDK implementations selected by
|
|
11
|
+
configuration — the provider choice is never a capability difference.
|
|
12
|
+
The port itself is never instantiated at runtime; implementations own
|
|
13
|
+
one completion request per attempt (attempt budgets live in the calling
|
|
14
|
+
engine) and map every service failure to
|
|
15
|
+
:class:`~prettyplay.failures.LlmUnavailableError` naming the provider.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
def generate_step_code( # noqa: PLR0913, PLR0917 — the signature is fixed by the port contract
|
|
19
|
+
self,
|
|
20
|
+
prompt: str,
|
|
21
|
+
user_instructions: str,
|
|
22
|
+
step_text: str,
|
|
23
|
+
previous_steps: list[str],
|
|
24
|
+
snapshot: str,
|
|
25
|
+
screenshot: bytes | None,
|
|
26
|
+
page_api: str,
|
|
27
|
+
existing_code: str | None,
|
|
28
|
+
error: str | None,
|
|
29
|
+
) -> str:
|
|
30
|
+
"""Generate step code of the fixed form working only through the driver facade.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
prompt: the system prompt text supplied by the calling engine;
|
|
34
|
+
applied verbatim as the system message.
|
|
35
|
+
user_instructions: the project's code style instructions supplied
|
|
36
|
+
by the calling engine from the generation_prompt setting;
|
|
37
|
+
empty — the request carries no instructions block, non-empty —
|
|
38
|
+
rendered verbatim as a separate USER INSTRUCTIONS block of the
|
|
39
|
+
user content, identically in both implementations.
|
|
40
|
+
step_text: the sentence of the step to generate.
|
|
41
|
+
previous_steps: the sentences of the previous steps of the test,
|
|
42
|
+
in execution order — scenario context.
|
|
43
|
+
snapshot: the accessibility snapshot of the current page.
|
|
44
|
+
screenshot: an optional PNG image of the page; passed only when
|
|
45
|
+
the project enables screenshots.
|
|
46
|
+
page_api: the exact page facade surface listing — the calls the
|
|
47
|
+
model may use.
|
|
48
|
+
existing_code: the existing step code that failed; non-empty only
|
|
49
|
+
on regeneration requests.
|
|
50
|
+
error: the failure description of the existing code; non-empty
|
|
51
|
+
only on regeneration requests.
|
|
52
|
+
|
|
53
|
+
Returns:
|
|
54
|
+
The generated step code of the fixed form.
|
|
55
|
+
|
|
56
|
+
Raises:
|
|
57
|
+
NotImplementedError: the port itself carries no implementation.
|
|
58
|
+
"""
|
|
59
|
+
raise NotImplementedError("LlmProvider is a port; use create_provider() to select an implementation")
|
|
60
|
+
|
|
61
|
+
def classify_failure( # noqa: PLR0913, PLR0917 — the signature is fixed by the port contract
|
|
62
|
+
self,
|
|
63
|
+
prompt: str,
|
|
64
|
+
step_text: str,
|
|
65
|
+
code: str,
|
|
66
|
+
error: str,
|
|
67
|
+
snapshot: str,
|
|
68
|
+
screenshot: bytes | None,
|
|
69
|
+
) -> FailureClassification:
|
|
70
|
+
"""Classify a failed cached step.
|
|
71
|
+
|
|
72
|
+
Args:
|
|
73
|
+
prompt: the system prompt text supplied by the calling engine;
|
|
74
|
+
applied verbatim as the system message.
|
|
75
|
+
step_text: the sentence of the failed step.
|
|
76
|
+
code: the existing step code that failed.
|
|
77
|
+
error: the human-readable description of the failure.
|
|
78
|
+
snapshot: the accessibility snapshot of the current page.
|
|
79
|
+
screenshot: an optional PNG image of the page; passed only when
|
|
80
|
+
the project enables screenshots.
|
|
81
|
+
|
|
82
|
+
Returns:
|
|
83
|
+
The classification verdict.
|
|
84
|
+
|
|
85
|
+
Raises:
|
|
86
|
+
NotImplementedError: the port itself carries no implementation.
|
|
87
|
+
"""
|
|
88
|
+
raise NotImplementedError("LlmProvider is a port; use create_provider() to select an implementation")
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def create_provider(config: Config) -> LlmProvider:
|
|
92
|
+
"""Select and construct the LLM provider from configuration.
|
|
93
|
+
|
|
94
|
+
Args:
|
|
95
|
+
config: project settings; the provider setting selects the SDK
|
|
96
|
+
implementation.
|
|
97
|
+
|
|
98
|
+
Returns:
|
|
99
|
+
The selected provider implementation.
|
|
100
|
+
|
|
101
|
+
Raises:
|
|
102
|
+
ValueError: the provider setting names no supported provider.
|
|
103
|
+
"""
|
|
104
|
+
# deferred: the implementations subclass the port defined in this module,
|
|
105
|
+
# so a top-level import here would be circular
|
|
106
|
+
from .anthropic_provider import AnthropicProvider # noqa: PLC0415
|
|
107
|
+
from .openai_provider import OpenAiProvider # noqa: PLC0415
|
|
108
|
+
|
|
109
|
+
if config.provider == "openai":
|
|
110
|
+
return OpenAiProvider(config)
|
|
111
|
+
|
|
112
|
+
if config.provider == "anthropic":
|
|
113
|
+
return AnthropicProvider(config)
|
|
114
|
+
|
|
115
|
+
raise ValueError(f"unsupported provider {config.provider!r}: expected one of openai, anthropic")
|
|
@@ -0,0 +1,41 @@
|
|
|
1
|
+
# Step and healing hooks
|
|
2
|
+
|
|
3
|
+
Domain: event callbacks of prettyplay. Audience: integrators building custom reporting, metrics or CI reactions on top of step execution.
|
|
4
|
+
|
|
5
|
+
StepHooks is a thin callback contract. The library calls the matching method synchronously while a step executes. The base implementation of every method is a no-op — override only the events you need. Hook implementations are registered on the main library object at the start of a test.
|
|
6
|
+
|
|
7
|
+
## Events
|
|
8
|
+
|
|
9
|
+
| Method | When | Payload |
|
|
10
|
+
|---|---|---|
|
|
11
|
+
| on_step_started | a step started executing | step_text, step_type (action or assertion) |
|
|
12
|
+
| on_step_passed | the step finished successfully | step_text, step_type |
|
|
13
|
+
| on_step_failed | the step failed | step_text, step_type, error (short description) |
|
|
14
|
+
| on_step_verdict | the terminal failure carried a verdict (fires after on_step_failed) | step_text, category (rot, product_defect, incurable), explanation, recommendation |
|
|
15
|
+
| on_generation_started | a generation attempt started | step_text, attempt (1-based) |
|
|
16
|
+
| on_healing_started | healing of a failed cached step started | step_text, category (rot, product_defect, incurable) |
|
|
17
|
+
| on_healed | the step healed, cache updated | step_text, explanation (why rot, what changed) |
|
|
18
|
+
| on_cache_saved | step code written to the cache | step_text, filename |
|
|
19
|
+
| on_cache_skipped | cache write skipped | step_text, reason (e.g. read-only cache) |
|
|
20
|
+
|
|
21
|
+
## Example
|
|
22
|
+
|
|
23
|
+
```python
|
|
24
|
+
from prettyplay.reporting import StepHooks
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class HealingMonitor(StepHooks):
|
|
28
|
+
def on_step_verdict(self, step_text: str, category: str, explanation: str, recommendation: str) -> None:
|
|
29
|
+
print(f"verdict [{category}]: {step_text} — {explanation} → {recommendation}")
|
|
30
|
+
|
|
31
|
+
def on_healed(self, step_text: str, explanation: str) -> None:
|
|
32
|
+
print(f"healed: {step_text} — {explanation}")
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## Rules
|
|
36
|
+
|
|
37
|
+
- Handlers run synchronously inside step execution: keep them fast
|
|
38
|
+
- A raising handler is logged as a warning and skipped — the test run never fails because of a hook
|
|
39
|
+
- on_step_verdict fires only when the verdict exists: LLM unavailability skips the verdict quietly (WARNING in the log)
|
|
40
|
+
- Payload values are plain strings; the attempt counter of on_generation_started is an int
|
|
41
|
+
- Step texts land in logs and hooks: never put secrets or personal data into a step sentence
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
Usages:
|
|
2
|
+
conventions: .goga/usages/conventions.md
|
|
3
|
+
|
|
4
|
+
Annotations: |
|
|
5
|
+
Use `conventions` for code writing rules and testing.
|
|
6
|
+
|
|
7
|
+
Visibility goes through the standard logging library: the logger is named prettyplay.
|
|
8
|
+
Log messages: lowercase, concise operational wording, stable event names, contextual metadata attached.
|
|
9
|
+
Never log secrets, credentials, tokens or personal sensitive data.
|
|
10
|
+
Step lifecycle events — including the verdict event — are logged at INFO; a skipped cache write and a failed hook call — WARNING.
|
|
11
|
+
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
"StepHooks()":
|
|
15
|
+
location: hooks.py
|
|
16
|
+
annotations: |
|
|
17
|
+
Thin callback contract for integrators: consumer-side reactions to step, generation, healing and cache events. The library calls the hooks synchronously; the integrator overrides the events of interest. No event bus, no queuing, no delivery retries.
|
|
18
|
+
|
|
19
|
+
The base implementation of every method is a no-op: override only the events you need.
|
|
20
|
+
methods:
|
|
21
|
+
"on_step_started(step_text: str, step_type: str)": |
|
|
22
|
+
A step started executing; `step_type` is action or assertion.
|
|
23
|
+
"on_step_passed(step_text: str, step_type: str)": |
|
|
24
|
+
The step finished successfully.
|
|
25
|
+
"on_step_failed(step_text: str, step_type: str, error: str)": |
|
|
26
|
+
The step failed; `error` is a short human-readable failure description.
|
|
27
|
+
"on_step_verdict(step_text: str, category: str, explanation: str, recommendation: str)": |
|
|
28
|
+
The terminal failure of the step carried a verdict; fires after on_step_failed.
|
|
29
|
+
|
|
30
|
+
`step_text`: the sentence of the failed step.
|
|
31
|
+
`category`: the verdict label — rot, product_defect or incurable.
|
|
32
|
+
`explanation`: what the LLM saw on the page at the moment of the failure.
|
|
33
|
+
`recommendation`: the recommended engineer action.
|
|
34
|
+
|
|
35
|
+
Requirements:
|
|
36
|
+
- Not fired when the verdict was skipped: LLM unavailability logs a WARNING and raises no event
|
|
37
|
+
- Payload values are plain strings, uniform with the other events
|
|
38
|
+
"on_generation_started(step_text: str, attempt: int)": |
|
|
39
|
+
A code generation attempt started; `attempt` is the 1-based attempt number.
|
|
40
|
+
"on_healing_started(step_text: str, category: str)": |
|
|
41
|
+
Healing of a failed cached step started; `category` is the classification label: rot, product_defect, incurable.
|
|
42
|
+
"on_healed(step_text: str, explanation: str)": |
|
|
43
|
+
The step was healed and the cache updated; `explanation` says why it was rot and what changed.
|
|
44
|
+
"on_cache_saved(step_text: str, filename: str)": |
|
|
45
|
+
Step code was written to the cache file `filename`.
|
|
46
|
+
"on_cache_skipped(step_text: str, reason: str)": |
|
|
47
|
+
The cache write was skipped; `reason` names the cause, e.g. a read-only cache.
|
|
48
|
+
|
|
49
|
+
"StepReporter(hooks: list[StepHooks])":
|
|
50
|
+
location: reporter.py
|
|
51
|
+
annotations: |
|
|
52
|
+
The single visibility point of the library: every step, generation, healing and cache event goes through the emit method — written to the logger prettyplay and forwarded to the registered `StepHooks` implementations.
|
|
53
|
+
|
|
54
|
+
`hooks`: callback implementations registered by the integrator; the list may be empty.
|
|
55
|
+
methods:
|
|
56
|
+
"emit(event: str, payload: dict[str, str | int])": |
|
|
57
|
+
Dispatch one event.
|
|
58
|
+
|
|
59
|
+
`event`: the event name — equals a `StepHooks` method name exactly.
|
|
60
|
+
`payload`: the payload fields of the event; values are strings, the attempt counter is an int.
|
|
61
|
+
|
|
62
|
+
Algorithm:
|
|
63
|
+
1. Write a structured log record to the logger prettyplay: the event name as the message, the payload fields as contextual metadata
|
|
64
|
+
2. For each registered hook, in registration order, call the method named `event` with the payload values as arguments
|
|
65
|
+
3. A hook that raises is logged as a warning and skipped; the run continues
|
|
66
|
+
|
|
67
|
+
Requirements:
|
|
68
|
+
- Log levels: step, generation and healing lifecycle — INFO; a skipped cache write and a failed hook call — WARNING
|
|
69
|
+
|
|
70
|
+
Constraints:
|
|
71
|
+
- No secrets in log records (see `conventions`)
|
|
72
|
+
- No own event bus: the emit call is a synchronous fan-out, nothing more
|
|
73
|
+
|
|
74
|
+
---
|
|
75
|
+
|
|
76
|
+
Author: Goga
|
|
77
|
+
CreatedAt: 07/09/26
|
|
78
|
+
Description: |
|
|
79
|
+
Visibility of prettyplay: the logger prettyplay plus the thin StepHooks callback contract.
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""Thin callback contract for integrator-side reactions to step execution events.
|
|
2
|
+
|
|
3
|
+
The library calls the matching method synchronously while a step executes; the
|
|
4
|
+
base implementation of every method is a no-op — override only the events you
|
|
5
|
+
need. No event bus, no queuing, no delivery retries.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
class StepHooks:
|
|
10
|
+
"""Callback contract of the nine prettyplay events; every base method is a no-op."""
|
|
11
|
+
|
|
12
|
+
def on_step_started(self, step_text: str, step_type: str) -> None:
|
|
13
|
+
"""A step started executing; ``step_type`` is action or assertion."""
|
|
14
|
+
|
|
15
|
+
def on_step_passed(self, step_text: str, step_type: str) -> None:
|
|
16
|
+
"""The step finished successfully."""
|
|
17
|
+
|
|
18
|
+
def on_step_failed(self, step_text: str, step_type: str, error: str) -> None:
|
|
19
|
+
"""The step failed; ``error`` is a short human-readable failure description."""
|
|
20
|
+
|
|
21
|
+
def on_step_verdict(self, step_text: str, category: str, explanation: str, recommendation: str) -> None:
|
|
22
|
+
"""The terminal failure of the step carried a verdict; fires after on_step_failed."""
|
|
23
|
+
|
|
24
|
+
def on_generation_started(self, step_text: str, attempt: int) -> None:
|
|
25
|
+
"""A code generation attempt started; ``attempt`` is the 1-based attempt number."""
|
|
26
|
+
|
|
27
|
+
def on_healing_started(self, step_text: str, category: str) -> None:
|
|
28
|
+
"""Healing of a failed cached step started; ``category`` is rot, product_defect or incurable."""
|
|
29
|
+
|
|
30
|
+
def on_healed(self, step_text: str, explanation: str) -> None:
|
|
31
|
+
"""The step was healed and the cache updated; ``explanation`` says why and what changed."""
|
|
32
|
+
|
|
33
|
+
def on_cache_saved(self, step_text: str, filename: str) -> None:
|
|
34
|
+
"""Step code was written to the cache file ``filename``."""
|
|
35
|
+
|
|
36
|
+
def on_cache_skipped(self, step_text: str, reason: str) -> None:
|
|
37
|
+
"""The cache write was skipped; ``reason`` names the cause, e.g. a read-only cache."""
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""The single visibility point of prettyplay: the ``prettyplay`` logger plus hook fan-out."""
|
|
2
|
+
|
|
3
|
+
import logging
|
|
4
|
+
|
|
5
|
+
from .hooks import StepHooks
|
|
6
|
+
|
|
7
|
+
#: Events written at WARNING; every other event is step/generation/healing lifecycle INFO.
|
|
8
|
+
_WARNING_EVENTS = frozenset({"on_cache_skipped"})
|
|
9
|
+
|
|
10
|
+
#: Attributes Logger.makeRecord forbids overwriting: collision keys get a ``ctx_`` prefix
|
|
11
|
+
#: in the log record only — hooks always receive the original payload kwargs.
|
|
12
|
+
_LOG_RECORD_RESERVED = frozenset(
|
|
13
|
+
{
|
|
14
|
+
"name",
|
|
15
|
+
"msg",
|
|
16
|
+
"message",
|
|
17
|
+
"args",
|
|
18
|
+
"levelname",
|
|
19
|
+
"levelno",
|
|
20
|
+
"pathname",
|
|
21
|
+
"filename",
|
|
22
|
+
"module",
|
|
23
|
+
"exc_info",
|
|
24
|
+
"funcName",
|
|
25
|
+
"lineno",
|
|
26
|
+
"created",
|
|
27
|
+
"msecs",
|
|
28
|
+
"relativeCreated",
|
|
29
|
+
"thread",
|
|
30
|
+
"threadName",
|
|
31
|
+
"process",
|
|
32
|
+
"processName",
|
|
33
|
+
"stack_info",
|
|
34
|
+
"asctime",
|
|
35
|
+
"taskName",
|
|
36
|
+
}
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class StepReporter:
|
|
41
|
+
"""Dispatches every step, generation, healing and cache event to the logger and hooks."""
|
|
42
|
+
|
|
43
|
+
def __init__(self, hooks: list[StepHooks]) -> None:
|
|
44
|
+
"""Keep the hooks list by reference in the public ``hooks`` attribute.
|
|
45
|
+
|
|
46
|
+
Args:
|
|
47
|
+
hooks: callback implementations registered by the integrator; may be empty.
|
|
48
|
+
``add_hooks`` of the main object appends into this same list.
|
|
49
|
+
"""
|
|
50
|
+
self.hooks = hooks
|
|
51
|
+
self._logger = logging.getLogger("prettyplay")
|
|
52
|
+
|
|
53
|
+
def emit(self, event: str, payload: dict[str, str | int]) -> None:
|
|
54
|
+
"""Dispatch one event to the logger and to every hook, in registration order.
|
|
55
|
+
|
|
56
|
+
Args:
|
|
57
|
+
event: the event name — equals a ``StepHooks`` method name exactly.
|
|
58
|
+
payload: the payload fields of the event; values are strings, the
|
|
59
|
+
attempt counter is an int.
|
|
60
|
+
"""
|
|
61
|
+
level = logging.WARNING if event in _WARNING_EVENTS else logging.INFO
|
|
62
|
+
extra = {f"ctx_{key}" if key in _LOG_RECORD_RESERVED else key: value for key, value in payload.items()}
|
|
63
|
+
|
|
64
|
+
self._logger.log(level, event, extra=extra)
|
|
65
|
+
|
|
66
|
+
for hook in self.hooks:
|
|
67
|
+
try:
|
|
68
|
+
getattr(hook, event)(**payload)
|
|
69
|
+
except Exception:
|
|
70
|
+
self._logger.warning("hook call failed", extra={"event": event, "hook": type(hook).__name__})
|
prettyplay/runtime.py
ADDED
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Per-test composition root of the library: the objects one test owns.
|
|
2
|
+
|
|
3
|
+
One :class:`PrettyplayRuntime` serves exactly one test: it composes the
|
|
4
|
+
validated settings, the per-test attempt registry, the browser driver and
|
|
5
|
+
the LLM provider of that single test. Nothing is shared between tests —
|
|
6
|
+
two tests in one process build two runtimes with fresh budgets and
|
|
7
|
+
independent browser sessions. The driver and the provider construct
|
|
8
|
+
lazily — building the runtime (and any test object on top of it) never
|
|
9
|
+
requires LLM credentials and never launches a browser.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import atexit
|
|
15
|
+
|
|
16
|
+
from .cache import RunBudgets
|
|
17
|
+
from .config import Config
|
|
18
|
+
from .driver import DriverSession, PageFacade
|
|
19
|
+
from .llm import LlmProvider, create_provider
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
class PrettyplayRuntime:
|
|
23
|
+
"""The composition root: the per-test objects one test composes.
|
|
24
|
+
|
|
25
|
+
The config and the attempt registry build eagerly — both are cheap and
|
|
26
|
+
credential-free. The browser driver and the LLM provider construct lazily
|
|
27
|
+
on first access; the provider client itself stays unconstructed until the
|
|
28
|
+
first request, so no LLM key is needed to build the runtime. Every
|
|
29
|
+
instance registers its own close with ``atexit``, so the browser and the
|
|
30
|
+
Playwright driver of the test stop synchronously before the process exits
|
|
31
|
+
even when the test never calls close explicitly. A manual ``close``
|
|
32
|
+
unregisters the hook — a closed runtime never stays pinned for the rest
|
|
33
|
+
of the process — and a driver started after a close re-arms it.
|
|
34
|
+
|
|
35
|
+
Attributes:
|
|
36
|
+
_config: validated project settings of the test.
|
|
37
|
+
_budgets: the per-test attempt registry owned by this runtime.
|
|
38
|
+
_driver: the browser driver of the test; ``None`` until first access.
|
|
39
|
+
_provider: the LLM provider of the test; ``None`` until first access.
|
|
40
|
+
"""
|
|
41
|
+
|
|
42
|
+
def __init__(self, config: Config) -> None:
|
|
43
|
+
"""Compose the eager per-test objects from the config.
|
|
44
|
+
|
|
45
|
+
Args:
|
|
46
|
+
config: validated project settings of the test.
|
|
47
|
+
"""
|
|
48
|
+
self._config = config
|
|
49
|
+
self._budgets = RunBudgets(config.generation_attempts, config.healing_attempts)
|
|
50
|
+
self._driver: DriverSession | None = None
|
|
51
|
+
self._provider: LlmProvider | None = None
|
|
52
|
+
atexit.register(self.close)
|
|
53
|
+
|
|
54
|
+
@property
|
|
55
|
+
def config(self) -> Config:
|
|
56
|
+
"""The validated settings of the test."""
|
|
57
|
+
return self._config
|
|
58
|
+
|
|
59
|
+
@property
|
|
60
|
+
def budgets(self) -> RunBudgets:
|
|
61
|
+
"""The per-test attempt registry owned by this one runtime."""
|
|
62
|
+
return self._budgets
|
|
63
|
+
|
|
64
|
+
@property
|
|
65
|
+
def driver(self) -> DriverSession:
|
|
66
|
+
"""The browser driver of the test, constructed lazily exactly once.
|
|
67
|
+
|
|
68
|
+
A construction after a ``close`` re-arms the ``atexit`` hook the
|
|
69
|
+
close dropped: a freshly started browser always has its exit stop.
|
|
70
|
+
"""
|
|
71
|
+
if self._driver is None:
|
|
72
|
+
self._driver = DriverSession(self._config)
|
|
73
|
+
atexit.register(self.close)
|
|
74
|
+
|
|
75
|
+
return self._driver
|
|
76
|
+
|
|
77
|
+
@property
|
|
78
|
+
def provider(self) -> LlmProvider:
|
|
79
|
+
"""The LLM provider of the test, constructed lazily exactly once."""
|
|
80
|
+
if self._provider is None:
|
|
81
|
+
self._provider = create_provider(self._config)
|
|
82
|
+
|
|
83
|
+
return self._provider
|
|
84
|
+
|
|
85
|
+
def open_page(self) -> PageFacade:
|
|
86
|
+
"""Open a fresh isolated page of the test driver.
|
|
87
|
+
|
|
88
|
+
Returns:
|
|
89
|
+
The facade of the new page of a fresh isolated context.
|
|
90
|
+
"""
|
|
91
|
+
return self.driver.open_context()
|
|
92
|
+
|
|
93
|
+
def close(self) -> None:
|
|
94
|
+
"""Stop the browser driver of the test; safe when nothing was started.
|
|
95
|
+
|
|
96
|
+
Idempotent: a call before any driver access and repeated calls are
|
|
97
|
+
no-ops. The call also drops the ``atexit`` hook registered in
|
|
98
|
+
``__init__`` — a closed runtime and its objects are collectible, not
|
|
99
|
+
pinned until process exit — and a driver started later re-arms the
|
|
100
|
+
hook. The runtime itself stays usable after the call: the next
|
|
101
|
+
driver access constructs a fresh session.
|
|
102
|
+
"""
|
|
103
|
+
atexit.unregister(self.close)
|
|
104
|
+
|
|
105
|
+
if self._driver is not None:
|
|
106
|
+
self._driver.close()
|
|
107
|
+
self._driver = None
|