prettyplay 0.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- prettyplay/.usages/lifecycle.md +50 -0
- prettyplay/.usages/steps.md +56 -0
- prettyplay/CODEMANIFEST +182 -0
- prettyplay/__init__.py +13 -0
- prettyplay/cache/.usages/addressing.md +31 -0
- prettyplay/cache/.usages/budgets.md +21 -0
- prettyplay/cache/.usages/storage.md +32 -0
- prettyplay/cache/CODEMANIFEST +152 -0
- prettyplay/cache/__init__.py +8 -0
- prettyplay/cache/budgets.py +73 -0
- prettyplay/cache/models.py +60 -0
- prettyplay/cache/store.py +260 -0
- prettyplay/cache/text.py +35 -0
- prettyplay/config/.usages/configuration.md +87 -0
- prettyplay/config/CODEMANIFEST +132 -0
- prettyplay/config/__init__.py +6 -0
- prettyplay/config/loader.py +192 -0
- prettyplay/config/models.py +92 -0
- prettyplay/driver/.usages/facade.md +66 -0
- prettyplay/driver/CODEMANIFEST +162 -0
- prettyplay/driver/__init__.py +6 -0
- prettyplay/driver/page.py +350 -0
- prettyplay/driver/session.py +288 -0
- prettyplay/engine/.usages/generation.md +45 -0
- prettyplay/engine/.usages/healing.md +31 -0
- prettyplay/engine/CODEMANIFEST +215 -0
- prettyplay/engine/__init__.py +8 -0
- prettyplay/engine/classification.py +69 -0
- prettyplay/engine/execution.py +25 -0
- prettyplay/engine/generator.py +318 -0
- prettyplay/engine/healer.py +116 -0
- prettyplay/engine/text.py +19 -0
- prettyplay/executor.py +109 -0
- prettyplay/failures/.usages/taxonomy.md +40 -0
- prettyplay/failures/CODEMANIFEST +117 -0
- prettyplay/failures/__init__.py +17 -0
- prettyplay/failures/errors.py +147 -0
- prettyplay/llm/.usages/classification.md +25 -0
- prettyplay/llm/.usages/providers.md +33 -0
- prettyplay/llm/CODEMANIFEST +125 -0
- prettyplay/llm/__init__.py +8 -0
- prettyplay/llm/_request.py +216 -0
- prettyplay/llm/anthropic_provider.py +213 -0
- prettyplay/llm/models.py +22 -0
- prettyplay/llm/openai_provider.py +187 -0
- prettyplay/llm/provider.py +115 -0
- prettyplay/reporting/.usages/hooks.md +41 -0
- prettyplay/reporting/CODEMANIFEST +79 -0
- prettyplay/reporting/__init__.py +6 -0
- prettyplay/reporting/hooks.py +37 -0
- prettyplay/reporting/reporter.py +70 -0
- prettyplay/runtime.py +107 -0
- prettyplay/scenario.py +291 -0
- prettyplay-0.0.0.dist-info/METADATA +236 -0
- prettyplay-0.0.0.dist-info/RECORD +58 -0
- prettyplay-0.0.0.dist-info/WHEEL +5 -0
- prettyplay-0.0.0.dist-info/licenses/LICENSE +28 -0
- prettyplay-0.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
Imports:
|
|
2
|
+
- Types:
|
|
3
|
+
- Config
|
|
4
|
+
From: prettyplay/config
|
|
5
|
+
- Types:
|
|
6
|
+
- LlmUnavailableError
|
|
7
|
+
From: prettyplay/failures
|
|
8
|
+
|
|
9
|
+
Usages:
|
|
10
|
+
conventions: .goga/usages/conventions.md
|
|
11
|
+
openai: .goga/usages/cooks/openai.md
|
|
12
|
+
anthropic: .goga/usages/cooks/anthropic.md
|
|
13
|
+
|
|
14
|
+
Annotations: |
|
|
15
|
+
Use `conventions` for code writing rules and testing.
|
|
16
|
+
Use `openai` and `anthropic` for the SDK call patterns and the error mapping of the two providers.
|
|
17
|
+
|
|
18
|
+
Provider parity is absolute: both providers expose the same operations, accept the same inputs, return the same output shapes and map failures to the same taxonomy; the provider choice is a configuration decision, never a capability difference.
|
|
19
|
+
The user instructions input participates in both provider implementations with identical semantics — a parity requirement, not a capability difference.
|
|
20
|
+
API keys come only from environment variables; never log keys or payloads containing secrets.
|
|
21
|
+
One completion request per attempt: attempt budgets are owned by the calling engine, never by a provider.
|
|
22
|
+
Cached step code never depends on the provider: the provider serves generation and classification only.
|
|
23
|
+
|
|
24
|
+
---
|
|
25
|
+
|
|
26
|
+
"LlmProvider()":
|
|
27
|
+
location: provider.py
|
|
28
|
+
annotations: |
|
|
29
|
+
The unified LLM port of the library: code generation for a step and failure classification for healing. One contract, two interchangeable implementations selected by configuration.
|
|
30
|
+
methods:
|
|
31
|
+
"generate_step_code(prompt: str, user_instructions: str, step_text: str, previous_steps: list[str], snapshot: str, screenshot: bytes | None, page_api: str, existing_code: str | None, error: str | None) -> code: str": |
|
|
32
|
+
Generate step code of the fixed form.
|
|
33
|
+
|
|
34
|
+
`prompt`: the system prompt text supplied by the calling engine — applied verbatim as the system message.
|
|
35
|
+
`user_instructions`: the project's code style instructions supplied by the calling engine from the generation_prompt setting; empty — the request carries no instructions block; non-empty — rendered by the provider implementations verbatim as a separate USER INSTRUCTIONS block of the user content, identically in both.
|
|
36
|
+
`step_text`: the sentence of the step to generate.
|
|
37
|
+
`previous_steps`: the sentences of the previous steps of the test, in execution order — scenario context.
|
|
38
|
+
`snapshot`: the accessibility snapshot of the current page.
|
|
39
|
+
`screenshot`: an optional PNG image of the page; passed only when the project enables screenshots.
|
|
40
|
+
`page_api`: the exact page facade surface listing — the list of calls the model may use.
|
|
41
|
+
`existing_code`: the existing step code that failed; non-empty only on regeneration requests.
|
|
42
|
+
`error`: the failure description of the existing code; non-empty only on regeneration requests.
|
|
43
|
+
`code`: the generated step code of the fixed form, working only through the driver facade; the first markdown-fenced block of the answer is unwrapped — an answer with no closed fence returns verbatim.
|
|
44
|
+
|
|
45
|
+
Requirements:
|
|
46
|
+
- A provider service failure (connectivity, timeout, rate limit, authentication) raises `LlmUnavailableError` naming the provider
|
|
47
|
+
- The generated code contains no provider-specific constructs
|
|
48
|
+
"classify_failure(prompt: str, step_text: str, code: str, error: str, snapshot: str, screenshot: bytes | None) -> classification: FailureClassification": |
|
|
49
|
+
Classify a failed cached step.
|
|
50
|
+
|
|
51
|
+
`prompt`: the system prompt text supplied by the calling engine — applied verbatim as the system message.
|
|
52
|
+
`step_text`: the sentence of the failed step.
|
|
53
|
+
`code`: the existing step code that failed.
|
|
54
|
+
`error`: the human-readable description of the failure.
|
|
55
|
+
`snapshot`: the accessibility snapshot of the current page.
|
|
56
|
+
`screenshot`: an optional PNG image of the page; passed only when the project enables screenshots.
|
|
57
|
+
`classification`: the `FailureClassification` verdict.
|
|
58
|
+
|
|
59
|
+
"LlmProvider::OpenAiProvider(config: Config)":
|
|
60
|
+
location: openai_provider.py
|
|
61
|
+
annotations: |
|
|
62
|
+
The openai SDK implementation of `LlmProvider` (see `openai`).
|
|
63
|
+
|
|
64
|
+
`config`: project settings; the generation model is the effective_generation_model of `config`, the classification model is the effective_classification_model of `config`; the base_url setting of `config` overrides the endpoint when set.
|
|
65
|
+
|
|
66
|
+
Algorithm (both operations):
|
|
67
|
+
1. Build the request: the prompt text as the system message, the user content carrying the inputs — a non-empty user_instructions renders as a separate USER INSTRUCTIONS block placed after the page API block, before the regeneration-only blocks (CODE, ERROR)
|
|
68
|
+
2. Send one completion request via the SDK
|
|
69
|
+
3. Extract the text answer; a generation answer unwraps its first markdown-fenced block — an unfenced answer passes through verbatim
|
|
70
|
+
4. An SDK error maps to `LlmUnavailableError` (see `openai`)
|
|
71
|
+
|
|
72
|
+
"LlmProvider::AnthropicProvider(config: Config)":
|
|
73
|
+
location: anthropic_provider.py
|
|
74
|
+
annotations: |
|
|
75
|
+
The anthropic SDK implementation of `LlmProvider` (see `anthropic`); full parity with the openai implementation.
|
|
76
|
+
|
|
77
|
+
`config`: the same settings semantics as the openai implementation.
|
|
78
|
+
|
|
79
|
+
Algorithm (both operations):
|
|
80
|
+
1. Build the request: the prompt text as the system message, the user content carrying the inputs — a non-empty user_instructions renders as a separate USER INSTRUCTIONS block placed after the page API block, before the regeneration-only blocks (CODE, ERROR)
|
|
81
|
+
2. Send one message request via the SDK
|
|
82
|
+
3. Extract the text answer; a generation answer unwraps its first markdown-fenced block — an unfenced answer passes through verbatim
|
|
83
|
+
4. An SDK error maps to `LlmUnavailableError` (see `anthropic`)
|
|
84
|
+
|
|
85
|
+
"create_provider(config: Config) -> provider: LlmProvider":
|
|
86
|
+
location: provider.py
|
|
87
|
+
annotations: |
|
|
88
|
+
Select and construct the LLM provider from configuration.
|
|
89
|
+
|
|
90
|
+
`config`: project settings.
|
|
91
|
+
`provider`: the selected provider implementation.
|
|
92
|
+
|
|
93
|
+
Algorithm:
|
|
94
|
+
1. Read the provider choice from `config`
|
|
95
|
+
2. Construct the matching provider implementation with `config`
|
|
96
|
+
3. Return it
|
|
97
|
+
|
|
98
|
+
Requirements:
|
|
99
|
+
- An unknown provider value fails loudly with an actionable message listing the supported providers
|
|
100
|
+
|
|
101
|
+
"FailureClassification(category: str, explanation: str, recommendation: str)":
|
|
102
|
+
location: models.py
|
|
103
|
+
annotations: |
|
|
104
|
+
The verdict of a failure classification: what kind of failure it is and what to do about it.
|
|
105
|
+
|
|
106
|
+
`category`: one of rot (the UI changed — regeneration is meaningful), product_defect (the expectation legitimately failed), incurable (regeneration cannot help).
|
|
107
|
+
`explanation`: why the failure got this category.
|
|
108
|
+
`recommendation`: the recommended engineer action.
|
|
109
|
+
|
|
110
|
+
Requirements:
|
|
111
|
+
- `category` is always one of the three labels
|
|
112
|
+
properties:
|
|
113
|
+
"category -> str": |
|
|
114
|
+
The classification label: rot, product_defect or incurable.
|
|
115
|
+
"explanation -> str": |
|
|
116
|
+
Why the failure got this category.
|
|
117
|
+
"recommendation -> str": |
|
|
118
|
+
The recommended engineer action.
|
|
119
|
+
|
|
120
|
+
---
|
|
121
|
+
|
|
122
|
+
Author: Goga
|
|
123
|
+
CreatedAt: 07/09/26
|
|
124
|
+
Description: |
|
|
125
|
+
The LLM port of prettyplay: one contract, the openai and anthropic SDK implementations in full parity, and the failure classification verdict.
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
"""Facade of the prettyplay.llm cell: the LLM port and its implementations."""
|
|
2
|
+
|
|
3
|
+
from .anthropic_provider import AnthropicProvider
|
|
4
|
+
from .models import FailureClassification
|
|
5
|
+
from .openai_provider import OpenAiProvider
|
|
6
|
+
from .provider import LlmProvider, create_provider
|
|
7
|
+
|
|
8
|
+
__all__ = ["AnthropicProvider", "FailureClassification", "LlmProvider", "OpenAiProvider", "create_provider"]
|
|
@@ -0,0 +1,216 @@
|
|
|
1
|
+
"""Shared request-field building for the LLM provider implementations."""
|
|
2
|
+
|
|
3
|
+
import base64
|
|
4
|
+
import re
|
|
5
|
+
|
|
6
|
+
from ..failures import LlmUnavailableError
|
|
7
|
+
|
|
8
|
+
#: The labels a classification category may take.
|
|
9
|
+
CATEGORY_ROT = "rot"
|
|
10
|
+
CATEGORY_PRODUCT_DEFECT = "product_defect"
|
|
11
|
+
CATEGORY_INCURABLE = "incurable"
|
|
12
|
+
|
|
13
|
+
#: The frozen set of the three classification labels.
|
|
14
|
+
CATEGORIES = frozenset({CATEGORY_ROT, CATEGORY_PRODUCT_DEFECT, CATEGORY_INCURABLE})
|
|
15
|
+
|
|
16
|
+
#: Field count of the one-line classification verdict.
|
|
17
|
+
VERDICT_FIELD_COUNT = 3
|
|
18
|
+
|
|
19
|
+
#: A fenced completion block: three backticks, an optional language tag, the body, the closing fence.
|
|
20
|
+
_FENCED_BLOCK = re.compile(r"```[a-zA-Z0-9_+-]*[ \t]*\r?\n(.*?)```", re.DOTALL)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def extract_code_block(answer: str) -> str:
|
|
24
|
+
"""Return the step code of a completion, unwrapping the markdown fence.
|
|
25
|
+
|
|
26
|
+
Models answer generation requests with a fenced python block even when told
|
|
27
|
+
to output only code, so the first fenced block of `answer` is the code; an
|
|
28
|
+
answer with no closed fence is passed through verbatim — an unfenced code
|
|
29
|
+
answer stays executable, an unparsable one keeps failing downstream.
|
|
30
|
+
|
|
31
|
+
Args:
|
|
32
|
+
answer: the non-empty completion text of a generation request.
|
|
33
|
+
|
|
34
|
+
Returns:
|
|
35
|
+
The code of the fixed form: the body of the first fenced block, or the
|
|
36
|
+
answer itself when it carries no closed fence.
|
|
37
|
+
"""
|
|
38
|
+
match = _FENCED_BLOCK.search(answer)
|
|
39
|
+
|
|
40
|
+
if match is None:
|
|
41
|
+
return answer
|
|
42
|
+
|
|
43
|
+
return match.group(1)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def require_completion_text(text: str | None, provider: str) -> str:
|
|
47
|
+
"""Return the completion text, refusing an empty answer as a service failure.
|
|
48
|
+
|
|
49
|
+
Args:
|
|
50
|
+
text: the raw text extracted from the provider response; ``None`` or an
|
|
51
|
+
empty string means the service returned no completion body.
|
|
52
|
+
provider: the provider name for the failure message.
|
|
53
|
+
|
|
54
|
+
Returns:
|
|
55
|
+
The non-empty completion text.
|
|
56
|
+
|
|
57
|
+
Raises:
|
|
58
|
+
LlmUnavailableError: the completion body is missing — a null/empty
|
|
59
|
+
content is an infrastructure shape, not a step verdict, so it maps
|
|
60
|
+
to the same taxonomy as any other service failure.
|
|
61
|
+
"""
|
|
62
|
+
if not text:
|
|
63
|
+
raise LlmUnavailableError(f"llm unavailable: {provider} returned empty completion")
|
|
64
|
+
|
|
65
|
+
return text
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def build_fields_text( # noqa: PLR0913, PLR0917 — the parameters mirror the fixed port signature
|
|
69
|
+
user_instructions: str,
|
|
70
|
+
step_text: str,
|
|
71
|
+
previous_steps: list[str],
|
|
72
|
+
snapshot: str,
|
|
73
|
+
page_api: str,
|
|
74
|
+
existing_code: str | None,
|
|
75
|
+
error: str | None,
|
|
76
|
+
) -> str:
|
|
77
|
+
"""Build the plain-text generation request fields shared by both providers.
|
|
78
|
+
|
|
79
|
+
Args:
|
|
80
|
+
user_instructions: the project's code style instructions from the
|
|
81
|
+
generation_prompt setting; empty — the request carries no
|
|
82
|
+
instructions block, non-empty — rendered verbatim as a separate
|
|
83
|
+
USER INSTRUCTIONS block after the page API block.
|
|
84
|
+
step_text: the sentence of the step to generate.
|
|
85
|
+
previous_steps: the sentences of the previous steps of the test, in
|
|
86
|
+
execution order — scenario context.
|
|
87
|
+
snapshot: the accessibility snapshot of the current page.
|
|
88
|
+
page_api: the exact page facade surface listing.
|
|
89
|
+
existing_code: the existing step code that failed; non-empty only on
|
|
90
|
+
regeneration requests.
|
|
91
|
+
error: the failure description of the existing code; non-empty only on
|
|
92
|
+
regeneration requests.
|
|
93
|
+
|
|
94
|
+
Returns:
|
|
95
|
+
The request fields as one text with STEP / PREVIOUS STEPS /
|
|
96
|
+
PAGE SNAPSHOT / PAGE API sections, the optional USER INSTRUCTIONS
|
|
97
|
+
section and, on regeneration requests, CODE / ERROR sections.
|
|
98
|
+
"""
|
|
99
|
+
sections = [
|
|
100
|
+
f"STEP:\n{step_text}",
|
|
101
|
+
_format_previous_steps(previous_steps),
|
|
102
|
+
f"PAGE SNAPSHOT:\n{snapshot}",
|
|
103
|
+
f"PAGE API:\n{page_api}",
|
|
104
|
+
]
|
|
105
|
+
|
|
106
|
+
if user_instructions:
|
|
107
|
+
sections.append(f"USER INSTRUCTIONS:\n{user_instructions}")
|
|
108
|
+
if existing_code is not None:
|
|
109
|
+
sections.append(f"CODE:\n{existing_code}")
|
|
110
|
+
if error is not None:
|
|
111
|
+
sections.append(f"ERROR:\n{error}")
|
|
112
|
+
|
|
113
|
+
return "\n\n".join(sections)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def build_classification_fields(step_text: str, code: str, error: str, snapshot: str) -> str:
|
|
117
|
+
"""Build the plain-text classification request fields shared by both providers.
|
|
118
|
+
|
|
119
|
+
Args:
|
|
120
|
+
step_text: the sentence of the failed step.
|
|
121
|
+
code: the existing step code that failed.
|
|
122
|
+
error: the human-readable description of the failure.
|
|
123
|
+
snapshot: the accessibility snapshot of the current page.
|
|
124
|
+
|
|
125
|
+
Returns:
|
|
126
|
+
The request fields as one text with STEP / CODE / ERROR / PAGE
|
|
127
|
+
SNAPSHOT sections.
|
|
128
|
+
"""
|
|
129
|
+
return "\n\n".join(
|
|
130
|
+
[
|
|
131
|
+
f"STEP:\n{step_text}",
|
|
132
|
+
f"CODE:\n{code}",
|
|
133
|
+
f"ERROR:\n{error}",
|
|
134
|
+
f"PAGE SNAPSHOT:\n{snapshot}",
|
|
135
|
+
]
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def encode_screenshot(screenshot: bytes) -> str:
|
|
140
|
+
"""Encode a PNG screenshot as the base64 payload of a content block.
|
|
141
|
+
|
|
142
|
+
Args:
|
|
143
|
+
screenshot: the raw PNG image bytes of the page.
|
|
144
|
+
|
|
145
|
+
Returns:
|
|
146
|
+
The base64 text of the image.
|
|
147
|
+
"""
|
|
148
|
+
return base64.b64encode(screenshot).decode("ascii")
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def openai_user_content(text: str, screenshot: bytes | None) -> str | list[dict]:
|
|
152
|
+
"""Wrap the request fields as an openai user content payload.
|
|
153
|
+
|
|
154
|
+
Args:
|
|
155
|
+
text: the plain-text request fields shared with the anthropic provider.
|
|
156
|
+
screenshot: an optional PNG image of the page; passed only when the
|
|
157
|
+
project enables screenshots.
|
|
158
|
+
|
|
159
|
+
Returns:
|
|
160
|
+
The plain text when no image is attached, otherwise the openai
|
|
161
|
+
content block list with the data-URI image block.
|
|
162
|
+
"""
|
|
163
|
+
if screenshot is None:
|
|
164
|
+
return text
|
|
165
|
+
|
|
166
|
+
return [
|
|
167
|
+
{"type": "text", "text": text},
|
|
168
|
+
{
|
|
169
|
+
"type": "image_url",
|
|
170
|
+
"image_url": {"url": f"data:image/png;base64,{encode_screenshot(screenshot)}"},
|
|
171
|
+
},
|
|
172
|
+
]
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def unparsable_classification() -> dict[str, str]:
|
|
176
|
+
"""Return the protective verdict fields for an unparsable provider answer.
|
|
177
|
+
|
|
178
|
+
Returns:
|
|
179
|
+
The incurable fallback verdict fields; both providers construct the
|
|
180
|
+
same defaults, so the field set is shared.
|
|
181
|
+
"""
|
|
182
|
+
return {
|
|
183
|
+
"category": CATEGORY_INCURABLE,
|
|
184
|
+
"explanation": "classification verdict unparsable",
|
|
185
|
+
"recommendation": "re-run the step or check the provider answer",
|
|
186
|
+
}
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def parse_classification_line(answer: str) -> tuple[str, str, str] | None:
|
|
190
|
+
"""Parse the one-line classification answer of the form ``category | explanation | recommendation``.
|
|
191
|
+
|
|
192
|
+
Args:
|
|
193
|
+
answer: the raw completion text.
|
|
194
|
+
|
|
195
|
+
Returns:
|
|
196
|
+
The stripped (category, explanation, recommendation) triple, or None
|
|
197
|
+
when the answer is not a line of the expected shape or names no
|
|
198
|
+
known category — the caller applies the protective default.
|
|
199
|
+
"""
|
|
200
|
+
line = next((stripped for stripped in (line.strip() for line in answer.splitlines()) if stripped), "")
|
|
201
|
+
parts = [part.strip() for part in line.split("|")]
|
|
202
|
+
|
|
203
|
+
if len(parts) == VERDICT_FIELD_COUNT and parts[0] in CATEGORIES:
|
|
204
|
+
return parts[0], parts[1], parts[2]
|
|
205
|
+
|
|
206
|
+
return None
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _format_previous_steps(previous_steps: list[str]) -> str:
|
|
210
|
+
"""Render the scenario context section; an empty history stays explicit."""
|
|
211
|
+
if not previous_steps:
|
|
212
|
+
return "PREVIOUS STEPS:\n(none)"
|
|
213
|
+
|
|
214
|
+
listed = "\n".join(f"- {sentence}" for sentence in previous_steps)
|
|
215
|
+
|
|
216
|
+
return f"PREVIOUS STEPS:\n{listed}"
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""The anthropic SDK implementation of the LlmProvider port."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
|
|
5
|
+
from anthropic import Anthropic, AnthropicError
|
|
6
|
+
|
|
7
|
+
from ..config import Config
|
|
8
|
+
from ..failures import LlmUnavailableError
|
|
9
|
+
from ._request import (
|
|
10
|
+
build_classification_fields,
|
|
11
|
+
build_fields_text,
|
|
12
|
+
encode_screenshot,
|
|
13
|
+
extract_code_block,
|
|
14
|
+
parse_classification_line,
|
|
15
|
+
require_completion_text,
|
|
16
|
+
unparsable_classification,
|
|
17
|
+
)
|
|
18
|
+
from .models import FailureClassification
|
|
19
|
+
from .provider import LlmProvider
|
|
20
|
+
|
|
21
|
+
REQUEST_MAX_TOKENS = 1024
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _first_text_block(response: object) -> str | None:
|
|
25
|
+
"""Extract the text of the first text block of an anthropic response.
|
|
26
|
+
|
|
27
|
+
Args:
|
|
28
|
+
response: the SDK response of a ``messages.create`` call.
|
|
29
|
+
|
|
30
|
+
Returns:
|
|
31
|
+
The text of the first text content block, or ``None`` when the
|
|
32
|
+
response carries no text block at all (empty or non-text content).
|
|
33
|
+
"""
|
|
34
|
+
for block in getattr(response, "content", None) or []:
|
|
35
|
+
if getattr(block, "type", None) == "text":
|
|
36
|
+
return getattr(block, "text", None)
|
|
37
|
+
|
|
38
|
+
return None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class AnthropicProvider(LlmProvider):
|
|
42
|
+
"""The LlmProvider implementation served by the anthropic SDK.
|
|
43
|
+
|
|
44
|
+
Full parity with :class:`OpenAiProvider`: the same operations, the same
|
|
45
|
+
inputs, the same output shapes. The constructor reads no environment
|
|
46
|
+
and constructs no client; the SDK client is created lazily on the first
|
|
47
|
+
request, so the library starts without LLM credentials. Every service
|
|
48
|
+
failure maps to :class:`~prettyplay.failures.LlmUnavailableError`
|
|
49
|
+
naming the provider; the API key is read from the environment only and
|
|
50
|
+
never logged.
|
|
51
|
+
"""
|
|
52
|
+
|
|
53
|
+
def __init__(self, config: Config) -> None:
|
|
54
|
+
"""Keep the config; the SDK client stays unconstructed until the first request.
|
|
55
|
+
|
|
56
|
+
Args:
|
|
57
|
+
config: project settings; the effective models and the optional
|
|
58
|
+
base_url endpoint override come from it.
|
|
59
|
+
"""
|
|
60
|
+
self._config = config
|
|
61
|
+
self._client: Anthropic | None = None
|
|
62
|
+
|
|
63
|
+
def _get_client(self) -> Anthropic:
|
|
64
|
+
"""Construct the SDK client on the first request.
|
|
65
|
+
|
|
66
|
+
Returns:
|
|
67
|
+
The lazily constructed anthropic SDK client.
|
|
68
|
+
|
|
69
|
+
Raises:
|
|
70
|
+
LlmUnavailableError: the ANTHROPIC_API_KEY environment variable
|
|
71
|
+
is missing or empty — generation and healing are blocked.
|
|
72
|
+
"""
|
|
73
|
+
if self._client is None:
|
|
74
|
+
api_key = os.environ.get("ANTHROPIC_API_KEY")
|
|
75
|
+
|
|
76
|
+
if not api_key:
|
|
77
|
+
raise LlmUnavailableError("llm unavailable: anthropic: ANTHROPIC_API_KEY is not set")
|
|
78
|
+
|
|
79
|
+
self._client = Anthropic(api_key=api_key, base_url=self._config.base_url or None)
|
|
80
|
+
return self._client
|
|
81
|
+
|
|
82
|
+
def _user_content(self, text: str, screenshot: bytes | None) -> str | list[dict]:
|
|
83
|
+
"""Wrap the request fields as an anthropic user content payload.
|
|
84
|
+
|
|
85
|
+
Args:
|
|
86
|
+
text: the plain-text request fields shared with the openai provider.
|
|
87
|
+
screenshot: an optional PNG image of the page; passed only when
|
|
88
|
+
the project enables screenshots.
|
|
89
|
+
|
|
90
|
+
Returns:
|
|
91
|
+
The plain text when no image is attached, otherwise the
|
|
92
|
+
anthropic content block list with the base64 image block.
|
|
93
|
+
"""
|
|
94
|
+
if screenshot is None:
|
|
95
|
+
return text
|
|
96
|
+
|
|
97
|
+
return [
|
|
98
|
+
{"type": "text", "text": text},
|
|
99
|
+
{
|
|
100
|
+
"type": "image",
|
|
101
|
+
"source": {
|
|
102
|
+
"type": "base64",
|
|
103
|
+
"media_type": "image/png",
|
|
104
|
+
"data": encode_screenshot(screenshot),
|
|
105
|
+
},
|
|
106
|
+
},
|
|
107
|
+
]
|
|
108
|
+
|
|
109
|
+
def generate_step_code( # noqa: PLR0913, PLR0917 — the signature is fixed by the port contract
|
|
110
|
+
self,
|
|
111
|
+
prompt: str,
|
|
112
|
+
user_instructions: str,
|
|
113
|
+
step_text: str,
|
|
114
|
+
previous_steps: list[str],
|
|
115
|
+
snapshot: str,
|
|
116
|
+
screenshot: bytes | None,
|
|
117
|
+
page_api: str,
|
|
118
|
+
existing_code: str | None,
|
|
119
|
+
error: str | None,
|
|
120
|
+
) -> str:
|
|
121
|
+
"""Generate step code of the fixed form working only through the driver facade.
|
|
122
|
+
|
|
123
|
+
Args:
|
|
124
|
+
prompt: the system prompt text supplied by the calling engine;
|
|
125
|
+
applied verbatim as the system parameter.
|
|
126
|
+
user_instructions: the project's code style instructions from the
|
|
127
|
+
generation_prompt setting; empty — the request carries no
|
|
128
|
+
instructions block, non-empty — rendered verbatim as a
|
|
129
|
+
separate USER INSTRUCTIONS block of the user content,
|
|
130
|
+
identically to the openai implementation.
|
|
131
|
+
step_text: the sentence of the step to generate.
|
|
132
|
+
previous_steps: the sentences of the previous steps of the test,
|
|
133
|
+
in execution order — scenario context.
|
|
134
|
+
snapshot: the accessibility snapshot of the current page.
|
|
135
|
+
screenshot: an optional PNG image of the page; passed only when
|
|
136
|
+
the project enables screenshots.
|
|
137
|
+
page_api: the exact page facade surface listing — the calls the
|
|
138
|
+
model may use.
|
|
139
|
+
existing_code: the existing step code that failed; non-empty only
|
|
140
|
+
on regeneration requests.
|
|
141
|
+
error: the failure description of the existing code; non-empty
|
|
142
|
+
only on regeneration requests.
|
|
143
|
+
|
|
144
|
+
Returns:
|
|
145
|
+
The generated step code of the fixed form.
|
|
146
|
+
|
|
147
|
+
Raises:
|
|
148
|
+
LlmUnavailableError: the SDK client is unavailable or the
|
|
149
|
+
service request failed.
|
|
150
|
+
"""
|
|
151
|
+
text = build_fields_text(user_instructions, step_text, previous_steps, snapshot, page_api, existing_code, error)
|
|
152
|
+
|
|
153
|
+
try:
|
|
154
|
+
response = self._get_client().messages.create(
|
|
155
|
+
model=self._config.effective_generation_model,
|
|
156
|
+
system=prompt,
|
|
157
|
+
max_tokens=REQUEST_MAX_TOKENS,
|
|
158
|
+
messages=[{"role": "user", "content": self._user_content(text, screenshot)}],
|
|
159
|
+
)
|
|
160
|
+
except AnthropicError as sdk_error:
|
|
161
|
+
raise LlmUnavailableError("llm unavailable: anthropic request failed") from sdk_error
|
|
162
|
+
|
|
163
|
+
return extract_code_block(require_completion_text(_first_text_block(response), "anthropic"))
|
|
164
|
+
|
|
165
|
+
def classify_failure( # noqa: PLR0913, PLR0917 — the signature is fixed by the port contract
|
|
166
|
+
self,
|
|
167
|
+
prompt: str,
|
|
168
|
+
step_text: str,
|
|
169
|
+
code: str,
|
|
170
|
+
error: str,
|
|
171
|
+
snapshot: str,
|
|
172
|
+
screenshot: bytes | None,
|
|
173
|
+
) -> FailureClassification:
|
|
174
|
+
"""Classify a failed cached step.
|
|
175
|
+
|
|
176
|
+
Args:
|
|
177
|
+
prompt: the system prompt text supplied by the calling engine;
|
|
178
|
+
applied verbatim as the system parameter.
|
|
179
|
+
step_text: the sentence of the failed step.
|
|
180
|
+
code: the existing step code that failed.
|
|
181
|
+
error: the human-readable description of the failure.
|
|
182
|
+
snapshot: the accessibility snapshot of the current page.
|
|
183
|
+
screenshot: an optional PNG image of the page; passed only when
|
|
184
|
+
the project enables screenshots.
|
|
185
|
+
|
|
186
|
+
Returns:
|
|
187
|
+
The classification verdict; an unparsable or unknown answer maps
|
|
188
|
+
protectively to the incurable category.
|
|
189
|
+
|
|
190
|
+
Raises:
|
|
191
|
+
LlmUnavailableError: the SDK client is unavailable or the
|
|
192
|
+
service request failed.
|
|
193
|
+
"""
|
|
194
|
+
text = build_classification_fields(step_text, code, error, snapshot)
|
|
195
|
+
|
|
196
|
+
try:
|
|
197
|
+
response = self._get_client().messages.create(
|
|
198
|
+
model=self._config.effective_classification_model,
|
|
199
|
+
system=prompt,
|
|
200
|
+
max_tokens=REQUEST_MAX_TOKENS,
|
|
201
|
+
messages=[{"role": "user", "content": self._user_content(text, screenshot)}],
|
|
202
|
+
)
|
|
203
|
+
except AnthropicError as sdk_error:
|
|
204
|
+
raise LlmUnavailableError("llm unavailable: anthropic request failed") from sdk_error
|
|
205
|
+
|
|
206
|
+
answer = require_completion_text(_first_text_block(response), "anthropic")
|
|
207
|
+
parsed = parse_classification_line(answer)
|
|
208
|
+
|
|
209
|
+
if parsed is None:
|
|
210
|
+
return FailureClassification(**unparsable_classification())
|
|
211
|
+
|
|
212
|
+
category, explanation, recommendation = parsed
|
|
213
|
+
return FailureClassification(category=category, explanation=explanation, recommendation=recommendation)
|
prettyplay/llm/models.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""Models of the prettyplay.llm cell: the failure classification verdict."""
|
|
2
|
+
|
|
3
|
+
from pydantic import BaseModel, ConfigDict
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
class FailureClassification(BaseModel):
|
|
7
|
+
"""The verdict of a failure classification.
|
|
8
|
+
|
|
9
|
+
Says what kind of failure a failed cached step hit and what the engineer
|
|
10
|
+
should do about it.
|
|
11
|
+
|
|
12
|
+
Attributes:
|
|
13
|
+
category: the classification label: rot, product_defect or incurable.
|
|
14
|
+
explanation: why the failure got this category.
|
|
15
|
+
recommendation: the recommended engineer action.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
model_config = ConfigDict(kw_only=True)
|
|
19
|
+
|
|
20
|
+
category: str
|
|
21
|
+
explanation: str
|
|
22
|
+
recommendation: str
|