foundry-testing-actor 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- foundry_testing_actor/__init__.py +56 -0
- foundry_testing_actor/cards/actor-data.yaml +68 -0
- foundry_testing_actor/cards/actor-message.yaml +32 -0
- foundry_testing_actor/cards/actor-synchronous-messaging.yaml +59 -0
- foundry_testing_actor/cards/actor.yaml +19 -0
- foundry_testing_actor/cli.py +181 -0
- foundry_testing_actor/config.py +525 -0
- foundry_testing_actor/conformance.py +133 -0
- foundry_testing_actor/correlation.py +193 -0
- foundry_testing_actor/engine.py +889 -0
- foundry_testing_actor/grounding.py +206 -0
- foundry_testing_actor/handler.py +248 -0
- foundry_testing_actor/instance.py +129 -0
- foundry_testing_actor/runner/Dockerfile +32 -0
- foundry_testing_actor/schemas/agentic-context.schema.yaml +168 -0
- foundry_testing_actor/serve.py +129 -0
- foundry_testing_actor-0.1.0.dist-info/METADATA +258 -0
- foundry_testing_actor-0.1.0.dist-info/RECORD +20 -0
- foundry_testing_actor-0.1.0.dist-info/WHEEL +4 -0
- foundry_testing_actor-0.1.0.dist-info/entry_points.txt +2 -0
|
@@ -0,0 +1,889 @@
|
|
|
1
|
+
"""`ClaudeCodeTesterEngine` — judgement by a headless Claude Code session against fresh clones.
|
|
2
|
+
|
|
3
|
+
TESTING AUTHORS A HARNESS; IT DOES NOT EXECUTE IT. No session this engine starts brings up docker
|
|
4
|
+
or docker compose, hits a live endpoint, or renders a pass/fail verdict. At `test-task` it writes
|
|
5
|
+
and extends black-box pytest source; at `propose-acceptance` it writes nothing at all. Standing up
|
|
6
|
+
infrastructure and rendering the verdict belongs to the orchestrating actor alone, against a real
|
|
7
|
+
ephemeral deployment, running the test image `handler.py` publishes.
|
|
8
|
+
|
|
9
|
+
WHY A CUSTOM ENGINE, NOT `papeete_actor_synchronous_messaging.engine.resolve("claude")`. The
|
|
10
|
+
built-in `ClaudeEngine` shells out to the raw `anthropic` Python SDK — `anthropic.Anthropic()`,
|
|
11
|
+
resolving `ANTHROPIC_API_KEY`, then `ANTHROPIC_AUTH_TOKEN`, then an `ant auth login` profile.
|
|
12
|
+
Every one of those is a metered API credential. Subscription OAuth tokens (`claude setup-token`
|
|
13
|
+
→ `CLAUDE_CODE_OAUTH_TOKEN`) are scoped to the `claude` CLI / claude.ai specifically and are NOT
|
|
14
|
+
among the credentials the raw SDK will resolve. So the only way to spend a subscription's own
|
|
15
|
+
credit rather than a separate metered key is to shell out to the `claude` CLI itself, which is
|
|
16
|
+
what this file does. It still satisfies the `Engine` port (`name` + `judge(system, prompt,
|
|
17
|
+
schema=None) -> dict`) — the port makes no promise about how long or heavy a judgement is.
|
|
18
|
+
|
|
19
|
+
**Do not set `ANTHROPIC_API_KEY` (or `ANTHROPIC_AUTH_TOKEN`) in a container running this engine.**
|
|
20
|
+
In `claude -p` non-interactive mode an API key present in the environment is ALWAYS preferred over
|
|
21
|
+
`CLAUDE_CODE_OAUTH_TOKEN`, silently routing every session through metered billing instead. There is
|
|
22
|
+
no warning and no visible difference in the transcript; the only symptom is the bill.
|
|
23
|
+
|
|
24
|
+
PAYLOAD-DRIVEN, NOT CARD-DRIVEN. The caller supplies everything a session needs to work — this
|
|
25
|
+
engine never clones a repo merely to look up what a task IS. A missing required field is refused
|
|
26
|
+
by the framework's own schema gate before any engine time is spent.
|
|
27
|
+
|
|
28
|
+
TWO CLONES, AND WHAT EACH DOOR MAY SEE OF THE SECOND. The testing repo is this actor's own, and
|
|
29
|
+
every session runs inside a private clone of it. The implementation repo is read, never written,
|
|
30
|
+
and the two doors read it for opposite reasons:
|
|
31
|
+
|
|
32
|
+
* `propose-acceptance` clones its DEFAULT BRANCH and shows it to the session — the capability as
|
|
33
|
+
it stands BEFORE the increment. It never checks out `impl/<task_id>`: a tester whose
|
|
34
|
+
assertions are derived from what was built can only ever confirm the build (ADR-FIA-0004's
|
|
35
|
+
"before or after", ADR-FTA-0002).
|
|
36
|
+
* `test-task` checks out `impl/<task_id>` and does NOT show it to the session. It is there so
|
|
37
|
+
`papeete_version.compute` can recompute, over identical inputs, the image refs the
|
|
38
|
+
implementation actor published — no tag is ever handed to this actor, and none has to be.
|
|
39
|
+
|
|
40
|
+
NO KNOWLEDGE TOOL IS NAMED HERE. The hand-written actor this replaces shelled out to two knowledge
|
|
41
|
+
tools by name, with the registry they resolved through as a module constant. Both are `ground_in`
|
|
42
|
+
entries in a use's sidecar now, run by `grounding.py` like any other argv (ADR-FTA-0001).
|
|
43
|
+
|
|
44
|
+
CARRIES NO CAPABILITY LITERAL. Every identifier it needs comes from `CapabilityConfig`, which
|
|
45
|
+
derives all of them from one capability id and one repo (see `config.py`).
|
|
46
|
+
|
|
47
|
+
WHAT THIS ENGINE DOES NOT DO. It never commits, pushes, builds, or opens a pull request — that is
|
|
48
|
+
`handler.py`'s job, and `handler.py`'s own containment check is the actual enforcement of the
|
|
49
|
+
write boundary this engine's prompt only *asks* the session to respect. On success at `test-task`
|
|
50
|
+
it deliberately does NOT clean up its clones: the handler still needs the testing clone to commit
|
|
51
|
+
and push, and is the one that removes both, on every path including containment failure.
|
|
52
|
+
"""
|
|
53
|
+
from __future__ import annotations
|
|
54
|
+
|
|
55
|
+
import json
|
|
56
|
+
import logging
|
|
57
|
+
import os
|
|
58
|
+
import re
|
|
59
|
+
import shutil
|
|
60
|
+
import subprocess
|
|
61
|
+
import tempfile
|
|
62
|
+
import threading
|
|
63
|
+
from pathlib import Path
|
|
64
|
+
|
|
65
|
+
from papeete_actor_synchronous_messaging.engine import EngineError
|
|
66
|
+
from papeete_version.version import compute as compute_version
|
|
67
|
+
from papeete_version.version import normalize_name
|
|
68
|
+
|
|
69
|
+
from . import correlation, grounding
|
|
70
|
+
from .config import CapabilityConfig, Component
|
|
71
|
+
|
|
72
|
+
DEFAULT_CLONE_TIMEOUT_S = 120
|
|
73
|
+
DEFAULT_SESSION_TIMEOUT_S = 1800
|
|
74
|
+
DEFAULT_MAX_TURNS = 60
|
|
75
|
+
|
|
76
|
+
# The propose door reads and answers; it never writes. Its budget is smaller than the test door's
|
|
77
|
+
# on every axis, and its tool list is the enforcement — a door that CANNOT write beats one asked
|
|
78
|
+
# not to. The same numbers as the implementation actor's assess door, deliberately: the two halves
|
|
79
|
+
# of one round should cost the same order of money (ADR-FIA-0004, ADR-FTA-0002).
|
|
80
|
+
DEFAULT_PROPOSE_TIMEOUT_S = 600
|
|
81
|
+
DEFAULT_PROPOSE_MAX_TURNS = 15
|
|
82
|
+
|
|
83
|
+
TEST_TOOLS = "Bash,Read,Edit,Write,Glob,Grep"
|
|
84
|
+
PROPOSE_TOOLS = "Read,Glob,Grep"
|
|
85
|
+
|
|
86
|
+
# The door ids this engine answers. They are the card's, not this module's invention — `judge()`
|
|
87
|
+
# is handed the id it must dispatch on (see `_door_from_prompt`).
|
|
88
|
+
TEST_DOOR = "test-task"
|
|
89
|
+
PROPOSE_DOOR = "propose-acceptance"
|
|
90
|
+
|
|
91
|
+
# The branch conventions both peers already follow, so nothing has to be looked up to know them.
|
|
92
|
+
IMPLEMENTATION_BRANCH = "impl/{task_id}"
|
|
93
|
+
TEST_BRANCH = "test/{task_id}"
|
|
94
|
+
|
|
95
|
+
# The name the implementation repo's default-branch clone is shown to the propose session under.
|
|
96
|
+
# A directory ADDED to the session beside its working directory, not a path inside the testing
|
|
97
|
+
# clone: a nested repository there would be one `git add` away from the tests this actor commits.
|
|
98
|
+
_CODE_CLONE_LABEL = "the implementation repository, default branch"
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
# ── stream-json projection: keeping every emitted log line inside Loki's max_line_size ────────
|
|
102
|
+
#
|
|
103
|
+
# WHY PROJECT RATHER THAN LOG THE RAW EVENT. `--output-format stream-json` emits one JSON object
|
|
104
|
+
# per turn — the inner conversation this actor's audit trail is made of. Two of its fields are
|
|
105
|
+
# unbounded: `tool_use.input` (for `Write`, the whole file body; for `Edit`, both sides of the
|
|
106
|
+
# replacement) and `tool_result.content` (for `Read`/`Grep`/`Bash`, the whole output). Nothing
|
|
107
|
+
# else is: a `text` block is a few hundred bytes. So clipping those two, and only those, keeps
|
|
108
|
+
# the entire conversation readable while bounding every line — and what gets clipped is
|
|
109
|
+
# recoverable from the branch `handler.py` pushes anyway.
|
|
110
|
+
#
|
|
111
|
+
# WHY IT MATTERS. Loki's `max_line_size` is 256KB with `max_line_size_truncate: false` — an
|
|
112
|
+
# oversized line is REJECTED OUTRIGHT, not trimmed, so a single large `Read` would silently
|
|
113
|
+
# delete exactly the turn worth reading while leaving the rest of the session intact. The budget
|
|
114
|
+
# below is a quarter of that, which also sidesteps Loki parsing `KB` as 1000 rather than 1024.
|
|
115
|
+
#
|
|
116
|
+
# AND WHY NOT RECORD ATTRIBUTES. The OTel `LoggingHandler` maps the formatted message to the OTLP
|
|
117
|
+
# body (the line Loki measures) and record attributes to log-record attributes, which land in Loki
|
|
118
|
+
# structured metadata under their own separate caps (`max_structured_metadata_size: 64KB`,
|
|
119
|
+
# `max_structured_metadata_entries_count: 128`). Payload cannot be smuggled out of the line limit
|
|
120
|
+
# by moving it there — that channel is for correlation ids, which `correlation.py` stamps onto
|
|
121
|
+
# every record this process emits without any call site here passing them.
|
|
122
|
+
|
|
123
|
+
LINE_BUDGET = 64 * 1024 # bytes per emitted log line
|
|
124
|
+
BLOB_HEAD = 2000 # bytes kept from the front of an unbounded value
|
|
125
|
+
BLOB_TAIL = 2000 # ...and from the back: a Bash failure lives in the tail, not the head
|
|
126
|
+
PROSE = 8000 # text/thinking/result — prose IS the audit trail, give it more room
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _clip(value, head: int = BLOB_HEAD, tail: int = BLOB_TAIL) -> str:
|
|
130
|
+
"""Head+tail slice of a value, budgeted in BYTES (not characters — the limit Loki enforces
|
|
131
|
+
is on the UTF-8 encoded line), with the true size recorded in the marker."""
|
|
132
|
+
if not isinstance(value, str):
|
|
133
|
+
value = json.dumps(value, default=str)
|
|
134
|
+
raw = value.encode("utf-8", "replace")
|
|
135
|
+
if len(raw) <= head + tail:
|
|
136
|
+
return value
|
|
137
|
+
return (raw[:head].decode("utf-8", "replace")
|
|
138
|
+
+ f"\n…[clipped {len(raw) - head - tail} of {len(raw)} bytes]…\n"
|
|
139
|
+
+ raw[-tail:].decode("utf-8", "replace"))
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _project(event: dict) -> dict | None:
|
|
143
|
+
"""One stream-json event -> a compact dict, or None to drop it entirely."""
|
|
144
|
+
kind = event.get("type")
|
|
145
|
+
|
|
146
|
+
if kind == "system" and event.get("subtype") == "init":
|
|
147
|
+
# The raw init event is ~2.2KB of tools/skills/slash_commands inventory. Four fields of
|
|
148
|
+
# it are worth keeping — `session_id` is the join key to the CLI's own full transcript,
|
|
149
|
+
# which it writes to $HOME/.claude/projects/<slug>/<session_id>.jsonl regardless of
|
|
150
|
+
# --output-format.
|
|
151
|
+
return {"event": "init", "session_id": event.get("session_id"),
|
|
152
|
+
"model": event.get("model"), "cwd": event.get("cwd")}
|
|
153
|
+
|
|
154
|
+
if kind == "result":
|
|
155
|
+
usage = event.get("usage") or {}
|
|
156
|
+
return {"event": "result", "session_id": event.get("session_id"),
|
|
157
|
+
"subtype": event.get("subtype"), "is_error": event.get("is_error"),
|
|
158
|
+
"num_turns": event.get("num_turns"), "duration_ms": event.get("duration_ms"),
|
|
159
|
+
"cost_usd": event.get("total_cost_usd"),
|
|
160
|
+
"input_tokens": usage.get("input_tokens"),
|
|
161
|
+
"output_tokens": usage.get("output_tokens"),
|
|
162
|
+
"result": _clip(event.get("result", ""), PROSE, PROSE)}
|
|
163
|
+
|
|
164
|
+
if kind not in ("assistant", "user"):
|
|
165
|
+
return None # rate_limit_event and friends carry no audit value
|
|
166
|
+
|
|
167
|
+
blocks = []
|
|
168
|
+
for block in (event.get("message") or {}).get("content", []):
|
|
169
|
+
if not isinstance(block, dict):
|
|
170
|
+
continue
|
|
171
|
+
btype = block.get("type")
|
|
172
|
+
if btype in ("text", "thinking"):
|
|
173
|
+
blocks.append({btype: _clip(block.get(btype, ""), PROSE, PROSE)})
|
|
174
|
+
elif btype == "tool_use":
|
|
175
|
+
blocks.append({"tool_use": block.get("name"), "id": block.get("id"),
|
|
176
|
+
"input": {k: _clip(v) for k, v in (block.get("input") or {}).items()}})
|
|
177
|
+
elif btype == "tool_result":
|
|
178
|
+
# `content` is a str for a text result and a list of blocks otherwise — `_clip`
|
|
179
|
+
# json-dumps the latter rather than this guessing at its shape.
|
|
180
|
+
blocks.append({"tool_result": block.get("tool_use_id"),
|
|
181
|
+
"is_error": block.get("is_error", False),
|
|
182
|
+
"content": _clip(block.get("content", ""))})
|
|
183
|
+
return {"event": kind, "blocks": blocks} if blocks else None
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _line(record: dict) -> str:
|
|
187
|
+
"""Serialize, then hard-enforce the budget.
|
|
188
|
+
|
|
189
|
+
The per-field clipping in `_project` is what keeps lines small; this is what makes "no line
|
|
190
|
+
exceeds the budget" a property of the code rather than a hope — a `tool_use` carrying a
|
|
191
|
+
hundred just-under-threshold keys would otherwise slip through the sum."""
|
|
192
|
+
line = json.dumps(record, default=str, ensure_ascii=False)
|
|
193
|
+
if len(line.encode("utf-8")) > LINE_BUDGET:
|
|
194
|
+
half = LINE_BUDGET // 2 - 200
|
|
195
|
+
line = json.dumps({"event": record.get("event"), "over_budget": True,
|
|
196
|
+
"clipped": _clip(line, half, half)}, ensure_ascii=False)
|
|
197
|
+
return line
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
# ```json … ``` or a bare ``` … ``` block. Non-greedy, DOTALL: one match per fence, in order.
|
|
201
|
+
_FENCE = re.compile(r"```(?:json)?\s*\n(.*?)```", re.DOTALL)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _redact(text: str, secret: str | None) -> str:
|
|
205
|
+
return text.replace(secret, "***") if secret else text
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _payload_from_prompt(prompt: str) -> dict:
|
|
209
|
+
"""Recover the full payload dict from `Actor.judge()`'s own fixed prompt format
|
|
210
|
+
(`papeete_actor_synchronous_messaging.actor.Actor.judge`):
|
|
211
|
+
|
|
212
|
+
f"verb: {verb}\\ndoor: {offer.id}\\n{as_prompt_json(situation)}"
|
|
213
|
+
|
|
214
|
+
where `situation["payload"]` is exactly what the caller sent at this door.
|
|
215
|
+
"""
|
|
216
|
+
lines = prompt.split("\n", 2)
|
|
217
|
+
if len(lines) < 3:
|
|
218
|
+
raise EngineError(f"prompt does not follow Actor.judge()'s fixed format: {prompt!r}")
|
|
219
|
+
try:
|
|
220
|
+
situation = json.loads(lines[2])
|
|
221
|
+
return situation["payload"]
|
|
222
|
+
except (json.JSONDecodeError, KeyError, TypeError) as e:
|
|
223
|
+
raise EngineError(f"could not recover payload from prompt: {e}") from e
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _door_from_prompt(prompt: str) -> str:
|
|
227
|
+
"""Recover the door id from `Actor.judge()`'s own fixed prompt format — its second line.
|
|
228
|
+
|
|
229
|
+
ONE ENGINE, TWO DOORS. The `Engine` port is a single `judge()`, and both this actor's doors
|
|
230
|
+
name the same engine key, so an instance registered under `claude-code` is asked to judge
|
|
231
|
+
both. `Actor.judge()` already puts the door id on line 2 of the prompt it builds, so the
|
|
232
|
+
dispatch needs nothing from the framework that is not already being handed over.
|
|
233
|
+
"""
|
|
234
|
+
lines = prompt.split("\n", 2)
|
|
235
|
+
if len(lines) < 2 or not lines[1].startswith("door: "):
|
|
236
|
+
raise EngineError(f"prompt does not follow Actor.judge()'s fixed format: {prompt!r}")
|
|
237
|
+
return lines[1][len("door: "):].strip()
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _extract_json(text: str) -> dict:
|
|
241
|
+
"""The single JSON object a judged answer ends with.
|
|
242
|
+
|
|
243
|
+
A session's final `result` is prose that HAPPENS to contain the answer, not the answer — it
|
|
244
|
+
reliably wraps it in a fence and unreliably says something either side of it. So: try every
|
|
245
|
+
fenced block, last first (the last one is the conclusion; an earlier one is usually the
|
|
246
|
+
session quoting what it was asked for), then the whole text for the case where it complied
|
|
247
|
+
exactly. Anything else is an `EngineError` — a door that cannot say what it decided has not
|
|
248
|
+
decided anything, and guessing on its behalf would be worse than failing.
|
|
249
|
+
"""
|
|
250
|
+
for block in reversed(_FENCE.findall(text)):
|
|
251
|
+
try:
|
|
252
|
+
parsed = json.loads(block)
|
|
253
|
+
except json.JSONDecodeError:
|
|
254
|
+
continue
|
|
255
|
+
if isinstance(parsed, dict):
|
|
256
|
+
return parsed
|
|
257
|
+
try:
|
|
258
|
+
parsed = json.loads(text.strip())
|
|
259
|
+
except json.JSONDecodeError:
|
|
260
|
+
parsed = None
|
|
261
|
+
if isinstance(parsed, dict):
|
|
262
|
+
return parsed
|
|
263
|
+
raise EngineError(
|
|
264
|
+
"the session produced no JSON object to read its judgement from; its answer ended: "
|
|
265
|
+
f"{text[-2000:]!r}"
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _as_list(value) -> list:
|
|
270
|
+
"""A session's idea of "a list of questions", as a list.
|
|
271
|
+
|
|
272
|
+
Absent or null is none; a lone string or object is one. Coerced rather than trusted: a session
|
|
273
|
+
reliably means the list and unreliably types it, and `Actor.receive()` would refuse the whole
|
|
274
|
+
proposal over one question that arrived bare.
|
|
275
|
+
"""
|
|
276
|
+
if value is None:
|
|
277
|
+
return []
|
|
278
|
+
if isinstance(value, list):
|
|
279
|
+
return value
|
|
280
|
+
return [value]
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _proposal(judged: dict) -> dict:
|
|
284
|
+
"""The reply `acceptance-proposed-result` can carry, and nothing else.
|
|
285
|
+
|
|
286
|
+
PROJECTED, NOT PASSED THROUGH. A door naming ONE completion message gets an open schema from
|
|
287
|
+
the framework (ADR-PAS-0009), so a session's helpful `"notes"` key beside its answer would not
|
|
288
|
+
be refused — it would travel to the orchestrating actor looking like part of a contract three
|
|
289
|
+
packages share, which is worse. And the day this door names a second outcome, the framework
|
|
290
|
+
closes the schema and that same key refuses the whole proposal. Only the two fields the message
|
|
291
|
+
references survive; everything else a session said is in the transcript.
|
|
292
|
+
|
|
293
|
+
WHAT IS CHECKED, AND WHY IT IS NOT MORE. `expectations` must be a list of objects, each with an
|
|
294
|
+
`id` and a `statement`, a `handle` key, and ids unique within the proposal — because a later
|
|
295
|
+
verdict names an expectation by its id, and the orchestrating actor attaches the implementer's
|
|
296
|
+
commitments by it too. Two expectations sharing one id cannot both be named. What a handle
|
|
297
|
+
CONTAINS is not checked: its shape depends on what is being addressed, and the card types the
|
|
298
|
+
surface as a list precisely so that stays prose.
|
|
299
|
+
"""
|
|
300
|
+
if "expectations" not in judged:
|
|
301
|
+
raise EngineError(
|
|
302
|
+
"the proposal named no `expectations` — the one thing the caller has to be able to "
|
|
303
|
+
f"send on; it answered: {judged!r}"
|
|
304
|
+
)
|
|
305
|
+
expectations = judged["expectations"]
|
|
306
|
+
if not isinstance(expectations, list):
|
|
307
|
+
raise EngineError(f"`expectations` is not a list: {expectations!r}")
|
|
308
|
+
seen: set[str] = set()
|
|
309
|
+
for index, expectation in enumerate(expectations):
|
|
310
|
+
if not isinstance(expectation, dict):
|
|
311
|
+
raise EngineError(f"expectations[{index}] is not an object: {expectation!r}")
|
|
312
|
+
missing = [key for key in ("id", "statement") if not expectation.get(key)]
|
|
313
|
+
if missing or "handle" not in expectation:
|
|
314
|
+
raise EngineError(
|
|
315
|
+
f"expectations[{index}] carries no {', '.join(missing or ['handle'])} — every "
|
|
316
|
+
f"expectation is an id, a statement and a handle: {expectation!r}"
|
|
317
|
+
)
|
|
318
|
+
identifier = str(expectation["id"])
|
|
319
|
+
if identifier in seen:
|
|
320
|
+
raise EngineError(
|
|
321
|
+
f"expectation id '{identifier}' is proposed twice — a verdict names an "
|
|
322
|
+
f"expectation by its id, so an id two expectations share names neither"
|
|
323
|
+
)
|
|
324
|
+
seen.add(identifier)
|
|
325
|
+
return {"expectations": expectations, "open_questions": _as_list(judged.get("open_questions"))}
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
class ClaudeCodeTesterEngine:
|
|
329
|
+
"""Judgement by shelling out to the `claude` CLI, against fresh private clones."""
|
|
330
|
+
|
|
331
|
+
def __init__(self, config: CapabilityConfig, *,
|
|
332
|
+
github_token: str | None = None, claude_bin: str = "claude",
|
|
333
|
+
clone_timeout: int = DEFAULT_CLONE_TIMEOUT_S,
|
|
334
|
+
fetch_timeout: int = grounding.DEFAULT_FETCH_TIMEOUT_S,
|
|
335
|
+
session_timeout: int = DEFAULT_SESSION_TIMEOUT_S,
|
|
336
|
+
max_turns: int = DEFAULT_MAX_TURNS,
|
|
337
|
+
propose_timeout: int = DEFAULT_PROPOSE_TIMEOUT_S,
|
|
338
|
+
propose_max_turns: int = DEFAULT_PROPOSE_MAX_TURNS):
|
|
339
|
+
self.config = config
|
|
340
|
+
# The engine's `name` is what the card's doors name, so it comes from the sidecar rather
|
|
341
|
+
# than being fixed here — the port requires the attribute, not a particular value.
|
|
342
|
+
self.name = config.engine
|
|
343
|
+
|
|
344
|
+
self.github_token = github_token or os.environ.get("GITHUB_TOKEN")
|
|
345
|
+
if not self.github_token:
|
|
346
|
+
raise RuntimeError(
|
|
347
|
+
"ClaudeCodeTesterEngine needs GITHUB_TOKEN in the environment — a fine-grained "
|
|
348
|
+
f"PAT with contents:write on {config.source_repo}, read-only contents on "
|
|
349
|
+
f"{config.implementation_repo}, plus read-only contents on "
|
|
350
|
+
f"{config.registry_repo} and whatever else this capability's `ground_in` fetches "
|
|
351
|
+
"resolve through."
|
|
352
|
+
)
|
|
353
|
+
self.claude_bin = claude_bin
|
|
354
|
+
self.clone_timeout = clone_timeout
|
|
355
|
+
self.fetch_timeout = fetch_timeout
|
|
356
|
+
self.session_timeout = session_timeout
|
|
357
|
+
self.max_turns = max_turns
|
|
358
|
+
# Constructor kwargs, not sidecar fields: how long this actor's own doors may think is
|
|
359
|
+
# operational tuning, not something a capability declares about itself.
|
|
360
|
+
self.propose_timeout = propose_timeout
|
|
361
|
+
self.propose_max_turns = propose_max_turns
|
|
362
|
+
self._configure_git_credentials()
|
|
363
|
+
|
|
364
|
+
# ── the one Engine method ──────────────────────────────────────────────────────────────
|
|
365
|
+
|
|
366
|
+
def judge(self, *, system: str, prompt: str, schema: dict | None = None) -> dict:
|
|
367
|
+
"""The one `Engine` method, serving both of this actor's doors.
|
|
368
|
+
|
|
369
|
+
The port is a single `judge()`, and both doors name the same engine key, so the dispatch
|
|
370
|
+
is on the door id `Actor.judge()` already puts on line 2 of the prompt. The two paths are
|
|
371
|
+
deliberately asymmetric: `test-task` hands its clones off live to the handler, which
|
|
372
|
+
commits, pushes, publishes and then removes them; `propose-acceptance` owns its clones
|
|
373
|
+
from end to end and writes nothing anywhere.
|
|
374
|
+
"""
|
|
375
|
+
door = _door_from_prompt(prompt)
|
|
376
|
+
if door == PROPOSE_DOOR:
|
|
377
|
+
return self._propose(_payload_from_prompt(prompt), schema)
|
|
378
|
+
if door == TEST_DOOR:
|
|
379
|
+
return self._test(system, _payload_from_prompt(prompt))
|
|
380
|
+
raise EngineError(
|
|
381
|
+
f"this engine answers {TEST_DOOR} and {PROPOSE_DOOR}, not '{door}' — a door naming "
|
|
382
|
+
f"this engine must be one it knows how to judge"
|
|
383
|
+
)
|
|
384
|
+
|
|
385
|
+
# ── test-task: the door that authors ───────────────────────────────────────────────────
|
|
386
|
+
|
|
387
|
+
def _test(self, system: str, payload: dict) -> dict:
|
|
388
|
+
task_id = payload["task_id"]
|
|
389
|
+
# The FIRST thing this door does, before anything that could fail: from here to the end
|
|
390
|
+
# of this request's own thread, every record — this module's, handler.py's, and the HTTP
|
|
391
|
+
# binding's own access line — carries both ids. `correlation_id()` reads the trace id the
|
|
392
|
+
# orchestrating actor already propagated, so every actor in the pipeline agrees on it
|
|
393
|
+
# without any of them passing it (see correlation.py).
|
|
394
|
+
correlation.bind(correlation_id=correlation.correlation_id(), task_id=task_id)
|
|
395
|
+
|
|
396
|
+
# Both resolved before a single clone: an undeclared component and an unset registry are
|
|
397
|
+
# facts about the request and the environment, and neither needs a network to find out.
|
|
398
|
+
components = self._components(payload["components"])
|
|
399
|
+
registry = _image_registry()
|
|
400
|
+
|
|
401
|
+
test_clone = Path(tempfile.mkdtemp(prefix=self.config.clone_prefix(task_id)))
|
|
402
|
+
code_clone = Path(tempfile.mkdtemp(prefix=self.config.code_clone_prefix(task_id)))
|
|
403
|
+
branch = TEST_BRANCH.format(task_id=task_id)
|
|
404
|
+
try:
|
|
405
|
+
with correlation.stage("clone-tests", repo=self.config.source_repo):
|
|
406
|
+
self._clone(test_clone, self.config.source_repo)
|
|
407
|
+
with correlation.stage("clone-code", repo=self.config.implementation_repo):
|
|
408
|
+
self._clone(code_clone, self.config.implementation_repo)
|
|
409
|
+
# The implementation actor committed, and computed its published tags, on
|
|
410
|
+
# impl/<task_id> — never on the default branch, because no pull request is opened
|
|
411
|
+
# until orchestration has a passing verdict. Checked out HERE ONLY, to recompute those
|
|
412
|
+
# tags; this clone is never shown to the session below.
|
|
413
|
+
self._git(code_clone, ["checkout", IMPLEMENTATION_BRANCH.format(task_id=task_id)])
|
|
414
|
+
self._git(test_clone, ["checkout", "-b", branch])
|
|
415
|
+
|
|
416
|
+
images = [self._resolve_image(code_clone, component.name, task_id, registry)
|
|
417
|
+
for component in components]
|
|
418
|
+
correlation.event("images-under-test", components=[c.name for c in components],
|
|
419
|
+
images=images)
|
|
420
|
+
|
|
421
|
+
self._ground(test_clone)
|
|
422
|
+
|
|
423
|
+
situational_prompt = self._situational_prompt(payload, components, images)
|
|
424
|
+
with correlation.stage("claude-session", branch=branch,
|
|
425
|
+
max_turns=self.max_turns, timeout_s=self.session_timeout):
|
|
426
|
+
summary = self._invoke_claude(test_clone, system, situational_prompt)
|
|
427
|
+
except BaseException:
|
|
428
|
+
# Every failure path removes both clones. Success does not: they are handed off live,
|
|
429
|
+
# and `handler.py` is what removes them once it has committed, pushed and published —
|
|
430
|
+
# or refused for containment.
|
|
431
|
+
_rmtree(test_clone)
|
|
432
|
+
_rmtree(code_clone)
|
|
433
|
+
raise
|
|
434
|
+
|
|
435
|
+
return {
|
|
436
|
+
"testable": True,
|
|
437
|
+
"clone_dir": str(test_clone),
|
|
438
|
+
"code_clone_dir": str(code_clone),
|
|
439
|
+
"branch": branch,
|
|
440
|
+
"summary": summary,
|
|
441
|
+
}
|
|
442
|
+
|
|
443
|
+
def _components(self, names: list) -> list[Component]:
|
|
444
|
+
"""The declared components a `test-task` payload names, in the order it names them.
|
|
445
|
+
|
|
446
|
+
REFUSED, NOT FILTERED. The hand-written actor this replaces silently dropped a name it had
|
|
447
|
+
no tests root for, and answered for the rest — which reads, downstream, exactly like a
|
|
448
|
+
component whose tests were written. A caller naming a component this use does not declare
|
|
449
|
+
is told so before any session time is spent.
|
|
450
|
+
"""
|
|
451
|
+
resolved, unknown = [], []
|
|
452
|
+
for name in names:
|
|
453
|
+
component = self.config.component(str(name))
|
|
454
|
+
(resolved if component else unknown).append(component or name)
|
|
455
|
+
if unknown:
|
|
456
|
+
declared = ", ".join(c.name for c in self.config.components)
|
|
457
|
+
raise EngineError(
|
|
458
|
+
f"no declared component named {', '.join(map(str, unknown))} — this use writes "
|
|
459
|
+
f"tests for {declared}"
|
|
460
|
+
)
|
|
461
|
+
if not resolved:
|
|
462
|
+
raise EngineError("the payload names no component to write tests for")
|
|
463
|
+
return resolved
|
|
464
|
+
|
|
465
|
+
def _resolve_image(self, code_clone: Path, component: str, task_id: str,
|
|
466
|
+
registry: str) -> str:
|
|
467
|
+
"""Recompute the ref the implementation actor published, over identical inputs.
|
|
468
|
+
|
|
469
|
+
Including `normalize_name(task_id)`, which that actor's own publisher passes. These two
|
|
470
|
+
must agree exactly: a divergence does not raise, it silently names an image that was never
|
|
471
|
+
built. The folder is `<component>/` in the implementation clone — see the schema's
|
|
472
|
+
`components` doc for why a component's name is also where its implementation lives.
|
|
473
|
+
"""
|
|
474
|
+
try:
|
|
475
|
+
version = compute_version(
|
|
476
|
+
folder=code_clone / component,
|
|
477
|
+
name=self.config.image_name(component),
|
|
478
|
+
label="feature",
|
|
479
|
+
feature_name=normalize_name(task_id),
|
|
480
|
+
)
|
|
481
|
+
except ValueError as e:
|
|
482
|
+
raise EngineError(f"could not resolve {component}'s published image: {e}") from e
|
|
483
|
+
return self.config.image_ref(registry, component, version)
|
|
484
|
+
|
|
485
|
+
# ── propose-acceptance: the door that only answers ─────────────────────────────────────
|
|
486
|
+
|
|
487
|
+
def _propose(self, payload: dict, schema: dict | None) -> dict:
|
|
488
|
+
"""Propose what this actor will assert, black-box, before anything is built. Write nothing.
|
|
489
|
+
|
|
490
|
+
THE CLONES ARE OWNED HERE, END TO END. `_test` hands its clones off live because the
|
|
491
|
+
handler still has to commit and push. This door has no handler and produces no artifact,
|
|
492
|
+
so the `finally` is unconditional — success removes both clones exactly as failure does.
|
|
493
|
+
|
|
494
|
+
NO BRANCH IS CHECKED OUT, IN EITHER CLONE. There is nothing to put on one in the testing
|
|
495
|
+
repo, and in the implementation repo there must not be: `impl/<task_id>` may already exist
|
|
496
|
+
from an earlier attempt, and a proposal read off it would be a report of the build dressed
|
|
497
|
+
as a proposal. The default branch is the state the increment is being proposed against.
|
|
498
|
+
"""
|
|
499
|
+
task_id = payload["task_id"]
|
|
500
|
+
correlation.bind(correlation_id=correlation.correlation_id(), task_id=task_id)
|
|
501
|
+
|
|
502
|
+
test_clone = Path(tempfile.mkdtemp(prefix=self.config.clone_prefix(task_id)))
|
|
503
|
+
code_clone = Path(tempfile.mkdtemp(prefix=self.config.code_clone_prefix(task_id)))
|
|
504
|
+
try:
|
|
505
|
+
with correlation.stage("clone-tests", repo=self.config.source_repo):
|
|
506
|
+
self._clone(test_clone, self.config.source_repo)
|
|
507
|
+
with correlation.stage("clone-code", repo=self.config.implementation_repo):
|
|
508
|
+
self._clone(code_clone, self.config.implementation_repo)
|
|
509
|
+
self._ground(test_clone)
|
|
510
|
+
|
|
511
|
+
with correlation.stage("propose-session", max_turns=self.propose_max_turns,
|
|
512
|
+
timeout_s=self.propose_timeout):
|
|
513
|
+
answer = self._invoke_claude(
|
|
514
|
+
test_clone, self._propose_system(),
|
|
515
|
+
self._proposal_prompt(payload, schema, code_clone),
|
|
516
|
+
allowed_tools=PROPOSE_TOOLS, max_turns=self.propose_max_turns,
|
|
517
|
+
timeout=self.propose_timeout, add_dirs=(code_clone,),
|
|
518
|
+
)
|
|
519
|
+
finally:
|
|
520
|
+
_rmtree(test_clone)
|
|
521
|
+
_rmtree(code_clone)
|
|
522
|
+
|
|
523
|
+
proposal = _proposal(_extract_json(answer))
|
|
524
|
+
correlation.event("acceptance-proposed",
|
|
525
|
+
expectations=[e["id"] for e in proposal["expectations"]],
|
|
526
|
+
open_questions=len(proposal["open_questions"]))
|
|
527
|
+
return proposal
|
|
528
|
+
|
|
529
|
+
def _propose_system(self) -> str:
|
|
530
|
+
"""The system prompt for the propose door.
|
|
531
|
+
|
|
532
|
+
NOT `Actor.judge()`'s own. That one is built from the whole card and ends with "Reply with
|
|
533
|
+
a single JSON object capturing your judgement" — which is right, and is passed through
|
|
534
|
+
unmodified for `test-task`. But it is handed to this engine as `system` on both doors, and
|
|
535
|
+
this door needs the session to spend its turns READING rather than answering from the
|
|
536
|
+
card's prose alone.
|
|
537
|
+
"""
|
|
538
|
+
return (
|
|
539
|
+
"You are proposing, not testing. Read the task, the capability context you have been "
|
|
540
|
+
"given, the existing test suite and the implementation as it stands today, then say "
|
|
541
|
+
"precisely what you will assert black-box once the increment is built — and what the "
|
|
542
|
+
"task leaves undetermined. Do not write, edit or create any file; you have no tools "
|
|
543
|
+
"to do so."
|
|
544
|
+
)
|
|
545
|
+
|
|
546
|
+
def _ground(self, clone_dir: Path) -> None:
|
|
547
|
+
"""Fetch every `ground_in` source into the clone and render its `CLAUDE.md`.
|
|
548
|
+
|
|
549
|
+
Grounding is a PRECONDITION, not a request — see grounding.py. It runs before either
|
|
550
|
+
door's prompt is built, and a failure here stops the request before any session time is
|
|
551
|
+
spent, which is the cheapest place for it to stop. Both doors ground identically: what
|
|
552
|
+
will be asserted needs the same standing context as writing the assertion.
|
|
553
|
+
"""
|
|
554
|
+
for entry in self.config.ground_in:
|
|
555
|
+
with correlation.stage(f"ground-{entry.name}", into=entry.into, load=entry.load):
|
|
556
|
+
envelope = grounding.fetch(self.config, entry, timeout=self.fetch_timeout)
|
|
557
|
+
grounding.write_envelope(self.config, entry, clone_dir, envelope)
|
|
558
|
+
grounding.render_claude_md(self.config, clone_dir)
|
|
559
|
+
|
|
560
|
+
# ── git ─────────────────────────────────────────────────────────────────────────────────
|
|
561
|
+
|
|
562
|
+
def _configure_git_credentials(self) -> None:
|
|
563
|
+
"""Make GITHUB_TOKEN available to subprocesses that do their own git clones.
|
|
564
|
+
|
|
565
|
+
A `ground_in` tool typically delegates git auth entirely to git's own credential
|
|
566
|
+
resolution and never takes a token itself. A global URL rewrite is the one hook available
|
|
567
|
+
to make those clones use this token too, without patching the tool. Idempotent; safe to
|
|
568
|
+
call on every construction.
|
|
569
|
+
"""
|
|
570
|
+
subprocess.run(
|
|
571
|
+
["git", "config", "--global",
|
|
572
|
+
f"url.https://x-access-token:{self.github_token}@github.com/.insteadOf",
|
|
573
|
+
"https://github.com/"],
|
|
574
|
+
check=True, capture_output=True, text=True,
|
|
575
|
+
)
|
|
576
|
+
|
|
577
|
+
def _clone(self, dest: Path, repo: str) -> None:
|
|
578
|
+
# Full clone, not --depth 1: `papeete_version.compute()`'s own `semver_base()` does `git
|
|
579
|
+
# describe --tags --match <name>/v*` against a clone — this actor's own, to version its
|
|
580
|
+
# test image, and the implementation's, to recompute the image under test — which needs
|
|
581
|
+
# the matching tag's commit reachable in local history. A shallow clone only has the tip
|
|
582
|
+
# commit, and breaks the moment the tag isn't that exact commit.
|
|
583
|
+
url = f"https://x-access-token:{self.github_token}@github.com/{repo}.git"
|
|
584
|
+
try:
|
|
585
|
+
subprocess.run(
|
|
586
|
+
["git", "clone", url, str(dest)],
|
|
587
|
+
check=True, capture_output=True, text=True, timeout=self.clone_timeout,
|
|
588
|
+
)
|
|
589
|
+
except subprocess.CalledProcessError as e:
|
|
590
|
+
raise EngineError(
|
|
591
|
+
f"could not clone {repo}: {_redact(e.stderr, self.github_token)}"
|
|
592
|
+
) from e
|
|
593
|
+
except subprocess.TimeoutExpired as e:
|
|
594
|
+
raise EngineError(f"cloning {repo} timed out after {self.clone_timeout}s") from e
|
|
595
|
+
|
|
596
|
+
def _git(self, clone_dir: Path, args: list[str]) -> str:
|
|
597
|
+
try:
|
|
598
|
+
result = subprocess.run(
|
|
599
|
+
["git", *args], cwd=clone_dir, check=True, capture_output=True, text=True,
|
|
600
|
+
)
|
|
601
|
+
except subprocess.CalledProcessError as e:
|
|
602
|
+
raise EngineError(
|
|
603
|
+
f"git {' '.join(args)} failed: {_redact(e.stderr, self.github_token)}"
|
|
604
|
+
) from e
|
|
605
|
+
return result.stdout
|
|
606
|
+
|
|
607
|
+
# ── situational prompt, built from the caller's own payload ───────────────────────────
|
|
608
|
+
|
|
609
|
+
def _situational_prompt(self, payload: dict, components: list[Component],
|
|
610
|
+
images: list[str]) -> str:
|
|
611
|
+
"""The task, and nothing else.
|
|
612
|
+
|
|
613
|
+
NOTE WHAT IS ABSENT: any instruction to go and read the capability's context, and any
|
|
614
|
+
path to it. The hand-written actor named two JSON files in a sibling tempdir and asked the
|
|
615
|
+
session to read them "before you start". It is a precondition now — the generated
|
|
616
|
+
`CLAUDE.md` is loaded before turn one — so asking for it again would be asking for
|
|
617
|
+
something already done.
|
|
618
|
+
|
|
619
|
+
THE `<COMPONENT>_URL` CONVENTION STAYS, AS A DEFAULT. It is the one addressing agreement
|
|
620
|
+
this actor and the orchestrating actor have always shared. An agreed acceptance surface
|
|
621
|
+
can now state a handle's own variable, and where it does, the surface is the more specific
|
|
622
|
+
statement and wins.
|
|
623
|
+
"""
|
|
624
|
+
config = self.config
|
|
625
|
+
task_id = payload["task_id"]
|
|
626
|
+
roots = ", ".join(c.tests for c in components)
|
|
627
|
+
image_lines = "\n".join(f" - {c.name}: {image}" for c, image in zip(components, images))
|
|
628
|
+
existing = "\n".join(f" - {c.tests}" for c in components)
|
|
629
|
+
|
|
630
|
+
sections = [
|
|
631
|
+
f"# Author/extend black-box tests for {task_id} of {config.capability}: "
|
|
632
|
+
f"{payload['title']}\n\n"
|
|
633
|
+
f"You are working inside your own private clone (branch already checked out). Write "
|
|
634
|
+
f"ONLY under {roots} — nothing else in this clone is yours to change. Do not "
|
|
635
|
+
f"`git add`, `git commit`, or `git push` — that is handled outside this session.\n\n"
|
|
636
|
+
f"This is a PERSISTENT, accumulated test suite, not a fresh one for this task alone "
|
|
637
|
+
f"— read what is already there first and extend or adjust it for {task_id}'s own "
|
|
638
|
+
f"Definition of Done, rather than re-deriving the whole black-box surface from "
|
|
639
|
+
f"scratch:\n{existing}\n\n"
|
|
640
|
+
f"The published image(s) this suite targets (already built elsewhere — you do NOT "
|
|
641
|
+
f"bring these up yourself, and you have NOT been given their source):\n{image_lines}"
|
|
642
|
+
f"\n\n"
|
|
643
|
+
f"Each test reads its target component's base URL from an environment variable named "
|
|
644
|
+
f"<COMPONENT>_URL (uppercase, e.g. BACKEND_URL for a component named backend), unless "
|
|
645
|
+
f"the agreed acceptance surface below names another one in an expectation's handle. "
|
|
646
|
+
f"The orchestrating actor sets it to the running deployment's real address when it "
|
|
647
|
+
f"runs the published test image; you never set or resolve it yourself.\n\n"
|
|
648
|
+
f"You must NOT bring up docker or docker compose, hit a live endpoint, or otherwise "
|
|
649
|
+
f"run these tests against running infrastructure — this session only authors pytest "
|
|
650
|
+
f"source. Ground the exact request/response shapes, status codes and event contracts "
|
|
651
|
+
f"you assert in this capability's standing context and the existing suite's own "
|
|
652
|
+
f"conventions, not by probing a live container. `pytest --collect-only` (no network "
|
|
653
|
+
f"calls) is fine as a syntax/import check; running the suite is not — that verdict "
|
|
654
|
+
f"belongs to the orchestrating actor alone, against a real ephemeral deployment."
|
|
655
|
+
]
|
|
656
|
+
if payload.get("context"):
|
|
657
|
+
sections.append(f"## Context\n{payload['context']}")
|
|
658
|
+
sections.append(
|
|
659
|
+
"## Definition of done\n"
|
|
660
|
+
+ "\n".join(f"- {item}" for item in payload["definition_of_done"])
|
|
661
|
+
)
|
|
662
|
+
if payload.get("acceptance_surface"):
|
|
663
|
+
# Agreed in the round before any of this was built: proposed at this actor's own
|
|
664
|
+
# propose-acceptance door, and accepted (with commitments) by the implementer. The ids
|
|
665
|
+
# are what a later verdict names, which is why every test has to carry one.
|
|
666
|
+
sections.append(
|
|
667
|
+
"## Agreed acceptance surface\n"
|
|
668
|
+
"This was agreed with the implementer before anything was built. Each expectation "
|
|
669
|
+
"below must be asserted by its `id`, addressing it exactly via its `handle` — "
|
|
670
|
+
"including any value committed to there. Where it is more specific than the "
|
|
671
|
+
"definition of done, it wins. Name the expectation id in each test that asserts "
|
|
672
|
+
"it (in the test's name or its docstring), so a failing verdict points at an "
|
|
673
|
+
"agreed expectation rather than at a line of output.\n\n"
|
|
674
|
+
+ json.dumps(payload["acceptance_surface"], indent=2, ensure_ascii=False)
|
|
675
|
+
)
|
|
676
|
+
if payload.get("remediation_context"):
|
|
677
|
+
sections.append(
|
|
678
|
+
"## Remediation — the prior attempt's failing criteria\n"
|
|
679
|
+
"A criterion failing could mean the test itself was wrong (bad expectation, "
|
|
680
|
+
"wrong endpoint, wrong shape) or that the implementation was wrong. Review each "
|
|
681
|
+
"one: fix the test here if the fault is in it; if the test looks correct against "
|
|
682
|
+
"this capability's standing context and the agreed acceptance surface, leave it "
|
|
683
|
+
"as-is (the implementation side should fix it instead) and say so in your "
|
|
684
|
+
"summary.\n"
|
|
685
|
+
f"{payload['remediation_context']}"
|
|
686
|
+
)
|
|
687
|
+
return "\n\n".join(sections)
|
|
688
|
+
|
|
689
|
+
# ── the proposal prompt ─────────────────────────────────────────────────────────────────
|
|
690
|
+
|
|
691
|
+
def _proposal_prompt(self, payload: dict, schema: dict | None, code_clone: Path) -> str:
|
|
692
|
+
"""What the propose door asks. Nothing about writing, because it cannot.
|
|
693
|
+
|
|
694
|
+
PROPOSE VALUES, DO NOT WAIT FOR THEM. The failure this round exists for is a test tree
|
|
695
|
+
carrying fixture ids "discovered black-box against the running container", against a task
|
|
696
|
+
that never named them (ADR-FIA-0004). Where a test will need a concrete value the task
|
|
697
|
+
leaves unnamed, the session proposes one here, and the implementer commits to it or
|
|
698
|
+
objects at its own assess door. A value a tester proposes before the build is a promise
|
|
699
|
+
someone must keep; the same value read off the build is a report.
|
|
700
|
+
|
|
701
|
+
QUESTIONS ARE NOT FAILURES. What neither the task nor the standing context determines —
|
|
702
|
+
and what no proposed value can responsibly settle, because it is a question of what the
|
|
703
|
+
behaviour SHOULD be — goes in `open_questions`, never into an invented statement. Any open
|
|
704
|
+
question stops the round and reaches a human, which is the point of asking it.
|
|
705
|
+
|
|
706
|
+
THE ANSWER'S SHAPE IS DERIVED, NOT INVENTED HERE. `Actor.judge()` computes it from the
|
|
707
|
+
door's own `completion_schema` and hands it over as `schema`; rendering that is how the
|
|
708
|
+
prompt and the card cannot come to disagree. When it is absent — an engine driven directly,
|
|
709
|
+
in a test or a probe — the fallback below says the same thing in words.
|
|
710
|
+
"""
|
|
711
|
+
config = self.config
|
|
712
|
+
task_id = payload["task_id"]
|
|
713
|
+
names = [str(name) for name in payload.get("components") or []]
|
|
714
|
+
component_lines = []
|
|
715
|
+
for name in names:
|
|
716
|
+
declared = config.component(name)
|
|
717
|
+
component_lines.append(
|
|
718
|
+
f" - {name}: its tests live under {declared.tests}" if declared else
|
|
719
|
+
f" - {name}: (this use declares no tests root for it — say so in open_questions "
|
|
720
|
+
f"if the task needs it asserted)")
|
|
721
|
+
|
|
722
|
+
sections = [
|
|
723
|
+
f"# What will you assert for {task_id} of {config.capability}: {payload['title']}?\n\n"
|
|
724
|
+
f"NOTHING HAS BEEN BUILT YET. The implementer has not started. Before it does, you "
|
|
725
|
+
f"propose the acceptance surface: exactly what you will assert, black-box, about this "
|
|
726
|
+
f"increment once it exists. The implementer then answers whether it can deliver each "
|
|
727
|
+
f"expectation, commits to the values a test will address, or objects.\n\n"
|
|
728
|
+
f"You are READING ONLY. Do not write, edit or create any file — you have no tools to "
|
|
729
|
+
f"do so, and there is no branch and no commit at this door.\n\n"
|
|
730
|
+
f"What you can read:\n"
|
|
731
|
+
f" - your working directory: this actor's own testing repository, with the existing "
|
|
732
|
+
f"accumulated suite under {', '.join(config.writes_only_under)}\n"
|
|
733
|
+
f" - {code_clone}: {_CODE_CLONE_LABEL} — the capability as it stands BEFORE this "
|
|
734
|
+
f"increment. Use it to align with what already exists; never treat it as the "
|
|
735
|
+
f"increment itself\n"
|
|
736
|
+
f" - this capability's standing context, already loaded for this session\n\n"
|
|
737
|
+
f"Propose each expectation as an object with:\n"
|
|
738
|
+
f" - `id`: short and stable (`E1`, or kebab-case), unique within this proposal — a "
|
|
739
|
+
f"later verdict names it\n"
|
|
740
|
+
f" - `statement`: what must hold, observable from outside the component\n"
|
|
741
|
+
f" - `handle`: exactly how a black-box test reaches it — the component, the endpoint "
|
|
742
|
+
f"(method and path), the event routing key, the environment variable carrying its "
|
|
743
|
+
f"base URL (<COMPONENT>_URL unless you need another)\n"
|
|
744
|
+
f" - `component` (optional): which of the components below it belongs to\n\n"
|
|
745
|
+
f"Where a test needs a concrete value the task does not name — a fixture's id, a "
|
|
746
|
+
f"seeded record, a path parameter, a routing key — PROPOSE A CONCRETE VALUE in the "
|
|
747
|
+
f"handle. The implementer will accept it and commit to it, or object. Do not leave it "
|
|
748
|
+
f"for the implementation to choose and for a test to discover afterwards.\n\n"
|
|
749
|
+
f"Anything the task and the standing context do not determine, and that no proposed "
|
|
750
|
+
f"value can settle because it is a question of what the behaviour should BE, goes in "
|
|
751
|
+
f"`open_questions` — never into an invented statement. An open question is not a "
|
|
752
|
+
f"failure: it stops this round and reaches a human, which is exactly what it is for."
|
|
753
|
+
]
|
|
754
|
+
sections.append(
|
|
755
|
+
"## Components this round concerns\n"
|
|
756
|
+
+ ("\n".join(component_lines) if component_lines else " (none named)")
|
|
757
|
+
)
|
|
758
|
+
if payload.get("context"):
|
|
759
|
+
sections.append(f"## Context\n{payload['context']}")
|
|
760
|
+
sections.append(
|
|
761
|
+
"## Definition of done\n"
|
|
762
|
+
+ "\n".join(f"- {item}" for item in payload["definition_of_done"])
|
|
763
|
+
)
|
|
764
|
+
sections.append(
|
|
765
|
+
"## Your answer\n"
|
|
766
|
+
"End with a single fenced ```json block and nothing after it, holding one object"
|
|
767
|
+
+ (f" conforming to:\n\n```json\n{json.dumps(schema, indent=2)}\n```"
|
|
768
|
+
if schema else
|
|
769
|
+
" with `expectations` (a list of objects, each with `id`, `statement` and "
|
|
770
|
+
"`handle`) and `open_questions` (a list of strings, or of objects with `about` and "
|
|
771
|
+
"`question`; empty when the task determines everything).")
|
|
772
|
+
)
|
|
773
|
+
return "\n\n".join(sections)
|
|
774
|
+
|
|
775
|
+
# ── the judgement itself: a claude -p session against the checked-out clone ────────────
|
|
776
|
+
|
|
777
|
+
def _invoke_claude(self, clone_dir: Path, system: str, situational_prompt: str, *,
|
|
778
|
+
allowed_tools: str = TEST_TOOLS,
|
|
779
|
+
max_turns: int | None = None,
|
|
780
|
+
timeout: int | None = None,
|
|
781
|
+
add_dirs: tuple[Path, ...] = ()) -> str:
|
|
782
|
+
"""Run the session, streaming every turn to the log as it happens.
|
|
783
|
+
|
|
784
|
+
`--output-format stream-json --verbose` rather than `--output-format json`: the latter
|
|
785
|
+
emits ONE object at the end, holding only the final assistant text, so the whole inner
|
|
786
|
+
conversation — what was read, what was run, what was decided — existed nowhere durable
|
|
787
|
+
once the pod went away. The projection above is what makes streaming it affordable.
|
|
788
|
+
|
|
789
|
+
`Popen` rather than `subprocess.run`: reading line by line is what lets each turn be
|
|
790
|
+
logged as it happens rather than after the session ends, and it stops a 30-minute
|
|
791
|
+
session's entire output being buffered in a pod capped at 2Gi.
|
|
792
|
+
"""
|
|
793
|
+
max_turns = self.max_turns if max_turns is None else max_turns
|
|
794
|
+
timeout = self.session_timeout if timeout is None else timeout
|
|
795
|
+
cmd = [self.claude_bin, "--print", "--output-format", "stream-json", "--verbose"]
|
|
796
|
+
for directory in add_dirs:
|
|
797
|
+
# BEFORE another option, never last. `--add-dir` takes a variadic list, and a
|
|
798
|
+
# variadic option followed by the positional prompt would swallow the prompt as a
|
|
799
|
+
# second directory. `--append-system-prompt` right after it closes the list.
|
|
800
|
+
cmd += ["--add-dir", str(directory)]
|
|
801
|
+
cmd += [
|
|
802
|
+
"--append-system-prompt", system,
|
|
803
|
+
"--permission-mode", "acceptEdits",
|
|
804
|
+
# The propose door passes a list with no Write, Edit or Bash in it. That is the
|
|
805
|
+
# enforcement, not the prompt's own "you are reading only" — the same discipline as
|
|
806
|
+
# handler.py's containment check standing behind the test door's write boundary.
|
|
807
|
+
"--allowedTools", allowed_tools,
|
|
808
|
+
"--max-turns", str(max_turns),
|
|
809
|
+
situational_prompt,
|
|
810
|
+
]
|
|
811
|
+
# stderr to a temp file, not a second pipe: nothing drains a second pipe while the
|
|
812
|
+
# stdout loop below runs, so a chatty stderr would fill its buffer and deadlock the
|
|
813
|
+
# session. Merging it into stdout is not an option either — it would corrupt the stream.
|
|
814
|
+
with tempfile.TemporaryFile("w+") as errfile:
|
|
815
|
+
try:
|
|
816
|
+
proc = subprocess.Popen(cmd, cwd=clone_dir, stdout=subprocess.PIPE,
|
|
817
|
+
stderr=errfile, text=True, bufsize=1)
|
|
818
|
+
except FileNotFoundError as e:
|
|
819
|
+
raise EngineError(
|
|
820
|
+
f"'{self.claude_bin}' is not on PATH — install @anthropic-ai/claude-code"
|
|
821
|
+
) from e
|
|
822
|
+
|
|
823
|
+
# A watchdog, not a deadline checked per line: a session that hangs having emitted
|
|
824
|
+
# nothing would never reach another loop iteration to be checked, and `Popen` has no
|
|
825
|
+
# equivalent of `subprocess.run(timeout=...)` while iterating its output.
|
|
826
|
+
timed_out = threading.Event()
|
|
827
|
+
|
|
828
|
+
def _expire() -> None:
|
|
829
|
+
timed_out.set()
|
|
830
|
+
proc.kill()
|
|
831
|
+
|
|
832
|
+
watchdog = threading.Timer(timeout, _expire)
|
|
833
|
+
watchdog.start()
|
|
834
|
+
|
|
835
|
+
final: dict | None = None
|
|
836
|
+
try:
|
|
837
|
+
for raw in proc.stdout:
|
|
838
|
+
try:
|
|
839
|
+
event = json.loads(raw)
|
|
840
|
+
except json.JSONDecodeError:
|
|
841
|
+
continue # a non-JSON line is noise, never the session's result
|
|
842
|
+
record = _project(event)
|
|
843
|
+
if record is not None:
|
|
844
|
+
# No %-args: `logging` only applies %-formatting when args are passed,
|
|
845
|
+
# so a stray % in a file body cannot raise here. No `extra=` either —
|
|
846
|
+
# `correlation.bind()` at the top of each door already stamps task_id and
|
|
847
|
+
# correlation_id onto every record emitted on this thread.
|
|
848
|
+
logging.info(_line(record))
|
|
849
|
+
if event.get("type") == "result":
|
|
850
|
+
final = event
|
|
851
|
+
proc.wait()
|
|
852
|
+
finally:
|
|
853
|
+
watchdog.cancel()
|
|
854
|
+
if proc.poll() is None:
|
|
855
|
+
proc.kill()
|
|
856
|
+
proc.wait()
|
|
857
|
+
proc.stdout.close()
|
|
858
|
+
|
|
859
|
+
if timed_out.is_set():
|
|
860
|
+
raise EngineError(
|
|
861
|
+
f"claude session for this task exceeded {timeout}s"
|
|
862
|
+
)
|
|
863
|
+
errfile.seek(0)
|
|
864
|
+
stderr = errfile.read()
|
|
865
|
+
|
|
866
|
+
if final is None:
|
|
867
|
+
raise EngineError(
|
|
868
|
+
f"claude (rc={proc.returncode}) produced no result event: {stderr[-2000:]}"
|
|
869
|
+
)
|
|
870
|
+
if final.get("is_error") or proc.returncode != 0:
|
|
871
|
+
raise EngineError(
|
|
872
|
+
f"claude session failed (subtype={final.get('subtype')}): "
|
|
873
|
+
f"{final.get('result', '')[:4000]}"
|
|
874
|
+
)
|
|
875
|
+
return final.get("result", "")
|
|
876
|
+
|
|
877
|
+
|
|
878
|
+
def _image_registry() -> str:
|
|
879
|
+
registry = os.environ.get("IMAGE_REGISTRY")
|
|
880
|
+
if not registry:
|
|
881
|
+
raise EngineError(
|
|
882
|
+
"no IMAGE_REGISTRY set — cannot name the image the implementation actor published, "
|
|
883
|
+
"nor the test image this door would publish"
|
|
884
|
+
)
|
|
885
|
+
return registry.rstrip("/")
|
|
886
|
+
|
|
887
|
+
|
|
888
|
+
def _rmtree(path: Path) -> None:
|
|
889
|
+
shutil.rmtree(path, ignore_errors=True)
|