foundry-testing-actor 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,889 @@
1
+ """`ClaudeCodeTesterEngine` — judgement by a headless Claude Code session against fresh clones.
2
+
3
+ TESTING AUTHORS A HARNESS; IT DOES NOT EXECUTE IT. No session this engine starts brings up docker
4
+ or docker compose, hits a live endpoint, or renders a pass/fail verdict. At `test-task` it writes
5
+ and extends black-box pytest source; at `propose-acceptance` it writes nothing at all. Standing up
6
+ infrastructure and rendering the verdict belongs to the orchestrating actor alone, against a real
7
+ ephemeral deployment, running the test image `handler.py` publishes.
8
+
9
+ WHY A CUSTOM ENGINE, NOT `papeete_actor_synchronous_messaging.engine.resolve("claude")`. The
10
+ built-in `ClaudeEngine` shells out to the raw `anthropic` Python SDK — `anthropic.Anthropic()`,
11
+ resolving `ANTHROPIC_API_KEY`, then `ANTHROPIC_AUTH_TOKEN`, then an `ant auth login` profile.
12
+ Every one of those is a metered API credential. Subscription OAuth tokens (`claude setup-token`
13
+ → `CLAUDE_CODE_OAUTH_TOKEN`) are scoped to the `claude` CLI / claude.ai specifically and are NOT
14
+ among the credentials the raw SDK will resolve. So the only way to spend a subscription's own
15
+ credit rather than a separate metered key is to shell out to the `claude` CLI itself, which is
16
+ what this file does. It still satisfies the `Engine` port (`name` + `judge(system, prompt,
17
+ schema=None) -> dict`) — the port makes no promise about how long or heavy a judgement is.
18
+
19
+ **Do not set `ANTHROPIC_API_KEY` (or `ANTHROPIC_AUTH_TOKEN`) in a container running this engine.**
20
+ In `claude -p` non-interactive mode an API key present in the environment is ALWAYS preferred over
21
+ `CLAUDE_CODE_OAUTH_TOKEN`, silently routing every session through metered billing instead. There is
22
+ no warning and no visible difference in the transcript; the only symptom is the bill.
23
+
24
+ PAYLOAD-DRIVEN, NOT CARD-DRIVEN. The caller supplies everything a session needs to work — this
25
+ engine never clones a repo merely to look up what a task IS. A missing required field is refused
26
+ by the framework's own schema gate before any engine time is spent.
27
+
28
+ TWO CLONES, AND WHAT EACH DOOR MAY SEE OF THE SECOND. The testing repo is this actor's own, and
29
+ every session runs inside a private clone of it. The implementation repo is read, never written,
30
+ and the two doors read it for opposite reasons:
31
+
32
+ * `propose-acceptance` clones its DEFAULT BRANCH and shows it to the session — the capability as
33
+ it stands BEFORE the increment. It never checks out `impl/<task_id>`: a tester whose
34
+ assertions are derived from what was built can only ever confirm the build (ADR-FIA-0004's
35
+ "before or after", ADR-FTA-0002).
36
+ * `test-task` checks out `impl/<task_id>` and does NOT show it to the session. It is there so
37
+ `papeete_version.compute` can recompute, over identical inputs, the image refs the
38
+ implementation actor published — no tag is ever handed to this actor, and none has to be.
39
+
40
+ NO KNOWLEDGE TOOL IS NAMED HERE. The hand-written actor this replaces shelled out to two knowledge
41
+ tools by name, with the registry they resolved through as a module constant. Both are `ground_in`
42
+ entries in a use's sidecar now, run by `grounding.py` like any other argv (ADR-FTA-0001).
43
+
44
+ CARRIES NO CAPABILITY LITERAL. Every identifier it needs comes from `CapabilityConfig`, which
45
+ derives all of them from one capability id and one repo (see `config.py`).
46
+
47
+ WHAT THIS ENGINE DOES NOT DO. It never commits, pushes, builds, or opens a pull request — that is
48
+ `handler.py`'s job, and `handler.py`'s own containment check is the actual enforcement of the
49
+ write boundary this engine's prompt only *asks* the session to respect. On success at `test-task`
50
+ it deliberately does NOT clean up its clones: the handler still needs the testing clone to commit
51
+ and push, and is the one that removes both, on every path including containment failure.
52
+ """
53
+ from __future__ import annotations
54
+
55
+ import json
56
+ import logging
57
+ import os
58
+ import re
59
+ import shutil
60
+ import subprocess
61
+ import tempfile
62
+ import threading
63
+ from pathlib import Path
64
+
65
+ from papeete_actor_synchronous_messaging.engine import EngineError
66
+ from papeete_version.version import compute as compute_version
67
+ from papeete_version.version import normalize_name
68
+
69
+ from . import correlation, grounding
70
+ from .config import CapabilityConfig, Component
71
+
72
+ DEFAULT_CLONE_TIMEOUT_S = 120
73
+ DEFAULT_SESSION_TIMEOUT_S = 1800
74
+ DEFAULT_MAX_TURNS = 60
75
+
76
+ # The propose door reads and answers; it never writes. Its budget is smaller than the test door's
77
+ # on every axis, and its tool list is the enforcement — a door that CANNOT write beats one asked
78
+ # not to. The same numbers as the implementation actor's assess door, deliberately: the two halves
79
+ # of one round should cost the same order of money (ADR-FIA-0004, ADR-FTA-0002).
80
+ DEFAULT_PROPOSE_TIMEOUT_S = 600
81
+ DEFAULT_PROPOSE_MAX_TURNS = 15
82
+
83
+ TEST_TOOLS = "Bash,Read,Edit,Write,Glob,Grep"
84
+ PROPOSE_TOOLS = "Read,Glob,Grep"
85
+
86
+ # The door ids this engine answers. They are the card's, not this module's invention — `judge()`
87
+ # is handed the id it must dispatch on (see `_door_from_prompt`).
88
+ TEST_DOOR = "test-task"
89
+ PROPOSE_DOOR = "propose-acceptance"
90
+
91
+ # The branch conventions both peers already follow, so nothing has to be looked up to know them.
92
+ IMPLEMENTATION_BRANCH = "impl/{task_id}"
93
+ TEST_BRANCH = "test/{task_id}"
94
+
95
+ # The name the implementation repo's default-branch clone is shown to the propose session under.
96
+ # A directory ADDED to the session beside its working directory, not a path inside the testing
97
+ # clone: a nested repository there would be one `git add` away from the tests this actor commits.
98
+ _CODE_CLONE_LABEL = "the implementation repository, default branch"
99
+
100
+
101
+ # ── stream-json projection: keeping every emitted log line inside Loki's max_line_size ────────
102
+ #
103
+ # WHY PROJECT RATHER THAN LOG THE RAW EVENT. `--output-format stream-json` emits one JSON object
104
+ # per turn — the inner conversation this actor's audit trail is made of. Two of its fields are
105
+ # unbounded: `tool_use.input` (for `Write`, the whole file body; for `Edit`, both sides of the
106
+ # replacement) and `tool_result.content` (for `Read`/`Grep`/`Bash`, the whole output). Nothing
107
+ # else is: a `text` block is a few hundred bytes. So clipping those two, and only those, keeps
108
+ # the entire conversation readable while bounding every line — and what gets clipped is
109
+ # recoverable from the branch `handler.py` pushes anyway.
110
+ #
111
+ # WHY IT MATTERS. Loki's `max_line_size` is 256KB with `max_line_size_truncate: false` — an
112
+ # oversized line is REJECTED OUTRIGHT, not trimmed, so a single large `Read` would silently
113
+ # delete exactly the turn worth reading while leaving the rest of the session intact. The budget
114
+ # below is a quarter of that, which also sidesteps Loki parsing `KB` as 1000 rather than 1024.
115
+ #
116
+ # AND WHY NOT RECORD ATTRIBUTES. The OTel `LoggingHandler` maps the formatted message to the OTLP
117
+ # body (the line Loki measures) and record attributes to log-record attributes, which land in Loki
118
+ # structured metadata under their own separate caps (`max_structured_metadata_size: 64KB`,
119
+ # `max_structured_metadata_entries_count: 128`). Payload cannot be smuggled out of the line limit
120
+ # by moving it there — that channel is for correlation ids, which `correlation.py` stamps onto
121
+ # every record this process emits without any call site here passing them.
122
+
123
+ LINE_BUDGET = 64 * 1024 # bytes per emitted log line
124
+ BLOB_HEAD = 2000 # bytes kept from the front of an unbounded value
125
+ BLOB_TAIL = 2000 # ...and from the back: a Bash failure lives in the tail, not the head
126
+ PROSE = 8000 # text/thinking/result — prose IS the audit trail, give it more room
127
+
128
+
129
+ def _clip(value, head: int = BLOB_HEAD, tail: int = BLOB_TAIL) -> str:
130
+ """Head+tail slice of a value, budgeted in BYTES (not characters — the limit Loki enforces
131
+ is on the UTF-8 encoded line), with the true size recorded in the marker."""
132
+ if not isinstance(value, str):
133
+ value = json.dumps(value, default=str)
134
+ raw = value.encode("utf-8", "replace")
135
+ if len(raw) <= head + tail:
136
+ return value
137
+ return (raw[:head].decode("utf-8", "replace")
138
+ + f"\n…[clipped {len(raw) - head - tail} of {len(raw)} bytes]…\n"
139
+ + raw[-tail:].decode("utf-8", "replace"))
140
+
141
+
142
+ def _project(event: dict) -> dict | None:
143
+ """One stream-json event -> a compact dict, or None to drop it entirely."""
144
+ kind = event.get("type")
145
+
146
+ if kind == "system" and event.get("subtype") == "init":
147
+ # The raw init event is ~2.2KB of tools/skills/slash_commands inventory. Four fields of
148
+ # it are worth keeping — `session_id` is the join key to the CLI's own full transcript,
149
+ # which it writes to $HOME/.claude/projects/<slug>/<session_id>.jsonl regardless of
150
+ # --output-format.
151
+ return {"event": "init", "session_id": event.get("session_id"),
152
+ "model": event.get("model"), "cwd": event.get("cwd")}
153
+
154
+ if kind == "result":
155
+ usage = event.get("usage") or {}
156
+ return {"event": "result", "session_id": event.get("session_id"),
157
+ "subtype": event.get("subtype"), "is_error": event.get("is_error"),
158
+ "num_turns": event.get("num_turns"), "duration_ms": event.get("duration_ms"),
159
+ "cost_usd": event.get("total_cost_usd"),
160
+ "input_tokens": usage.get("input_tokens"),
161
+ "output_tokens": usage.get("output_tokens"),
162
+ "result": _clip(event.get("result", ""), PROSE, PROSE)}
163
+
164
+ if kind not in ("assistant", "user"):
165
+ return None # rate_limit_event and friends carry no audit value
166
+
167
+ blocks = []
168
+ for block in (event.get("message") or {}).get("content", []):
169
+ if not isinstance(block, dict):
170
+ continue
171
+ btype = block.get("type")
172
+ if btype in ("text", "thinking"):
173
+ blocks.append({btype: _clip(block.get(btype, ""), PROSE, PROSE)})
174
+ elif btype == "tool_use":
175
+ blocks.append({"tool_use": block.get("name"), "id": block.get("id"),
176
+ "input": {k: _clip(v) for k, v in (block.get("input") or {}).items()}})
177
+ elif btype == "tool_result":
178
+ # `content` is a str for a text result and a list of blocks otherwise — `_clip`
179
+ # json-dumps the latter rather than this guessing at its shape.
180
+ blocks.append({"tool_result": block.get("tool_use_id"),
181
+ "is_error": block.get("is_error", False),
182
+ "content": _clip(block.get("content", ""))})
183
+ return {"event": kind, "blocks": blocks} if blocks else None
184
+
185
+
186
+ def _line(record: dict) -> str:
187
+ """Serialize, then hard-enforce the budget.
188
+
189
+ The per-field clipping in `_project` is what keeps lines small; this is what makes "no line
190
+ exceeds the budget" a property of the code rather than a hope — a `tool_use` carrying a
191
+ hundred just-under-threshold keys would otherwise slip through the sum."""
192
+ line = json.dumps(record, default=str, ensure_ascii=False)
193
+ if len(line.encode("utf-8")) > LINE_BUDGET:
194
+ half = LINE_BUDGET // 2 - 200
195
+ line = json.dumps({"event": record.get("event"), "over_budget": True,
196
+ "clipped": _clip(line, half, half)}, ensure_ascii=False)
197
+ return line
198
+
199
+
200
+ # ```json … ``` or a bare ``` … ``` block. Non-greedy, DOTALL: one match per fence, in order.
201
+ _FENCE = re.compile(r"```(?:json)?\s*\n(.*?)```", re.DOTALL)
202
+
203
+
204
+ def _redact(text: str, secret: str | None) -> str:
205
+ return text.replace(secret, "***") if secret else text
206
+
207
+
208
+ def _payload_from_prompt(prompt: str) -> dict:
209
+ """Recover the full payload dict from `Actor.judge()`'s own fixed prompt format
210
+ (`papeete_actor_synchronous_messaging.actor.Actor.judge`):
211
+
212
+ f"verb: {verb}\\ndoor: {offer.id}\\n{as_prompt_json(situation)}"
213
+
214
+ where `situation["payload"]` is exactly what the caller sent at this door.
215
+ """
216
+ lines = prompt.split("\n", 2)
217
+ if len(lines) < 3:
218
+ raise EngineError(f"prompt does not follow Actor.judge()'s fixed format: {prompt!r}")
219
+ try:
220
+ situation = json.loads(lines[2])
221
+ return situation["payload"]
222
+ except (json.JSONDecodeError, KeyError, TypeError) as e:
223
+ raise EngineError(f"could not recover payload from prompt: {e}") from e
224
+
225
+
226
+ def _door_from_prompt(prompt: str) -> str:
227
+ """Recover the door id from `Actor.judge()`'s own fixed prompt format — its second line.
228
+
229
+ ONE ENGINE, TWO DOORS. The `Engine` port is a single `judge()`, and both this actor's doors
230
+ name the same engine key, so an instance registered under `claude-code` is asked to judge
231
+ both. `Actor.judge()` already puts the door id on line 2 of the prompt it builds, so the
232
+ dispatch needs nothing from the framework that is not already being handed over.
233
+ """
234
+ lines = prompt.split("\n", 2)
235
+ if len(lines) < 2 or not lines[1].startswith("door: "):
236
+ raise EngineError(f"prompt does not follow Actor.judge()'s fixed format: {prompt!r}")
237
+ return lines[1][len("door: "):].strip()
238
+
239
+
240
+ def _extract_json(text: str) -> dict:
241
+ """The single JSON object a judged answer ends with.
242
+
243
+ A session's final `result` is prose that HAPPENS to contain the answer, not the answer — it
244
+ reliably wraps it in a fence and unreliably says something either side of it. So: try every
245
+ fenced block, last first (the last one is the conclusion; an earlier one is usually the
246
+ session quoting what it was asked for), then the whole text for the case where it complied
247
+ exactly. Anything else is an `EngineError` — a door that cannot say what it decided has not
248
+ decided anything, and guessing on its behalf would be worse than failing.
249
+ """
250
+ for block in reversed(_FENCE.findall(text)):
251
+ try:
252
+ parsed = json.loads(block)
253
+ except json.JSONDecodeError:
254
+ continue
255
+ if isinstance(parsed, dict):
256
+ return parsed
257
+ try:
258
+ parsed = json.loads(text.strip())
259
+ except json.JSONDecodeError:
260
+ parsed = None
261
+ if isinstance(parsed, dict):
262
+ return parsed
263
+ raise EngineError(
264
+ "the session produced no JSON object to read its judgement from; its answer ended: "
265
+ f"{text[-2000:]!r}"
266
+ )
267
+
268
+
269
+ def _as_list(value) -> list:
270
+ """A session's idea of "a list of questions", as a list.
271
+
272
+ Absent or null is none; a lone string or object is one. Coerced rather than trusted: a session
273
+ reliably means the list and unreliably types it, and `Actor.receive()` would refuse the whole
274
+ proposal over one question that arrived bare.
275
+ """
276
+ if value is None:
277
+ return []
278
+ if isinstance(value, list):
279
+ return value
280
+ return [value]
281
+
282
+
283
+ def _proposal(judged: dict) -> dict:
284
+ """The reply `acceptance-proposed-result` can carry, and nothing else.
285
+
286
+ PROJECTED, NOT PASSED THROUGH. A door naming ONE completion message gets an open schema from
287
+ the framework (ADR-PAS-0009), so a session's helpful `"notes"` key beside its answer would not
288
+ be refused — it would travel to the orchestrating actor looking like part of a contract three
289
+ packages share, which is worse. And the day this door names a second outcome, the framework
290
+ closes the schema and that same key refuses the whole proposal. Only the two fields the message
291
+ references survive; everything else a session said is in the transcript.
292
+
293
+ WHAT IS CHECKED, AND WHY IT IS NOT MORE. `expectations` must be a list of objects, each with an
294
+ `id` and a `statement`, a `handle` key, and ids unique within the proposal — because a later
295
+ verdict names an expectation by its id, and the orchestrating actor attaches the implementer's
296
+ commitments by it too. Two expectations sharing one id cannot both be named. What a handle
297
+ CONTAINS is not checked: its shape depends on what is being addressed, and the card types the
298
+ surface as a list precisely so that stays prose.
299
+ """
300
+ if "expectations" not in judged:
301
+ raise EngineError(
302
+ "the proposal named no `expectations` — the one thing the caller has to be able to "
303
+ f"send on; it answered: {judged!r}"
304
+ )
305
+ expectations = judged["expectations"]
306
+ if not isinstance(expectations, list):
307
+ raise EngineError(f"`expectations` is not a list: {expectations!r}")
308
+ seen: set[str] = set()
309
+ for index, expectation in enumerate(expectations):
310
+ if not isinstance(expectation, dict):
311
+ raise EngineError(f"expectations[{index}] is not an object: {expectation!r}")
312
+ missing = [key for key in ("id", "statement") if not expectation.get(key)]
313
+ if missing or "handle" not in expectation:
314
+ raise EngineError(
315
+ f"expectations[{index}] carries no {', '.join(missing or ['handle'])} — every "
316
+ f"expectation is an id, a statement and a handle: {expectation!r}"
317
+ )
318
+ identifier = str(expectation["id"])
319
+ if identifier in seen:
320
+ raise EngineError(
321
+ f"expectation id '{identifier}' is proposed twice — a verdict names an "
322
+ f"expectation by its id, so an id two expectations share names neither"
323
+ )
324
+ seen.add(identifier)
325
+ return {"expectations": expectations, "open_questions": _as_list(judged.get("open_questions"))}
326
+
327
+
328
+ class ClaudeCodeTesterEngine:
329
+ """Judgement by shelling out to the `claude` CLI, against fresh private clones."""
330
+
331
+ def __init__(self, config: CapabilityConfig, *,
332
+ github_token: str | None = None, claude_bin: str = "claude",
333
+ clone_timeout: int = DEFAULT_CLONE_TIMEOUT_S,
334
+ fetch_timeout: int = grounding.DEFAULT_FETCH_TIMEOUT_S,
335
+ session_timeout: int = DEFAULT_SESSION_TIMEOUT_S,
336
+ max_turns: int = DEFAULT_MAX_TURNS,
337
+ propose_timeout: int = DEFAULT_PROPOSE_TIMEOUT_S,
338
+ propose_max_turns: int = DEFAULT_PROPOSE_MAX_TURNS):
339
+ self.config = config
340
+ # The engine's `name` is what the card's doors name, so it comes from the sidecar rather
341
+ # than being fixed here — the port requires the attribute, not a particular value.
342
+ self.name = config.engine
343
+
344
+ self.github_token = github_token or os.environ.get("GITHUB_TOKEN")
345
+ if not self.github_token:
346
+ raise RuntimeError(
347
+ "ClaudeCodeTesterEngine needs GITHUB_TOKEN in the environment — a fine-grained "
348
+ f"PAT with contents:write on {config.source_repo}, read-only contents on "
349
+ f"{config.implementation_repo}, plus read-only contents on "
350
+ f"{config.registry_repo} and whatever else this capability's `ground_in` fetches "
351
+ "resolve through."
352
+ )
353
+ self.claude_bin = claude_bin
354
+ self.clone_timeout = clone_timeout
355
+ self.fetch_timeout = fetch_timeout
356
+ self.session_timeout = session_timeout
357
+ self.max_turns = max_turns
358
+ # Constructor kwargs, not sidecar fields: how long this actor's own doors may think is
359
+ # operational tuning, not something a capability declares about itself.
360
+ self.propose_timeout = propose_timeout
361
+ self.propose_max_turns = propose_max_turns
362
+ self._configure_git_credentials()
363
+
364
+ # ── the one Engine method ──────────────────────────────────────────────────────────────
365
+
366
+ def judge(self, *, system: str, prompt: str, schema: dict | None = None) -> dict:
367
+ """The one `Engine` method, serving both of this actor's doors.
368
+
369
+ The port is a single `judge()`, and both doors name the same engine key, so the dispatch
370
+ is on the door id `Actor.judge()` already puts on line 2 of the prompt. The two paths are
371
+ deliberately asymmetric: `test-task` hands its clones off live to the handler, which
372
+ commits, pushes, publishes and then removes them; `propose-acceptance` owns its clones
373
+ from end to end and writes nothing anywhere.
374
+ """
375
+ door = _door_from_prompt(prompt)
376
+ if door == PROPOSE_DOOR:
377
+ return self._propose(_payload_from_prompt(prompt), schema)
378
+ if door == TEST_DOOR:
379
+ return self._test(system, _payload_from_prompt(prompt))
380
+ raise EngineError(
381
+ f"this engine answers {TEST_DOOR} and {PROPOSE_DOOR}, not '{door}' — a door naming "
382
+ f"this engine must be one it knows how to judge"
383
+ )
384
+
385
+ # ── test-task: the door that authors ───────────────────────────────────────────────────
386
+
387
+ def _test(self, system: str, payload: dict) -> dict:
388
+ task_id = payload["task_id"]
389
+ # The FIRST thing this door does, before anything that could fail: from here to the end
390
+ # of this request's own thread, every record — this module's, handler.py's, and the HTTP
391
+ # binding's own access line — carries both ids. `correlation_id()` reads the trace id the
392
+ # orchestrating actor already propagated, so every actor in the pipeline agrees on it
393
+ # without any of them passing it (see correlation.py).
394
+ correlation.bind(correlation_id=correlation.correlation_id(), task_id=task_id)
395
+
396
+ # Both resolved before a single clone: an undeclared component and an unset registry are
397
+ # facts about the request and the environment, and neither needs a network to find out.
398
+ components = self._components(payload["components"])
399
+ registry = _image_registry()
400
+
401
+ test_clone = Path(tempfile.mkdtemp(prefix=self.config.clone_prefix(task_id)))
402
+ code_clone = Path(tempfile.mkdtemp(prefix=self.config.code_clone_prefix(task_id)))
403
+ branch = TEST_BRANCH.format(task_id=task_id)
404
+ try:
405
+ with correlation.stage("clone-tests", repo=self.config.source_repo):
406
+ self._clone(test_clone, self.config.source_repo)
407
+ with correlation.stage("clone-code", repo=self.config.implementation_repo):
408
+ self._clone(code_clone, self.config.implementation_repo)
409
+ # The implementation actor committed, and computed its published tags, on
410
+ # impl/<task_id> — never on the default branch, because no pull request is opened
411
+ # until orchestration has a passing verdict. Checked out HERE ONLY, to recompute those
412
+ # tags; this clone is never shown to the session below.
413
+ self._git(code_clone, ["checkout", IMPLEMENTATION_BRANCH.format(task_id=task_id)])
414
+ self._git(test_clone, ["checkout", "-b", branch])
415
+
416
+ images = [self._resolve_image(code_clone, component.name, task_id, registry)
417
+ for component in components]
418
+ correlation.event("images-under-test", components=[c.name for c in components],
419
+ images=images)
420
+
421
+ self._ground(test_clone)
422
+
423
+ situational_prompt = self._situational_prompt(payload, components, images)
424
+ with correlation.stage("claude-session", branch=branch,
425
+ max_turns=self.max_turns, timeout_s=self.session_timeout):
426
+ summary = self._invoke_claude(test_clone, system, situational_prompt)
427
+ except BaseException:
428
+ # Every failure path removes both clones. Success does not: they are handed off live,
429
+ # and `handler.py` is what removes them once it has committed, pushed and published —
430
+ # or refused for containment.
431
+ _rmtree(test_clone)
432
+ _rmtree(code_clone)
433
+ raise
434
+
435
+ return {
436
+ "testable": True,
437
+ "clone_dir": str(test_clone),
438
+ "code_clone_dir": str(code_clone),
439
+ "branch": branch,
440
+ "summary": summary,
441
+ }
442
+
443
+ def _components(self, names: list) -> list[Component]:
444
+ """The declared components a `test-task` payload names, in the order it names them.
445
+
446
+ REFUSED, NOT FILTERED. The hand-written actor this replaces silently dropped a name it had
447
+ no tests root for, and answered for the rest — which reads, downstream, exactly like a
448
+ component whose tests were written. A caller naming a component this use does not declare
449
+ is told so before any session time is spent.
450
+ """
451
+ resolved, unknown = [], []
452
+ for name in names:
453
+ component = self.config.component(str(name))
454
+ (resolved if component else unknown).append(component or name)
455
+ if unknown:
456
+ declared = ", ".join(c.name for c in self.config.components)
457
+ raise EngineError(
458
+ f"no declared component named {', '.join(map(str, unknown))} — this use writes "
459
+ f"tests for {declared}"
460
+ )
461
+ if not resolved:
462
+ raise EngineError("the payload names no component to write tests for")
463
+ return resolved
464
+
465
+ def _resolve_image(self, code_clone: Path, component: str, task_id: str,
466
+ registry: str) -> str:
467
+ """Recompute the ref the implementation actor published, over identical inputs.
468
+
469
+ Including `normalize_name(task_id)`, which that actor's own publisher passes. These two
470
+ must agree exactly: a divergence does not raise, it silently names an image that was never
471
+ built. The folder is `<component>/` in the implementation clone — see the schema's
472
+ `components` doc for why a component's name is also where its implementation lives.
473
+ """
474
+ try:
475
+ version = compute_version(
476
+ folder=code_clone / component,
477
+ name=self.config.image_name(component),
478
+ label="feature",
479
+ feature_name=normalize_name(task_id),
480
+ )
481
+ except ValueError as e:
482
+ raise EngineError(f"could not resolve {component}'s published image: {e}") from e
483
+ return self.config.image_ref(registry, component, version)
484
+
485
+ # ── propose-acceptance: the door that only answers ─────────────────────────────────────
486
+
487
+ def _propose(self, payload: dict, schema: dict | None) -> dict:
488
+ """Propose what this actor will assert, black-box, before anything is built. Write nothing.
489
+
490
+ THE CLONES ARE OWNED HERE, END TO END. `_test` hands its clones off live because the
491
+ handler still has to commit and push. This door has no handler and produces no artifact,
492
+ so the `finally` is unconditional — success removes both clones exactly as failure does.
493
+
494
+ NO BRANCH IS CHECKED OUT, IN EITHER CLONE. There is nothing to put on one in the testing
495
+ repo, and in the implementation repo there must not be: `impl/<task_id>` may already exist
496
+ from an earlier attempt, and a proposal read off it would be a report of the build dressed
497
+ as a proposal. The default branch is the state the increment is being proposed against.
498
+ """
499
+ task_id = payload["task_id"]
500
+ correlation.bind(correlation_id=correlation.correlation_id(), task_id=task_id)
501
+
502
+ test_clone = Path(tempfile.mkdtemp(prefix=self.config.clone_prefix(task_id)))
503
+ code_clone = Path(tempfile.mkdtemp(prefix=self.config.code_clone_prefix(task_id)))
504
+ try:
505
+ with correlation.stage("clone-tests", repo=self.config.source_repo):
506
+ self._clone(test_clone, self.config.source_repo)
507
+ with correlation.stage("clone-code", repo=self.config.implementation_repo):
508
+ self._clone(code_clone, self.config.implementation_repo)
509
+ self._ground(test_clone)
510
+
511
+ with correlation.stage("propose-session", max_turns=self.propose_max_turns,
512
+ timeout_s=self.propose_timeout):
513
+ answer = self._invoke_claude(
514
+ test_clone, self._propose_system(),
515
+ self._proposal_prompt(payload, schema, code_clone),
516
+ allowed_tools=PROPOSE_TOOLS, max_turns=self.propose_max_turns,
517
+ timeout=self.propose_timeout, add_dirs=(code_clone,),
518
+ )
519
+ finally:
520
+ _rmtree(test_clone)
521
+ _rmtree(code_clone)
522
+
523
+ proposal = _proposal(_extract_json(answer))
524
+ correlation.event("acceptance-proposed",
525
+ expectations=[e["id"] for e in proposal["expectations"]],
526
+ open_questions=len(proposal["open_questions"]))
527
+ return proposal
528
+
529
+ def _propose_system(self) -> str:
530
+ """The system prompt for the propose door.
531
+
532
+ NOT `Actor.judge()`'s own. That one is built from the whole card and ends with "Reply with
533
+ a single JSON object capturing your judgement" — which is right, and is passed through
534
+ unmodified for `test-task`. But it is handed to this engine as `system` on both doors, and
535
+ this door needs the session to spend its turns READING rather than answering from the
536
+ card's prose alone.
537
+ """
538
+ return (
539
+ "You are proposing, not testing. Read the task, the capability context you have been "
540
+ "given, the existing test suite and the implementation as it stands today, then say "
541
+ "precisely what you will assert black-box once the increment is built — and what the "
542
+ "task leaves undetermined. Do not write, edit or create any file; you have no tools "
543
+ "to do so."
544
+ )
545
+
546
+ def _ground(self, clone_dir: Path) -> None:
547
+ """Fetch every `ground_in` source into the clone and render its `CLAUDE.md`.
548
+
549
+ Grounding is a PRECONDITION, not a request — see grounding.py. It runs before either
550
+ door's prompt is built, and a failure here stops the request before any session time is
551
+ spent, which is the cheapest place for it to stop. Both doors ground identically: what
552
+ will be asserted needs the same standing context as writing the assertion.
553
+ """
554
+ for entry in self.config.ground_in:
555
+ with correlation.stage(f"ground-{entry.name}", into=entry.into, load=entry.load):
556
+ envelope = grounding.fetch(self.config, entry, timeout=self.fetch_timeout)
557
+ grounding.write_envelope(self.config, entry, clone_dir, envelope)
558
+ grounding.render_claude_md(self.config, clone_dir)
559
+
560
+ # ── git ─────────────────────────────────────────────────────────────────────────────────
561
+
562
+ def _configure_git_credentials(self) -> None:
563
+ """Make GITHUB_TOKEN available to subprocesses that do their own git clones.
564
+
565
+ A `ground_in` tool typically delegates git auth entirely to git's own credential
566
+ resolution and never takes a token itself. A global URL rewrite is the one hook available
567
+ to make those clones use this token too, without patching the tool. Idempotent; safe to
568
+ call on every construction.
569
+ """
570
+ subprocess.run(
571
+ ["git", "config", "--global",
572
+ f"url.https://x-access-token:{self.github_token}@github.com/.insteadOf",
573
+ "https://github.com/"],
574
+ check=True, capture_output=True, text=True,
575
+ )
576
+
577
+ def _clone(self, dest: Path, repo: str) -> None:
578
+ # Full clone, not --depth 1: `papeete_version.compute()`'s own `semver_base()` does `git
579
+ # describe --tags --match <name>/v*` against a clone — this actor's own, to version its
580
+ # test image, and the implementation's, to recompute the image under test — which needs
581
+ # the matching tag's commit reachable in local history. A shallow clone only has the tip
582
+ # commit, and breaks the moment the tag isn't that exact commit.
583
+ url = f"https://x-access-token:{self.github_token}@github.com/{repo}.git"
584
+ try:
585
+ subprocess.run(
586
+ ["git", "clone", url, str(dest)],
587
+ check=True, capture_output=True, text=True, timeout=self.clone_timeout,
588
+ )
589
+ except subprocess.CalledProcessError as e:
590
+ raise EngineError(
591
+ f"could not clone {repo}: {_redact(e.stderr, self.github_token)}"
592
+ ) from e
593
+ except subprocess.TimeoutExpired as e:
594
+ raise EngineError(f"cloning {repo} timed out after {self.clone_timeout}s") from e
595
+
596
+ def _git(self, clone_dir: Path, args: list[str]) -> str:
597
+ try:
598
+ result = subprocess.run(
599
+ ["git", *args], cwd=clone_dir, check=True, capture_output=True, text=True,
600
+ )
601
+ except subprocess.CalledProcessError as e:
602
+ raise EngineError(
603
+ f"git {' '.join(args)} failed: {_redact(e.stderr, self.github_token)}"
604
+ ) from e
605
+ return result.stdout
606
+
607
+ # ── situational prompt, built from the caller's own payload ───────────────────────────
608
+
609
+ def _situational_prompt(self, payload: dict, components: list[Component],
610
+ images: list[str]) -> str:
611
+ """The task, and nothing else.
612
+
613
+ NOTE WHAT IS ABSENT: any instruction to go and read the capability's context, and any
614
+ path to it. The hand-written actor named two JSON files in a sibling tempdir and asked the
615
+ session to read them "before you start". It is a precondition now — the generated
616
+ `CLAUDE.md` is loaded before turn one — so asking for it again would be asking for
617
+ something already done.
618
+
619
+ THE `<COMPONENT>_URL` CONVENTION STAYS, AS A DEFAULT. It is the one addressing agreement
620
+ this actor and the orchestrating actor have always shared. An agreed acceptance surface
621
+ can now state a handle's own variable, and where it does, the surface is the more specific
622
+ statement and wins.
623
+ """
624
+ config = self.config
625
+ task_id = payload["task_id"]
626
+ roots = ", ".join(c.tests for c in components)
627
+ image_lines = "\n".join(f" - {c.name}: {image}" for c, image in zip(components, images))
628
+ existing = "\n".join(f" - {c.tests}" for c in components)
629
+
630
+ sections = [
631
+ f"# Author/extend black-box tests for {task_id} of {config.capability}: "
632
+ f"{payload['title']}\n\n"
633
+ f"You are working inside your own private clone (branch already checked out). Write "
634
+ f"ONLY under {roots} — nothing else in this clone is yours to change. Do not "
635
+ f"`git add`, `git commit`, or `git push` — that is handled outside this session.\n\n"
636
+ f"This is a PERSISTENT, accumulated test suite, not a fresh one for this task alone "
637
+ f"— read what is already there first and extend or adjust it for {task_id}'s own "
638
+ f"Definition of Done, rather than re-deriving the whole black-box surface from "
639
+ f"scratch:\n{existing}\n\n"
640
+ f"The published image(s) this suite targets (already built elsewhere — you do NOT "
641
+ f"bring these up yourself, and you have NOT been given their source):\n{image_lines}"
642
+ f"\n\n"
643
+ f"Each test reads its target component's base URL from an environment variable named "
644
+ f"<COMPONENT>_URL (uppercase, e.g. BACKEND_URL for a component named backend), unless "
645
+ f"the agreed acceptance surface below names another one in an expectation's handle. "
646
+ f"The orchestrating actor sets it to the running deployment's real address when it "
647
+ f"runs the published test image; you never set or resolve it yourself.\n\n"
648
+ f"You must NOT bring up docker or docker compose, hit a live endpoint, or otherwise "
649
+ f"run these tests against running infrastructure — this session only authors pytest "
650
+ f"source. Ground the exact request/response shapes, status codes and event contracts "
651
+ f"you assert in this capability's standing context and the existing suite's own "
652
+ f"conventions, not by probing a live container. `pytest --collect-only` (no network "
653
+ f"calls) is fine as a syntax/import check; running the suite is not — that verdict "
654
+ f"belongs to the orchestrating actor alone, against a real ephemeral deployment."
655
+ ]
656
+ if payload.get("context"):
657
+ sections.append(f"## Context\n{payload['context']}")
658
+ sections.append(
659
+ "## Definition of done\n"
660
+ + "\n".join(f"- {item}" for item in payload["definition_of_done"])
661
+ )
662
+ if payload.get("acceptance_surface"):
663
+ # Agreed in the round before any of this was built: proposed at this actor's own
664
+ # propose-acceptance door, and accepted (with commitments) by the implementer. The ids
665
+ # are what a later verdict names, which is why every test has to carry one.
666
+ sections.append(
667
+ "## Agreed acceptance surface\n"
668
+ "This was agreed with the implementer before anything was built. Each expectation "
669
+ "below must be asserted by its `id`, addressing it exactly via its `handle` — "
670
+ "including any value committed to there. Where it is more specific than the "
671
+ "definition of done, it wins. Name the expectation id in each test that asserts "
672
+ "it (in the test's name or its docstring), so a failing verdict points at an "
673
+ "agreed expectation rather than at a line of output.\n\n"
674
+ + json.dumps(payload["acceptance_surface"], indent=2, ensure_ascii=False)
675
+ )
676
+ if payload.get("remediation_context"):
677
+ sections.append(
678
+ "## Remediation — the prior attempt's failing criteria\n"
679
+ "A criterion failing could mean the test itself was wrong (bad expectation, "
680
+ "wrong endpoint, wrong shape) or that the implementation was wrong. Review each "
681
+ "one: fix the test here if the fault is in it; if the test looks correct against "
682
+ "this capability's standing context and the agreed acceptance surface, leave it "
683
+ "as-is (the implementation side should fix it instead) and say so in your "
684
+ "summary.\n"
685
+ f"{payload['remediation_context']}"
686
+ )
687
+ return "\n\n".join(sections)
688
+
689
+ # ── the proposal prompt ─────────────────────────────────────────────────────────────────
690
+
691
+ def _proposal_prompt(self, payload: dict, schema: dict | None, code_clone: Path) -> str:
692
+ """What the propose door asks. Nothing about writing, because it cannot.
693
+
694
+ PROPOSE VALUES, DO NOT WAIT FOR THEM. The failure this round exists for is a test tree
695
+ carrying fixture ids "discovered black-box against the running container", against a task
696
+ that never named them (ADR-FIA-0004). Where a test will need a concrete value the task
697
+ leaves unnamed, the session proposes one here, and the implementer commits to it or
698
+ objects at its own assess door. A value a tester proposes before the build is a promise
699
+ someone must keep; the same value read off the build is a report.
700
+
701
+ QUESTIONS ARE NOT FAILURES. What neither the task nor the standing context determines —
702
+ and what no proposed value can responsibly settle, because it is a question of what the
703
+ behaviour SHOULD be — goes in `open_questions`, never into an invented statement. Any open
704
+ question stops the round and reaches a human, which is the point of asking it.
705
+
706
+ THE ANSWER'S SHAPE IS DERIVED, NOT INVENTED HERE. `Actor.judge()` computes it from the
707
+ door's own `completion_schema` and hands it over as `schema`; rendering that is how the
708
+ prompt and the card cannot come to disagree. When it is absent — an engine driven directly,
709
+ in a test or a probe — the fallback below says the same thing in words.
710
+ """
711
+ config = self.config
712
+ task_id = payload["task_id"]
713
+ names = [str(name) for name in payload.get("components") or []]
714
+ component_lines = []
715
+ for name in names:
716
+ declared = config.component(name)
717
+ component_lines.append(
718
+ f" - {name}: its tests live under {declared.tests}" if declared else
719
+ f" - {name}: (this use declares no tests root for it — say so in open_questions "
720
+ f"if the task needs it asserted)")
721
+
722
+ sections = [
723
+ f"# What will you assert for {task_id} of {config.capability}: {payload['title']}?\n\n"
724
+ f"NOTHING HAS BEEN BUILT YET. The implementer has not started. Before it does, you "
725
+ f"propose the acceptance surface: exactly what you will assert, black-box, about this "
726
+ f"increment once it exists. The implementer then answers whether it can deliver each "
727
+ f"expectation, commits to the values a test will address, or objects.\n\n"
728
+ f"You are READING ONLY. Do not write, edit or create any file — you have no tools to "
729
+ f"do so, and there is no branch and no commit at this door.\n\n"
730
+ f"What you can read:\n"
731
+ f" - your working directory: this actor's own testing repository, with the existing "
732
+ f"accumulated suite under {', '.join(config.writes_only_under)}\n"
733
+ f" - {code_clone}: {_CODE_CLONE_LABEL} — the capability as it stands BEFORE this "
734
+ f"increment. Use it to align with what already exists; never treat it as the "
735
+ f"increment itself\n"
736
+ f" - this capability's standing context, already loaded for this session\n\n"
737
+ f"Propose each expectation as an object with:\n"
738
+ f" - `id`: short and stable (`E1`, or kebab-case), unique within this proposal — a "
739
+ f"later verdict names it\n"
740
+ f" - `statement`: what must hold, observable from outside the component\n"
741
+ f" - `handle`: exactly how a black-box test reaches it — the component, the endpoint "
742
+ f"(method and path), the event routing key, the environment variable carrying its "
743
+ f"base URL (<COMPONENT>_URL unless you need another)\n"
744
+ f" - `component` (optional): which of the components below it belongs to\n\n"
745
+ f"Where a test needs a concrete value the task does not name — a fixture's id, a "
746
+ f"seeded record, a path parameter, a routing key — PROPOSE A CONCRETE VALUE in the "
747
+ f"handle. The implementer will accept it and commit to it, or object. Do not leave it "
748
+ f"for the implementation to choose and for a test to discover afterwards.\n\n"
749
+ f"Anything the task and the standing context do not determine, and that no proposed "
750
+ f"value can settle because it is a question of what the behaviour should BE, goes in "
751
+ f"`open_questions` — never into an invented statement. An open question is not a "
752
+ f"failure: it stops this round and reaches a human, which is exactly what it is for."
753
+ ]
754
+ sections.append(
755
+ "## Components this round concerns\n"
756
+ + ("\n".join(component_lines) if component_lines else " (none named)")
757
+ )
758
+ if payload.get("context"):
759
+ sections.append(f"## Context\n{payload['context']}")
760
+ sections.append(
761
+ "## Definition of done\n"
762
+ + "\n".join(f"- {item}" for item in payload["definition_of_done"])
763
+ )
764
+ sections.append(
765
+ "## Your answer\n"
766
+ "End with a single fenced ```json block and nothing after it, holding one object"
767
+ + (f" conforming to:\n\n```json\n{json.dumps(schema, indent=2)}\n```"
768
+ if schema else
769
+ " with `expectations` (a list of objects, each with `id`, `statement` and "
770
+ "`handle`) and `open_questions` (a list of strings, or of objects with `about` and "
771
+ "`question`; empty when the task determines everything).")
772
+ )
773
+ return "\n\n".join(sections)
774
+
775
+ # ── the judgement itself: a claude -p session against the checked-out clone ────────────
776
+
777
+ def _invoke_claude(self, clone_dir: Path, system: str, situational_prompt: str, *,
778
+ allowed_tools: str = TEST_TOOLS,
779
+ max_turns: int | None = None,
780
+ timeout: int | None = None,
781
+ add_dirs: tuple[Path, ...] = ()) -> str:
782
+ """Run the session, streaming every turn to the log as it happens.
783
+
784
+ `--output-format stream-json --verbose` rather than `--output-format json`: the latter
785
+ emits ONE object at the end, holding only the final assistant text, so the whole inner
786
+ conversation — what was read, what was run, what was decided — existed nowhere durable
787
+ once the pod went away. The projection above is what makes streaming it affordable.
788
+
789
+ `Popen` rather than `subprocess.run`: reading line by line is what lets each turn be
790
+ logged as it happens rather than after the session ends, and it stops a 30-minute
791
+ session's entire output being buffered in a pod capped at 2Gi.
792
+ """
793
+ max_turns = self.max_turns if max_turns is None else max_turns
794
+ timeout = self.session_timeout if timeout is None else timeout
795
+ cmd = [self.claude_bin, "--print", "--output-format", "stream-json", "--verbose"]
796
+ for directory in add_dirs:
797
+ # BEFORE another option, never last. `--add-dir` takes a variadic list, and a
798
+ # variadic option followed by the positional prompt would swallow the prompt as a
799
+ # second directory. `--append-system-prompt` right after it closes the list.
800
+ cmd += ["--add-dir", str(directory)]
801
+ cmd += [
802
+ "--append-system-prompt", system,
803
+ "--permission-mode", "acceptEdits",
804
+ # The propose door passes a list with no Write, Edit or Bash in it. That is the
805
+ # enforcement, not the prompt's own "you are reading only" — the same discipline as
806
+ # handler.py's containment check standing behind the test door's write boundary.
807
+ "--allowedTools", allowed_tools,
808
+ "--max-turns", str(max_turns),
809
+ situational_prompt,
810
+ ]
811
+ # stderr to a temp file, not a second pipe: nothing drains a second pipe while the
812
+ # stdout loop below runs, so a chatty stderr would fill its buffer and deadlock the
813
+ # session. Merging it into stdout is not an option either — it would corrupt the stream.
814
+ with tempfile.TemporaryFile("w+") as errfile:
815
+ try:
816
+ proc = subprocess.Popen(cmd, cwd=clone_dir, stdout=subprocess.PIPE,
817
+ stderr=errfile, text=True, bufsize=1)
818
+ except FileNotFoundError as e:
819
+ raise EngineError(
820
+ f"'{self.claude_bin}' is not on PATH — install @anthropic-ai/claude-code"
821
+ ) from e
822
+
823
+ # A watchdog, not a deadline checked per line: a session that hangs having emitted
824
+ # nothing would never reach another loop iteration to be checked, and `Popen` has no
825
+ # equivalent of `subprocess.run(timeout=...)` while iterating its output.
826
+ timed_out = threading.Event()
827
+
828
+ def _expire() -> None:
829
+ timed_out.set()
830
+ proc.kill()
831
+
832
+ watchdog = threading.Timer(timeout, _expire)
833
+ watchdog.start()
834
+
835
+ final: dict | None = None
836
+ try:
837
+ for raw in proc.stdout:
838
+ try:
839
+ event = json.loads(raw)
840
+ except json.JSONDecodeError:
841
+ continue # a non-JSON line is noise, never the session's result
842
+ record = _project(event)
843
+ if record is not None:
844
+ # No %-args: `logging` only applies %-formatting when args are passed,
845
+ # so a stray % in a file body cannot raise here. No `extra=` either —
846
+ # `correlation.bind()` at the top of each door already stamps task_id and
847
+ # correlation_id onto every record emitted on this thread.
848
+ logging.info(_line(record))
849
+ if event.get("type") == "result":
850
+ final = event
851
+ proc.wait()
852
+ finally:
853
+ watchdog.cancel()
854
+ if proc.poll() is None:
855
+ proc.kill()
856
+ proc.wait()
857
+ proc.stdout.close()
858
+
859
+ if timed_out.is_set():
860
+ raise EngineError(
861
+ f"claude session for this task exceeded {timeout}s"
862
+ )
863
+ errfile.seek(0)
864
+ stderr = errfile.read()
865
+
866
+ if final is None:
867
+ raise EngineError(
868
+ f"claude (rc={proc.returncode}) produced no result event: {stderr[-2000:]}"
869
+ )
870
+ if final.get("is_error") or proc.returncode != 0:
871
+ raise EngineError(
872
+ f"claude session failed (subtype={final.get('subtype')}): "
873
+ f"{final.get('result', '')[:4000]}"
874
+ )
875
+ return final.get("result", "")
876
+
877
+
878
+ def _image_registry() -> str:
879
+ registry = os.environ.get("IMAGE_REGISTRY")
880
+ if not registry:
881
+ raise EngineError(
882
+ "no IMAGE_REGISTRY set — cannot name the image the implementation actor published, "
883
+ "nor the test image this door would publish"
884
+ )
885
+ return registry.rstrip("/")
886
+
887
+
888
+ def _rmtree(path: Path) -> None:
889
+ shutil.rmtree(path, ignore_errors=True)