foundry-testing-actor 0.1.1__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/PKG-INFO +25 -7
  2. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/README.md +24 -6
  3. foundry_testing_actor-0.2.0/adr/ADR-FTA-0003-every-expectation-says-how-its-data-comes-to-exist.md +136 -0
  4. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/adr/README.md +1 -0
  5. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/examples/ACME.PARTS.CAP.SUP.007.WID-testing/Dockerfile +2 -2
  6. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/examples/README.md +18 -4
  7. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/pyproject.toml +1 -1
  8. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cards/actor-data.yaml +12 -0
  9. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cards/actor-message.yaml +6 -5
  10. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cards/actor-synchronous-messaging.yaml +8 -2
  11. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/engine.py +141 -10
  12. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_cards.py +7 -4
  13. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_engine.py +82 -4
  14. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/uv.lock +1 -1
  15. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/.github/workflows/ci.yml +0 -0
  16. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/.github/workflows/release.yml +0 -0
  17. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/.gitignore +0 -0
  18. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/CLAUDE.md +0 -0
  19. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/adr/ADR-FTA-0001-the-machinery-leaves-the-capability.md +0 -0
  20. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/adr/ADR-FTA-0002-this-actors-half-of-the-three-amigos-round.md +0 -0
  21. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/adr/template.md +0 -0
  22. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/docker/Dockerfile +0 -0
  23. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/examples/ACME.PARTS.CAP.SUP.007.WID-testing/actor-agentic-context.yaml +0 -0
  24. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/scripts/probe_grounding.py +0 -0
  25. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/__init__.py +0 -0
  26. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cards/actor.yaml +0 -0
  27. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cli.py +0 -0
  28. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/config.py +0 -0
  29. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/conformance.py +0 -0
  30. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/correlation.py +0 -0
  31. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/grounding.py +0 -0
  32. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/handler.py +0 -0
  33. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/instance.py +0 -0
  34. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/runner/Dockerfile +0 -0
  35. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/schemas/agentic-context.schema.yaml +0 -0
  36. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/serve.py +0 -0
  37. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/conftest.py +0 -0
  38. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/fixtures/broken/actor-agentic-context.yaml +0 -0
  39. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_cli.py +0 -0
  40. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_config.py +0 -0
  41. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_conformance.py +0 -0
  42. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_grounding.py +0 -0
  43. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_handler.py +0 -0
  44. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_instance.py +0 -0
  45. {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_portability.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: foundry-testing-actor
3
- Version: 0.1.1
3
+ Version: 0.2.0
4
4
  Summary: Runs a headless Claude Code black-box test-authoring session against one capability's own testing repo — a papeete-actor for one use, with the capability supplied by a sidecar.
5
5
  Project-URL: Homepage, https://github.com/papeete-hub/foundry-testing-actor
6
6
  Author-email: Papeete Consulting <yoann.remy@outlook.com>
@@ -77,7 +77,7 @@ ground_in:
77
77
  and, beside it, a four-line Dockerfile:
78
78
 
79
79
  ```dockerfile
80
- FROM ghcr.io/papeete-hub/foundry-testing-actor:0.1.1
80
+ FROM ghcr.io/papeete-hub/foundry-testing-actor:0.2.0
81
81
  RUN pip install --no-cache-dir kpack==2.0.1 kontract==0.1.0 # what this sidecar's ground_in names
82
82
  COPY actor-agentic-context.yaml /actor/
83
83
  RUN foundry-testing-actor render-cards /actor && foundry-testing-actor lint /actor
@@ -137,9 +137,11 @@ Both name the same engine. `Actor.judge()` hands it the door id, and it dispatch
137
137
 
138
138
  ```
139
139
  orchestration ──▶ testing: propose-acceptance {task_id, title, definition_of_done, components, context?}
140
- ◀── {expectations: [{id, statement, handle, component?}], open_questions: [...]}
141
- orchestration ──▶ implementation: assess-task {..., acceptance_surface: expectations}
140
+ ◀── {expectations: [{id, statement, handle, component?}],
141
+ datasets: [{expectation, via, …}], open_questions: [...]}
142
+ orchestration ──▶ implementation: assess-task {..., acceptance_surface: expectations} ← never datasets
142
143
  ◀── {feasible, objections, commitments}
144
+ orchestration ──▶ testing: test-task {..., acceptance_surface, datasets}
143
145
  ```
144
146
 
145
147
  A **query**, read-only: a clone of the testing repo, a clone of the implementation repo's **default
@@ -154,9 +156,25 @@ implementer to commit to or object to; and anything the task does not determine
154
156
  `open_questions`, **never** into an invented statement. A non-empty `open_questions` stops the round
155
157
  and reaches a human. See `adr/ADR-FTA-0002-*.md`.
156
158
 
157
- The reply is **projected** onto `{expectations, open_questions}` — a session's stray keys never
158
- reach the orchestrating actor dressed as contract — after checking that every expectation carries
159
- an `id`, a `statement` and a `handle`, with ids unique. `open_questions` is always a list.
159
+ **Every expectation says how its data comes to exist** (`adr/ADR-FTA-0003-*.md`), in a separate
160
+ `datasets` list that is **private to this actor** — the implementer never sees it. One entry per
161
+ expectation, `via` one of:
162
+
163
+ | `via` | the test's state comes from | must carry |
164
+ |---|---|---|
165
+ | `none` | nothing — written down, never inferred from a missing entry | — |
166
+ | `command` | the capability's own commands, called by the test first (preferred) | `steps` |
167
+ | `event` | upstream events the test publishes; broker address in `AMQP_URL` | `steps` |
168
+ | `seed` | pre-existing data that is itself under test | `because`, `provided_by` (a proposed expectation) |
169
+
170
+ Never a component's storage. Anything a dataset needs that the contract does not already offer — a
171
+ seed, a command this task introduces — must also be proposed as an **expectation**, because that is
172
+ the only thing the implementer assesses.
173
+
174
+ The reply is **projected** onto `{expectations, datasets, open_questions}` — a session's stray keys
175
+ never reach the orchestrating actor dressed as contract — after checking that every expectation
176
+ carries an `id`, a `statement` and a `handle`, with ids unique, and has exactly one valid dataset.
177
+ `open_questions` is always a list.
160
178
 
161
179
  ### `test-task`
162
180
 
@@ -54,7 +54,7 @@ ground_in:
54
54
  and, beside it, a four-line Dockerfile:
55
55
 
56
56
  ```dockerfile
57
- FROM ghcr.io/papeete-hub/foundry-testing-actor:0.1.1
57
+ FROM ghcr.io/papeete-hub/foundry-testing-actor:0.2.0
58
58
  RUN pip install --no-cache-dir kpack==2.0.1 kontract==0.1.0 # what this sidecar's ground_in names
59
59
  COPY actor-agentic-context.yaml /actor/
60
60
  RUN foundry-testing-actor render-cards /actor && foundry-testing-actor lint /actor
@@ -114,9 +114,11 @@ Both name the same engine. `Actor.judge()` hands it the door id, and it dispatch
114
114
 
115
115
  ```
116
116
  orchestration ──▶ testing: propose-acceptance {task_id, title, definition_of_done, components, context?}
117
- ◀── {expectations: [{id, statement, handle, component?}], open_questions: [...]}
118
- orchestration ──▶ implementation: assess-task {..., acceptance_surface: expectations}
117
+ ◀── {expectations: [{id, statement, handle, component?}],
118
+ datasets: [{expectation, via, …}], open_questions: [...]}
119
+ orchestration ──▶ implementation: assess-task {..., acceptance_surface: expectations} ← never datasets
119
120
  ◀── {feasible, objections, commitments}
121
+ orchestration ──▶ testing: test-task {..., acceptance_surface, datasets}
120
122
  ```
121
123
 
122
124
  A **query**, read-only: a clone of the testing repo, a clone of the implementation repo's **default
@@ -131,9 +133,25 @@ implementer to commit to or object to; and anything the task does not determine
131
133
  `open_questions`, **never** into an invented statement. A non-empty `open_questions` stops the round
132
134
  and reaches a human. See `adr/ADR-FTA-0002-*.md`.
133
135
 
134
- The reply is **projected** onto `{expectations, open_questions}` — a session's stray keys never
135
- reach the orchestrating actor dressed as contract — after checking that every expectation carries
136
- an `id`, a `statement` and a `handle`, with ids unique. `open_questions` is always a list.
136
+ **Every expectation says how its data comes to exist** (`adr/ADR-FTA-0003-*.md`), in a separate
137
+ `datasets` list that is **private to this actor** — the implementer never sees it. One entry per
138
+ expectation, `via` one of:
139
+
140
+ | `via` | the test's state comes from | must carry |
141
+ |---|---|---|
142
+ | `none` | nothing — written down, never inferred from a missing entry | — |
143
+ | `command` | the capability's own commands, called by the test first (preferred) | `steps` |
144
+ | `event` | upstream events the test publishes; broker address in `AMQP_URL` | `steps` |
145
+ | `seed` | pre-existing data that is itself under test | `because`, `provided_by` (a proposed expectation) |
146
+
147
+ Never a component's storage. Anything a dataset needs that the contract does not already offer — a
148
+ seed, a command this task introduces — must also be proposed as an **expectation**, because that is
149
+ the only thing the implementer assesses.
150
+
151
+ The reply is **projected** onto `{expectations, datasets, open_questions}` — a session's stray keys
152
+ never reach the orchestrating actor dressed as contract — after checking that every expectation
153
+ carries an `id`, a `statement` and a `handle`, with ids unique, and has exactly one valid dataset.
154
+ `open_questions` is always a list.
137
155
 
138
156
  ### `test-task`
139
157
 
@@ -0,0 +1,136 @@
1
+ ---
2
+ id: ADR-FTA-0003
3
+ title: "Every expectation says how its data comes to exist — privately, and anything the contract cannot already provide becomes an expectation"
4
+ status: Proposed
5
+ date: 2026-09-14
6
+ supersedes: []
7
+ references:
8
+ - ../src/foundry_testing_actor/cards/actor-data.yaml
9
+ - ../src/foundry_testing_actor/cards/actor-synchronous-messaging.yaml
10
+ - ../src/foundry_testing_actor/engine.py
11
+ - ./ADR-FTA-0002-this-actors-half-of-the-three-amigos-round.md
12
+ - https://github.com/papeete-hub/foundry-task-orchestration-actor/blob/main/adr/ADR-FTOA-0003-datasets-travel-to-the-tester-only.md
13
+ - https://github.com/papeete-hub/foundry-implementation-actor/blob/main/adr/ADR-FIA-0004-the-three-amigos-round.md
14
+ ---
15
+
16
+ # ADR-FTA-0003 — Every expectation says how its data comes to exist
17
+
18
+ ## Context
19
+
20
+ ADR-FTA-0002 made this actor propose its acceptance surface before anything is built, and told it
21
+ to **propose a concrete value** wherever a test needs one the task leaves unnamed: "a fixture's id,
22
+ a seeded record". It said nothing about *where the data a test runs against comes from*, and so the
23
+ default was the one the whole round exists to remove — data that already exists, found by reading.
24
+
25
+ The second live run showed it. Asked for a deliberately vague task, the propose session read the
26
+ implementation's `stub/fixtures/anchors.json`, proposed the three anchors it found there at their
27
+ existing ids, and stopped the round on a question about how a consumer would ever learn those ids.
28
+ Every expectation depended on pre-existing records. None needed to: the capability's own contract
29
+ has a command that creates an anchor.
30
+
31
+ Two things are true of every capability this actor will ever serve, not only of the one that
32
+ showed it:
33
+
34
+ 1. **A black-box test can almost always build its own state** through the contract under test —
35
+ the commands it exposes, the upstream events it consumes. A test that does so needs no agreed
36
+ fixture, cannot be broken by someone regenerating one, and does not care what the database
37
+ already holds.
38
+ 2. **Where it cannot**, that is a fact the implementer has to know about: a command the task must
39
+ introduce, or seeded data at fixed values that someone has to put there and keep there.
40
+
41
+ A first design put the dataset inside each shared expectation (`given`). It was rejected in
42
+ review: for the ordinary case — build state through what the contract already offers — the
43
+ implementer has no use for it, and the one case that does concern the implementer is better stated
44
+ as an expectation of its own than buried in a private detail of another.
45
+
46
+ ## Decision
47
+
48
+ **1. `propose-acceptance` answers a second list, `datasets`, private to this actor.** One entry per
49
+ expectation, keyed by the expectation's `id`:
50
+
51
+ ```json
52
+ {"expectation": "E3",
53
+ "via": "command",
54
+ "steps": ["POST $BACKEND_URL/anchors {valid mint payload} → internal_id",
55
+ "POST $BACKEND_URL/anchors/{internal_id}/archive"]}
56
+ ```
57
+
58
+ `via` is one of four, and nothing else:
59
+
60
+ | `via` | the test's state comes from | the entry must carry |
61
+ |---|---|---|
62
+ | `none` | nothing — the expectation needs no prior state | — (saying so is the point) |
63
+ | `command` | the capability's own commands, called by the test | `steps` |
64
+ | `event` | upstream events the test publishes on the bus | `steps` |
65
+ | `seed` | data that exists before the test runs | `because`, and `provided_by` naming the expectation that states the seed |
66
+
67
+ **2. Every expectation has exactly one dataset — and this is checked, not asked.** The engine
68
+ refuses a proposal where an expectation has no entry, an entry names an unknown expectation or has
69
+ two, `via` is outside the four, `command`/`event` has no `steps`, or `seed` has no `because` or a
70
+ `provided_by` that is not a proposed expectation. The refusal happens at this door, where a
71
+ proposal is still cheap; `none` must be written, not inferred, because "needs nothing" is a claim
72
+ that can be wrong.
73
+
74
+ **3. Anything a dataset needs that the contract does not already offer is promoted to an
75
+ expectation.** A seed is the plainest case: `provided_by` must point at an expectation whose
76
+ `statement` and `handle` describe the seeded data — which is what the implementer assesses and
77
+ commits to. The same holds for a command or a subscription the task itself introduces: it is
78
+ something the increment must deliver, so it is an expectation, and a dataset may then use it.
79
+ `seed` is for data whose existence *is* what is being tested (a stub's canned records); the prompt
80
+ says to prefer `command` and `event` everywhere else.
81
+
82
+ **4. `datasets` never reaches the implementer.** The shared expectation stays `{id, statement,
83
+ handle}` with optional `component`. The orchestrating actor relays `datasets` unread to
84
+ `test-task`, and never to `assess-task` or `implement-task` (ADR-FTOA-0003).
85
+
86
+ **5. `test-task` takes the datasets and builds state exactly as they say.** `test-task-cmd` gains an
87
+ optional `datasets`. Each test arranges its state as its expectation's dataset states; it does not
88
+ rely on records it did not create, except through a `seed` dataset. Without datasets — a caller that
89
+ skipped round 0 — the prompt still says to build state through the contract's commands.
90
+
91
+ **6. How a test reaches the bus is a convention, like `<COMPONENT>_URL`.** An `event` dataset
92
+ publishes to the broker whose address arrives in `AMQP_URL`. The orchestrating actor sets it on the
93
+ test Job when its use declares one (ADR-FTOA-0003).
94
+
95
+ **7. No `via: database`.** A test never writes a component's storage. Doing so would bind the
96
+ tests to the implementer's private schema — the coupling black-box testing exists to prevent — and
97
+ a state the contract cannot produce is a question about the contract, which belongs in
98
+ `open_questions`.
99
+
100
+ ## Rationale
101
+
102
+ **Why a separate list rather than a field on the expectation.** The expectation is the part of the
103
+ proposal two actors agree on. How a test sets itself up is how this actor does its job; putting it
104
+ on the shared object would put it in front of an implementer that can only ignore it, and — worse —
105
+ would let a genuine dev↔test dependency ride along unseen instead of being stated where it gets
106
+ assessed.
107
+
108
+ **Why keyed by id and validated here.** It is the same reason ids are unique (ADR-FTA-0002 §6): the
109
+ orchestrating actor and a verdict both address expectations by id. A dataset that names nothing, or
110
+ an expectation with none, is exactly the silent gap this ADR closes, and this door is the last place
111
+ it costs nothing to refuse.
112
+
113
+ **Why relay through the orchestrator instead of re-deriving at `test-task`.** The plan a test is
114
+ written against should be the one that existed when the implementer agreed to the surface. A
115
+ `test-task` session re-deciding how to arrange data, after the increment is built, would be free to
116
+ read state off the build again — the *after* ADR-FIA-0004 excludes.
117
+
118
+ **Why the promotion rule rather than trusting the dataset.** A private plan is only safe if it can
119
+ not hide a demand on someone else. The rule makes every such demand public in the one shape the
120
+ implementer already assesses. Where a session breaks it anyway — assumes a command that does not
121
+ exist — the first attempt's tests fail against the real deployment, and remediation names them.
122
+
123
+ ## Consequences
124
+
125
+ - **The completion shape of `propose-acceptance` changes**: `datasets` is required. A use pinned to
126
+ an older image keeps the older door; a use that takes this version gets it through its `FROM`
127
+ line and `render-cards`, with nothing written by hand. Released as `0.2.0`.
128
+ - **A proposal can now be refused for a missing dataset**, which the orchestrating actor reports as
129
+ a round-0 failure — `testing did not answer`, with the reason. That is intended: a proposal with
130
+ no plan for its data is not a proposal.
131
+ - **`remediation_context` gains a failure class**: a dataset that could not be built. The
132
+ `test-task` prompt already asks the session to judge whether a failure is the test's or the
133
+ implementation's; it now also names this one.
134
+ - **Not decided here**: whether datasets should be committed beside the tests that use them (they
135
+ are in the tests' code, and in the transcript), and whether `open_questions` should be split into
136
+ blocking and non-blocking — raised by the same run, still open.
@@ -17,6 +17,7 @@ are cited rather than restated.
17
17
  |----|-------|--------|
18
18
  | [ADR-FTA-0001](./ADR-FTA-0001-the-machinery-leaves-the-capability.md) | The machinery leaves the capability — a published testing actor any capability can instantiate | Proposed |
19
19
  | [ADR-FTA-0002](./ADR-FTA-0002-this-actors-half-of-the-three-amigos-round.md) | This actor's half of the three amigos round — propose before anything is built, read-only, and hand the unknowns to a human | Proposed |
20
+ | [ADR-FTA-0003](./ADR-FTA-0003-every-expectation-says-how-its-data-comes-to-exist.md) | Every expectation says how its data comes to exist — privately, and anything the contract cannot already provide becomes an expectation | Proposed |
20
21
 
21
22
  ## Authoring
22
23
 
@@ -13,8 +13,8 @@
13
13
  # The default names the product registry rather than GHCR, as the implementation actor's example
14
14
  # does, because it is the copy an in-cluster builder can pull (ADR-FIA-0006's arrangement, copied).
15
15
  # Override it if you are somewhere else:
16
- # docker build --build-arg ACTOR_IMAGE=ghcr.io/papeete-hub/foundry-testing-actor:0.1.1 .
17
- ARG ACTOR_IMAGE=papeetefoundry.azurecr.io/foundry/foundry-testing-actor:0.1.1
16
+ # docker build --build-arg ACTOR_IMAGE=ghcr.io/papeete-hub/foundry-testing-actor:0.2.0 .
17
+ ARG ACTOR_IMAGE=papeetefoundry.azurecr.io/foundry/foundry-testing-actor:0.2.0
18
18
  FROM ${ACTOR_IMAGE}
19
19
 
20
20
  # THE KNOWLEDGE TOOLS THIS SIDECAR NAMES, and the one thing the base image deliberately does not
@@ -22,7 +22,7 @@ repo and publishes them as a runnable image.**
22
22
  definition_of_done, ground the session ─▶ the knowledge tools
23
23
  components, context? propose expectations the sidecar names
24
24
  (Read/Glob/Grep only)
25
- ◀────────────────────────────── {expectations: [{id, statement, handle}], open_questions}
25
+ ◀────────────────────────────── {expectations, datasets (private), open_questions}
26
26
 
27
27
  POST /test-task ─────▶ clone testing repo → test/TASK-NNN
28
28
  ...the same, plus clone implementation → impl/TASK-NNN ─▶ recompute images under test
@@ -75,8 +75,8 @@ python scripts/probe_grounding.py examples/ACME.PARTS.CAP.SUP.007.WID-testing
75
75
  **Build it** — this needs the base image, from a registry you can pull, or one you built:
76
76
 
77
77
  ```bash
78
- uv build && docker build -f docker/Dockerfile -t foundry-testing-actor:0.1.1 .
79
- docker build --build-arg ACTOR_IMAGE=foundry-testing-actor:0.1.1 -t acme-wid-tester \
78
+ uv build && docker build -f docker/Dockerfile -t foundry-testing-actor:0.2.0 .
79
+ docker build --build-arg ACTOR_IMAGE=foundry-testing-actor:0.2.0 -t acme-wid-tester \
80
80
  examples/ACME.PARTS.CAP.SUP.007.WID-testing
81
81
  docker run --rm acme-wid-tester foundry-testing-actor lint /actor
82
82
  ```
@@ -102,10 +102,24 @@ curl -X POST http://<actor>/propose-acceptance -H 'Content-Type: application/jso
102
102
  "handle": {"component": "stub", "base_url_env": "STUB_URL", "method": "GET",
103
103
  "path": "/widgets/{id}",
104
104
  "ids": {"ACTIVE": "w-0001", "ARCHIVED": "w-0002", "RETIRED": "w-0003"}},
105
- "component": "stub"}],
105
+ "component": "stub"},
106
+ {"id": "retired-widget-refuses-update",
107
+ "statement": "PATCH /widgets/{id} on a retired widget answers 409 WIDGET_RETIRED",
108
+ "handle": "PATCH $BACKEND_URL/widgets/{id}", "component": "backend"}],
109
+ "datasets": [
110
+ {"expectation": "stub-seeded-widgets", "via": "seed",
111
+ "because": "the stub's canned widgets are themselves what is under test",
112
+ "provided_by": "stub-seeded-widgets"},
113
+ {"expectation": "retired-widget-refuses-update", "via": "command",
114
+ "steps": ["POST $BACKEND_URL/widgets {valid payload} → id",
115
+ "POST $BACKEND_URL/widgets/{id}/retire"]}],
106
116
  "open_questions": []}
107
117
  ```
108
118
 
119
+ `datasets` stays with this actor: the orchestrating actor relays it to `test-task` and never to the
120
+ implementer. The second test creates the widget it asserts on, so it needs no agreed fixture; the
121
+ first relies on seeded data, which is why that data is an expectation the implementer assesses.
122
+
109
123
  The ids are **proposed**, not discovered: the implementer commits to them at its `assess-task`
110
124
  door, or objects. Then `test-task` is sent the agreed surface as `acceptance_surface`, and
111
125
  answers `{"accepted": true, "branch": "test/TASK-014", "images":
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "foundry-testing-actor"
3
- version = "0.1.1"
3
+ version = "0.2.0"
4
4
  description = "Runs a headless Claude Code black-box test-authoring session against one capability's own testing repo — a papeete-actor for one use, with the capability supplied by a sidecar."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -45,6 +45,18 @@ items:
45
45
  reaches it, and optionally the `component` it belongs to. Proposed BEFORE anything is built,
46
46
  so a concrete value in a handle is a proposal the implementer commits to or objects to, never
47
47
  a report of what was found
48
+ - name: datasets
49
+ type: list
50
+ description: >-
51
+ PRIVATE TO THIS ACTOR — how the state each proposed expectation's test runs against comes to
52
+ exist, one object per expectation: `expectation` (its id) and `via`, one of `none` (needs no
53
+ prior state), `command` (the test calls the capability's own commands first; with `steps`),
54
+ `event` (the test publishes upstream events, broker address in AMQP_URL; with `steps`) or
55
+ `seed` (pre-existing data that is itself under test; with `because`, and `provided_by`
56
+ naming the proposed expectation that states it). Never a component's storage. Relayed
57
+ unread to test-task by the orchestrating actor and never shown to the implementer: anything
58
+ a dataset needs that the contract does not already offer is proposed as an expectation too
59
+ (ADR-FTA-0003)
48
60
  - name: open_questions
49
61
  type: list
50
62
  description: >-
@@ -12,16 +12,17 @@ messages:
12
12
 
13
13
  - name: acceptance-proposed-result
14
14
  intent: >-
15
- answer with the proposed expectations, each with a stable id, a statement and a handle, and
16
- the questions the task leaves open
17
- references: [expectations, open_questions]
15
+ answer with the proposed expectations, each with a stable id, a statement and a handle;
16
+ how the state each one's test runs against comes to exist (private to this actor); and the
17
+ questions the task leaves open
18
+ references: [expectations, datasets, open_questions]
18
19
  optional: [open_questions]
19
20
 
20
21
  - name: test-task-cmd
21
22
  intent: ask this actor to author black-box tests for the published image(s) of a TASK-NNN card
22
23
  references: [task_id, title, definition_of_done, components, context, remediation_context,
23
- acceptance_surface]
24
- optional: [context, remediation_context, acceptance_surface]
24
+ acceptance_surface, datasets]
25
+ optional: [context, remediation_context, acceptance_surface, datasets]
25
26
 
26
27
  - name: test-task-result
27
28
  intent: report that the tests were written, pushed, and their touched components' test images published
@@ -19,14 +19,16 @@ actions:
19
19
  means: >-
20
20
  the door for "author black-box tests for TASK-NNN of the capability I serve, now that its
21
21
  increment is built". Send it here with the task id, title, definition of done, components,
22
- and (optionally) context, remediation_context and the agreed `acceptance_surface` — the
22
+ and (optionally) context, remediation_context, the agreed `acceptance_surface` and the
23
+ `datasets` I proposed beside it — the
23
24
  caller supplies everything, no task card is looked up here. I clone my capability's testing
24
25
  repository into my own private copy, and read-only clone its implementation repository at
25
26
  impl/TASK-NNN solely to recompute, by convention, the image refs published there (never
26
27
  passed to me, and never shown to my session); ground myself in that capability's own
27
28
  standing context; and extend the persistent suite under each named component's declared
28
29
  tests root — asserting every agreed expectation by its id, through its handle, where a
29
- surface was sent. I never bring up the images, hit a live endpoint, or render a verdict.
30
+ surface was sent, each test building its state as its dataset says, or through the
31
+ contract's own commands where none was sent. I never bring up the images, hit a live endpoint, or render a verdict.
30
32
  Then I commit and push to test/TASK-NNN, and build each touched component's test image in
31
33
  the cluster's shared buildkit and push it to the registry, named and versioned by
32
34
  convention. I never open a pull request — an orchestrating actor runs that image and does.
@@ -51,6 +53,10 @@ queries:
51
53
  reaches it (the component, the endpoint, the event routing key, the environment variable its
52
54
  base URL arrives in — with a concrete value proposed wherever a test needs one the task does
53
55
  not name, for the implementer to commit to or object to), and optionally its `component`;
56
+ for EVERY expectation, a dataset saying how the state its test needs comes to exist — `none`,
57
+ `command` (preferred), `event`, or `seed` only where pre-existing data is itself under test
58
+ and a proposed expectation states it — kept private to me: anything a dataset needs that the
59
+ contract does not already offer is proposed as an expectation too (ADR-FTA-0003);
54
60
  and the open questions the task and the standing context leave undetermined, each a string
55
61
  or an `about`/`question` object, never an invented answer. Any open question is meant to
56
62
  stop the round and reach a human rather than a session.
@@ -86,6 +86,14 @@ DEFAULT_MAX_TURNS = 60
86
86
  DEFAULT_PROPOSE_TIMEOUT_S = 900
87
87
  DEFAULT_PROPOSE_MAX_TURNS = 30
88
88
 
89
+ # How the state a test runs against comes to exist (ADR-FTA-0003). Closed on purpose: a fifth way
90
+ # is a decision, not a session's improvisation — and `database` is deliberately not one of them.
91
+ DATASET_VIA = ("none", "command", "event", "seed")
92
+
93
+ # Where an `event` dataset reaches the bus: the broker address arrives in this variable, set on the
94
+ # test Job by the orchestrating actor, the way <COMPONENT>_URL is (ADR-FTA-0003 §6, ADR-FTOA-0003).
95
+ BROKER_URL_ENV = "AMQP_URL"
96
+
89
97
  TEST_TOOLS = "Bash,Read,Edit,Write,Glob,Grep"
90
98
  PROPOSE_TOOLS = "Read,Glob,Grep"
91
99
 
@@ -293,8 +301,8 @@ def _proposal(judged: dict) -> dict:
293
301
  the framework (ADR-PAS-0009), so a session's helpful `"notes"` key beside its answer would not
294
302
  be refused — it would travel to the orchestrating actor looking like part of a contract three
295
303
  packages share, which is worse. And the day this door names a second outcome, the framework
296
- closes the schema and that same key refuses the whole proposal. Only the two fields the message
297
- references survive; everything else a session said is in the transcript.
304
+ closes the schema and that same key refuses the whole proposal. Only the three fields the
305
+ message references survive; everything else a session said is in the transcript.
298
306
 
299
307
  WHAT IS CHECKED, AND WHY IT IS NOT MORE. `expectations` must be a list of objects, each with an
300
308
  `id` and a `statement`, a `handle` key, and ids unique within the proposal — because a later
@@ -328,7 +336,81 @@ def _proposal(judged: dict) -> dict:
328
336
  f"expectation by its id, so an id two expectations share names neither"
329
337
  )
330
338
  seen.add(identifier)
331
- return {"expectations": expectations, "open_questions": _as_list(judged.get("open_questions"))}
339
+ datasets = _datasets(judged.get("datasets"), [str(e["id"]) for e in expectations])
340
+ return {"expectations": expectations, "datasets": datasets,
341
+ "open_questions": _as_list(judged.get("open_questions"))}
342
+
343
+
344
+ def _datasets(raw, expectation_ids: list[str]) -> list:
345
+ """Every expectation's dataset, checked — how the state its test runs against comes to exist.
346
+
347
+ PRIVATE TO THIS ACTOR (ADR-FTA-0003). The implementer never sees this list; the orchestrating
348
+ actor relays it to `test-task` unread. So the checks that matter are the ones that stop a
349
+ private plan from hiding a gap:
350
+
351
+ - exactly one entry per proposed expectation, none for an expectation nobody proposed — "needs
352
+ nothing" is `via: none`, written down, never an absent entry;
353
+ - `via` is one of `DATASET_VIA`, and nothing else — in particular never a component's database;
354
+ - `command` and `event` say how, as `steps`;
355
+ - `seed` says why (`because`) and points `provided_by` at a PROPOSED expectation that states
356
+ the seeded data — because data that must exist before a test runs is something the
357
+ implementer has to put there, and the only place the implementer looks is the expectations.
358
+
359
+ What a step SAYS is not checked, for the reason a handle's content is not: it depends on what
360
+ is being addressed, and a test failing against the real deployment is the check that can.
361
+ """
362
+ if not expectation_ids and raw in (None, []):
363
+ return []
364
+ if raw is None:
365
+ raise EngineError(
366
+ "the proposal carries no `datasets` — every expectation says how the state its test "
367
+ "runs against comes to exist, even when that is `via: none` (ADR-FTA-0003)"
368
+ )
369
+ if not isinstance(raw, list):
370
+ raise EngineError(f"`datasets` is not a list: {raw!r}")
371
+ proposed = set(expectation_ids)
372
+ covered: set[str] = set()
373
+ for index, entry in enumerate(raw):
374
+ where = f"datasets[{index}]"
375
+ if not isinstance(entry, dict):
376
+ raise EngineError(f"{where} is not an object: {entry!r}")
377
+ expectation = str(entry.get("expectation") or "")
378
+ if expectation not in proposed:
379
+ raise EngineError(
380
+ f"{where} is for expectation '{expectation}', which this proposal does not "
381
+ f"propose — a dataset belongs to exactly one proposed expectation")
382
+ if expectation in covered:
383
+ raise EngineError(
384
+ f"expectation '{expectation}' has two datasets — its test builds its state one way")
385
+ covered.add(expectation)
386
+ via = entry.get("via")
387
+ if via not in DATASET_VIA:
388
+ raise EngineError(
389
+ f"{where} (for {expectation}) says via {via!r}; a test's state comes from one of "
390
+ f"{', '.join(DATASET_VIA)} — never a component's own storage")
391
+ if via in ("command", "event"):
392
+ steps = entry.get("steps")
393
+ if not isinstance(steps, list) or not steps:
394
+ raise EngineError(
395
+ f"{where} (for {expectation}) is via {via} but lists no `steps` — say which "
396
+ f"{'commands the test calls' if via == 'command' else 'events it publishes'}")
397
+ if via == "seed":
398
+ if not entry.get("because"):
399
+ raise EngineError(
400
+ f"{where} (for {expectation}) relies on seeded data without saying `because` — "
401
+ f"a seed is for data whose existence is itself under test; everything else is "
402
+ f"built through the contract")
403
+ if str(entry.get("provided_by") or "") not in proposed:
404
+ raise EngineError(
405
+ f"{where} (for {expectation}) relies on seeded data but `provided_by` names no "
406
+ f"proposed expectation — data that must exist before a test runs is something "
407
+ f"the implementer has to provide, so it has to be an expectation of its own")
408
+ missing = [i for i in expectation_ids if i not in covered]
409
+ if missing:
410
+ raise EngineError(
411
+ f"no dataset for expectation(s) {', '.join(missing)} — say how the state each test "
412
+ f"runs against comes to exist, even when that is `via: none`")
413
+ return raw
332
414
 
333
415
 
334
416
  class ClaudeCodeTesterEngine:
@@ -529,6 +611,7 @@ class ClaudeCodeTesterEngine:
529
611
  proposal = _proposal(_extract_json(answer))
530
612
  correlation.event("acceptance-proposed",
531
613
  expectations=[e["id"] for e in proposal["expectations"]],
614
+ datasets={d["expectation"]: d["via"] for d in proposal["datasets"]},
532
615
  open_questions=len(proposal["open_questions"]))
533
616
  return proposal
534
617
 
@@ -679,11 +762,35 @@ class ClaudeCodeTesterEngine:
679
762
  "agreed expectation rather than at a line of output.\n\n"
680
763
  + json.dumps(payload["acceptance_surface"], indent=2, ensure_ascii=False)
681
764
  )
765
+ if payload.get("datasets"):
766
+ # Private to this actor, and relayed unread by the orchestrating actor: the plan this
767
+ # actor proposed for each expectation's data BEFORE the build, so the tests are not
768
+ # free to read state off the build now (ADR-FTA-0003).
769
+ sections.append(
770
+ "## How each test builds the data it runs against\n"
771
+ "You planned this yourself before anything was built. Each test arranges its "
772
+ "state exactly as its expectation's dataset says: `none` needs nothing; "
773
+ "`command` calls the listed commands first and asserts on what they returned; "
774
+ f"`event` publishes the listed events to the broker at `{BROKER_URL_ENV}`; "
775
+ "`seed` relies on the seeded data its `provided_by` expectation states. Do not "
776
+ "rely on any record the test did not create, except through a `seed` dataset, and "
777
+ "never write a component's storage directly.\n\n"
778
+ + json.dumps(payload["datasets"], indent=2, ensure_ascii=False)
779
+ )
780
+ else:
781
+ sections.append(
782
+ "## The data each test runs against\n"
783
+ "Build each test's state through this capability's own contract — call its "
784
+ "commands, or publish the upstream events it consumes (broker address in "
785
+ f"`{BROKER_URL_ENV}`) — and assert on what the test created. Do not rely on "
786
+ "records that happen to exist, and never write a component's storage directly."
787
+ )
682
788
  if payload.get("remediation_context"):
683
789
  sections.append(
684
790
  "## Remediation — the prior attempt's failing criteria\n"
685
791
  "A criterion failing could mean the test itself was wrong (bad expectation, "
686
- "wrong endpoint, wrong shape) or that the implementation was wrong. Review each "
792
+ "wrong endpoint, wrong shape, a dataset it could not build) or that the "
793
+ "implementation was wrong. Review each "
687
794
  "one: fix the test here if the fault is in it; if the test looks correct against "
688
795
  "this capability's standing context and the agreed acceptance surface, leave it "
689
796
  "as-is (the implementation side should fix it instead) and say so in your "
@@ -748,10 +855,33 @@ class ClaudeCodeTesterEngine:
748
855
  f"(method and path), the event routing key, the environment variable carrying its "
749
856
  f"base URL (<COMPONENT>_URL unless you need another)\n"
750
857
  f" - `component` (optional): which of the components below it belongs to\n\n"
751
- f"Where a test needs a concrete value the task does not name — a fixture's id, a "
752
- f"seeded record, a path parameter, a routing key — PROPOSE A CONCRETE VALUE in the "
753
- f"handle. The implementer will accept it and commit to it, or object. Do not leave it "
754
- f"for the implementation to choose and for a test to discover afterwards.\n\n"
858
+ f"Where a test needs a concrete value the task does not name — a path parameter, a "
859
+ f"routing key, a field value — PROPOSE A CONCRETE VALUE in the handle. The implementer "
860
+ f"will accept it and commit to it, or object. Do not leave it for the implementation "
861
+ f"to choose and for a test to discover afterwards.\n\n"
862
+ f"## The data each test runs against\n"
863
+ f"For EVERY expectation, say how the state its test needs comes to exist, in a "
864
+ f"separate `datasets` list — one entry per expectation. This list is yours: the "
865
+ f"implementer never sees it. Each entry is `{{\"expectation\": <id>, \"via\": ...}}`, "
866
+ f"with `via` one of:\n"
867
+ f" - `none`: the expectation needs no prior state (e.g. a health check). Write it "
868
+ f"down; do not leave the entry out.\n"
869
+ f" - `command` (PREFERRED): the test builds its own state by calling this "
870
+ f"capability's own commands first — add `steps`, the calls in order. A test that "
871
+ f"creates what it asserts on needs no agreed fixture and breaks when nobody "
872
+ f"regenerates one.\n"
873
+ f" - `event`: the test publishes the upstream events this capability consumes — add "
874
+ f"`steps`. The broker's address arrives in `{BROKER_URL_ENV}`.\n"
875
+ f" - `seed`: ONLY when pre-existing data is itself what is under test (a stub's "
876
+ f"canned records, say) — add `because`, and `provided_by`: the id of a proposed "
877
+ f"expectation stating that seeded data, so the implementer assesses it.\n"
878
+ f"Never a component's database: a test does not write the implementation's storage.\n"
879
+ f"Existing fixtures in the implementation repository are NOT a reason to use `seed` — "
880
+ f"prefer building the state through the contract.\n"
881
+ f"RULE: anything a dataset needs that this capability's contract does not ALREADY "
882
+ f"offer — a command or subscription this task introduces, seeded data at fixed values "
883
+ f"— must also be proposed as an expectation of its own, because the implementer only "
884
+ f"reads the expectations.\n\n"
755
885
  f"Anything the task and the standing context do not determine, and that no proposed "
756
886
  f"value can settle because it is a question of what the behaviour should BE, goes in "
757
887
  f"`open_questions` — never into an invented statement. An open question is not a "
@@ -773,8 +903,9 @@ class ClaudeCodeTesterEngine:
773
903
  + (f" conforming to:\n\n```json\n{json.dumps(schema, indent=2)}\n```"
774
904
  if schema else
775
905
  " with `expectations` (a list of objects, each with `id`, `statement` and "
776
- "`handle`) and `open_questions` (a list of strings, or of objects with `about` and "
777
- "`question`; empty when the task determines everything).")
906
+ "`handle`), `datasets` (one object per expectation, with `expectation` and `via`, "
907
+ "as described above) and `open_questions` (a list of strings, or of objects with "
908
+ "`about` and `question`; empty when the task determines everything).")
778
909
  )
779
910
  return "\n\n".join(sections)
780
911
 
@@ -78,19 +78,21 @@ def test_propose_acceptance_accepts_exactly_the_contracted_payload():
78
78
  assert offer.request_schema["properties"]["components"]["type"] == "array"
79
79
 
80
80
 
81
- def test_propose_acceptance_replies_with_expectations_and_open_questions():
81
+ def test_propose_acceptance_replies_with_expectations_datasets_and_open_questions():
82
+ """`datasets` is required: every proposal says how each test's state comes to exist
83
+ (ADR-FTA-0003)."""
82
84
  offer = card.load(cards_path()).queries["propose-acceptance"]
83
85
  (completion,) = offer.completion_schema
84
86
  fields, required = _fields(completion)
85
- assert fields == {"expectations", "open_questions"}
86
- assert required == {"expectations"}
87
+ assert fields == {"expectations", "datasets", "open_questions"}
88
+ assert required == {"expectations", "datasets"}
87
89
 
88
90
 
89
91
  def test_test_task_accepts_exactly_the_contracted_payload():
90
92
  offer = card.load(cards_path()).actions["test-task"]
91
93
  fields, required = _fields(offer.request_schema)
92
94
  assert fields == {"task_id", "title", "definition_of_done", "components", "context",
93
- "remediation_context", "acceptance_surface"}
95
+ "remediation_context", "acceptance_surface", "datasets"}
94
96
  assert required == {"task_id", "title", "definition_of_done", "components"}
95
97
 
96
98
 
@@ -114,6 +116,7 @@ def test_the_propose_door_is_answered_by_its_engine_with_no_handler_registered()
114
116
  def judge(self, *, system, prompt, schema=None):
115
117
  assert prompt.splitlines()[1] == "door: propose-acceptance"
116
118
  return {"expectations": [{"id": "E1", "statement": "s", "handle": {}}],
119
+ "datasets": [{"expectation": "E1", "via": "none"}],
117
120
  "open_questions": []}
118
121
 
119
122
  actor = Actor.from_card(cards_path(), engines={"claude-code": StubEngine()})
@@ -17,7 +17,7 @@ from papeete_actor_synchronous_messaging.engine import Engine, EngineError
17
17
 
18
18
  from foundry_testing_actor import engine as engine_module
19
19
  from foundry_testing_actor.engine import (
20
- LINE_BUDGET, PROPOSE_TOOLS, TEST_TOOLS, ClaudeCodeTesterEngine, _door_from_prompt,
20
+ BROKER_URL_ENV, DATASET_VIA, LINE_BUDGET, PROPOSE_TOOLS, TEST_TOOLS, ClaudeCodeTesterEngine, _door_from_prompt,
21
21
  _extract_json, _line, _payload_from_prompt, _project, _proposal,
22
22
  )
23
23
 
@@ -270,13 +270,88 @@ def test_open_questions_always_arrive_as_a_list(given, expected):
270
270
 
271
271
  def test_a_sessions_extra_keys_do_not_reach_the_caller():
272
272
  proposal = _proposal({"expectations": [], "notes": "I also looked at the stub", "x": 1})
273
- assert set(proposal) == {"expectations", "open_questions"}
273
+ assert set(proposal) == {"expectations", "datasets", "open_questions"}
274
274
 
275
275
 
276
276
  def test_extra_keys_inside_an_expectation_survive():
277
277
  """The surface is typed only as a list; an entry's own extra keys are the contract's to allow."""
278
278
  entry = {"id": "E1", "statement": "s", "handle": {}, "component": "backend", "why": "DoD 1"}
279
- assert _proposal({"expectations": [entry]})["expectations"] == [entry]
279
+ proposal = _proposal({"expectations": [entry], "datasets": [NONE("E1")]})
280
+ assert proposal["expectations"] == [entry]
281
+
282
+
283
+ # ── datasets: how each test's state comes to exist (ADR-FTA-0003) ───────────────────────────
284
+
285
+ def NONE(expectation):
286
+ return {"expectation": expectation, "via": "none"}
287
+
288
+
289
+ TWO = [{"id": "E1", "statement": "an archived anchor refuses an update", "handle": "PATCH"},
290
+ {"id": "E2", "statement": "the stub serves anchor 0001", "handle": "GET"}]
291
+
292
+
293
+ def test_every_via_the_contract_allows_is_accepted():
294
+ datasets = [
295
+ {"expectation": "E1", "via": "command", "steps": ["POST /anchors", "POST …/archive"]},
296
+ {"expectation": "E2", "via": "seed", "because": "the canned record IS the subject",
297
+ "provided_by": "E2"},
298
+ ]
299
+ assert _proposal({"expectations": TWO, "datasets": datasets})["datasets"] == datasets
300
+ assert set(DATASET_VIA) == {"none", "command", "event", "seed"}
301
+
302
+
303
+ def test_a_proposal_with_expectations_and_no_datasets_is_refused():
304
+ with pytest.raises(EngineError, match="no `datasets`"):
305
+ _proposal({"expectations": TWO})
306
+
307
+
308
+ def test_an_expectation_without_a_dataset_is_refused_even_when_it_needs_nothing():
309
+ """`none` is a claim that can be wrong, so it is written — never inferred from an absence."""
310
+ with pytest.raises(EngineError, match="no dataset for expectation.*E2"):
311
+ _proposal({"expectations": TWO, "datasets": [NONE("E1")]})
312
+
313
+
314
+ @pytest.mark.parametrize("datasets, match", [
315
+ ([NONE("E1"), NONE("E2"), NONE("E9")], "does not propose"),
316
+ ([NONE("E1"), NONE("E1"), NONE("E2")], "two datasets"),
317
+ ([NONE("E1"), {"expectation": "E2", "via": "database"}], "never a component's own storage"),
318
+ ([NONE("E1"), {"expectation": "E2", "via": "command"}], "lists no `steps`"),
319
+ ([NONE("E1"), {"expectation": "E2", "via": "event", "steps": []}], "lists no `steps`"),
320
+ ([NONE("E1"), {"expectation": "E2", "via": "seed", "provided_by": "E2"}], "without saying"),
321
+ ([NONE("E1"), {"expectation": "E2", "via": "seed", "because": "canned"}], "provided_by"),
322
+ ([NONE("E1"), {"expectation": "E2", "via": "seed", "because": "canned",
323
+ "provided_by": "E7"}], "provided_by"),
324
+ ([NONE("E1"), "E2: none"], "not an object"),
325
+ ])
326
+ def test_a_dataset_that_hides_a_gap_is_refused(datasets, match):
327
+ with pytest.raises(EngineError, match=match):
328
+ _proposal({"expectations": TWO, "datasets": datasets})
329
+
330
+
331
+ def test_an_empty_proposal_needs_no_datasets():
332
+ assert _proposal({"expectations": []})["datasets"] == []
333
+
334
+
335
+ def test_the_proposal_prompt_asks_for_a_dataset_per_expectation_and_prefers_commands(engine,
336
+ tmp_path):
337
+ prompt = engine._proposal_prompt(PAYLOAD, None, tmp_path)
338
+ assert "`datasets`" in prompt and "one entry per expectation" in prompt
339
+ assert "`command` (PREFERRED)" in prompt
340
+ assert "the implementer never sees it" in prompt
341
+ assert "Never a component's database" in prompt
342
+ assert "must also be proposed as an expectation of its own" in prompt
343
+ assert BROKER_URL_ENV in prompt
344
+
345
+
346
+ def test_the_test_prompt_builds_state_as_the_datasets_say(engine, config):
347
+ datasets = [{"expectation": "EXP-001", "via": "command", "steps": ["POST /widgets"]}]
348
+ with_plan = _prompt(engine, config, {**PAYLOAD, "datasets": datasets})
349
+ assert "How each test builds the data it runs against" in with_plan
350
+ assert '"POST /widgets"' in with_plan
351
+
352
+ without = _prompt(engine, config, PAYLOAD)
353
+ assert "Build each test's state through this capability's own contract" in without
354
+ assert "never write a component's storage directly" in without
280
355
 
281
356
 
282
357
  # ── the propose door end to end, without a session ──────────────────────────────────────────
@@ -307,7 +382,8 @@ def proposed(engine, monkeypatch):
307
382
  return _run
308
383
 
309
384
 
310
- ANSWER = '```json\n{"expectations": [{"id": "E1", "statement": "s", "handle": {}}]}\n```'
385
+ ANSWER = ('```json\n{"expectations": [{"id": "E1", "statement": "s", "handle": {}}],'
386
+ ' "datasets": [{"expectation": "E1", "via": "none"}]}\n```')
311
387
 
312
388
 
313
389
  def test_the_propose_session_is_given_no_tool_that_writes(proposed):
@@ -363,8 +439,10 @@ def test_the_propose_door_removes_both_clones_on_failure(proposed):
363
439
  def test_the_proposal_comes_back_as_the_reply(proposed):
364
440
  reply, _ = proposed(
365
441
  '```json\n{"expectations": [{"id": "E1", "statement": "s", "handle": {"path": "/w"}}],'
442
+ ' "datasets": [{"expectation": "E1", "via": "command", "steps": ["POST /w"]}],'
366
443
  ' "open_questions": "which status codes?"}\n```')
367
444
  assert reply == {"expectations": [{"id": "E1", "statement": "s", "handle": {"path": "/w"}}],
445
+ "datasets": [{"expectation": "E1", "via": "command", "steps": ["POST /w"]}],
368
446
  "open_questions": ["which status codes?"]}
369
447
 
370
448
 
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "foundry-testing-actor"
21
- version = "0.1.1"
21
+ version = "0.2.0"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "papeete-actor-synchronous-messaging" },