foundry-testing-actor 0.1.1__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/PKG-INFO +25 -7
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/README.md +24 -6
- foundry_testing_actor-0.2.0/adr/ADR-FTA-0003-every-expectation-says-how-its-data-comes-to-exist.md +136 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/adr/README.md +1 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/examples/ACME.PARTS.CAP.SUP.007.WID-testing/Dockerfile +2 -2
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/examples/README.md +18 -4
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/pyproject.toml +1 -1
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cards/actor-data.yaml +12 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cards/actor-message.yaml +6 -5
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cards/actor-synchronous-messaging.yaml +8 -2
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/engine.py +141 -10
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_cards.py +7 -4
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_engine.py +82 -4
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/uv.lock +1 -1
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/.github/workflows/ci.yml +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/.github/workflows/release.yml +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/.gitignore +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/CLAUDE.md +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/adr/ADR-FTA-0001-the-machinery-leaves-the-capability.md +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/adr/ADR-FTA-0002-this-actors-half-of-the-three-amigos-round.md +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/adr/template.md +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/docker/Dockerfile +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/examples/ACME.PARTS.CAP.SUP.007.WID-testing/actor-agentic-context.yaml +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/scripts/probe_grounding.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/__init__.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cards/actor.yaml +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cli.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/config.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/conformance.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/correlation.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/grounding.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/handler.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/instance.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/runner/Dockerfile +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/schemas/agentic-context.schema.yaml +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/serve.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/conftest.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/fixtures/broken/actor-agentic-context.yaml +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_cli.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_config.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_conformance.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_grounding.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_handler.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_instance.py +0 -0
- {foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/tests/test_portability.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: foundry-testing-actor
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: Runs a headless Claude Code black-box test-authoring session against one capability's own testing repo — a papeete-actor for one use, with the capability supplied by a sidecar.
|
|
5
5
|
Project-URL: Homepage, https://github.com/papeete-hub/foundry-testing-actor
|
|
6
6
|
Author-email: Papeete Consulting <yoann.remy@outlook.com>
|
|
@@ -77,7 +77,7 @@ ground_in:
|
|
|
77
77
|
and, beside it, a four-line Dockerfile:
|
|
78
78
|
|
|
79
79
|
```dockerfile
|
|
80
|
-
FROM ghcr.io/papeete-hub/foundry-testing-actor:0.
|
|
80
|
+
FROM ghcr.io/papeete-hub/foundry-testing-actor:0.2.0
|
|
81
81
|
RUN pip install --no-cache-dir kpack==2.0.1 kontract==0.1.0 # what this sidecar's ground_in names
|
|
82
82
|
COPY actor-agentic-context.yaml /actor/
|
|
83
83
|
RUN foundry-testing-actor render-cards /actor && foundry-testing-actor lint /actor
|
|
@@ -137,9 +137,11 @@ Both name the same engine. `Actor.judge()` hands it the door id, and it dispatch
|
|
|
137
137
|
|
|
138
138
|
```
|
|
139
139
|
orchestration ──▶ testing: propose-acceptance {task_id, title, definition_of_done, components, context?}
|
|
140
|
-
◀── {expectations: [{id, statement, handle, component?}],
|
|
141
|
-
|
|
140
|
+
◀── {expectations: [{id, statement, handle, component?}],
|
|
141
|
+
datasets: [{expectation, via, …}], open_questions: [...]}
|
|
142
|
+
orchestration ──▶ implementation: assess-task {..., acceptance_surface: expectations} ← never datasets
|
|
142
143
|
◀── {feasible, objections, commitments}
|
|
144
|
+
orchestration ──▶ testing: test-task {..., acceptance_surface, datasets}
|
|
143
145
|
```
|
|
144
146
|
|
|
145
147
|
A **query**, read-only: a clone of the testing repo, a clone of the implementation repo's **default
|
|
@@ -154,9 +156,25 @@ implementer to commit to or object to; and anything the task does not determine
|
|
|
154
156
|
`open_questions`, **never** into an invented statement. A non-empty `open_questions` stops the round
|
|
155
157
|
and reaches a human. See `adr/ADR-FTA-0002-*.md`.
|
|
156
158
|
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
159
|
+
**Every expectation says how its data comes to exist** (`adr/ADR-FTA-0003-*.md`), in a separate
|
|
160
|
+
`datasets` list that is **private to this actor** — the implementer never sees it. One entry per
|
|
161
|
+
expectation, `via` one of:
|
|
162
|
+
|
|
163
|
+
| `via` | the test's state comes from | must carry |
|
|
164
|
+
|---|---|---|
|
|
165
|
+
| `none` | nothing — written down, never inferred from a missing entry | — |
|
|
166
|
+
| `command` | the capability's own commands, called by the test first (preferred) | `steps` |
|
|
167
|
+
| `event` | upstream events the test publishes; broker address in `AMQP_URL` | `steps` |
|
|
168
|
+
| `seed` | pre-existing data that is itself under test | `because`, `provided_by` (a proposed expectation) |
|
|
169
|
+
|
|
170
|
+
Never a component's storage. Anything a dataset needs that the contract does not already offer — a
|
|
171
|
+
seed, a command this task introduces — must also be proposed as an **expectation**, because that is
|
|
172
|
+
the only thing the implementer assesses.
|
|
173
|
+
|
|
174
|
+
The reply is **projected** onto `{expectations, datasets, open_questions}` — a session's stray keys
|
|
175
|
+
never reach the orchestrating actor dressed as contract — after checking that every expectation
|
|
176
|
+
carries an `id`, a `statement` and a `handle`, with ids unique, and has exactly one valid dataset.
|
|
177
|
+
`open_questions` is always a list.
|
|
160
178
|
|
|
161
179
|
### `test-task`
|
|
162
180
|
|
|
@@ -54,7 +54,7 @@ ground_in:
|
|
|
54
54
|
and, beside it, a four-line Dockerfile:
|
|
55
55
|
|
|
56
56
|
```dockerfile
|
|
57
|
-
FROM ghcr.io/papeete-hub/foundry-testing-actor:0.
|
|
57
|
+
FROM ghcr.io/papeete-hub/foundry-testing-actor:0.2.0
|
|
58
58
|
RUN pip install --no-cache-dir kpack==2.0.1 kontract==0.1.0 # what this sidecar's ground_in names
|
|
59
59
|
COPY actor-agentic-context.yaml /actor/
|
|
60
60
|
RUN foundry-testing-actor render-cards /actor && foundry-testing-actor lint /actor
|
|
@@ -114,9 +114,11 @@ Both name the same engine. `Actor.judge()` hands it the door id, and it dispatch
|
|
|
114
114
|
|
|
115
115
|
```
|
|
116
116
|
orchestration ──▶ testing: propose-acceptance {task_id, title, definition_of_done, components, context?}
|
|
117
|
-
◀── {expectations: [{id, statement, handle, component?}],
|
|
118
|
-
|
|
117
|
+
◀── {expectations: [{id, statement, handle, component?}],
|
|
118
|
+
datasets: [{expectation, via, …}], open_questions: [...]}
|
|
119
|
+
orchestration ──▶ implementation: assess-task {..., acceptance_surface: expectations} ← never datasets
|
|
119
120
|
◀── {feasible, objections, commitments}
|
|
121
|
+
orchestration ──▶ testing: test-task {..., acceptance_surface, datasets}
|
|
120
122
|
```
|
|
121
123
|
|
|
122
124
|
A **query**, read-only: a clone of the testing repo, a clone of the implementation repo's **default
|
|
@@ -131,9 +133,25 @@ implementer to commit to or object to; and anything the task does not determine
|
|
|
131
133
|
`open_questions`, **never** into an invented statement. A non-empty `open_questions` stops the round
|
|
132
134
|
and reaches a human. See `adr/ADR-FTA-0002-*.md`.
|
|
133
135
|
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
136
|
+
**Every expectation says how its data comes to exist** (`adr/ADR-FTA-0003-*.md`), in a separate
|
|
137
|
+
`datasets` list that is **private to this actor** — the implementer never sees it. One entry per
|
|
138
|
+
expectation, `via` one of:
|
|
139
|
+
|
|
140
|
+
| `via` | the test's state comes from | must carry |
|
|
141
|
+
|---|---|---|
|
|
142
|
+
| `none` | nothing — written down, never inferred from a missing entry | — |
|
|
143
|
+
| `command` | the capability's own commands, called by the test first (preferred) | `steps` |
|
|
144
|
+
| `event` | upstream events the test publishes; broker address in `AMQP_URL` | `steps` |
|
|
145
|
+
| `seed` | pre-existing data that is itself under test | `because`, `provided_by` (a proposed expectation) |
|
|
146
|
+
|
|
147
|
+
Never a component's storage. Anything a dataset needs that the contract does not already offer — a
|
|
148
|
+
seed, a command this task introduces — must also be proposed as an **expectation**, because that is
|
|
149
|
+
the only thing the implementer assesses.
|
|
150
|
+
|
|
151
|
+
The reply is **projected** onto `{expectations, datasets, open_questions}` — a session's stray keys
|
|
152
|
+
never reach the orchestrating actor dressed as contract — after checking that every expectation
|
|
153
|
+
carries an `id`, a `statement` and a `handle`, with ids unique, and has exactly one valid dataset.
|
|
154
|
+
`open_questions` is always a list.
|
|
137
155
|
|
|
138
156
|
### `test-task`
|
|
139
157
|
|
foundry_testing_actor-0.2.0/adr/ADR-FTA-0003-every-expectation-says-how-its-data-comes-to-exist.md
ADDED
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
---
|
|
2
|
+
id: ADR-FTA-0003
|
|
3
|
+
title: "Every expectation says how its data comes to exist — privately, and anything the contract cannot already provide becomes an expectation"
|
|
4
|
+
status: Proposed
|
|
5
|
+
date: 2026-09-14
|
|
6
|
+
supersedes: []
|
|
7
|
+
references:
|
|
8
|
+
- ../src/foundry_testing_actor/cards/actor-data.yaml
|
|
9
|
+
- ../src/foundry_testing_actor/cards/actor-synchronous-messaging.yaml
|
|
10
|
+
- ../src/foundry_testing_actor/engine.py
|
|
11
|
+
- ./ADR-FTA-0002-this-actors-half-of-the-three-amigos-round.md
|
|
12
|
+
- https://github.com/papeete-hub/foundry-task-orchestration-actor/blob/main/adr/ADR-FTOA-0003-datasets-travel-to-the-tester-only.md
|
|
13
|
+
- https://github.com/papeete-hub/foundry-implementation-actor/blob/main/adr/ADR-FIA-0004-the-three-amigos-round.md
|
|
14
|
+
---
|
|
15
|
+
|
|
16
|
+
# ADR-FTA-0003 — Every expectation says how its data comes to exist
|
|
17
|
+
|
|
18
|
+
## Context
|
|
19
|
+
|
|
20
|
+
ADR-FTA-0002 made this actor propose its acceptance surface before anything is built, and told it
|
|
21
|
+
to **propose a concrete value** wherever a test needs one the task leaves unnamed: "a fixture's id,
|
|
22
|
+
a seeded record". It said nothing about *where the data a test runs against comes from*, and so the
|
|
23
|
+
default was the one the whole round exists to remove — data that already exists, found by reading.
|
|
24
|
+
|
|
25
|
+
The second live run showed it. Asked for a deliberately vague task, the propose session read the
|
|
26
|
+
implementation's `stub/fixtures/anchors.json`, proposed the three anchors it found there at their
|
|
27
|
+
existing ids, and stopped the round on a question about how a consumer would ever learn those ids.
|
|
28
|
+
Every expectation depended on pre-existing records. None needed to: the capability's own contract
|
|
29
|
+
has a command that creates an anchor.
|
|
30
|
+
|
|
31
|
+
Two things are true of every capability this actor will ever serve, not only of the one that
|
|
32
|
+
showed it:
|
|
33
|
+
|
|
34
|
+
1. **A black-box test can almost always build its own state** through the contract under test —
|
|
35
|
+
the commands it exposes, the upstream events it consumes. A test that does so needs no agreed
|
|
36
|
+
fixture, cannot be broken by someone regenerating one, and does not care what the database
|
|
37
|
+
already holds.
|
|
38
|
+
2. **Where it cannot**, that is a fact the implementer has to know about: a command the task must
|
|
39
|
+
introduce, or seeded data at fixed values that someone has to put there and keep there.
|
|
40
|
+
|
|
41
|
+
A first design put the dataset inside each shared expectation (`given`). It was rejected in
|
|
42
|
+
review: for the ordinary case — build state through what the contract already offers — the
|
|
43
|
+
implementer has no use for it, and the one case that does concern the implementer is better stated
|
|
44
|
+
as an expectation of its own than buried in a private detail of another.
|
|
45
|
+
|
|
46
|
+
## Decision
|
|
47
|
+
|
|
48
|
+
**1. `propose-acceptance` answers a second list, `datasets`, private to this actor.** One entry per
|
|
49
|
+
expectation, keyed by the expectation's `id`:
|
|
50
|
+
|
|
51
|
+
```json
|
|
52
|
+
{"expectation": "E3",
|
|
53
|
+
"via": "command",
|
|
54
|
+
"steps": ["POST $BACKEND_URL/anchors {valid mint payload} → internal_id",
|
|
55
|
+
"POST $BACKEND_URL/anchors/{internal_id}/archive"]}
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
`via` is one of four, and nothing else:
|
|
59
|
+
|
|
60
|
+
| `via` | the test's state comes from | the entry must carry |
|
|
61
|
+
|---|---|---|
|
|
62
|
+
| `none` | nothing — the expectation needs no prior state | — (saying so is the point) |
|
|
63
|
+
| `command` | the capability's own commands, called by the test | `steps` |
|
|
64
|
+
| `event` | upstream events the test publishes on the bus | `steps` |
|
|
65
|
+
| `seed` | data that exists before the test runs | `because`, and `provided_by` naming the expectation that states the seed |
|
|
66
|
+
|
|
67
|
+
**2. Every expectation has exactly one dataset — and this is checked, not asked.** The engine
|
|
68
|
+
refuses a proposal where an expectation has no entry, an entry names an unknown expectation or has
|
|
69
|
+
two, `via` is outside the four, `command`/`event` has no `steps`, or `seed` has no `because` or a
|
|
70
|
+
`provided_by` that is not a proposed expectation. The refusal happens at this door, where a
|
|
71
|
+
proposal is still cheap; `none` must be written, not inferred, because "needs nothing" is a claim
|
|
72
|
+
that can be wrong.
|
|
73
|
+
|
|
74
|
+
**3. Anything a dataset needs that the contract does not already offer is promoted to an
|
|
75
|
+
expectation.** A seed is the plainest case: `provided_by` must point at an expectation whose
|
|
76
|
+
`statement` and `handle` describe the seeded data — which is what the implementer assesses and
|
|
77
|
+
commits to. The same holds for a command or a subscription the task itself introduces: it is
|
|
78
|
+
something the increment must deliver, so it is an expectation, and a dataset may then use it.
|
|
79
|
+
`seed` is for data whose existence *is* what is being tested (a stub's canned records); the prompt
|
|
80
|
+
says to prefer `command` and `event` everywhere else.
|
|
81
|
+
|
|
82
|
+
**4. `datasets` never reaches the implementer.** The shared expectation stays `{id, statement,
|
|
83
|
+
handle}` with optional `component`. The orchestrating actor relays `datasets` unread to
|
|
84
|
+
`test-task`, and never to `assess-task` or `implement-task` (ADR-FTOA-0003).
|
|
85
|
+
|
|
86
|
+
**5. `test-task` takes the datasets and builds state exactly as they say.** `test-task-cmd` gains an
|
|
87
|
+
optional `datasets`. Each test arranges its state as its expectation's dataset states; it does not
|
|
88
|
+
rely on records it did not create, except through a `seed` dataset. Without datasets — a caller that
|
|
89
|
+
skipped round 0 — the prompt still says to build state through the contract's commands.
|
|
90
|
+
|
|
91
|
+
**6. How a test reaches the bus is a convention, like `<COMPONENT>_URL`.** An `event` dataset
|
|
92
|
+
publishes to the broker whose address arrives in `AMQP_URL`. The orchestrating actor sets it on the
|
|
93
|
+
test Job when its use declares one (ADR-FTOA-0003).
|
|
94
|
+
|
|
95
|
+
**7. No `via: database`.** A test never writes a component's storage. Doing so would bind the
|
|
96
|
+
tests to the implementer's private schema — the coupling black-box testing exists to prevent — and
|
|
97
|
+
a state the contract cannot produce is a question about the contract, which belongs in
|
|
98
|
+
`open_questions`.
|
|
99
|
+
|
|
100
|
+
## Rationale
|
|
101
|
+
|
|
102
|
+
**Why a separate list rather than a field on the expectation.** The expectation is the part of the
|
|
103
|
+
proposal two actors agree on. How a test sets itself up is how this actor does its job; putting it
|
|
104
|
+
on the shared object would put it in front of an implementer that can only ignore it, and — worse —
|
|
105
|
+
would let a genuine dev↔test dependency ride along unseen instead of being stated where it gets
|
|
106
|
+
assessed.
|
|
107
|
+
|
|
108
|
+
**Why keyed by id and validated here.** It is the same reason ids are unique (ADR-FTA-0002 §6): the
|
|
109
|
+
orchestrating actor and a verdict both address expectations by id. A dataset that names nothing, or
|
|
110
|
+
an expectation with none, is exactly the silent gap this ADR closes, and this door is the last place
|
|
111
|
+
it costs nothing to refuse.
|
|
112
|
+
|
|
113
|
+
**Why relay through the orchestrator instead of re-deriving at `test-task`.** The plan a test is
|
|
114
|
+
written against should be the one that existed when the implementer agreed to the surface. A
|
|
115
|
+
`test-task` session re-deciding how to arrange data, after the increment is built, would be free to
|
|
116
|
+
read state off the build again — the *after* ADR-FIA-0004 excludes.
|
|
117
|
+
|
|
118
|
+
**Why the promotion rule rather than trusting the dataset.** A private plan is only safe if it can
|
|
119
|
+
not hide a demand on someone else. The rule makes every such demand public in the one shape the
|
|
120
|
+
implementer already assesses. Where a session breaks it anyway — assumes a command that does not
|
|
121
|
+
exist — the first attempt's tests fail against the real deployment, and remediation names them.
|
|
122
|
+
|
|
123
|
+
## Consequences
|
|
124
|
+
|
|
125
|
+
- **The completion shape of `propose-acceptance` changes**: `datasets` is required. A use pinned to
|
|
126
|
+
an older image keeps the older door; a use that takes this version gets it through its `FROM`
|
|
127
|
+
line and `render-cards`, with nothing written by hand. Released as `0.2.0`.
|
|
128
|
+
- **A proposal can now be refused for a missing dataset**, which the orchestrating actor reports as
|
|
129
|
+
a round-0 failure — `testing did not answer`, with the reason. That is intended: a proposal with
|
|
130
|
+
no plan for its data is not a proposal.
|
|
131
|
+
- **`remediation_context` gains a failure class**: a dataset that could not be built. The
|
|
132
|
+
`test-task` prompt already asks the session to judge whether a failure is the test's or the
|
|
133
|
+
implementation's; it now also names this one.
|
|
134
|
+
- **Not decided here**: whether datasets should be committed beside the tests that use them (they
|
|
135
|
+
are in the tests' code, and in the transcript), and whether `open_questions` should be split into
|
|
136
|
+
blocking and non-blocking — raised by the same run, still open.
|
|
@@ -17,6 +17,7 @@ are cited rather than restated.
|
|
|
17
17
|
|----|-------|--------|
|
|
18
18
|
| [ADR-FTA-0001](./ADR-FTA-0001-the-machinery-leaves-the-capability.md) | The machinery leaves the capability — a published testing actor any capability can instantiate | Proposed |
|
|
19
19
|
| [ADR-FTA-0002](./ADR-FTA-0002-this-actors-half-of-the-three-amigos-round.md) | This actor's half of the three amigos round — propose before anything is built, read-only, and hand the unknowns to a human | Proposed |
|
|
20
|
+
| [ADR-FTA-0003](./ADR-FTA-0003-every-expectation-says-how-its-data-comes-to-exist.md) | Every expectation says how its data comes to exist — privately, and anything the contract cannot already provide becomes an expectation | Proposed |
|
|
20
21
|
|
|
21
22
|
## Authoring
|
|
22
23
|
|
|
@@ -13,8 +13,8 @@
|
|
|
13
13
|
# The default names the product registry rather than GHCR, as the implementation actor's example
|
|
14
14
|
# does, because it is the copy an in-cluster builder can pull (ADR-FIA-0006's arrangement, copied).
|
|
15
15
|
# Override it if you are somewhere else:
|
|
16
|
-
# docker build --build-arg ACTOR_IMAGE=ghcr.io/papeete-hub/foundry-testing-actor:0.
|
|
17
|
-
ARG ACTOR_IMAGE=papeetefoundry.azurecr.io/foundry/foundry-testing-actor:0.
|
|
16
|
+
# docker build --build-arg ACTOR_IMAGE=ghcr.io/papeete-hub/foundry-testing-actor:0.2.0 .
|
|
17
|
+
ARG ACTOR_IMAGE=papeetefoundry.azurecr.io/foundry/foundry-testing-actor:0.2.0
|
|
18
18
|
FROM ${ACTOR_IMAGE}
|
|
19
19
|
|
|
20
20
|
# THE KNOWLEDGE TOOLS THIS SIDECAR NAMES, and the one thing the base image deliberately does not
|
|
@@ -22,7 +22,7 @@ repo and publishes them as a runnable image.**
|
|
|
22
22
|
definition_of_done, ground the session ─▶ the knowledge tools
|
|
23
23
|
components, context? propose expectations the sidecar names
|
|
24
24
|
(Read/Glob/Grep only)
|
|
25
|
-
◀────────────────────────────── {expectations
|
|
25
|
+
◀────────────────────────────── {expectations, datasets (private), open_questions}
|
|
26
26
|
|
|
27
27
|
POST /test-task ─────▶ clone testing repo → test/TASK-NNN
|
|
28
28
|
...the same, plus clone implementation → impl/TASK-NNN ─▶ recompute images under test
|
|
@@ -75,8 +75,8 @@ python scripts/probe_grounding.py examples/ACME.PARTS.CAP.SUP.007.WID-testing
|
|
|
75
75
|
**Build it** — this needs the base image, from a registry you can pull, or one you built:
|
|
76
76
|
|
|
77
77
|
```bash
|
|
78
|
-
uv build && docker build -f docker/Dockerfile -t foundry-testing-actor:0.
|
|
79
|
-
docker build --build-arg ACTOR_IMAGE=foundry-testing-actor:0.
|
|
78
|
+
uv build && docker build -f docker/Dockerfile -t foundry-testing-actor:0.2.0 .
|
|
79
|
+
docker build --build-arg ACTOR_IMAGE=foundry-testing-actor:0.2.0 -t acme-wid-tester \
|
|
80
80
|
examples/ACME.PARTS.CAP.SUP.007.WID-testing
|
|
81
81
|
docker run --rm acme-wid-tester foundry-testing-actor lint /actor
|
|
82
82
|
```
|
|
@@ -102,10 +102,24 @@ curl -X POST http://<actor>/propose-acceptance -H 'Content-Type: application/jso
|
|
|
102
102
|
"handle": {"component": "stub", "base_url_env": "STUB_URL", "method": "GET",
|
|
103
103
|
"path": "/widgets/{id}",
|
|
104
104
|
"ids": {"ACTIVE": "w-0001", "ARCHIVED": "w-0002", "RETIRED": "w-0003"}},
|
|
105
|
-
"component": "stub"}
|
|
105
|
+
"component": "stub"},
|
|
106
|
+
{"id": "retired-widget-refuses-update",
|
|
107
|
+
"statement": "PATCH /widgets/{id} on a retired widget answers 409 WIDGET_RETIRED",
|
|
108
|
+
"handle": "PATCH $BACKEND_URL/widgets/{id}", "component": "backend"}],
|
|
109
|
+
"datasets": [
|
|
110
|
+
{"expectation": "stub-seeded-widgets", "via": "seed",
|
|
111
|
+
"because": "the stub's canned widgets are themselves what is under test",
|
|
112
|
+
"provided_by": "stub-seeded-widgets"},
|
|
113
|
+
{"expectation": "retired-widget-refuses-update", "via": "command",
|
|
114
|
+
"steps": ["POST $BACKEND_URL/widgets {valid payload} → id",
|
|
115
|
+
"POST $BACKEND_URL/widgets/{id}/retire"]}],
|
|
106
116
|
"open_questions": []}
|
|
107
117
|
```
|
|
108
118
|
|
|
119
|
+
`datasets` stays with this actor: the orchestrating actor relays it to `test-task` and never to the
|
|
120
|
+
implementer. The second test creates the widget it asserts on, so it needs no agreed fixture; the
|
|
121
|
+
first relies on seeded data, which is why that data is an expectation the implementer assesses.
|
|
122
|
+
|
|
109
123
|
The ids are **proposed**, not discovered: the implementer commits to them at its `assess-task`
|
|
110
124
|
door, or objects. Then `test-task` is sent the agreed surface as `acceptance_surface`, and
|
|
111
125
|
answers `{"accepted": true, "branch": "test/TASK-014", "images":
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "foundry-testing-actor"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.2.0"
|
|
4
4
|
description = "Runs a headless Claude Code black-box test-authoring session against one capability's own testing repo — a papeete-actor for one use, with the capability supplied by a sidecar."
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
requires-python = ">=3.11"
|
|
@@ -45,6 +45,18 @@ items:
|
|
|
45
45
|
reaches it, and optionally the `component` it belongs to. Proposed BEFORE anything is built,
|
|
46
46
|
so a concrete value in a handle is a proposal the implementer commits to or objects to, never
|
|
47
47
|
a report of what was found
|
|
48
|
+
- name: datasets
|
|
49
|
+
type: list
|
|
50
|
+
description: >-
|
|
51
|
+
PRIVATE TO THIS ACTOR — how the state each proposed expectation's test runs against comes to
|
|
52
|
+
exist, one object per expectation: `expectation` (its id) and `via`, one of `none` (needs no
|
|
53
|
+
prior state), `command` (the test calls the capability's own commands first; with `steps`),
|
|
54
|
+
`event` (the test publishes upstream events, broker address in AMQP_URL; with `steps`) or
|
|
55
|
+
`seed` (pre-existing data that is itself under test; with `because`, and `provided_by`
|
|
56
|
+
naming the proposed expectation that states it). Never a component's storage. Relayed
|
|
57
|
+
unread to test-task by the orchestrating actor and never shown to the implementer: anything
|
|
58
|
+
a dataset needs that the contract does not already offer is proposed as an expectation too
|
|
59
|
+
(ADR-FTA-0003)
|
|
48
60
|
- name: open_questions
|
|
49
61
|
type: list
|
|
50
62
|
description: >-
|
|
@@ -12,16 +12,17 @@ messages:
|
|
|
12
12
|
|
|
13
13
|
- name: acceptance-proposed-result
|
|
14
14
|
intent: >-
|
|
15
|
-
answer with the proposed expectations, each with a stable id, a statement and a handle
|
|
16
|
-
the
|
|
17
|
-
|
|
15
|
+
answer with the proposed expectations, each with a stable id, a statement and a handle;
|
|
16
|
+
how the state each one's test runs against comes to exist (private to this actor); and the
|
|
17
|
+
questions the task leaves open
|
|
18
|
+
references: [expectations, datasets, open_questions]
|
|
18
19
|
optional: [open_questions]
|
|
19
20
|
|
|
20
21
|
- name: test-task-cmd
|
|
21
22
|
intent: ask this actor to author black-box tests for the published image(s) of a TASK-NNN card
|
|
22
23
|
references: [task_id, title, definition_of_done, components, context, remediation_context,
|
|
23
|
-
acceptance_surface]
|
|
24
|
-
optional: [context, remediation_context, acceptance_surface]
|
|
24
|
+
acceptance_surface, datasets]
|
|
25
|
+
optional: [context, remediation_context, acceptance_surface, datasets]
|
|
25
26
|
|
|
26
27
|
- name: test-task-result
|
|
27
28
|
intent: report that the tests were written, pushed, and their touched components' test images published
|
|
@@ -19,14 +19,16 @@ actions:
|
|
|
19
19
|
means: >-
|
|
20
20
|
the door for "author black-box tests for TASK-NNN of the capability I serve, now that its
|
|
21
21
|
increment is built". Send it here with the task id, title, definition of done, components,
|
|
22
|
-
and (optionally) context, remediation_context
|
|
22
|
+
and (optionally) context, remediation_context, the agreed `acceptance_surface` and the
|
|
23
|
+
`datasets` I proposed beside it — the
|
|
23
24
|
caller supplies everything, no task card is looked up here. I clone my capability's testing
|
|
24
25
|
repository into my own private copy, and read-only clone its implementation repository at
|
|
25
26
|
impl/TASK-NNN solely to recompute, by convention, the image refs published there (never
|
|
26
27
|
passed to me, and never shown to my session); ground myself in that capability's own
|
|
27
28
|
standing context; and extend the persistent suite under each named component's declared
|
|
28
29
|
tests root — asserting every agreed expectation by its id, through its handle, where a
|
|
29
|
-
surface was sent
|
|
30
|
+
surface was sent, each test building its state as its dataset says, or through the
|
|
31
|
+
contract's own commands where none was sent. I never bring up the images, hit a live endpoint, or render a verdict.
|
|
30
32
|
Then I commit and push to test/TASK-NNN, and build each touched component's test image in
|
|
31
33
|
the cluster's shared buildkit and push it to the registry, named and versioned by
|
|
32
34
|
convention. I never open a pull request — an orchestrating actor runs that image and does.
|
|
@@ -51,6 +53,10 @@ queries:
|
|
|
51
53
|
reaches it (the component, the endpoint, the event routing key, the environment variable its
|
|
52
54
|
base URL arrives in — with a concrete value proposed wherever a test needs one the task does
|
|
53
55
|
not name, for the implementer to commit to or object to), and optionally its `component`;
|
|
56
|
+
for EVERY expectation, a dataset saying how the state its test needs comes to exist — `none`,
|
|
57
|
+
`command` (preferred), `event`, or `seed` only where pre-existing data is itself under test
|
|
58
|
+
and a proposed expectation states it — kept private to me: anything a dataset needs that the
|
|
59
|
+
contract does not already offer is proposed as an expectation too (ADR-FTA-0003);
|
|
54
60
|
and the open questions the task and the standing context leave undetermined, each a string
|
|
55
61
|
or an `about`/`question` object, never an invented answer. Any open question is meant to
|
|
56
62
|
stop the round and reach a human rather than a session.
|
{foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/engine.py
RENAMED
|
@@ -86,6 +86,14 @@ DEFAULT_MAX_TURNS = 60
|
|
|
86
86
|
DEFAULT_PROPOSE_TIMEOUT_S = 900
|
|
87
87
|
DEFAULT_PROPOSE_MAX_TURNS = 30
|
|
88
88
|
|
|
89
|
+
# How the state a test runs against comes to exist (ADR-FTA-0003). Closed on purpose: a fifth way
|
|
90
|
+
# is a decision, not a session's improvisation — and `database` is deliberately not one of them.
|
|
91
|
+
DATASET_VIA = ("none", "command", "event", "seed")
|
|
92
|
+
|
|
93
|
+
# Where an `event` dataset reaches the bus: the broker address arrives in this variable, set on the
|
|
94
|
+
# test Job by the orchestrating actor, the way <COMPONENT>_URL is (ADR-FTA-0003 §6, ADR-FTOA-0003).
|
|
95
|
+
BROKER_URL_ENV = "AMQP_URL"
|
|
96
|
+
|
|
89
97
|
TEST_TOOLS = "Bash,Read,Edit,Write,Glob,Grep"
|
|
90
98
|
PROPOSE_TOOLS = "Read,Glob,Grep"
|
|
91
99
|
|
|
@@ -293,8 +301,8 @@ def _proposal(judged: dict) -> dict:
|
|
|
293
301
|
the framework (ADR-PAS-0009), so a session's helpful `"notes"` key beside its answer would not
|
|
294
302
|
be refused — it would travel to the orchestrating actor looking like part of a contract three
|
|
295
303
|
packages share, which is worse. And the day this door names a second outcome, the framework
|
|
296
|
-
closes the schema and that same key refuses the whole proposal. Only the
|
|
297
|
-
references survive; everything else a session said is in the transcript.
|
|
304
|
+
closes the schema and that same key refuses the whole proposal. Only the three fields the
|
|
305
|
+
message references survive; everything else a session said is in the transcript.
|
|
298
306
|
|
|
299
307
|
WHAT IS CHECKED, AND WHY IT IS NOT MORE. `expectations` must be a list of objects, each with an
|
|
300
308
|
`id` and a `statement`, a `handle` key, and ids unique within the proposal — because a later
|
|
@@ -328,7 +336,81 @@ def _proposal(judged: dict) -> dict:
|
|
|
328
336
|
f"expectation by its id, so an id two expectations share names neither"
|
|
329
337
|
)
|
|
330
338
|
seen.add(identifier)
|
|
331
|
-
|
|
339
|
+
datasets = _datasets(judged.get("datasets"), [str(e["id"]) for e in expectations])
|
|
340
|
+
return {"expectations": expectations, "datasets": datasets,
|
|
341
|
+
"open_questions": _as_list(judged.get("open_questions"))}
|
|
342
|
+
|
|
343
|
+
|
|
344
|
+
def _datasets(raw, expectation_ids: list[str]) -> list:
|
|
345
|
+
"""Every expectation's dataset, checked — how the state its test runs against comes to exist.
|
|
346
|
+
|
|
347
|
+
PRIVATE TO THIS ACTOR (ADR-FTA-0003). The implementer never sees this list; the orchestrating
|
|
348
|
+
actor relays it to `test-task` unread. So the checks that matter are the ones that stop a
|
|
349
|
+
private plan from hiding a gap:
|
|
350
|
+
|
|
351
|
+
- exactly one entry per proposed expectation, none for an expectation nobody proposed — "needs
|
|
352
|
+
nothing" is `via: none`, written down, never an absent entry;
|
|
353
|
+
- `via` is one of `DATASET_VIA`, and nothing else — in particular never a component's database;
|
|
354
|
+
- `command` and `event` say how, as `steps`;
|
|
355
|
+
- `seed` says why (`because`) and points `provided_by` at a PROPOSED expectation that states
|
|
356
|
+
the seeded data — because data that must exist before a test runs is something the
|
|
357
|
+
implementer has to put there, and the only place the implementer looks is the expectations.
|
|
358
|
+
|
|
359
|
+
What a step SAYS is not checked, for the reason a handle's content is not: it depends on what
|
|
360
|
+
is being addressed, and a test failing against the real deployment is the check that can.
|
|
361
|
+
"""
|
|
362
|
+
if not expectation_ids and raw in (None, []):
|
|
363
|
+
return []
|
|
364
|
+
if raw is None:
|
|
365
|
+
raise EngineError(
|
|
366
|
+
"the proposal carries no `datasets` — every expectation says how the state its test "
|
|
367
|
+
"runs against comes to exist, even when that is `via: none` (ADR-FTA-0003)"
|
|
368
|
+
)
|
|
369
|
+
if not isinstance(raw, list):
|
|
370
|
+
raise EngineError(f"`datasets` is not a list: {raw!r}")
|
|
371
|
+
proposed = set(expectation_ids)
|
|
372
|
+
covered: set[str] = set()
|
|
373
|
+
for index, entry in enumerate(raw):
|
|
374
|
+
where = f"datasets[{index}]"
|
|
375
|
+
if not isinstance(entry, dict):
|
|
376
|
+
raise EngineError(f"{where} is not an object: {entry!r}")
|
|
377
|
+
expectation = str(entry.get("expectation") or "")
|
|
378
|
+
if expectation not in proposed:
|
|
379
|
+
raise EngineError(
|
|
380
|
+
f"{where} is for expectation '{expectation}', which this proposal does not "
|
|
381
|
+
f"propose — a dataset belongs to exactly one proposed expectation")
|
|
382
|
+
if expectation in covered:
|
|
383
|
+
raise EngineError(
|
|
384
|
+
f"expectation '{expectation}' has two datasets — its test builds its state one way")
|
|
385
|
+
covered.add(expectation)
|
|
386
|
+
via = entry.get("via")
|
|
387
|
+
if via not in DATASET_VIA:
|
|
388
|
+
raise EngineError(
|
|
389
|
+
f"{where} (for {expectation}) says via {via!r}; a test's state comes from one of "
|
|
390
|
+
f"{', '.join(DATASET_VIA)} — never a component's own storage")
|
|
391
|
+
if via in ("command", "event"):
|
|
392
|
+
steps = entry.get("steps")
|
|
393
|
+
if not isinstance(steps, list) or not steps:
|
|
394
|
+
raise EngineError(
|
|
395
|
+
f"{where} (for {expectation}) is via {via} but lists no `steps` — say which "
|
|
396
|
+
f"{'commands the test calls' if via == 'command' else 'events it publishes'}")
|
|
397
|
+
if via == "seed":
|
|
398
|
+
if not entry.get("because"):
|
|
399
|
+
raise EngineError(
|
|
400
|
+
f"{where} (for {expectation}) relies on seeded data without saying `because` — "
|
|
401
|
+
f"a seed is for data whose existence is itself under test; everything else is "
|
|
402
|
+
f"built through the contract")
|
|
403
|
+
if str(entry.get("provided_by") or "") not in proposed:
|
|
404
|
+
raise EngineError(
|
|
405
|
+
f"{where} (for {expectation}) relies on seeded data but `provided_by` names no "
|
|
406
|
+
f"proposed expectation — data that must exist before a test runs is something "
|
|
407
|
+
f"the implementer has to provide, so it has to be an expectation of its own")
|
|
408
|
+
missing = [i for i in expectation_ids if i not in covered]
|
|
409
|
+
if missing:
|
|
410
|
+
raise EngineError(
|
|
411
|
+
f"no dataset for expectation(s) {', '.join(missing)} — say how the state each test "
|
|
412
|
+
f"runs against comes to exist, even when that is `via: none`")
|
|
413
|
+
return raw
|
|
332
414
|
|
|
333
415
|
|
|
334
416
|
class ClaudeCodeTesterEngine:
|
|
@@ -529,6 +611,7 @@ class ClaudeCodeTesterEngine:
|
|
|
529
611
|
proposal = _proposal(_extract_json(answer))
|
|
530
612
|
correlation.event("acceptance-proposed",
|
|
531
613
|
expectations=[e["id"] for e in proposal["expectations"]],
|
|
614
|
+
datasets={d["expectation"]: d["via"] for d in proposal["datasets"]},
|
|
532
615
|
open_questions=len(proposal["open_questions"]))
|
|
533
616
|
return proposal
|
|
534
617
|
|
|
@@ -679,11 +762,35 @@ class ClaudeCodeTesterEngine:
|
|
|
679
762
|
"agreed expectation rather than at a line of output.\n\n"
|
|
680
763
|
+ json.dumps(payload["acceptance_surface"], indent=2, ensure_ascii=False)
|
|
681
764
|
)
|
|
765
|
+
if payload.get("datasets"):
|
|
766
|
+
# Private to this actor, and relayed unread by the orchestrating actor: the plan this
|
|
767
|
+
# actor proposed for each expectation's data BEFORE the build, so the tests are not
|
|
768
|
+
# free to read state off the build now (ADR-FTA-0003).
|
|
769
|
+
sections.append(
|
|
770
|
+
"## How each test builds the data it runs against\n"
|
|
771
|
+
"You planned this yourself before anything was built. Each test arranges its "
|
|
772
|
+
"state exactly as its expectation's dataset says: `none` needs nothing; "
|
|
773
|
+
"`command` calls the listed commands first and asserts on what they returned; "
|
|
774
|
+
f"`event` publishes the listed events to the broker at `{BROKER_URL_ENV}`; "
|
|
775
|
+
"`seed` relies on the seeded data its `provided_by` expectation states. Do not "
|
|
776
|
+
"rely on any record the test did not create, except through a `seed` dataset, and "
|
|
777
|
+
"never write a component's storage directly.\n\n"
|
|
778
|
+
+ json.dumps(payload["datasets"], indent=2, ensure_ascii=False)
|
|
779
|
+
)
|
|
780
|
+
else:
|
|
781
|
+
sections.append(
|
|
782
|
+
"## The data each test runs against\n"
|
|
783
|
+
"Build each test's state through this capability's own contract — call its "
|
|
784
|
+
"commands, or publish the upstream events it consumes (broker address in "
|
|
785
|
+
f"`{BROKER_URL_ENV}`) — and assert on what the test created. Do not rely on "
|
|
786
|
+
"records that happen to exist, and never write a component's storage directly."
|
|
787
|
+
)
|
|
682
788
|
if payload.get("remediation_context"):
|
|
683
789
|
sections.append(
|
|
684
790
|
"## Remediation — the prior attempt's failing criteria\n"
|
|
685
791
|
"A criterion failing could mean the test itself was wrong (bad expectation, "
|
|
686
|
-
"wrong endpoint, wrong shape
|
|
792
|
+
"wrong endpoint, wrong shape, a dataset it could not build) or that the "
|
|
793
|
+
"implementation was wrong. Review each "
|
|
687
794
|
"one: fix the test here if the fault is in it; if the test looks correct against "
|
|
688
795
|
"this capability's standing context and the agreed acceptance surface, leave it "
|
|
689
796
|
"as-is (the implementation side should fix it instead) and say so in your "
|
|
@@ -748,10 +855,33 @@ class ClaudeCodeTesterEngine:
|
|
|
748
855
|
f"(method and path), the event routing key, the environment variable carrying its "
|
|
749
856
|
f"base URL (<COMPONENT>_URL unless you need another)\n"
|
|
750
857
|
f" - `component` (optional): which of the components below it belongs to\n\n"
|
|
751
|
-
f"Where a test needs a concrete value the task does not name — a
|
|
752
|
-
f"
|
|
753
|
-
f"
|
|
754
|
-
f"
|
|
858
|
+
f"Where a test needs a concrete value the task does not name — a path parameter, a "
|
|
859
|
+
f"routing key, a field value — PROPOSE A CONCRETE VALUE in the handle. The implementer "
|
|
860
|
+
f"will accept it and commit to it, or object. Do not leave it for the implementation "
|
|
861
|
+
f"to choose and for a test to discover afterwards.\n\n"
|
|
862
|
+
f"## The data each test runs against\n"
|
|
863
|
+
f"For EVERY expectation, say how the state its test needs comes to exist, in a "
|
|
864
|
+
f"separate `datasets` list — one entry per expectation. This list is yours: the "
|
|
865
|
+
f"implementer never sees it. Each entry is `{{\"expectation\": <id>, \"via\": ...}}`, "
|
|
866
|
+
f"with `via` one of:\n"
|
|
867
|
+
f" - `none`: the expectation needs no prior state (e.g. a health check). Write it "
|
|
868
|
+
f"down; do not leave the entry out.\n"
|
|
869
|
+
f" - `command` (PREFERRED): the test builds its own state by calling this "
|
|
870
|
+
f"capability's own commands first — add `steps`, the calls in order. A test that "
|
|
871
|
+
f"creates what it asserts on needs no agreed fixture and breaks when nobody "
|
|
872
|
+
f"regenerates one.\n"
|
|
873
|
+
f" - `event`: the test publishes the upstream events this capability consumes — add "
|
|
874
|
+
f"`steps`. The broker's address arrives in `{BROKER_URL_ENV}`.\n"
|
|
875
|
+
f" - `seed`: ONLY when pre-existing data is itself what is under test (a stub's "
|
|
876
|
+
f"canned records, say) — add `because`, and `provided_by`: the id of a proposed "
|
|
877
|
+
f"expectation stating that seeded data, so the implementer assesses it.\n"
|
|
878
|
+
f"Never a component's database: a test does not write the implementation's storage.\n"
|
|
879
|
+
f"Existing fixtures in the implementation repository are NOT a reason to use `seed` — "
|
|
880
|
+
f"prefer building the state through the contract.\n"
|
|
881
|
+
f"RULE: anything a dataset needs that this capability's contract does not ALREADY "
|
|
882
|
+
f"offer — a command or subscription this task introduces, seeded data at fixed values "
|
|
883
|
+
f"— must also be proposed as an expectation of its own, because the implementer only "
|
|
884
|
+
f"reads the expectations.\n\n"
|
|
755
885
|
f"Anything the task and the standing context do not determine, and that no proposed "
|
|
756
886
|
f"value can settle because it is a question of what the behaviour should BE, goes in "
|
|
757
887
|
f"`open_questions` — never into an invented statement. An open question is not a "
|
|
@@ -773,8 +903,9 @@ class ClaudeCodeTesterEngine:
|
|
|
773
903
|
+ (f" conforming to:\n\n```json\n{json.dumps(schema, indent=2)}\n```"
|
|
774
904
|
if schema else
|
|
775
905
|
" with `expectations` (a list of objects, each with `id`, `statement` and "
|
|
776
|
-
"`handle`)
|
|
777
|
-
"`
|
|
906
|
+
"`handle`), `datasets` (one object per expectation, with `expectation` and `via`, "
|
|
907
|
+
"as described above) and `open_questions` (a list of strings, or of objects with "
|
|
908
|
+
"`about` and `question`; empty when the task determines everything).")
|
|
778
909
|
)
|
|
779
910
|
return "\n\n".join(sections)
|
|
780
911
|
|
|
@@ -78,19 +78,21 @@ def test_propose_acceptance_accepts_exactly_the_contracted_payload():
|
|
|
78
78
|
assert offer.request_schema["properties"]["components"]["type"] == "array"
|
|
79
79
|
|
|
80
80
|
|
|
81
|
-
def
|
|
81
|
+
def test_propose_acceptance_replies_with_expectations_datasets_and_open_questions():
|
|
82
|
+
"""`datasets` is required: every proposal says how each test's state comes to exist
|
|
83
|
+
(ADR-FTA-0003)."""
|
|
82
84
|
offer = card.load(cards_path()).queries["propose-acceptance"]
|
|
83
85
|
(completion,) = offer.completion_schema
|
|
84
86
|
fields, required = _fields(completion)
|
|
85
|
-
assert fields == {"expectations", "open_questions"}
|
|
86
|
-
assert required == {"expectations"}
|
|
87
|
+
assert fields == {"expectations", "datasets", "open_questions"}
|
|
88
|
+
assert required == {"expectations", "datasets"}
|
|
87
89
|
|
|
88
90
|
|
|
89
91
|
def test_test_task_accepts_exactly_the_contracted_payload():
|
|
90
92
|
offer = card.load(cards_path()).actions["test-task"]
|
|
91
93
|
fields, required = _fields(offer.request_schema)
|
|
92
94
|
assert fields == {"task_id", "title", "definition_of_done", "components", "context",
|
|
93
|
-
"remediation_context", "acceptance_surface"}
|
|
95
|
+
"remediation_context", "acceptance_surface", "datasets"}
|
|
94
96
|
assert required == {"task_id", "title", "definition_of_done", "components"}
|
|
95
97
|
|
|
96
98
|
|
|
@@ -114,6 +116,7 @@ def test_the_propose_door_is_answered_by_its_engine_with_no_handler_registered()
|
|
|
114
116
|
def judge(self, *, system, prompt, schema=None):
|
|
115
117
|
assert prompt.splitlines()[1] == "door: propose-acceptance"
|
|
116
118
|
return {"expectations": [{"id": "E1", "statement": "s", "handle": {}}],
|
|
119
|
+
"datasets": [{"expectation": "E1", "via": "none"}],
|
|
117
120
|
"open_questions": []}
|
|
118
121
|
|
|
119
122
|
actor = Actor.from_card(cards_path(), engines={"claude-code": StubEngine()})
|
|
@@ -17,7 +17,7 @@ from papeete_actor_synchronous_messaging.engine import Engine, EngineError
|
|
|
17
17
|
|
|
18
18
|
from foundry_testing_actor import engine as engine_module
|
|
19
19
|
from foundry_testing_actor.engine import (
|
|
20
|
-
LINE_BUDGET, PROPOSE_TOOLS, TEST_TOOLS, ClaudeCodeTesterEngine, _door_from_prompt,
|
|
20
|
+
BROKER_URL_ENV, DATASET_VIA, LINE_BUDGET, PROPOSE_TOOLS, TEST_TOOLS, ClaudeCodeTesterEngine, _door_from_prompt,
|
|
21
21
|
_extract_json, _line, _payload_from_prompt, _project, _proposal,
|
|
22
22
|
)
|
|
23
23
|
|
|
@@ -270,13 +270,88 @@ def test_open_questions_always_arrive_as_a_list(given, expected):
|
|
|
270
270
|
|
|
271
271
|
def test_a_sessions_extra_keys_do_not_reach_the_caller():
|
|
272
272
|
proposal = _proposal({"expectations": [], "notes": "I also looked at the stub", "x": 1})
|
|
273
|
-
assert set(proposal) == {"expectations", "open_questions"}
|
|
273
|
+
assert set(proposal) == {"expectations", "datasets", "open_questions"}
|
|
274
274
|
|
|
275
275
|
|
|
276
276
|
def test_extra_keys_inside_an_expectation_survive():
|
|
277
277
|
"""The surface is typed only as a list; an entry's own extra keys are the contract's to allow."""
|
|
278
278
|
entry = {"id": "E1", "statement": "s", "handle": {}, "component": "backend", "why": "DoD 1"}
|
|
279
|
-
|
|
279
|
+
proposal = _proposal({"expectations": [entry], "datasets": [NONE("E1")]})
|
|
280
|
+
assert proposal["expectations"] == [entry]
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
# ── datasets: how each test's state comes to exist (ADR-FTA-0003) ───────────────────────────
|
|
284
|
+
|
|
285
|
+
def NONE(expectation):
|
|
286
|
+
return {"expectation": expectation, "via": "none"}
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
TWO = [{"id": "E1", "statement": "an archived anchor refuses an update", "handle": "PATCH"},
|
|
290
|
+
{"id": "E2", "statement": "the stub serves anchor 0001", "handle": "GET"}]
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
def test_every_via_the_contract_allows_is_accepted():
|
|
294
|
+
datasets = [
|
|
295
|
+
{"expectation": "E1", "via": "command", "steps": ["POST /anchors", "POST …/archive"]},
|
|
296
|
+
{"expectation": "E2", "via": "seed", "because": "the canned record IS the subject",
|
|
297
|
+
"provided_by": "E2"},
|
|
298
|
+
]
|
|
299
|
+
assert _proposal({"expectations": TWO, "datasets": datasets})["datasets"] == datasets
|
|
300
|
+
assert set(DATASET_VIA) == {"none", "command", "event", "seed"}
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
def test_a_proposal_with_expectations_and_no_datasets_is_refused():
|
|
304
|
+
with pytest.raises(EngineError, match="no `datasets`"):
|
|
305
|
+
_proposal({"expectations": TWO})
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def test_an_expectation_without_a_dataset_is_refused_even_when_it_needs_nothing():
|
|
309
|
+
"""`none` is a claim that can be wrong, so it is written — never inferred from an absence."""
|
|
310
|
+
with pytest.raises(EngineError, match="no dataset for expectation.*E2"):
|
|
311
|
+
_proposal({"expectations": TWO, "datasets": [NONE("E1")]})
|
|
312
|
+
|
|
313
|
+
|
|
314
|
+
@pytest.mark.parametrize("datasets, match", [
|
|
315
|
+
([NONE("E1"), NONE("E2"), NONE("E9")], "does not propose"),
|
|
316
|
+
([NONE("E1"), NONE("E1"), NONE("E2")], "two datasets"),
|
|
317
|
+
([NONE("E1"), {"expectation": "E2", "via": "database"}], "never a component's own storage"),
|
|
318
|
+
([NONE("E1"), {"expectation": "E2", "via": "command"}], "lists no `steps`"),
|
|
319
|
+
([NONE("E1"), {"expectation": "E2", "via": "event", "steps": []}], "lists no `steps`"),
|
|
320
|
+
([NONE("E1"), {"expectation": "E2", "via": "seed", "provided_by": "E2"}], "without saying"),
|
|
321
|
+
([NONE("E1"), {"expectation": "E2", "via": "seed", "because": "canned"}], "provided_by"),
|
|
322
|
+
([NONE("E1"), {"expectation": "E2", "via": "seed", "because": "canned",
|
|
323
|
+
"provided_by": "E7"}], "provided_by"),
|
|
324
|
+
([NONE("E1"), "E2: none"], "not an object"),
|
|
325
|
+
])
|
|
326
|
+
def test_a_dataset_that_hides_a_gap_is_refused(datasets, match):
|
|
327
|
+
with pytest.raises(EngineError, match=match):
|
|
328
|
+
_proposal({"expectations": TWO, "datasets": datasets})
|
|
329
|
+
|
|
330
|
+
|
|
331
|
+
def test_an_empty_proposal_needs_no_datasets():
|
|
332
|
+
assert _proposal({"expectations": []})["datasets"] == []
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def test_the_proposal_prompt_asks_for_a_dataset_per_expectation_and_prefers_commands(engine,
|
|
336
|
+
tmp_path):
|
|
337
|
+
prompt = engine._proposal_prompt(PAYLOAD, None, tmp_path)
|
|
338
|
+
assert "`datasets`" in prompt and "one entry per expectation" in prompt
|
|
339
|
+
assert "`command` (PREFERRED)" in prompt
|
|
340
|
+
assert "the implementer never sees it" in prompt
|
|
341
|
+
assert "Never a component's database" in prompt
|
|
342
|
+
assert "must also be proposed as an expectation of its own" in prompt
|
|
343
|
+
assert BROKER_URL_ENV in prompt
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def test_the_test_prompt_builds_state_as_the_datasets_say(engine, config):
|
|
347
|
+
datasets = [{"expectation": "EXP-001", "via": "command", "steps": ["POST /widgets"]}]
|
|
348
|
+
with_plan = _prompt(engine, config, {**PAYLOAD, "datasets": datasets})
|
|
349
|
+
assert "How each test builds the data it runs against" in with_plan
|
|
350
|
+
assert '"POST /widgets"' in with_plan
|
|
351
|
+
|
|
352
|
+
without = _prompt(engine, config, PAYLOAD)
|
|
353
|
+
assert "Build each test's state through this capability's own contract" in without
|
|
354
|
+
assert "never write a component's storage directly" in without
|
|
280
355
|
|
|
281
356
|
|
|
282
357
|
# ── the propose door end to end, without a session ──────────────────────────────────────────
|
|
@@ -307,7 +382,8 @@ def proposed(engine, monkeypatch):
|
|
|
307
382
|
return _run
|
|
308
383
|
|
|
309
384
|
|
|
310
|
-
ANSWER = '```json\n{"expectations": [{"id": "E1", "statement": "s", "handle": {}}]
|
|
385
|
+
ANSWER = ('```json\n{"expectations": [{"id": "E1", "statement": "s", "handle": {}}],'
|
|
386
|
+
' "datasets": [{"expectation": "E1", "via": "none"}]}\n```')
|
|
311
387
|
|
|
312
388
|
|
|
313
389
|
def test_the_propose_session_is_given_no_tool_that_writes(proposed):
|
|
@@ -363,8 +439,10 @@ def test_the_propose_door_removes_both_clones_on_failure(proposed):
|
|
|
363
439
|
def test_the_proposal_comes_back_as_the_reply(proposed):
|
|
364
440
|
reply, _ = proposed(
|
|
365
441
|
'```json\n{"expectations": [{"id": "E1", "statement": "s", "handle": {"path": "/w"}}],'
|
|
442
|
+
' "datasets": [{"expectation": "E1", "via": "command", "steps": ["POST /w"]}],'
|
|
366
443
|
' "open_questions": "which status codes?"}\n```')
|
|
367
444
|
assert reply == {"expectations": [{"id": "E1", "statement": "s", "handle": {"path": "/w"}}],
|
|
445
|
+
"datasets": [{"expectation": "E1", "via": "command", "steps": ["POST /w"]}],
|
|
368
446
|
"open_questions": ["which status codes?"]}
|
|
369
447
|
|
|
370
448
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/cli.py
RENAMED
|
File without changes
|
{foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/config.py
RENAMED
|
File without changes
|
{foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/conformance.py
RENAMED
|
File without changes
|
{foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/correlation.py
RENAMED
|
File without changes
|
{foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/grounding.py
RENAMED
|
File without changes
|
{foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/handler.py
RENAMED
|
File without changes
|
{foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/instance.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{foundry_testing_actor-0.1.1 → foundry_testing_actor-0.2.0}/src/foundry_testing_actor/serve.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|