foundry-implementation-actor 0.5.1__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/CLAUDE.md +9 -1
  2. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/PKG-INFO +26 -1
  3. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/README.md +25 -0
  4. foundry_implementation_actor-0.7.0/adr/ADR-FIA-0007-a-sessions-budget-is-an-environment-setting.md +105 -0
  5. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/adr/README.md +1 -0
  6. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/docker/Dockerfile +6 -0
  7. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/pyproject.toml +1 -1
  8. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/__init__.py +7 -0
  9. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/cli.py +6 -1
  10. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/config.py +30 -9
  11. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/engine.py +36 -15
  12. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/schemas/agentic-context.schema.yaml +2 -1
  13. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/serve.py +15 -2
  14. foundry_implementation_actor-0.7.0/src/foundry_implementation_actor/settings.py +102 -0
  15. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/test_cli.py +18 -0
  16. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/test_config.py +42 -0
  17. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/test_engine.py +88 -2
  18. foundry_implementation_actor-0.7.0/tests/test_settings.py +107 -0
  19. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/uv.lock +1 -1
  20. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/.github/workflows/ci.yml +0 -0
  21. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/.github/workflows/release.yml +0 -0
  22. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/.gitignore +0 -0
  23. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/adr/ADR-FIA-0001-the-machinery-leaves-the-capability.md +0 -0
  24. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/adr/ADR-FIA-0002-testing-leaves-this-actors-contract.md +0 -0
  25. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/adr/ADR-FIA-0003-the-dev-test-agreement-has-no-home-yet.md +0 -0
  26. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/adr/ADR-FIA-0004-the-three-amigos-round.md +0 -0
  27. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/adr/ADR-FIA-0005-the-actor-ships-an-image.md +0 -0
  28. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/adr/ADR-FIA-0006-the-image-is-published-to-two-registries.md +0 -0
  29. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/adr/template.md +0 -0
  30. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/examples/ACME.PARTS.CAP.SUP.007.WID-implementation/Dockerfile +0 -0
  31. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/examples/ACME.PARTS.CAP.SUP.007.WID-implementation/actor-agentic-context.yaml +0 -0
  32. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/examples/README.md +0 -0
  33. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/scripts/probe_grounding.py +0 -0
  34. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/cards/actor-data.yaml +0 -0
  35. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/cards/actor-message.yaml +0 -0
  36. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/cards/actor-synchronous-messaging.yaml +0 -0
  37. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/cards/actor.yaml +0 -0
  38. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/conformance.py +0 -0
  39. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/correlation.py +0 -0
  40. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/grounding.py +0 -0
  41. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/handler.py +0 -0
  42. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/src/foundry_implementation_actor/instance.py +0 -0
  43. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/conftest.py +0 -0
  44. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/fixtures/broken/actor-agentic-context.yaml +0 -0
  45. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/test_cards.py +0 -0
  46. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/test_conformance.py +0 -0
  47. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/test_grounding.py +0 -0
  48. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/test_handler.py +0 -0
  49. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/test_instance.py +0 -0
  50. {foundry_implementation_actor-0.5.1 → foundry_implementation_actor-0.7.0}/tests/test_portability.py +0 -0
@@ -29,7 +29,7 @@ There is no separate lint/format command configured in this repo.
29
29
 
30
30
  ## Architecture
31
31
 
32
- Seven modules under `src/foundry_implementation_actor/`:
32
+ Eight modules under `src/foundry_implementation_actor/`:
33
33
 
34
34
  - **`config.py`** — `CapabilityConfig`. The heart. Loads the sidecar and derives **every**
35
35
  rendering of the capability id from two declared fields (`capability`, `source_repo`). Also
@@ -48,6 +48,10 @@ Seven modules under `src/foundry_implementation_actor/`:
48
48
  the derived wire contract only (door ids, each door's request/completion schema and engine).
49
49
  Prose and `actor.yaml`'s `name:` are deliberately not compared — a use should name its own
50
50
  capability. `lint` runs it beside the sidecar gate.
51
+ - **`settings.py`** — `Settings`. How much a session may spend: turns and seconds, per door, as
52
+ one `field → environment variable` table with a `from_env` that refuses a value it cannot read.
53
+ `serve` fills the engine from it (ADR-FIA-0007). The defaults live here, not in `engine.py`,
54
+ because `engine.py` names the variable when a budget runs out.
51
55
  - **`cli.py`** — argparse wiring only, no logic of its own.
52
56
 
53
57
  Beside them, two folders of committed contract, both shipped in the wheel:
@@ -93,6 +97,10 @@ Beside them, two folders of committed contract, both shipped in the wheel:
93
97
  authenticate (ADR-FIA-0006). `release.yml` `docker tag`s one build into both so they hold one
94
98
  digest — never add a second `docker build`, and never hardcode a registry: the product names
95
99
  itself in `vars.PRODUCT_IMAGE`, the same reason `src/` names no capability.
100
+ - **A budget is an environment setting, and a door that runs out says what to change.** Turns and
101
+ timeouts are `Settings` fields with an `ENV` entry each, not constructor defaults `serve` never
102
+ passes (ADR-FIA-0007). Never add a knob without an entry in `ENV` and `ENGINE_KWARGS` — and never
103
+ put one in the sidecar: a capability declares what it is, not how long its actor may think.
96
104
  - **Never set `ANTHROPIC_API_KEY` in a container running this.** In `claude -p` non-interactive
97
105
  mode an API key in the environment is always preferred over `CLAUDE_CODE_OAUTH_TOKEN`, silently
98
106
  routing every session through metered billing. There is no warning; the only symptom is the bill.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: foundry-implementation-actor
3
- Version: 0.5.1
3
+ Version: 0.7.0
4
4
  Summary: Runs a headless Claude Code implementation session against one capability's own repo — a papeete-actor for one use, with the capability supplied by a sidecar.
5
5
  Project-URL: Homepage, https://github.com/papeete-hub/foundry-implementation-actor
6
6
  Author-email: Papeete Consulting <yoann.remy@outlook.com>
@@ -287,6 +287,31 @@ Publishing additionally needs `IMAGE_REGISTRY` and `BUILDKIT_HOST`. There is no
287
287
  no docker socket anywhere in this design — `buildctl` is a client, which is why an actor running
288
288
  this can be an ordinary Pod.
289
289
 
290
+ ## What a session may spend
291
+
292
+ Environment, read once at boot by `serve`; constructor keywords on `Settings` for an embedder. The
293
+ budget is **not** a sidecar field: a capability declares what it is, not how long its actor may
294
+ think (ADR-FIA-0007).
295
+
296
+ | variable | default | what it costs to raise |
297
+ |---|---|---|
298
+ | `MAX_TURNS` | `60` | `implement-task`'s turns. A turn is a model call plus a tool call; raising it buys a slower-to-navigate repo more room, and buys a session that has lost the plot more room to keep losing it |
299
+ | `SESSION_TIMEOUT_S` | `1800` | `implement-task`'s wall clock. The caller's own door timeout has to exceed it, or a slow success arrives as "did not answer" |
300
+ | `ASSESS_MAX_TURNS` | `15` | `assess-task`'s turns. It reads and answers; it cannot write |
301
+ | `ASSESS_TIMEOUT_S` | `600` | `assess-task`'s wall clock. Round 0 blocks on it, before anything is built |
302
+ | `CLONE_TIMEOUT_S` | `120` | the full clone, per door call |
303
+ | `FETCH_TIMEOUT_S` | `120` | each `ground_in` fetch, per door call |
304
+
305
+ **The defaults have not moved** since these became reachable; they are what every use was already
306
+ running. A value that is not a positive integer is refused at boot, naming itself, rather than
307
+ silently falling back — so a raised budget that was misspelt crash-loops with the reason on stdout
308
+ instead of changing nothing. The four session knobs are on the `actor-started` record too, so a run
309
+ that ran out of budget can be read against the budget it actually had.
310
+
311
+ A door that runs out says so in those terms: `implement-task ran out of turns (max_turns=60) and
312
+ was stopped mid-work, so nothing it produced is kept — raise MAX_TURNS on this actor's Deployment,
313
+ or narrow the task.`
314
+
290
315
  ## CLI
291
316
 
292
317
  ```bash
@@ -264,6 +264,31 @@ Publishing additionally needs `IMAGE_REGISTRY` and `BUILDKIT_HOST`. There is no
264
264
  no docker socket anywhere in this design — `buildctl` is a client, which is why an actor running
265
265
  this can be an ordinary Pod.
266
266
 
267
+ ## What a session may spend
268
+
269
+ Environment, read once at boot by `serve`; constructor keywords on `Settings` for an embedder. The
270
+ budget is **not** a sidecar field: a capability declares what it is, not how long its actor may
271
+ think (ADR-FIA-0007).
272
+
273
+ | variable | default | what it costs to raise |
274
+ |---|---|---|
275
+ | `MAX_TURNS` | `60` | `implement-task`'s turns. A turn is a model call plus a tool call; raising it buys a slower-to-navigate repo more room, and buys a session that has lost the plot more room to keep losing it |
276
+ | `SESSION_TIMEOUT_S` | `1800` | `implement-task`'s wall clock. The caller's own door timeout has to exceed it, or a slow success arrives as "did not answer" |
277
+ | `ASSESS_MAX_TURNS` | `15` | `assess-task`'s turns. It reads and answers; it cannot write |
278
+ | `ASSESS_TIMEOUT_S` | `600` | `assess-task`'s wall clock. Round 0 blocks on it, before anything is built |
279
+ | `CLONE_TIMEOUT_S` | `120` | the full clone, per door call |
280
+ | `FETCH_TIMEOUT_S` | `120` | each `ground_in` fetch, per door call |
281
+
282
+ **The defaults have not moved** since these became reachable; they are what every use was already
283
+ running. A value that is not a positive integer is refused at boot, naming itself, rather than
284
+ silently falling back — so a raised budget that was misspelt crash-loops with the reason on stdout
285
+ instead of changing nothing. The four session knobs are on the `actor-started` record too, so a run
286
+ that ran out of budget can be read against the budget it actually had.
287
+
288
+ A door that runs out says so in those terms: `implement-task ran out of turns (max_turns=60) and
289
+ was stopped mid-work, so nothing it produced is kept — raise MAX_TURNS on this actor's Deployment,
290
+ or narrow the task.`
291
+
267
292
  ## CLI
268
293
 
269
294
  ```bash
@@ -0,0 +1,105 @@
1
+ ---
2
+ id: ADR-FIA-0007
3
+ title: "A session's budget is an environment setting — and a door that runs out says what to change"
4
+ status: Accepted
5
+ date: 2026-09-16
6
+ supersedes: []
7
+ references:
8
+ - src/foundry_implementation_actor/settings.py
9
+ - src/foundry_implementation_actor/serve.py
10
+ - src/foundry_implementation_actor/engine.py
11
+ ---
12
+
13
+ # ADR-FIA-0007 — A session's budget is an environment setting
14
+
15
+ ## Context
16
+
17
+ `ClaudeCodeEngine` has always taken its budget as constructor keywords: `max_turns`,
18
+ `session_timeout`, `assess_max_turns`, `assess_timeout`, plus `clone_timeout` and `fetch_timeout`.
19
+ `serve.py` constructed it as `ClaudeCodeEngine(config)` and passed none of them.
20
+
21
+ So for every actor running from the image this package publishes — which is every actor, since
22
+ ADR-FIA-0005 made a use one sidecar and no Python — the budget was whatever `engine.py` said, and
23
+ the only way to move it was to edit and re-release the package. The keywords were reachable in
24
+ principle and unreachable in practice.
25
+
26
+ A live end-to-end run on 2026-09-15 turned that into a real cost. An `implement-task` session hit
27
+ `--max-turns 60` after 470s of a 1800s timeout. It had made the actual fix by turn 14, then spent
28
+ turns 18–60 building a unit test nobody had asked for, and was killed with the work uncommitted.
29
+ The repo in question is a .NET solution that restores packages on many of its turns, so 60 turns
30
+ buys it materially less reading than it buys a Python one — exactly the case an operator should be
31
+ able to answer from a Deployment.
32
+
33
+ The failure said none of this:
34
+
35
+ ```
36
+ EngineError: claude session failed (subtype=error_max_turns):
37
+ ```
38
+
39
+ Not what the limit was, not which of the two doors hit it, not what to change. The only way to
40
+ learn any of it was to read the CLI's own transcript out of the Pod's `/tmp/.claude/projects`
41
+ before the Pod went away.
42
+
43
+ `foundry-task-orchestration-actor` had already answered the first half of this for its own knobs: a
44
+ `settings.py` holding one `field name → environment variable` table, a `from_env` that reads it,
45
+ and a `serve` that passes the result.
46
+
47
+ ## Decision
48
+
49
+ 1. **The budget is an environment setting, not a sidecar field.** A new `settings.py`, modelled on
50
+ the orchestration actor's: a frozen `Settings` dataclass, one `ENV` table, `from_env(environ)`,
51
+ and a `SettingsError` for a value that cannot be read. Six variables — `MAX_TURNS`,
52
+ `SESSION_TIMEOUT_S`, `ASSESS_MAX_TURNS`, `ASSESS_TIMEOUT_S`, `CLONE_TIMEOUT_S`,
53
+ `FETCH_TIMEOUT_S`. `serve` reads it at boot and splats `settings.engine_kwargs()` into the
54
+ engine; an embedder passes its own `Settings` and never touches the environment.
55
+ 2. **A value that is not a positive integer is refused at boot**, naming the variable and what it
56
+ was set to. A misspelt budget crash-loops with the reason on stdout rather than silently
57
+ running under the default it was raised from.
58
+ 3. **The `actor-started` record carries the four session knobs**, so a run that ran out is
59
+ readable against the budget it actually had.
60
+ 4. **A door that runs out of budget names the budget, the door and the variable.** Both the
61
+ out-of-turns and the timeout paths, keyed off `ENV` so the remedy cannot drift from what
62
+ `from_env` reads.
63
+ 5. **The defaults do not move.** 60/1800 to implement, 15/600 to assess — unchanged.
64
+
65
+ ## Rationale
66
+
67
+ **Why not the sidecar.** The sidecar declares facts about a capability: its id, its repo, its
68
+ components, what it grounds itself in. How many turns its actor may take is not one of them. The
69
+ same capability deployed twice may want two answers; two capabilities of identical shape in
70
+ different languages want different ones for the same task. ADR-FIA-0002 drew this line already when
71
+ it took `components[].tests` out of the sidecar — a component declares what is built, not how it is
72
+ verified — and this is the same line one field over: a capability declares what it is, not how long
73
+ its actor may think.
74
+
75
+ **Why not raise the default instead.** Raising 60 to 90 would have made this particular run pass
76
+ and taught nothing. The run that failed was not short of turns in general; it was short of turns
77
+ *for that repository*, and it also spent two thirds of its budget on work outside its task. Those
78
+ are two different problems with two different owners, and a default that hides the first also hides
79
+ the second. Making the budget reachable lets the one use that needs more say so, in its own
80
+ Deployment, where the reason can be written next to the number.
81
+
82
+ **Why the failure matters as much as the setting.** A knob nobody can find is not much better than
83
+ a knob that does not exist. The evidence for this decision is an incident whose diagnosis required
84
+ `kubectl exec` into a Pod; the failure message is what makes the next one diagnosable from the log
85
+ line alone.
86
+
87
+ **Why `engine.py` imports the defaults from `settings.py` rather than the reverse.** The engine's
88
+ own failure message names the environment variable, so it needs `ENV`. One table, in one direction,
89
+ is what keeps the message, `from_env` and the README from disagreeing.
90
+
91
+ ## Consequences
92
+
93
+ - `serve.py` is no longer the place that decides a budget. An operator raises one on a Deployment.
94
+ - `DEFAULT_MAX_TURNS` and its four siblings moved from `engine.py` to `settings.py`. They are
95
+ still importable, under a new module — an embedder that imported them from `engine` must follow.
96
+ - The engine's constructor signature is unchanged, so an embedder that passed keywords is
97
+ untouched.
98
+ - `tests/test_settings.py` pins the defaults, every variable, the refusals, and that `ENV`,
99
+ `ENGINE_KWARGS` and the dataclass's own fields name the same set — a field added without a
100
+ variable cannot go quiet.
101
+ - **Follow-up: `foundry-testing-actor` carries the same shape** — `DEFAULT_MAX_TURNS = 60`, a
102
+ `propose_max_turns` beside it, and a `serve` that passes neither. It deserves the identical
103
+ change, as its own pass and its own release; its doors are not the ones that failed here.
104
+ - Not addressed: a session that spends its budget on work outside its task. That is a prompt
105
+ question, and turning it into a budget question is what this ADR declines to do.
@@ -17,6 +17,7 @@ are named or placed (`ADR-ECO-*` in
17
17
  | [ADR-FIA-0004](./ADR-FIA-0004-the-three-amigos-round.md) | The three amigos round — the tester proposes, this actor answers, a human breaks the tie | Accepted |
18
18
  | [ADR-FIA-0005](./ADR-FIA-0005-the-actor-ships-an-image.md) | The actor ships an image, and renders its own cards into it — a use is one sidecar | Proposed |
19
19
  | [ADR-FIA-0006](./ADR-FIA-0006-the-image-is-published-to-two-registries.md) | The image is published to two registries, and names neither in its source | Proposed |
20
+ | [ADR-FIA-0007](./ADR-FIA-0007-a-sessions-budget-is-an-environment-setting.md) | A session's budget is an environment setting — and a door that runs out says what to change | Accepted |
20
21
 
21
22
  ## Authoring
22
23
 
@@ -97,6 +97,12 @@ ENV HOME=/home/actor \
97
97
  # BUILDKIT_HOST, IMAGE_REGISTRY, DOCKER_CONFIG where images are built, pushed, and the
98
98
  # credential buildctl resolves CLIENT-side before handing it to the
99
99
  # daemon (the daemon does not authenticate on a client's behalf).
100
+ #
101
+ # And, optionally, how much a session may spend: MAX_TURNS / SESSION_TIMEOUT_S (implement-task),
102
+ # ASSESS_MAX_TURNS / ASSESS_TIMEOUT_S (assess-task), CLONE_TIMEOUT_S, FETCH_TIMEOUT_S. Deliberately
103
+ # NOT given defaults here: `settings.py` holds them, so a use reads one number in one place, and an
104
+ # ENV line in this image would be a second copy to keep in step (ADR-FIA-0007). A value that is not
105
+ # a positive integer is refused at boot, naming itself.
100
106
  EXPOSE 8080
101
107
 
102
108
  # `serve`, not a copied-in app.py. Sixty lines of observability wiring used to live in every use's
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "foundry-implementation-actor"
3
- version = "0.5.1"
3
+ version = "0.7.0"
4
4
  description = "Runs a headless Claude Code implementation session against one capability's own repo — a papeete-actor for one use, with the capability supplied by a sidecar."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.11"
@@ -15,6 +15,10 @@ Wiring one up is four lines:
15
15
  engines={config.engine: ClaudeCodeEngine(config)},
16
16
  actions={"implement-task": make_implement_task(config)})
17
17
 
18
+ `Settings` carries how much a session may spend — turns and seconds, per door. `serve` reads it
19
+ from the environment; the four lines above take the defaults, and an embedder passes
20
+ `**Settings(...).engine_kwargs()` to choose its own (ADR-FIA-0007).
21
+
18
22
  `correlation` is exported too — an entrypoint installs its filter on the root logger's handlers
19
23
  after configuring observability, so every record the process emits carries this request's ids.
20
24
 
@@ -28,6 +32,7 @@ from .engine import ClaudeCodeEngine
28
32
  from .handler import HandlerError, make_implement_task
29
33
  from .instance import render_cards
30
34
  from .serve import ServeError, serve
35
+ from .settings import Settings, SettingsError
31
36
  from . import conformance, correlation, grounding, instance
32
37
 
33
38
  __all__ = [
@@ -48,5 +53,7 @@ __all__ = [
48
53
  "render_cards",
49
54
  "serve",
50
55
  "ServeError",
56
+ "Settings",
57
+ "SettingsError",
51
58
  "version",
52
59
  ]
@@ -22,6 +22,7 @@ from . import conformance
22
22
  from .config import CapabilityConfig, ConfigError, lint, version
23
23
  from .instance import render_cards
24
24
  from .serve import DEFAULT_PORT, ServeError, serve
25
+ from .settings import SettingsError
25
26
 
26
27
  _REGISTRY_PLACEHOLDER = "<registry>"
27
28
 
@@ -115,9 +116,13 @@ def _cmd_serve(args: argparse.Namespace) -> int:
115
116
  # No try/except around the boot itself. A misconfigured actor that starts anyway and refuses
116
117
  # every caller at the door is strictly worse than a pod that crash-loops with the reason on
117
118
  # stdout, which is what an uncaught ConfigError produces here.
119
+ #
120
+ # `SettingsError` is the same kind of thing one env var over — `MAX_TURNS=ninety` is a
121
+ # misconfiguration an operator has just made and is watching for, and one line naming the
122
+ # variable reads better in `kubectl logs` than the traceback under it.
118
123
  try:
119
124
  serve(Path(args.folder), port=args.port)
120
- except (ConfigError, ServeError) as e:
125
+ except (ConfigError, ServeError, SettingsError) as e:
121
126
  print(f" FAIL {e}", file=sys.stderr)
122
127
  return 2
123
128
  return 0
@@ -12,7 +12,7 @@ They are now derivations of two fields. Nothing in this package spells a capabil
12
12
  capability ACME.PARTS.CAP.SUP.007.WID ← the only id anyone writes
13
13
  source_repo <owner>/ACME.PARTS.CAP.SUP.007.WID-impl ← and the only repo
14
14
 
15
- actor_name <repo half of source_repo>
15
+ actor_name {capability}-implementation
16
16
  actor_slug same, lowercased, dots to hyphens
17
17
  git_author_name actor_name
18
18
  git_author_email {actor_slug}@users.noreply.github.com
@@ -21,6 +21,15 @@ They are now derivations of two fields. Nothing in this package spells a capabil
21
21
  image_name(c) {capability lowercased}-{c}
22
22
  image_ref(r, c, v) {r}/{capability_path}/{c}:{v}
23
23
 
24
+ IDENTITY IS `capability` + ROLE, NOT THE REPO HALF. `actor_name` used to be whatever came after
25
+ the `/` in `source_repo`, which reads as a derivation but is really an assumption: that one
26
+ repository holds exactly one actor. Every sidecar in existence satisfies
27
+ `source_repo == "<owner>/" + capability + "-" + ROLE`, so deriving the name from the two facts it
28
+ was always shorthand for produces the identical string — and it keeps producing the right one when
29
+ a capability's three actors come to share a repository, where the repo half would name all three
30
+ the same thing. `source_repo` is unchanged and still required: it is where this actor clones from
31
+ and pushes to, which is a different question from who it is.
32
+
24
33
  THE IMAGE REF IS A THREE-WAY CONTRACT. The testing actor recomputes the identical string and the
25
34
  orchestration actor parses it back apart. `image_ref` must therefore stay byte-identical to what
26
35
  those two agree on; it is derivation output or nothing, and no tag scheme is invented here.
@@ -53,6 +62,10 @@ _CARDS_PATH = Path(__file__).resolve().parent / "cards"
53
62
  # segments rather than concatenated.
54
63
  _CAPABILITY_SEGMENT = "cap"
55
64
 
65
+ # The role this package plays for the capability it serves. Half of this actor's identity — see
66
+ # `CapabilityConfig.actor_name` — and the half that is a property of the package, not of the use.
67
+ ROLE = "implementation"
68
+
56
69
  # What an unsubstituted placeholder looks like: a bare lowercase word in braces, and nothing else.
57
70
  # Narrow on purpose — see `CapabilityConfig.expand`.
58
71
  _PLACEHOLDER = re.compile(r"\{([a-z_]+)\}")
@@ -187,9 +200,17 @@ class CapabilityConfig:
187
200
  ground_in = tuple(_grounding(entry, source, i)
188
201
  for i, entry in enumerate(raw["ground_in"] or ()))
189
202
 
203
+ source_repo = str(raw["source_repo"])
204
+ owner, _, repo = source_repo.partition("/")
205
+ if not owner or not repo:
206
+ # `actor_name` used to be the repo half and validated this shape on the way past. It
207
+ # is derived from `capability` now, so the field every clone and push URL is built
208
+ # from needs checking in its own right rather than by a side effect.
209
+ raise ConfigError(f"{source}: source_repo '{source_repo}' is not '<owner>/<repo>'")
210
+
190
211
  config = cls(
191
212
  capability=str(raw["capability"]),
192
- source_repo=str(raw["source_repo"]),
213
+ source_repo=source_repo,
193
214
  registry_repo=str(raw["registry_repo"]),
194
215
  engine=str(raw["engine"]),
195
216
  components=components,
@@ -204,13 +225,13 @@ class CapabilityConfig:
204
225
 
205
226
  @property
206
227
  def actor_name(self) -> str:
207
- """The repo half of `source_repo` — this actor's own name, and its git author name."""
208
- owner, _, repo = self.source_repo.partition("/")
209
- if not owner or not repo:
210
- raise ConfigError(
211
- f"source_repo '{self.source_repo}' is not '<owner>/<repo>'"
212
- )
213
- return repo
228
+ """`{capability}-{ROLE}` — this actor's own name, and its git author name.
229
+
230
+ Not the repo half of `source_repo`, which is the same string for every sidecar that
231
+ exists but stops being this actor's name alone the moment a capability's actors share a
232
+ repository. See this module's own docstring.
233
+ """
234
+ return f"{self.capability}-{ROLE}"
214
235
 
215
236
  @property
216
237
  def actor_slug(self) -> str:
@@ -48,16 +48,9 @@ from papeete_actor_synchronous_messaging.engine import EngineError
48
48
 
49
49
  from . import correlation, grounding
50
50
  from .config import CapabilityConfig
51
-
52
- DEFAULT_CLONE_TIMEOUT_S = 120
53
- DEFAULT_SESSION_TIMEOUT_S = 1800
54
- DEFAULT_MAX_TURNS = 60
55
-
56
- # The assess door reads and answers; it never writes. Its budget is smaller than the implement
57
- # door's on every axis, and its tool list is the enforcement — a door that CANNOT write beats one
58
- # asked not to. See ADR-FIA-0004.
59
- DEFAULT_ASSESS_TIMEOUT_S = 600
60
- DEFAULT_ASSESS_MAX_TURNS = 15
51
+ from .settings import (DEFAULT_ASSESS_MAX_TURNS, DEFAULT_ASSESS_TIMEOUT_S,
52
+ DEFAULT_CLONE_TIMEOUT_S, DEFAULT_MAX_TURNS,
53
+ DEFAULT_SESSION_TIMEOUT_S, ENV)
61
54
 
62
55
  IMPLEMENT_TOOLS = "Bash,Read,Edit,Write,Glob,Grep"
63
56
  ASSESS_TOOLS = "Read,Glob,Grep"
@@ -67,6 +60,13 @@ ASSESS_TOOLS = "Read,Glob,Grep"
67
60
  IMPLEMENT_DOOR = "implement-task"
68
61
  ASSESS_DOOR = "assess-task"
69
62
 
63
+ # Which environment variable moves each door's budget. Keyed off `ENV` rather than spelled again,
64
+ # so a run that says "raise MAX_TURNS" is naming the variable `Settings.from_env` actually reads.
65
+ BUDGET_VARS = {
66
+ IMPLEMENT_DOOR: (ENV["max_turns"], ENV["session_timeout_s"]),
67
+ ASSESS_DOOR: (ENV["assess_max_turns"], ENV["assess_timeout_s"]),
68
+ }
69
+
70
70
 
71
71
  # ── stream-json projection: keeping every emitted log line inside Loki's max_line_size ────────
72
72
  #
@@ -275,7 +275,9 @@ class ClaudeCodeEngine:
275
275
  self.session_timeout = session_timeout
276
276
  self.max_turns = max_turns
277
277
  # Constructor kwargs, not sidecar fields: how long this actor's own doors may think is
278
- # operational tuning, not something a capability declares about itself (ADR-FIA-0002).
278
+ # operational tuning, not something a capability declares about itself (ADR-FIA-0007).
279
+ # `serve` fills every one of them from the environment through `Settings.from_env()`; an
280
+ # embedder passes its own and never touches the environment.
279
281
  self.assess_timeout = assess_timeout
280
282
  self.assess_max_turns = assess_max_turns
281
283
  self._configure_git_credentials()
@@ -326,7 +328,8 @@ class ClaudeCodeEngine:
326
328
  situational_prompt = self._situational_prompt(payload)
327
329
  with correlation.stage("claude-session", branch=branch,
328
330
  max_turns=self.max_turns, timeout_s=self.session_timeout):
329
- summary = self._invoke_claude(clone_dir, system, situational_prompt)
331
+ summary = self._invoke_claude(clone_dir, system, situational_prompt,
332
+ door=IMPLEMENT_DOOR)
330
333
  except BaseException:
331
334
  # Every failure path removes the clone. Success does not: the clone is handed off
332
335
  # live, and `handler.py` is what removes it once it has committed and pushed — or
@@ -368,8 +371,8 @@ class ClaudeCodeEngine:
368
371
  timeout_s=self.assess_timeout):
369
372
  answer = self._invoke_claude(
370
373
  clone_dir, self._assess_system(), self._assessment_prompt(payload, schema),
371
- allowed_tools=ASSESS_TOOLS, max_turns=self.assess_max_turns,
372
- timeout=self.assess_timeout,
374
+ door=ASSESS_DOOR, allowed_tools=ASSESS_TOOLS,
375
+ max_turns=self.assess_max_turns, timeout=self.assess_timeout,
373
376
  )
374
377
  finally:
375
378
  _rmtree(clone_dir)
@@ -587,6 +590,7 @@ class ClaudeCodeEngine:
587
590
  # ── the judgement itself: a claude -p session against the checked-out clone ────────────
588
591
 
589
592
  def _invoke_claude(self, clone_dir: Path, system: str, situational_prompt: str, *,
593
+ door: str = IMPLEMENT_DOOR,
590
594
  allowed_tools: str = IMPLEMENT_TOOLS,
591
595
  max_turns: int | None = None,
592
596
  timeout: int | None = None) -> str:
@@ -603,6 +607,7 @@ class ClaudeCodeEngine:
603
607
  """
604
608
  max_turns = self.max_turns if max_turns is None else max_turns
605
609
  timeout = self.session_timeout if timeout is None else timeout
610
+ turns_var, timeout_var = BUDGET_VARS.get(door, BUDGET_VARS[IMPLEMENT_DOOR])
606
611
  cmd = [
607
612
  self.claude_bin, "--print", "--output-format", "stream-json", "--verbose",
608
613
  "--append-system-prompt", system,
@@ -669,7 +674,8 @@ class ClaudeCodeEngine:
669
674
 
670
675
  if timed_out.is_set():
671
676
  raise EngineError(
672
- f"claude session for this task exceeded {timeout}s"
677
+ f"{door} exceeded its {timeout}s session budget and was killed mid-work — "
678
+ f"raise {timeout_var} on this actor's Deployment, or narrow the task"
673
679
  )
674
680
  errfile.seek(0)
675
681
  stderr = errfile.read()
@@ -678,6 +684,21 @@ class ClaudeCodeEngine:
678
684
  raise EngineError(
679
685
  f"claude (rc={proc.returncode}) produced no result event: {stderr[-2000:]}"
680
686
  )
687
+ if final.get("subtype") == "error_max_turns":
688
+ # The one failure an operator can actually act on, and the one that used to say
689
+ # least: a session cut off with the work half-done leaves a clone that is about to be
690
+ # removed and a transcript nobody is looking at, and `subtype=error_max_turns` named
691
+ # neither the budget, nor the door that hit it, nor the variable that moves it.
692
+ #
693
+ # `result` is usually empty here — the session was stopped, not finished — so it is
694
+ # quoted only when there is something in it.
695
+ said = (final.get("result") or "")[:2000]
696
+ raise EngineError(
697
+ f"{door} ran out of turns (max_turns={max_turns}) and was stopped mid-work, so "
698
+ f"nothing it produced is kept — raise {turns_var} on this actor's Deployment, or "
699
+ f"narrow the task."
700
+ + (f" Its last words: {said}" if said else "")
701
+ )
681
702
  if final.get("is_error") or proc.returncode != 0:
682
703
  raise EngineError(
683
704
  f"claude session failed (subtype={final.get('subtype')}): "
@@ -76,7 +76,8 @@ fields:
76
76
  type: string
77
77
  doc: >-
78
78
  `<owner>/<repo>` of the repository this actor clones, writes to, and pushes a branch to.
79
- The actor's own name and its git commit identity are derived from the repo half.
79
+ Where it works, not who it is: the actor's own name and its git commit identity are
80
+ `<capability>-implementation`, derived from `capability` and this package's own role.
80
81
 
81
82
  registry_repo:
82
83
  type: string
@@ -26,6 +26,7 @@ from .config import CapabilityConfig
26
26
  from .engine import ClaudeCodeEngine
27
27
  from .handler import make_implement_task
28
28
  from .instance import CARD_FILES, render_cards
29
+ from .settings import Settings
29
30
 
30
31
  DEFAULT_PORT = 8080
31
32
 
@@ -104,6 +105,9 @@ def serve(folder: str | Path = ".", port: int | None = None) -> None:
104
105
  folder = Path(folder)
105
106
  # The sidecar is the only thing in that folder this package did not put there.
106
107
  config = CapabilityConfig.load(folder)
108
+ # Read before the mailbox is built: a misspelt budget is a crash-loop with the reason on
109
+ # stdout, not a surprise at the first door call.
110
+ settings = Settings.from_env()
107
111
  cards = _cards_for(config, folder)
108
112
 
109
113
  # PORT, not a hardcoded default: an env var is how an environment moves it without editing an
@@ -116,7 +120,9 @@ def serve(folder: str | Path = ".", port: int | None = None) -> None:
116
120
  # One engine instance serves every door that names one: `Actor.judge()` hands it the door
117
121
  # id and it dispatches on that. The key comes from the sidecar rather than a literal here,
118
122
  # so a use whose sidecar names another engine is caught by its own card.
119
- engines={config.engine: ClaudeCodeEngine(config)},
123
+ # ...constructed with the budget the environment set, so `serve` is no longer the place
124
+ # that decides how long a session may think (ADR-FIA-0007).
125
+ engines={config.engine: ClaudeCodeEngine(config, **settings.engine_kwargs())},
120
126
  # `assess-task` needs no entry: it is a query with an engine and no handler, so the
121
127
  # engine's own judgement is the reply and there is no deterministic half to contain.
122
128
  actions={"implement-task": make_implement_task(config)},
@@ -124,6 +130,13 @@ def serve(folder: str | Path = ".", port: int | None = None) -> None:
124
130
  # An `event` record rather than a `print`: a restart in the middle of a run is one of the most
125
131
  # explanatory things a pipeline panel can show, and `print` reaches only the container's own
126
132
  # stdout — never the OTLP handler, so never the log backend.
133
+ # The budget is on this record because a run that ran out of it has to be diagnosable from
134
+ # the log alone: "60 turns" in the failure means nothing unless the boot line says whether 60
135
+ # was what this Deployment asked for.
127
136
  correlation.event("actor-started", actor=actor.name, port=port, capability=config.capability,
128
- cards=str(cards))
137
+ cards=str(cards),
138
+ max_turns=settings.max_turns,
139
+ session_timeout_s=settings.session_timeout_s,
140
+ assess_max_turns=settings.assess_max_turns,
141
+ assess_timeout_s=settings.assess_timeout_s)
129
142
  mailbox.serve_forever()
@@ -0,0 +1,102 @@
1
+ """`Settings` — how much a session may spend, as opposed to which capability it serves.
2
+
3
+ WHY THIS IS NOT THE SIDECAR. The sidecar declares facts about a capability: its id, its repo, its
4
+ components, what it grounds itself in. How many turns a door may take and how long it may take
5
+ them are facts about an ENVIRONMENT — a capability whose code is slow to navigate (a solution that
6
+ restores packages on every build, a monorepo whose grep is expensive) needs a larger budget for
7
+ exactly the same task definition, and the same capability deployed twice may want two answers.
8
+ A capability declares what it is, not how long its actor may think (ADR-FIA-0007; ADR-FIA-0002
9
+ drew the same line when it took testing out of the sidecar).
10
+
11
+ WHY NOT CONSTRUCTOR KEYWORDS ALONE. They were exactly that, and `serve` constructed the engine
12
+ with none of them — so the budget of every actor running from the published image was whatever
13
+ this file says, and an operator facing an out-of-turns failure had no way to move it short of
14
+ editing the package. They are still constructor keywords, for an embedder; `from_env` is how
15
+ `serve` fills them from the Pod spec, which is where an operator can actually reach.
16
+
17
+ WHERE THE DEFAULTS LIVE. Here, not in `engine.py`, because `engine.py` names `ENV` when it
18
+ reports a budget that ran out — and one table is what keeps the failure message, `from_env` and
19
+ the README from disagreeing about what an operator should set.
20
+ """
21
+ from __future__ import annotations
22
+
23
+ import os
24
+ from dataclasses import dataclass, fields, replace
25
+
26
+ from .grounding import DEFAULT_FETCH_TIMEOUT_S
27
+
28
+ DEFAULT_CLONE_TIMEOUT_S = 120
29
+ DEFAULT_SESSION_TIMEOUT_S = 1800
30
+ DEFAULT_MAX_TURNS = 60
31
+
32
+ # The assess door reads and answers; it never writes. Its budget is smaller than the implement
33
+ # door's on every axis, and its tool list is the enforcement — a door that CANNOT write beats one
34
+ # asked not to. See ADR-FIA-0004.
35
+ DEFAULT_ASSESS_TIMEOUT_S = 600
36
+ DEFAULT_ASSESS_MAX_TURNS = 15
37
+
38
+
39
+ class SettingsError(ValueError):
40
+ """An environment variable was set to something this actor cannot use."""
41
+
42
+
43
+ # field name → environment variable. One table, so `from_env`, the engine's own out-of-budget
44
+ # message and the README cannot disagree about what to set.
45
+ ENV = {
46
+ "max_turns": "MAX_TURNS",
47
+ "session_timeout_s": "SESSION_TIMEOUT_S",
48
+ "assess_max_turns": "ASSESS_MAX_TURNS",
49
+ "assess_timeout_s": "ASSESS_TIMEOUT_S",
50
+ "clone_timeout_s": "CLONE_TIMEOUT_S",
51
+ "fetch_timeout_s": "FETCH_TIMEOUT_S",
52
+ }
53
+
54
+ # Settings field → the `ClaudeCodeEngine` constructor keyword it fills. The two spellings differ
55
+ # on purpose: `_s` says "seconds" to whoever reads a Deployment, and the engine's keywords are
56
+ # already published API.
57
+ ENGINE_KWARGS = {
58
+ "max_turns": "max_turns",
59
+ "session_timeout_s": "session_timeout",
60
+ "assess_max_turns": "assess_max_turns",
61
+ "assess_timeout_s": "assess_timeout",
62
+ "clone_timeout_s": "clone_timeout",
63
+ "fetch_timeout_s": "fetch_timeout",
64
+ }
65
+
66
+
67
+ @dataclass(frozen=True)
68
+ class Settings:
69
+ max_turns: int = DEFAULT_MAX_TURNS
70
+ session_timeout_s: int = DEFAULT_SESSION_TIMEOUT_S
71
+ assess_max_turns: int = DEFAULT_ASSESS_MAX_TURNS
72
+ assess_timeout_s: int = DEFAULT_ASSESS_TIMEOUT_S
73
+ clone_timeout_s: int = DEFAULT_CLONE_TIMEOUT_S
74
+ fetch_timeout_s: int = DEFAULT_FETCH_TIMEOUT_S
75
+
76
+ @classmethod
77
+ def from_env(cls, environ: dict | None = None) -> Settings:
78
+ """Defaults, overridden by whichever of `ENV`'s variables are set and non-empty.
79
+
80
+ A typo raises `SettingsError` — at boot, where it is a crash-loop with the reason on
81
+ stdout, rather than silently falling back to the default and leaving an operator who
82
+ raised a budget wondering why nothing changed.
83
+ """
84
+ environ = os.environ if environ is None else environ
85
+ values: dict = {}
86
+ for field in fields(cls):
87
+ variable = ENV[field.name]
88
+ raw = environ.get(variable)
89
+ if raw is None or raw == "":
90
+ continue
91
+ try:
92
+ value = int(raw)
93
+ except ValueError as e:
94
+ raise SettingsError(f"{variable}={raw!r} is not an integer") from e
95
+ if value < 1:
96
+ raise SettingsError(f"{variable}={raw!r} must be at least 1")
97
+ values[field.name] = value
98
+ return replace(cls(), **values)
99
+
100
+ def engine_kwargs(self) -> dict:
101
+ """What `serve` splats into `ClaudeCodeEngine(config, **...)`."""
102
+ return {keyword: getattr(self, field) for field, keyword in ENGINE_KWARGS.items()}
@@ -48,6 +48,24 @@ def test_show_expands_the_fetch_argv(config, tmp_path, capsys):
48
48
  assert "{capability}" not in out
49
49
 
50
50
 
51
+ def test_serve_refuses_a_budget_it_cannot_read(config, tmp_path, monkeypatch, capsys):
52
+ """A misspelt budget is a misconfiguration an operator has just made and is watching for. One
53
+ line naming the variable reads better in a pod's log than the traceback under it."""
54
+ import importlib
55
+ # `foundry_implementation_actor.serve` is the FUNCTION on the package — the module is only
56
+ # reachable by name.
57
+ serve_module = importlib.import_module("foundry_implementation_actor.serve")
58
+ # The wire half is an extra this suite does not install, and it is imported before anything
59
+ # this test is about. Stand it down; the boot then reaches the budget, which is the point.
60
+ monkeypatch.setattr(serve_module, "_imports",
61
+ lambda: (object(), object(), lambda: None))
62
+ monkeypatch.setenv("MAX_TURNS", "ninety")
63
+ assert cli.main(["serve", str(tmp_path)]) == 2
64
+ err = capsys.readouterr().err
65
+ assert "FAIL" in err
66
+ assert "MAX_TURNS" in err and "ninety" in err
67
+
68
+
51
69
  def test_the_cli_needs_a_subcommand(capsys):
52
70
  with pytest.raises(SystemExit):
53
71
  cli.main([])
@@ -205,3 +205,45 @@ def test_components_may_not_be_empty(sidecar_dict, write_sidecar):
205
205
  sidecar_dict["components"] = []
206
206
  with pytest.raises(ConfigError, match="empty"):
207
207
  CapabilityConfig.load(write_sidecar(sidecar_dict))
208
+
209
+
210
+ # ── identity: capability + role, not the repo half ──────────────────────────────────────────
211
+
212
+ def test_the_actor_name_is_exactly_what_the_repo_half_used_to_be(config):
213
+ """The no-op half of the change, pinned.
214
+
215
+ `actor_name` was `source_repo.partition("/")[2]`. It is `{capability}-{ROLE}` now. Every
216
+ sidecar that exists satisfies `source_repo == "<owner>/" + capability + "-" + ROLE`, so the
217
+ two derivations agree on all of them — which is what makes this releasable on its own, ahead
218
+ of any repository moving.
219
+ """
220
+ assert config.source_repo == f"acme-lab/{config.capability}-implementation"
221
+ assert config.actor_name == config.source_repo.partition("/")[2]
222
+ assert config.actor_name == "ACME.PARTS.CAP.SUP.007.WID-implementation"
223
+
224
+
225
+ def test_the_actor_name_no_longer_follows_the_repository(sidecar_dict, write_sidecar):
226
+ """The point of the change: three actors in one repository still have three names.
227
+
228
+ A consolidated capability repository is named for the capability alone, with no role suffix,
229
+ because all three of its actors live in it. Under the old derivation this actor would have
230
+ been called `ACME.PARTS.CAP.SUP.007.WID` — and so would both of its siblings.
231
+ """
232
+ sidecar_dict["source_repo"] = "acme-lab/ACME.PARTS.CAP.SUP.007.WID"
233
+ config = CapabilityConfig.load(write_sidecar(sidecar_dict))
234
+ assert config.actor_name == "ACME.PARTS.CAP.SUP.007.WID-implementation"
235
+ assert config.git_author_name == config.actor_name
236
+ assert config.actor_slug == "acme-parts-cap-sup-007-wid-implementation"
237
+ assert config.clone_prefix("TASK-042") == \
238
+ "acme-parts-cap-sup-007-wid-implementation-TASK-042-"
239
+
240
+
241
+ def test_a_malformed_source_repo_is_still_refused(sidecar_dict, write_sidecar):
242
+ """`actor_name` used to validate this shape on its way past, and nothing else did.
243
+
244
+ `source_repo` is still what every clone and push URL is built from, so a `source_repo` that
245
+ is not `<owner>/<repo>` has to keep failing at load rather than at the first push.
246
+ """
247
+ sidecar_dict["source_repo"] = "just-a-name"
248
+ with pytest.raises(ConfigError, match="<owner>/<repo>"):
249
+ CapabilityConfig.load(write_sidecar(sidecar_dict))
@@ -14,8 +14,8 @@ import pytest
14
14
  from papeete_actor_synchronous_messaging.engine import Engine, EngineError
15
15
 
16
16
  from foundry_implementation_actor.engine import (
17
- ASSESS_TOOLS, IMPLEMENT_TOOLS, LINE_BUDGET, ClaudeCodeEngine, _door_from_prompt,
18
- _extract_json, _line, _payload_from_prompt, _project,
17
+ ASSESS_DOOR, ASSESS_TOOLS, IMPLEMENT_DOOR, IMPLEMENT_TOOLS, LINE_BUDGET, ClaudeCodeEngine,
18
+ _door_from_prompt, _extract_json, _line, _payload_from_prompt, _project,
19
19
  )
20
20
 
21
21
  PAYLOAD = {
@@ -308,6 +308,92 @@ def test_the_tool_list_removes_tools_rather_than_only_approving_some(config, tmp
308
308
  assert argv[argv.index("--tools") + 2].startswith("--")
309
309
 
310
310
 
311
+ # ── a budget that ran out says what to change ───────────────────────────────────────────────
312
+
313
+ # `--max-turns` was reached: the CLI still emits a result event, with this subtype and no result
314
+ # text worth reading. The whole point of the branch under test is that the failure has to say
315
+ # something the event itself does not.
316
+ FAKE_OUT_OF_TURNS = """
317
+ import json, sys
318
+ print(json.dumps({"type": "result", "subtype": "error_max_turns", "is_error": True,
319
+ "result": "", "num_turns": 60}))
320
+ sys.exit(1)
321
+ """
322
+
323
+ FAKE_HANGS = """
324
+ import time
325
+ time.sleep(30)
326
+ """
327
+
328
+
329
+ def _fake_claude(tmp_path, body):
330
+ script = tmp_path / "claude"
331
+ script.write_text(f"#!{sys.executable}\n{body}")
332
+ script.chmod(0o755)
333
+ return script
334
+
335
+
336
+ @pytest.fixture
337
+ def budget_engine(config, tmp_path, monkeypatch):
338
+ monkeypatch.setenv("GIT_CONFIG_GLOBAL", str(tmp_path / "gitconfig"))
339
+
340
+ def _build(body):
341
+ return ClaudeCodeEngine(config, github_token="ghs_fake",
342
+ claude_bin=str(_fake_claude(tmp_path, body)))
343
+ return _build
344
+
345
+
346
+ def test_running_out_of_turns_names_the_limit_the_door_and_the_variable(budget_engine, tmp_path):
347
+ """The failure this whole change exists for. A live run died at `error_max_turns` with the fix
348
+ written but not committed, and `claude session failed (subtype=error_max_turns):` said neither
349
+ what the limit was, nor which door hit it, nor what an operator could do about it."""
350
+ engine = budget_engine(FAKE_OUT_OF_TURNS)
351
+ with pytest.raises(EngineError) as excinfo:
352
+ engine._invoke_claude(tmp_path, "system", "prompt", door=IMPLEMENT_DOOR)
353
+ message = str(excinfo.value)
354
+ assert "implement-task" in message
355
+ assert f"max_turns={engine.max_turns}" in message
356
+ assert "MAX_TURNS" in message
357
+ # ...and it says what was lost, because the clone is removed on this path.
358
+ assert "nothing it produced is kept" in message
359
+ # `result` is empty on this subtype, and an empty quotation is noise.
360
+ assert "last words" not in message
361
+
362
+
363
+ def test_the_assess_door_names_its_own_variable_not_the_implement_one(budget_engine, tmp_path):
364
+ """Two doors, two budgets. An operator told to raise `MAX_TURNS` for an assess session that
365
+ ran out would move the wrong number and see no change."""
366
+ engine = budget_engine(FAKE_OUT_OF_TURNS)
367
+ with pytest.raises(EngineError) as excinfo:
368
+ engine._invoke_claude(tmp_path, "system", "prompt", door=ASSESS_DOOR,
369
+ allowed_tools=ASSESS_TOOLS, max_turns=engine.assess_max_turns,
370
+ timeout=engine.assess_timeout)
371
+ message = str(excinfo.value)
372
+ assert "assess-task" in message
373
+ assert f"max_turns={engine.assess_max_turns}" in message
374
+ assert "raise ASSESS_MAX_TURNS" in message
375
+ # ...and not the implement door's, which is a substring of it — hence the whole remedy phrase.
376
+ assert "raise MAX_TURNS" not in message
377
+
378
+
379
+ def test_a_timeout_names_its_own_variable_too(budget_engine, tmp_path):
380
+ engine = budget_engine(FAKE_HANGS)
381
+ with pytest.raises(EngineError) as excinfo:
382
+ engine._invoke_claude(tmp_path, "system", "prompt", door=IMPLEMENT_DOOR, timeout=1)
383
+ message = str(excinfo.value)
384
+ assert "implement-task" in message
385
+ assert "1s" in message
386
+ assert "SESSION_TIMEOUT_S" in message
387
+
388
+
389
+ def test_another_session_failure_is_left_alone(budget_engine, tmp_path):
390
+ """Only the out-of-budget paths gained a remedy. A session that failed for its own reasons
391
+ still reports the subtype and what it said — inventing a remedy for those would be worse."""
392
+ engine = budget_engine(FAKE_OUT_OF_TURNS.replace("error_max_turns", "error_during_execution"))
393
+ with pytest.raises(EngineError, match="error_during_execution"):
394
+ engine._invoke_claude(tmp_path, "system", "prompt")
395
+
396
+
311
397
  def test_the_assess_session_gets_a_smaller_budget(assessed, engine):
312
398
  _, seen = assessed('```json\n{"feasible": true}\n```')
313
399
  assert seen["max_turns"] == engine.assess_max_turns < engine.max_turns
@@ -0,0 +1,107 @@
1
+ """The session budget, as something an environment can move.
2
+
3
+ It used to be six constructor keywords that `serve` passed none of, so the only budget any actor
4
+ running from the published image could have was the one written here. These tests pin the three
5
+ things that makes true: the defaults did not change, every one of them has a variable, and a typo
6
+ is refused at boot rather than silently ignored.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import pytest
11
+
12
+ from foundry_implementation_actor.engine import BUDGET_VARS, ClaudeCodeEngine
13
+ from foundry_implementation_actor.grounding import DEFAULT_FETCH_TIMEOUT_S
14
+ from foundry_implementation_actor.settings import ENGINE_KWARGS, ENV, Settings, SettingsError
15
+
16
+
17
+ # ── the defaults, which this pass deliberately did not retune ───────────────────────────────
18
+
19
+ def test_an_empty_environment_is_the_documented_default():
20
+ """60 turns / 1800s to implement, 15 / 600 to assess. Making the budget movable is not the
21
+ same act as moving it, and these are the numbers every live use is already running."""
22
+ settings = Settings.from_env({})
23
+ assert (settings.max_turns, settings.session_timeout_s) == (60, 1800)
24
+ assert (settings.assess_max_turns, settings.assess_timeout_s) == (15, 600)
25
+ assert settings.clone_timeout_s == 120
26
+ assert settings.fetch_timeout_s == DEFAULT_FETCH_TIMEOUT_S
27
+
28
+
29
+ def test_the_assess_door_is_the_smaller_budget_by_default():
30
+ """It reads and answers; it never writes. Smaller on every axis — see ADR-FIA-0004."""
31
+ settings = Settings.from_env({})
32
+ assert settings.assess_max_turns < settings.max_turns
33
+ assert settings.assess_timeout_s < settings.session_timeout_s
34
+
35
+
36
+ # ── every field has a variable, and every variable moves its field ──────────────────────────
37
+
38
+ def test_every_field_is_reachable_from_the_environment():
39
+ """`ENV` is the one table. A field added without an entry would raise inside `from_env`
40
+ rather than quietly becoming unreachable, which is the failure this whole change exists to
41
+ remove — so it is asserted here too."""
42
+ from dataclasses import fields
43
+ assert {f.name for f in fields(Settings)} == set(ENV) == set(ENGINE_KWARGS)
44
+
45
+
46
+ @pytest.mark.parametrize("field,variable", sorted(ENV.items()))
47
+ def test_each_variable_overrides_its_own_field(field, variable):
48
+ settings = Settings.from_env({variable: "97"})
49
+ assert getattr(settings, field) == 97
50
+ # ...and nothing else moved with it.
51
+ others = {f: getattr(settings, f) for f in ENV if f != field}
52
+ assert others == {f: getattr(Settings(), f) for f in ENV if f != field}
53
+
54
+
55
+ def test_an_unset_or_empty_variable_leaves_the_default():
56
+ """An empty string is what a Deployment leaves behind when someone blanks a value rather than
57
+ deleting the `env:` entry. It means "unset", not "zero"."""
58
+ assert Settings.from_env({"MAX_TURNS": ""}).max_turns == 60
59
+
60
+
61
+ # ── a typo fails at boot, naming itself ─────────────────────────────────────────────────────
62
+
63
+ def test_a_non_integer_is_refused_and_names_the_variable():
64
+ with pytest.raises(SettingsError) as excinfo:
65
+ Settings.from_env({"MAX_TURNS": "ninety"})
66
+ assert "MAX_TURNS" in str(excinfo.value)
67
+ assert "ninety" in str(excinfo.value)
68
+
69
+
70
+ def test_zero_and_negative_are_refused():
71
+ """A budget of zero is not a budget; it is a door that cannot answer. Refuse it where the
72
+ reason reaches an operator, not at the first request."""
73
+ for value in ("0", "-1"):
74
+ with pytest.raises(SettingsError, match="ASSESS_MAX_TURNS"):
75
+ Settings.from_env({"ASSESS_MAX_TURNS": value})
76
+
77
+
78
+ def test_the_refusal_names_the_variable_that_was_wrong_not_the_first_one():
79
+ with pytest.raises(SettingsError) as excinfo:
80
+ Settings.from_env({"MAX_TURNS": "90", "ASSESS_TIMEOUT_S": "10 minutes"})
81
+ assert "ASSESS_TIMEOUT_S" in str(excinfo.value)
82
+ assert "MAX_TURNS" not in str(excinfo.value)
83
+
84
+
85
+ # ── what `serve` splats into the engine ─────────────────────────────────────────────────────
86
+
87
+ def test_engine_kwargs_are_the_engines_own_keywords(config, tmp_path, monkeypatch):
88
+ """The two spellings differ — `session_timeout_s` in a Deployment, `session_timeout` in the
89
+ constructor — so this checks the mapping against the real signature rather than a copy."""
90
+ monkeypatch.setenv("GIT_CONFIG_GLOBAL", str(tmp_path / "gitconfig"))
91
+ settings = Settings.from_env({"MAX_TURNS": "90", "ASSESS_TIMEOUT_S": "300"})
92
+ engine = ClaudeCodeEngine(config, github_token="ghs_fake", **settings.engine_kwargs())
93
+ assert engine.max_turns == 90
94
+ assert engine.assess_timeout == 300
95
+ assert engine.session_timeout == 1800
96
+ assert engine.clone_timeout == 120
97
+
98
+
99
+ # ── the failure message and the table cannot drift apart ────────────────────────────────────
100
+
101
+ def test_both_doors_name_variables_that_actually_exist():
102
+ """`BUDGET_VARS` is what an out-of-budget failure tells an operator to set. It is keyed off
103
+ `ENV`, and this is what keeps that true."""
104
+ named = {name for pair in BUDGET_VARS.values() for name in pair}
105
+ assert named <= set(ENV.values())
106
+ assert BUDGET_VARS["implement-task"] == ("MAX_TURNS", "SESSION_TIMEOUT_S")
107
+ assert BUDGET_VARS["assess-task"] == ("ASSESS_MAX_TURNS", "ASSESS_TIMEOUT_S")
@@ -18,7 +18,7 @@ wheels = [
18
18
 
19
19
  [[package]]
20
20
  name = "foundry-implementation-actor"
21
- version = "0.5.1"
21
+ version = "0.7.0"
22
22
  source = { editable = "." }
23
23
  dependencies = [
24
24
  { name = "papeete-actor-synchronous-messaging" },