autonomous-sdlc-harness 0.2.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. package/README.md +2 -2
  2. package/dist/cli.js +0 -0
  3. package/dist/commands/docs.js +2 -1
  4. package/dist/commands/docs.js.map +1 -1
  5. package/dist/commands/doctor.js +27 -6
  6. package/dist/commands/doctor.js.map +1 -1
  7. package/dist/commands/init.js +102 -20
  8. package/dist/commands/init.js.map +1 -1
  9. package/dist/config/check.js +11 -4
  10. package/dist/config/check.js.map +1 -1
  11. package/dist/config/model.js +25 -0
  12. package/dist/config/model.js.map +1 -1
  13. package/dist/core/git.js +29 -0
  14. package/dist/core/git.js.map +1 -1
  15. package/dist/core/paths.js +22 -2
  16. package/dist/core/paths.js.map +1 -1
  17. package/dist/core/prompt.js +6 -2
  18. package/dist/core/prompt.js.map +1 -1
  19. package/dist/core/writer.js +2 -0
  20. package/dist/core/writer.js.map +1 -1
  21. package/dist/doctor/checks.js +343 -28
  22. package/dist/doctor/checks.js.map +1 -1
  23. package/dist/generators/githubWorkflows.js +66 -0
  24. package/dist/generators/githubWorkflows.js.map +1 -0
  25. package/dist/generators/notifications.js +105 -19
  26. package/dist/generators/notifications.js.map +1 -1
  27. package/dist/generators/outerLoopScripts.js +17 -3
  28. package/dist/generators/outerLoopScripts.js.map +1 -1
  29. package/dist/generators/permissionProfile.js +101 -12
  30. package/dist/generators/permissionProfile.js.map +1 -1
  31. package/dist/generators/repoRoot.js +18 -2
  32. package/dist/generators/repoRoot.js.map +1 -1
  33. package/dist/generators/stateDir.js +9 -3
  34. package/dist/generators/stateDir.js.map +1 -1
  35. package/dist/machine/paths.js +4 -1
  36. package/dist/machine/paths.js.map +1 -1
  37. package/dist/remote/githubActions.js +86 -0
  38. package/dist/remote/githubActions.js.map +1 -0
  39. package/dist/retrieval/queryLog.js +6 -4
  40. package/dist/retrieval/queryLog.js.map +1 -1
  41. package/dist/retrieval/runtime.js +12 -22
  42. package/dist/retrieval/runtime.js.map +1 -1
  43. package/dist/retrieval/search.js +21 -10
  44. package/dist/retrieval/search.js.map +1 -1
  45. package/dist/retrieval/server.js +1 -1
  46. package/dist/retrieval/setup.js +2 -1
  47. package/dist/retrieval/setup.js.map +1 -1
  48. package/package.json +1 -1
  49. package/templates/README.md +3 -2
  50. package/templates/claude/context/conventions.md +1 -1
  51. package/templates/claude/context/layer.md +1 -1
  52. package/templates/claude/push-notify.env.example +7 -2
  53. package/templates/claude/settings.autonomous.json +1 -1
  54. package/templates/github/workflows/harness-resume.yml +124 -0
  55. package/templates/github/workflows/harness-run.yml +446 -0
  56. package/templates/repo/gitignore +5 -0
  57. package/templates/scripts/README.md +1 -1
  58. package/templates/scripts/autonomous-watcher.sh +1141 -143
  59. package/templates/scripts/flow-walker.sh +629 -0
  60. package/templates/scripts/flows/task_plan_writing.graph.json +192 -0
  61. package/templates/scripts/lib/flow-walker-gates.sh +165 -0
  62. package/templates/scripts/lib/harness-run-lib.sh +433 -18
  63. package/templates/scripts/remote-run.sh +1785 -0
  64. package/templates/scripts/restart-watcher.sh +21 -1
  65. package/templates/scripts/run-test-suite.sh +182 -0
  66. package/templates/scripts/scratch-run.sh +2 -1
  67. package/templates/state-dir/README-root.md +1 -1
  68. package/templates/state-dir/autonomous_logs/README.md +1 -1
  69. package/templates/state-dir/improvement_observations/README.md +2 -0
  70. package/templates/state-dir/scratch/README.md +1 -1
  71. package/templates/state-dir/test_fix_plan_reviews/README.md +9 -0
  72. package/templates/state-dir/test_fix_plans/README.md +9 -0
  73. package/templates/state-dir/test_fix_point_reviews/README.md +9 -0
  74. package/templates/state-dir/test_run_logs/README.md +11 -0
@@ -53,6 +53,10 @@
53
53
  # * the run registry (`<state_dir>/autonomous_logs/registry.json`, shaped
54
54
  # `{"runs": {"<branch>": {"status": …}}}`) is the SOURCE OF TRUTH: a record
55
55
  # whose status is `running`, `parked` or `park_loop` is a run in flight;
56
+ # * a record whose `execution` is `github-actions` is the exception: that run
57
+ # executes on GitHub, beyond the local process tree a restart tears down, so
58
+ # it is listed as `remote (not affected by a restart): <branch> <status>`
59
+ # and never blocks;
56
60
  # * a process probe for the agent binary — `${HARNESS_AGENT_CLI:-claude}`, the
57
61
  # same variable the watcher launches through — is a BACKSTOP, for a run OF
58
62
  # THIS PROJECT whose record has not been written yet or was written by a
@@ -127,6 +131,10 @@
127
131
  # idle printf '%s' '{"runs":{"feat_x":{"status":"completed"}}}' > "$reg"
128
132
  # rm -f "$w/calls"; run
129
133
  # -> those same two lines and exit 0
134
+ # remote printf '%s' '{"runs":{"feat_r":{"status":"running","execution":"github-actions"}}}' > "$reg"
135
+ # rm -f "$w/calls"; run
136
+ # -> the remote line for feat_r, no refusal,
137
+ # the restart proceeds, exit 0
130
138
  # never ran rm -f "$reg"; run -> the restart proceeds, exit 0
131
139
  # unreadable printf 'x' > "$reg"; run
132
140
  # -> "cannot prove no run is in flight", exit 2
@@ -233,6 +241,7 @@ fi
233
241
  active=""
234
242
  unknown=""
235
243
  registry=""
244
+ remote=""
236
245
 
237
246
  # The library's 1/2 split is kept apart here, because the two mean different
238
247
  # things to an operator: a repository with NO configuration has no harness
@@ -258,12 +267,23 @@ else
258
267
  # document that has no such wrapper is a registry this cannot enumerate, so
259
268
  # `jq` fails and the refusal below fires. Reading such a file as "no active
260
269
  # runs" is the one misreading that costs a run.
261
- elif ! active="$(jq -r '.runs | to_entries[] | select(.value.status == "running" or .value.status == "parked" or .value.status == "park_loop") | "\(.value.status) \(.key)"' "$registry" 2>/dev/null)"; then
270
+ elif ! active="$(jq -r '.runs | to_entries[] | select((.value.status == "running" or .value.status == "parked" or .value.status == "park_loop") and (.value.execution != "github-actions")) | "\(.value.status) \(.key)"' "$registry" 2>/dev/null)"; then
262
271
  active=""
263
272
  unknown="'$registry' could not be read as a run registry (invalid JSON, or no .runs wrapper)"
273
+ else
274
+ remote="$(jq -r '.runs | to_entries[] | select((.value.status == "running" or .value.status == "parked" or .value.status == "park_loop") and (.value.execution == "github-actions")) | "\(.key) \(.value.status)"' "$registry" 2>/dev/null || true)"
264
275
  fi
265
276
  fi
266
277
 
278
+ if [ -n "$remote" ]; then
279
+ while IFS= read -r line; do
280
+ [ -n "$line" ] || continue
281
+ printf 'restart-watcher.sh: remote (not affected by a restart): %s\n' "$line"
282
+ done <<EOF
283
+ $remote
284
+ EOF
285
+ fi
286
+
267
287
  # The backstop probe. The basename is matched so a binary named by an absolute
268
288
  # path still matches the command line it was spawned with, and the bracket in
269
289
  # `<name>[ ]-p` keeps the pattern from matching the process running this probe.
@@ -0,0 +1,182 @@
1
+ #!/usr/bin/env bash
2
+ # run-test-suite.sh — run the configured `commands.test` string once and decide
3
+ # one verdict for it, `pass` or `fail`, so the Run gates phase's orchestrator
4
+ # reads that verdict and never the suite's output.
5
+ #
6
+ # OUTPUT CONTRACT — the run form prints exactly one stdout line: `pass` (exit 0),
7
+ # `fail <log path>` (exit 1), or — when a run of the same label is already in
8
+ # flight — `pending` (exit 3), the log path repo-relative. Nothing the command
9
+ # prints reaches stdout or stderr; all of it goes to the log. The orchestrator
10
+ # never reads the log — it hands the path on to whatever plans the fix.
11
+ #
12
+ # THE LOG IS VERSIONED PER ROUND, NOT OVERWRITTEN:
13
+ # <state_dir>/test_run_logs/<sanitized branch>/<label>.log
14
+ # The caller passes `<gate_key>_round_<gate_round>` as the label, so each round
15
+ # keeps its own log and a fix planned against round N still finds round N's
16
+ # output after round N+1 ran. Only a re-run of the SAME label — a resumed round —
17
+ # truncates that label's log.
18
+ #
19
+ # THE WAIT FORM, AND WHY IT EXISTS. A headless session is torn down when its turn
20
+ # ends (`plugin/instructions/autonomous_pause_and_ledger.md` → §2.5), so a caller
21
+ # whose run the Bash tool moved to the background must not end its turn, sleep,
22
+ # or hand the wait to a `Monitor` to collect the verdict. It issues
23
+ # `--wait <label>` as a foreground call instead, repeatedly, until that prints a
24
+ # verdict. Each call polls for at most one wait slice and then prints `pending`.
25
+ # No caller's correctness depends on the slice's length: `pending` means only
26
+ # "issue the same call again", so the suite may take any length of time.
27
+ #
28
+ # ONE RUN PER LABEL. The run form never starts a second run of a label whose
29
+ # run is live: it collects that run's verdict as the wait form would. A run
30
+ # removes a `.running` file only while it still holds its own PID.
31
+ #
32
+ # BESIDE THE LOG, in the same directory:
33
+ # <label>.running the run form's own PID, present while the command runs
34
+ # <label>.verdict the verdict line, renamed into place whole once the command
35
+ # exits, so a reader never sees half a line
36
+ # A refusal writes neither file, and no log.
37
+ #
38
+ # Usage:
39
+ # run-test-suite.sh <label> run the suite; label ^[A-Za-z0-9][A-Za-z0-9._-]*$
40
+ # run-test-suite.sh --wait <label> read the verdict of a run of <label>; never runs anything
41
+ #
42
+ # Exit map:
43
+ # 0 pass (run form, or wait form reading a `pass` verdict)
44
+ # 1 fail (likewise)
45
+ # 2 refusal — one stderr line `run-test-suite.sh: <reason>`, nothing on stdout:
46
+ # bad arguments, a label outside the pattern, an unresolvable configuration,
47
+ # `commands.test` unset, no current branch, or (wait form) no run in flight
48
+ # 3 pending — the wait form, or the run form finding a run of the same label already in flight; issue the wait form
49
+
50
+ # Deliberately no `-e`: this script has to outlive the command's failure long
51
+ # enough to write and print the verdict.
52
+ set -uo pipefail
53
+
54
+ # The wait form's poll bound. Far inside the Bash tool's default foreground
55
+ # window, so a `--wait` call is never itself moved to the background.
56
+ WAIT_SLICE_SECONDS=60
57
+ if [[ "${RUN_TEST_SUITE_WAIT_SLICE-}" =~ ^[0-9]+$ ]]; then
58
+ WAIT_SLICE_SECONDS="$RUN_TEST_SUITE_WAIT_SLICE"
59
+ fi
60
+
61
+ LOG_SUBDIR='test_run_logs'
62
+ LABEL_PATTERN='^[A-Za-z0-9][A-Za-z0-9._-]*$'
63
+
64
+ refuse() {
65
+ echo "run-test-suite.sh: $1" >&2
66
+ exit 2
67
+ }
68
+
69
+ mode=run
70
+ case "$#" in
71
+ 1) label="$1" ;;
72
+ 2)
73
+ [ "$1" = "--wait" ] || refuse "usage: run-test-suite.sh <label> | run-test-suite.sh --wait <label>"
74
+ mode=wait
75
+ label="$2"
76
+ ;;
77
+ *) refuse "usage: run-test-suite.sh <label> | run-test-suite.sh --wait <label>" ;;
78
+ esac
79
+ [[ "$label" =~ $LABEL_PATTERN ]] || refuse "label '$label' does not match $LABEL_PATTERN"
80
+
81
+ script_dir="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
82
+ hr_lib="$script_dir/lib/harness-run-lib.sh"
83
+ [ -r "$hr_lib" ] || refuse "cannot read '$hr_lib'"
84
+ # shellcheck source=lib/harness-run-lib.sh
85
+ . "$hr_lib"
86
+
87
+ root="$(hr_repo_root "$script_dir")" || refuse "'$script_dir' is not inside a git repository"
88
+ cd "$root" || refuse "cannot enter '$root'"
89
+
90
+ state_dir="$(hr_state_dir "$root")" || refuse "cannot resolve '$root/harness.config.json'"
91
+
92
+ branch="$(hr_current_branch "$root")"
93
+ [ -n "$branch" ] || refuse "no current branch (detached HEAD?)"
94
+ safe_branch="$(hr_sanitize_branch "$branch")" || refuse "no current branch (detached HEAD?)"
95
+
96
+ log_dir="$state_dir/$LOG_SUBDIR/$safe_branch"
97
+ log_file="$log_dir/$label.log"
98
+ running_file="$log_dir/$label.running"
99
+ verdict_file="$log_dir/$label.verdict"
100
+
101
+ # Print a verdict file's line and exit with its status.
102
+ emit_verdict() {
103
+ local line
104
+ line="$(cat "$verdict_file" 2>/dev/null)" || refuse "cannot read '$verdict_file'"
105
+ case "$line" in
106
+ pass) echo "$line"; exit 0 ;;
107
+ "fail "?*) echo "$line"; exit 1 ;;
108
+ esac
109
+ refuse "'$verdict_file' holds no verdict line"
110
+ }
111
+
112
+ run_alive() {
113
+ local pid
114
+ [ -f "$running_file" ] || return 1
115
+ pid="$(cat "$running_file" 2>/dev/null)" || return 1
116
+ [[ "$pid" =~ ^[0-9]+$ ]] || return 1
117
+ kill -0 "$pid" 2>/dev/null
118
+ }
119
+
120
+ # Poll for the verdict of the run of <label> that is in flight, for at most one
121
+ # wait slice; print it, or `pending`.
122
+ wait_for_verdict() {
123
+ local deadline=$((SECONDS + WAIT_SLICE_SECONDS))
124
+ while :; do
125
+ [ -f "$verdict_file" ] && emit_verdict
126
+ if ! run_alive; then
127
+ # The verdict is renamed into place before `.running` is removed, so a
128
+ # run that finished between the two tests above has left it.
129
+ [ -f "$verdict_file" ] && emit_verdict
130
+ refuse "no run in flight for $label"
131
+ fi
132
+ [ "$SECONDS" -lt "$deadline" ] || break
133
+ sleep 1
134
+ done
135
+ echo pending
136
+ exit 3
137
+ }
138
+
139
+ if [ "$mode" = wait ]; then
140
+ wait_for_verdict
141
+ fi
142
+
143
+ command_line="$(hr_command "$root" test)"
144
+ case $? in
145
+ 0) ;;
146
+ 2) refuse "cannot resolve '$root/harness.config.json'" ;;
147
+ *) command_line="" ;;
148
+ esac
149
+ [ -n "$command_line" ] || refuse "commands.test is not set in harness.config.json"
150
+
151
+ # ONE RUN PER LABEL. A live run of this label — a session that re-entered the
152
+ # phase while its earlier, backgrounded run still runs — is never joined by a
153
+ # second: collect that run's verdict instead, exactly as the wait form would.
154
+ if run_alive; then
155
+ wait_for_verdict
156
+ fi
157
+
158
+ mkdir -p "$log_dir" || refuse "cannot create '$log_dir'"
159
+ rm -f "$verdict_file" "$verdict_file.tmp"
160
+ printf '%s\n' "$$" > "$running_file"
161
+ : > "$log_file"
162
+
163
+ # `set +u +o pipefail` inside the subshell grades the command under the options
164
+ # a plain `bash -c` would give it, not this script's.
165
+ (
166
+ set +u +o pipefail
167
+ eval "$command_line"
168
+ ) < /dev/null > "$log_file" 2>&1
169
+ status=$?
170
+
171
+ if [ "$status" -eq 0 ]; then
172
+ verdict="pass"
173
+ else
174
+ verdict="fail $log_file"
175
+ fi
176
+
177
+ printf '%s\n' "$verdict" > "$verdict_file.tmp" && mv -f "$verdict_file.tmp" "$verdict_file"
178
+ if [ "$(cat "$running_file" 2>/dev/null)" = "$$" ]; then rm -f "$running_file"; fi
179
+
180
+ echo "$verdict"
181
+ [ "$status" -eq 0 ] && exit 0
182
+ exit 1
@@ -97,7 +97,8 @@
97
97
  # a negation for that directory's own README — so a probe cannot be committed by
98
98
  # accident. The other half of that duty belongs to the agent rather than to this
99
99
  # script: a MUTATION CHECK is REVERTED before the task's own verification runs,
100
- # because the committer will see that suite and it has to be clean.
100
+ # because the Run gates phase runs the full suite over the committed tree, and a
101
+ # mutation left in place fails it.
101
102
  #
102
103
  # IT IS DELIBERATELY NOT A WRAPPER. It carries no adopter command line — it is
103
104
  # not a `WRAPPER_SCRIPTS` row, answers to no `commands.*` key, and has no
@@ -2,7 +2,7 @@
2
2
 
3
3
  Everything the delivery flow writes and reads lives here: the task prompt a branch starts from, the plans made for it, the findings of every review round, the progress ledger a resumed run reads back, the end-of-run statistics, and the two long-lived ledgers at the root of this directory. **The tree is committed** — these artifacts are the record of how each branch was planned, reviewed and fixed, and a reviewer or a resumed run reads them out of the repository rather than out of one machine's scratch space.
4
4
 
5
- What git ignores is the machine-local part of it, and it is a short list: the run daemon's per-run transcripts, event logs and run registry under `<state_dir>/autonomous_logs/`; the prompts dropped into `<state_dir>/autonomous_inbox/`; the questions a parked run asks with the answers it is given, under `<state_dir>/clarifications/`; and the throwaway files an agent runs a probe or a mutation check from, under `<state_dir>/scratch/`. All four are ignored **by their contents**, so the committed `README.md` in each survives the rule and the directory keeps its contract. Ignored with them are the stop, pause and dispatch-count files a run leaves flat at the root of this directory while it is in flight — a committed `STOP` being the one that would halt every run for everyone who clones, at a step whose own instruction forbids deleting the file.
5
+ What git ignores is the machine-local part of it, and it is a short list: the run daemon's per-run transcripts, event logs and run registry under `<state_dir>/autonomous_logs/`; the prompts dropped into `<state_dir>/autonomous_inbox/`; the questions a parked run asks with the answers it is given, under `<state_dir>/clarifications/`; and the throwaway files an agent runs a probe or a mutation check from, under `<state_dir>/scratch/`; and the full output of each test-gate run, one log per round, under `<state_dir>/test_run_logs/`. All five are ignored **by their contents**, so the committed `README.md` in each survives the rule and the directory keeps its contract. Ignored with them are the stop, pause and dispatch-count files a run leaves flat at the root of this directory while it is in flight — a committed `STOP` being the one that would halt every run for everyone who clones, at a step whose own instruction forbids deleting the file.
6
6
 
7
7
  **Do not rename this directory to a dot-name.** That is this tree's own hard constraint, not a preference. It is tempting to tidy the tree out of sight, and it is the one change that risks unattended operation: this tree has to be writable by an unattended run, and a dot-path is where a host reserves directories an unattended run may not write to — the measured case is the host's own `.claude/**`, where an unattended run completes reporting success with nothing written. The name is configured as `stateDir` in `harness.config.json`, which refuses a dot at the start of any of its path segments rather than trusting which dot-paths a given host reserves — so a value that only reaches a dot-directory by traversal is refused as well as a plainly dot-named one.
8
8
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  One readable transcript `<branch>.log` and one raw event log `<branch>.stream.jsonl` per dispatched run, plus the daemon's own `watcher.log`, the service manager's capture of its standard output and error as `watcher.out.log` and `watcher.err.log`, and `registry.json` — the run registry that indexes every run's status. All of it resolves to `<state_dir>/autonomous_logs/` **in the main checkout**, whichever sibling working copy a run actually executes in, so one repository has one set of these files however many runs it has in flight.
4
4
 
5
- Everything here is written by the run daemon and read by the operator watching or diagnosing a run: the transcript is formatted live and tailable while the run is going, and the raw stream beside it is the deep-debug copy and the one thing the usage gate can parse, since a run cannot read the stream it is itself producing. The registry has readers beyond the operator, and that is what makes it different in kind from its neighbours: it is the single record of which runs exist and what state each is in, and the harness's status command, its pause and resume commands, `<scripts_dir>/restart-watcher.sh` and `<scripts_dir>/cleanup-merged-worktrees.sh` all read it before they act.
5
+ Everything here is written by the run daemon and read by the operator watching or diagnosing a run: the transcript is formatted live and tailable while the run is going, and the raw stream beside it is the deep-debug copy and the one thing the usage gate can parse, since a run cannot read the stream it is itself producing. The registry has readers beyond the operator, and that is what makes it different in kind from its neighbours: it is the single record of which runs exist and what state each is in, and the harness's status command, its pause and resume commands, `<scripts_dir>/restart-watcher.sh` and `<scripts_dir>/cleanup-merged-worktrees.sh` all read it before they act, and `<scripts_dir>/remote-run.sh` reads and writes it for a run executing on GitHub Actions.
6
6
 
7
7
  A run's two files appear when it launches and are **appended** to, not replaced — a resumed run, a later review round or a re-launch on the same branch adds to the same pair, so one transcript may hold several sessions end to end. The registry likewise holds one record per branch, re-used across a branch's runs rather than accumulating one per launch. None of it is committed: the contents are **machine-local and gitignored**, the directory is ignored by its contents rather than as a directory, and this README survives by an explicit negation, the same way the unattended loop's inbox does.
8
8
 
@@ -17,3 +17,5 @@ One block, illustrative rather than real, in the shape every entry takes — as
17
17
  - **cost this run:** two of five tasks carry no type-check result.
18
18
  - **hypothesis:** (guess) the entry predates runs executing outside the main checkout.
19
19
  ```
20
+
21
+ Every path an entry names — quoted command output included — is repo-relative: written relative to the checkout it sits in, never with the machine's home directory or a checkout's root in front of it, because this file is committed and merged into the default branch, where a path from one machine means nothing to any other reader and discloses that machine's layout. The run checks the written file for those locations before it commits it, and rewrites any it finds. The rule is `${CLAUDE_PLUGIN_ROOT}/instructions/improvement_observations_instructions.md` → `## The entry format`'s; the check is that file's → `## Commit mechanics`'s.
@@ -6,6 +6,6 @@ A file here is written by the dispatched implementer or reviewer that needs the
6
6
 
7
7
  Its contents are **machine-local and gitignored**: the directory is ignored by its contents rather than as a directory, and this README survives by an explicit negation, which makes it the only committed file here.
8
8
 
9
- Two boundaries an agent does not cross. They are rules for the agent rather than a containment claim: the runner's own `WHAT THIS DOES NOT CONTAIN` paragraph records that its fence is over **which file runs** and over nothing that file does. A probe file lives **here and nowhere else** — the runner refuses any path that does not resolve inside this directory, so putting one somewhere more convenient does not get it run, it gets it refused. And a mutation check is **reverted before the task's own verification runs**: the point of one is a suite that fails, and the committer has to see that suite clean, so the revert belongs with the check rather than at the end of the task. What a file run from here reaches follows from that fence: it executes with the session's own privileges and reaches whatever the permission surfaces withhold from a command string, and the `pre-push` git hook is the layer that still holds — driven rather than asserted in the harness's `docs/outer-loop-verification.md` §3, under *What that `allow` reaches*.
9
+ Two boundaries an agent does not cross. They are rules for the agent rather than a containment claim: the runner's own `WHAT THIS DOES NOT CONTAIN` paragraph records that its fence is over **which file runs** and over nothing that file does. A probe file lives **here and nowhere else** — the runner refuses any path that does not resolve inside this directory, so putting one somewhere more convenient does not get it run, it gets it refused. And a mutation check is **reverted before the task's own verification runs**: the Run gates phase runs the full suite over the committed tree, so a mutation left in place fails it, and the revert belongs with the check rather than at the end of the task. What a file run from here reaches follows from that fence: it executes with the session's own privileges and reaches whatever the permission surfaces withhold from a command string, and the `pre-push` git hook is the layer that still holds — driven rather than asserted in the harness's `docs/outer-loop-verification.md` §3, under *What that `allow` reaches*.
10
10
 
11
11
  The mistake worth naming is treating a file here as evidence. It is gitignored by construction, so it never reaches a commit and no later reader can open it: a finding, a verdict or a plan sentence that rests on a probe records **what the probe showed** and the command that produced it, never a pointer to the file that showed it. That holds whether or not the file is still on disk when the task ends.
@@ -0,0 +1,9 @@
1
+ # test_fix_plan_reviews/
2
+
3
+ One `<branch>_<gate_key>_round_<gate_round>/review_<i>.md` per failed plan review, holding the findings against one round's test fix plan. The folder name repeats the suffix of the fix-plan index it reviews, under `<state_dir>/test_fix_plans/`. Inside it, `<i>` starts at `review_0.md` and increments once per failed review; a later number is added beside the earlier files, never written over them.
4
+
5
+ A file here is written by the architecture reviewer at its fourth insertion point, and read by the test fix-plan writer on its next iteration, which applies the Must Fix items to the index or to the named finding files. The reviewer creates the folder itself and only when it has findings to write, so a plan approved on the first pass leaves no folder behind.
6
+
7
+ Files appear one per failed review and stop when the plan is approved, or when the loop reaches its cap. Nothing supersedes an earlier file — the numbered set is how that round's plan converged — and the whole directory is committed with the branch, since no ignore rule reaches it.
8
+
9
+ The mistake worth naming is reading a file here as a review of the code. It holds the architecture gate's findings against one round's **plan**, before any of its fixes is implemented; the review of a finished fix is in `<state_dir>/test_fix_point_reviews/`.
@@ -0,0 +1,9 @@
1
+ # test_fix_plans/
2
+
3
+ One `<branch>_<gate_key>_round_<gate_round>.md` index plus one `<branch>_<gate_key>_round_<gate_round>/finding_<N>.md` per fix, planned from one failing gate run. `<gate_key>` is `task` in the task flow and `review_<n>` in the user-review fix flow, and `<gate_round>` is the round whose log failed, so the index and its folder always carry the same suffix as the log they answer. The index is thin — a context paragraph, the `## Phase 2 Readiness — Ordered Fix List`, and one pointer per fix — while each finding file carries the failure it addresses and the concrete fix, self-contained enough to implement from alone. A failure judged not fixable on the branch is listed under the index's `## Not fixable on this branch` rather than given a finding file.
4
+
5
+ The set is written by the test fix-plan writer, from the round's log. The architecture reviewer approves it in plan-review mode before any fix is implemented; the unit loop's row `G.4` walks the readiness list and hands one implementer one `finding_<N>.md`; the committer flips the item's readiness entry as the fix lands.
6
+
7
+ An index appears when a gate run fails and a fix round opens. **Each round is a new index, never an edit of the previous one**: the next failure is planned under the next round's suffix, beside the earlier ones, and the accumulated rounds are the branch's gate-fix history. The whole directory is committed with the branch, since no ignore rule reaches it.
8
+
9
+ The mistake worth naming is reading an index here as a review. It is a fix plan: nobody graded the branch to produce it. It is the writer's plan for making one failing run pass, and the Run gates phase's next run is what verifies it.
@@ -0,0 +1,9 @@
1
+ # test_fix_point_reviews/
2
+
3
+ One `<branch>_<gate_key>_round_<gate_round>/item_<N>/review_<i>.md` per failed unit review, holding the findings against a single implemented fix. `<N>` is the fix's position in the round's readiness list — the first entry is `item_1`, the second `item_2` — which is not necessarily the `K` its `finding_<K>.md` carries under `<state_dir>/test_fix_plans/`, because the list is sorted by ship order. Inside the item's folder, `<i>` starts at `review_0.md` and increments once per failed review of that fix; the counter is per item, so it restarts rather than running on across the round.
4
+
5
+ A file here is written by the layer reviewer the fix's layer routed to, and read by that fix's layer implementer on the fix iteration, which is handed the folder's most recent file. The reviewer creates the folder itself and only when it has findings to write, so a fix that passed on the first review leaves no folder behind.
6
+
7
+ Nothing supersedes an earlier file — the numbered set is that fix's convergence history — and the whole directory is committed with the branch, since no ignore rule reaches it.
8
+
9
+ The mistake worth naming is reading an empty directory as a missed step. Files appear only where the flow's per-unit review is on; in a flow where it is off, fixes go straight from implementer to committer, and this directory stays empty by design.
@@ -0,0 +1,11 @@
1
+ # test_run_logs/
2
+
3
+ One `<branch>/<gate_key>_round_<gate_round>.log` per gate run, holding the full output of one run of the configured test command. `<branch>` is the branch name sanitized for use as a path; `<gate_key>` is `task` in the task flow and `review_<n>` in the user-review fix flow; `<gate_round>` counts the Run gates phase's runs for that key. Beside each log sit `<label>.running`, the in-flight run's PID, and `<label>.verdict`, the one line the run printed — `pass` or `fail <log path>`.
4
+
5
+ A log is written by the harness's test-suite wrapper, `run-test-suite.sh`, and by nothing else; the wrapper creates the `<branch>/` subdirectory itself. The orchestrator of the Run gates phase never reads a log — it reads the verdict line and hands a failing log's path on. The test fix-plan writer reads it, and on a later round reads the previous round's log too, to tell a regression the last fix introduced from a failure that persisted.
6
+
7
+ A log appears on each gate run and is **versioned per round, never overwritten across rounds**: each round runs under its own label, so round N's log is still there after round N+1 ran. Only a re-run of the same label — a resumed round — truncates that label's log. Nothing rotates or archives the directory.
8
+
9
+ Its contents are **machine-local and gitignored**: the directory is ignored by its contents rather than as a directory, and this README survives by an explicit negation, which makes it the only committed file here.
10
+
11
+ The mistake worth naming is treating a log as evidence a later reader can open. It never reaches a commit, because the machine paths it carries are what the self-containment gate refuses. A fix plan quotes **what the log showed**, with machine paths rewritten — paths under the checkout root made repo-relative, any other home-directory path replaced by `<home>` — because the fix plan is committed and a raw path would carry into it exactly what keeps the log itself out.