okstra 0.213.0 → 0.214.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/lead/okstra-lead-contract.md +3 -0
- package/runtime/python/okstra_ctl/blocking_checks.py +11 -0
- package/runtime/python/okstra_ctl/phases/implementation/host-rules.md +1 -1
- package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-deliverable.md +33 -5
- package/runtime/python/okstra_ctl/phases/implementation/profile.md +1 -0
- package/runtime/python/okstra_ctl/phases/implementation/report_assets/implementation.template.html +6 -0
- package/runtime/python/okstra_ctl/phases/implementation/spec.md +2 -1
- package/runtime/python/okstra_ctl/phases/implementation/validation.py +91 -0
- package/runtime/python/okstra_ctl/phases/implementation_planning/entry.py +5 -5
- package/runtime/python/okstra_ctl/stage_fix_carry.py +4 -0
- package/runtime/python/okstra_ctl/wizard/registry.py +3 -1
- package/runtime/python/okstra_ctl/wizard/steps_options.py +11 -0
- package/runtime/schemas/final-report-v3.0.schema.json +93 -0
- package/runtime/skills/okstra-run/SKILL.md +1 -1
- package/runtime/templates/reports/html/i18n/en.json +6 -0
- package/runtime/templates/reports/html/i18n/ko.json +6 -0
- package/runtime/validators/validate-run.py +51 -0
package/package.json
CHANGED
package/runtime/BUILD.json
CHANGED
|
@@ -137,6 +137,7 @@ Every `okstra` command the lead documents cite, grouped by phase, each spelled w
|
|
|
137
137
|
| `okstra plan-items next-dispatch --state <path>` | Decide whether the round opens another worker batch | `plan-body-verification` "Round protocol" |
|
|
138
138
|
| `okstra plan-items resolve-dissent --state <path> --item <P-id> --decision-file <path>` | Record an evidence-based lead decision after the single self-fix | `plan-body-verification` "Round protocol" |
|
|
139
139
|
| `okstra plan-verify --report <path>` | Score the plan-body gate; never tally votes in a script | `plan-body-verification` "Round protocol" |
|
|
140
|
+
| `okstra worktree-lock --worktree <dir> --command <text>` | Lead minor fix (implementation only): re-run each Tier 1 command and failing verifier check that covers a fixed finding, in the stage worktree, under its lock | `_implementation-deliverable.md` "Lead minor fix" |
|
|
140
141
|
|
|
141
142
|
### Phase 7 — persist
|
|
142
143
|
|
|
@@ -354,6 +355,8 @@ For `--task-type implementation` runs, the task bundle additionally pins one of
|
|
|
354
355
|
|
|
355
356
|
Lead MUST dispatch Edit/Write-bearing work only through that executor binding: use the host primitive with `hostModelValue` for `runner=native-session`. For `runner=cli-wrapper`, use `okstra team dispatch` when `terminalBackend` is `cmux-pane`, or `okstra worker-dispatch` with `modelExecutionValue` otherwise. The other providers in the roster still run as read-only verifiers in the same run; the executor's own provider does not, because its worker ID materializes as the executor on every dispatch — so the diff is reviewed context-isolated by the remaining verifiers. Session isolation is the primary self-review safeguard — a verifier reusing the executor's model variant is acceptable in a distinct session. A different model variant (e.g. executor=opus / Claude verifier=sonnet) is recommended but not mandatory.
|
|
356
357
|
|
|
358
|
+
The one exception is the **lead minor fix**: after the verifier round and Phase 5.5 convergence, and before the report-writer dispatch, the lead edits and commits in the stage worktree itself to fix verifier FAIL findings that meet every eligibility rule in `scripts/okstra_ctl/phases/implementation/instructions/_implementation-deliverable.md` "Lead minor fix" — a local fix at the cited location, at most 30 changed lines in total, no change to a test's expected values, dependencies, schemas, migrations or config of record, only files inside the stage's planned paths or already changed by the executor, one pass per run. It then re-runs the covering checks through `okstra worktree-lock --worktree <dir> --command <text>` and records them in `implementation.leadFixes`. No worker dispatch is open at that point, so no read-only worker's mutation audit window covers the edit. Any finding outside those rules goes through the executor in a fix run. **Enforced:** `validators/validate-run.py` `_validate_lead_fixes` fails a fix recorded as passing with a failing check, a fix over 30 lines, and a stage settled over a FAIL that no passing fix answers; the remaining eligibility rules are instructions only.
|
|
359
|
+
|
|
357
360
|
Executor is chosen at run-prep time via `--executor <claude|codex|antigravity>` (or `OKSTRA_DEFAULT_EXECUTOR`, fallback `claude`); the model used by the executor is taken from the corresponding worker model flag (`--claude-model` / `--codex-model` / `--antigravity-model`). For CLI-backed executors, the underlying file mutation happens inside the executor CLI's own auto-edit mode (e.g. `codex exec --sandbox danger-full-access`), not through the lead runtime's `write_artifact` operation.
|
|
358
361
|
|
|
359
362
|
**Enforced:** every non-executor assignment is dispatched under `writePolicy.sourcePolicy.mode` `source-readonly`, and `scripts/okstra_ctl/execution_mutation_audit.py` `_source_changes` compares the before/after snapshots and reports a read-only worker that mutated source; `assert_compatible_batch` refuses a batch that mixes a mutating worker with read-only ones.
|
|
@@ -216,6 +216,17 @@ _BLOCKING: tuple[tuple[str, str], ...] = (
|
|
|
216
216
|
"stage_fix_carry.py 가 통과 판정만 보므로 거부된 작업이 "
|
|
217
217
|
"release-handoff 로 넘어간다.",
|
|
218
218
|
),
|
|
219
|
+
# 검증자 FAIL 이 남은 stage 를 `done` 으로 정착시키는 길은 재실행 검사가
|
|
220
|
+
# 전부 통과한 리드 수정 하나뿐이다(`_implementation-deliverable.md`
|
|
221
|
+
# "Lead minor fix"). 기록이 그 조건을 어기면 stage_fix_carry.py 가 통과한
|
|
222
|
+
# 리드 수정이 답한 FAIL 행을 해소된 것으로 읽어 fix run 을 열지 않는다.
|
|
223
|
+
(
|
|
224
|
+
"lead-fix BLOCKING:",
|
|
225
|
+
"검증자가 거부했고 통과 기록이 없는 head 가 done 으로 정착된다 — "
|
|
226
|
+
"consumers.py 의 done 행으로 후속 stage 가 그 head 에서 분기하고, "
|
|
227
|
+
"stage_fix_carry.py 는 통과로 기록된 리드 수정이 답한 FAIL 을 다시 "
|
|
228
|
+
"fix run 으로 넘기지 않는다.",
|
|
229
|
+
),
|
|
219
230
|
(
|
|
220
231
|
"implementation run declares stage-",
|
|
221
232
|
"stage 캐리 사이드카가 디스크에 없다 — consumers.py 의 "
|
|
@@ -54,7 +54,7 @@ If the anchor (`implementation_base_commit`) is reported unresolvable, run the s
|
|
|
54
54
|
Because of the dependency closure, the chain queue **may include a stage that another implementation run has occupied as started/reserved.** That stage's `render-bundle` is rejected with `--stage N already in progress or reserved by another run` (StageTargetError). This is **not** an exception gate needing human judgment but a "next stage not yet ready" situation. On this rejection, **terminate the chain normally** and report the remaining queue to the user (e.g. `remaining queue: stage 4, 5 — resume with okstra-run after occupancy is released`). This is a different branch from the exception gate below (data corruption·concurrent-occupancy conflict confirmation).
|
|
55
55
|
|
|
56
56
|
### Stage ended FAIL — stop the queue and report (not an exception gate)
|
|
57
|
-
When a stage's synthesised verdict is `FAIL`, Phase 6 writes no carry sidecar and appends a `status:"failed"` row in place of `done` (`scripts/okstra_ctl/phases/implementation/instructions/_implementation-deliverable.md` "Lead post-stage persistence"). **Stop the queue at that stage** and report the failed stage, its report path, and the remaining queue (e.g. `stage 1 FAIL — remaining queue: stage 2, 3, 5; re-enter with okstra-run --stage 1 after the fix`). Do **not** continue to the next stage even when that stage is dependency-independent: an unattended chain that keeps building past a confirmed regression stacks later work on top of it. The `failed` row releases the stage's occupancy, so `--stage <N>` re-enters the same stage on its preserved worktree and branch — there is nothing to unblock by hand.
|
|
57
|
+
A verifier FAIL that a passing lead minor fix answered does not make the verdict `FAIL` (`_implementation-deliverable.md` "Lead minor fix"); that stage settles `done` and the queue continues. When a stage's synthesised verdict is `FAIL`, Phase 6 writes no carry sidecar and appends a `status:"failed"` row in place of `done` (`scripts/okstra_ctl/phases/implementation/instructions/_implementation-deliverable.md` "Lead post-stage persistence"). **Stop the queue at that stage** and report the failed stage, its report path, and the remaining queue (e.g. `stage 1 FAIL — remaining queue: stage 2, 3, 5; re-enter with okstra-run --stage 1 after the fix`). Do **not** continue to the next stage even when that stage is dependency-independent: an unattended chain that keeps building past a confirmed regression stacks later work on top of it. The `failed` row releases the stage's occupancy, so `--stage <N>` re-enters the same stage on its preserved worktree and branch — there is nothing to unblock by hand.
|
|
58
58
|
|
|
59
59
|
### Exception gate during chaining
|
|
60
60
|
If `render-bundle` raises Step 5's concurrent-run conflict detection (concurrent-run branch) or git stale-SHA reconciliation (git-reconcile branch), **stop the chain at that stage** and present the gate to the user exactly as Step 5 prescribes. Once the user resolves the gate, resume the chain in place (continue with the remaining queue). Data corruption·concurrent-occupancy conflicts are confirmed by a human — this is the safety boundary of unattended chaining. (Unlike the "not ready" rejection above, these two branches do not discard the queue; they wait for user resolution.)
|
package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-deliverable.md
CHANGED
|
@@ -6,12 +6,13 @@ are collected and convergence finished. Phase 1-5 do not need it.
|
|
|
6
6
|
|
|
7
7
|
# Implementation profile — Deliverable sidecar
|
|
8
8
|
|
|
9
|
-
> **When to read**: lead reads this file ONCE at the start of Phase 6 (after Phase 5.5 convergence completes, before constructing the report-writer dispatch prompt). Carries the final-report deliverable shape, lead's post-stage persistence rules, and the self-review checklist.
|
|
9
|
+
> **When to read**: lead reads this file ONCE at the start of Phase 6 (after Phase 5.5 convergence completes, before constructing the report-writer dispatch prompt). Carries the final-report deliverable shape, the lead minor fix, lead's post-stage persistence rules, and the self-review checklist.
|
|
10
10
|
|
|
11
11
|
## Required deliverable shape (final report, in addition to the standard sections)
|
|
12
12
|
|
|
13
13
|
- **Plan link & approval evidence**: path to the approved `final-report.md`, the exact quoted approval marker, AND the executed stage number / title quoted from the Stage Map row. For a selected-direction plan, also quote `selectedDirectionRef.optionId`, `snapshotPath`, and the validated snapshot digest; for a legacy plan, quote the effective `implementation-option` or the Recommended Option fallback.
|
|
14
|
-
- **Commit list**: each commit's SHA (or short SHA), message, and the plan step(s) / TDD cycle it satisfies
|
|
14
|
+
- **Commit list**: each commit's SHA (or short SHA), message, and the plan step(s) / TDD cycle it satisfies. A lead minor fix commit is listed too, with `planStep` `Lead minor fix (<finding ids>)`.
|
|
15
|
+
- **Lead minor fixes**: when the lead fixed verifier findings in this run, pass each `implementation.leadFixes[]` entry to the report writer as recorded (§"Lead minor fix"). Omit the field when there was none.
|
|
15
16
|
- **Diff summary**: capture `git diff --stat <base>..<captured-head>` and `git diff --numstat <base>..<captured-head>` for the full implementation stage, including after a fix run. Supply both outputs to the report writer; put each file's added/deleted counts alone in `diffSummary.files[].lines` (`+150/-0`) and one sentence on what the change in that file does in `diffSummary.files[].summary` — the report renders `summary` as the file's "what it does" column and `lines` as a separate number column. Do not substitute a fix-run aggregate or carry forward older statistics. If the captured commits are unavailable, identify the missing reference in the evidence instead of inventing counts.
|
|
16
17
|
- **Change narrative scope**: `userNarrative.changeExplanation` describes the same scope as `diffSummary` — the whole stage from its base, every file the table lists. On a fix run, state what the fix run itself changed (its commits and files) as a separate sentence; never present the fix-run delta as the stage's change.
|
|
17
18
|
- **Out-of-plan edits block**: every file edited that was not in the approved plan's file list, with rationale (empty block is acceptable and preferred)
|
|
@@ -26,7 +27,7 @@ are collected and convergence finished. Phase 1-5 do not need it.
|
|
|
26
27
|
- **independent validation re-run results** — per plan-validation command: command line, exit code, and tail of output captured by the verifier (not the executor); any divergence from the executor's reported result MUST be called out as a `Discrepancy` line citing both sides — and, when the divergence rests on a plan `validationChecklist` row, naming that row with its `phase` (`VC-003 (phase: mid)`; `_validate_verifier_discrepancy_names_checklist_phase` fails a row cited without it) — and when nothing diverged the line is omitted rather than written as `None` — an empty `discrepancy` is how "no divergence" is recorded, and non-empty text there with verdict `PASS` fails the run,
|
|
27
28
|
- **style / lint / type-check results** — each check-only tool the verifier ran, its exit code, and the count of new findings attributable to lines this run introduced. When no tool is configured for a touched language, record the single line `no lint/style tool configured for <language>`,
|
|
28
29
|
- any fix recommendations the verifier declined to apply.
|
|
29
|
-
The Okstra lead synthesises a unified verdict but MUST preserve dissent — do not collapse opinions into one paragraph. External Tier 3 advisory results are excluded from this aggregate promotion and remain user-owned follow-up evidence. If any other verifier issued `FAIL` on a `Discrepancy` line, the synthesised verdict MUST be `FAIL`.
|
|
30
|
+
The Okstra lead synthesises a unified verdict but MUST preserve dissent — do not collapse opinions into one paragraph. External Tier 3 advisory results are excluded from this aggregate promotion and remain user-owned follow-up evidence. If any other verifier issued `FAIL` on a `Discrepancy` line, the synthesised verdict MUST be `FAIL`. The one exception is a passing lead minor fix (§"Lead minor fix" below), which answers the FAIL inside this run; there is no other override: `_validate_verifier_fail_blocks_verdict` in `validators/validate-run.py` fails any report whose `finalVerdict.verdictToken` passes while a `verifierResults[]` row records `FAIL`, `_validate_lead_fixes` reports (advisory) a settled stage whose `FAIL` row no passing `leadFixes[]` entry names, and both read the rows alone — a rationale written beside them changes nothing. The verifier's `FAIL` row stays in `verifierResults[]` after a lead fix; the fix is recorded beside it, not over it. A divergence the lead believes is not the code's (a committed flaky-test record, a documented environment delta) belongs in the routing recommendation and the user-owned follow-up, and is settled in the next fix run where the verifier re-checks the finding and cites it `resolved`.
|
|
30
31
|
- **Rollback verification** (advisory — never blocks): a human-facing record of whether the plan's rollback path is still valid after the changes. A rollback is executed by a human, not by an okstra worker/verifier, so nothing here blocks the run, forces a `contract-violated` outcome, or routes back to planning. Each `rollbackVerification` row's `result` is `ok` (verified), `not-applicable` (nothing to roll back), or `advisory — human-run` (could not verify here; a human owns it). Strength of the record depends on the change category:
|
|
31
32
|
- **Pure code changes** (no persisted state, no infra mutation): a reachable revert SHA is sufficient. Record the exact `git revert <SHA>` command that would undo the change, and confirm `git rev-parse <SHA>` resolves.
|
|
32
33
|
- **Feature-flag-gated changes**: prefer confirming the off-switch path was exercised in this run's validation evidence (i.e. one of the validation commands ran with the flag off and succeeded). If the off-path was not exercised here, record the row as `advisory — human-run` rather than treating it as a blocking requirement.
|
|
@@ -41,6 +42,33 @@ are collected and convergence finished. Phase 1-5 do not need it.
|
|
|
41
42
|
- **Routing recommendation**: `implementation.routingRecommendation` is an **object** with exactly two fields — `target`, one of `final-verification`, `error-analysis`, `implementation-planning`, `implementation`, and `rationale`, one or two sentences on why that target and nothing else. It is not a prose note: Phase 7 projects `workflow.nextRecommendedPhase` from `target` alone, so a phase named only in the prose does not route the task. Record the stage completion, validation outcome, failure-cause certainty, and plan-fit evidence; the lead chooses the destination using `prompts/lead/phase-routing.md` §implementation. **Enforced:** `schemas/final-report-v2.0.schema.json` rejects a `target` outside the enum, a missing `rationale`, and a string in place of the object.
|
|
42
43
|
- **Follow-up tasks (Section 4 of the final report)**: every item discovered during this run that was *not* delivered MUST appear in the final report's `## 4. Follow-up Tasks` table with a concrete `Origin`, `New Task ID`, `Suggested task-type`, `Scope`, and `Reason / Why deferred`. Sources include: out-of-scope discoveries that the executor consciously chose not to fold into this run, verifier concerns the executor declined to fix in-place, scope-boundary items from the approved plan that turned out to need their own ticket, and any unresolved `## 1. Clarification Items` row carried over from the approved plan (`Status` ∈ `{open, answered}` at approval time). An empty section is acceptable but only when expressed as the single line `- No follow-up tasks.` — silence is treated as a contract violation. Rows with `Auto-spawn? = yes` will be materialised by `scripts/okstra-spawn-followups.py` in Phase 7; rows with `Auto-spawn? = no` MUST also appear in `Section 3. Recommended Next Steps` so the user knows to act manually.
|
|
43
44
|
|
|
45
|
+
## Lead minor fix (decided after convergence, before the report-writer dispatch)
|
|
46
|
+
|
|
47
|
+
When the synthesised verdict would be `FAIL` only because of verifier findings that are small and local, the lead fixes them itself in this run instead of ending it and opening a fix run. Decide after Phase 5.5 convergence, before building the report-writer dispatch prompt and before "Lead post-stage persistence" below. No worker dispatch is open then, so the edit falls inside no read-only worker's mutation audit window. Never edit while a verifier, convergence, or report-writer dispatch is running: its audit reports `readonly source changed` and discards that worker's result.
|
|
48
|
+
|
|
49
|
+
**Eligibility.** Every rule holds for every blocking finding of the round, or the lead fixes none of them and the normal FAIL → fix-run path applies:
|
|
50
|
+
|
|
51
|
+
1. The fix is a code change at, or next to, the location the verifier cited.
|
|
52
|
+
2. The lead's total diff is at most 30 changed lines (added plus deleted, `git diff --numstat`).
|
|
53
|
+
3. It changes no test's expected values or an assertion's meaning, adds no dependency, and changes no schema, migration, or config of record.
|
|
54
|
+
4. It touches only files inside this stage's `plannedPaths` or files the executor already changed in this stage.
|
|
55
|
+
5. One lead-fix pass per run. A failing re-run check gets no second attempt.
|
|
56
|
+
|
|
57
|
+
A finding that needs no code change (convergence settled it as a verifier misreading) is not a lead fix; it stays in the routing recommendation and the verdict stays `FAIL`.
|
|
58
|
+
|
|
59
|
+
**Procedure.**
|
|
60
|
+
|
|
61
|
+
1. Edit in the stage worktree (`approvedPlanReference.executorWorktreePath`).
|
|
62
|
+
2. Commit with a Conventional Commit (`fix(<scope>): …`) whose body names the finding id(s). Run the executor sidecar's ignored-file check on the staged paths first; it must print nothing.
|
|
63
|
+
3. Re-run, each through `okstra worktree-lock --worktree <stage worktree> --command '<command>'`, the plan's Tier 1 command(s) that cover each fixed finding and the check the verifier reported failing. Keep each command line, exit code, and output tail.
|
|
64
|
+
4. Every check exits 0: `result: pass`. The FAIL rows the fix answers no longer make the verdict `FAIL`, and the stage settles like any non-`FAIL` run. Any check exits non-zero: `result: fail`. Revert nothing; the verdict stays `FAIL`, the stage is withheld, and the next fix run starts from the lead commit.
|
|
65
|
+
|
|
66
|
+
**Record.** One `implementation.leadFixes[]` entry per fixed finding group: `findingIds`, `verifiers` (the `verifierResults[].verifier` names whose FAIL it answers — every `FAIL` row is named by a passing entry before the stage may settle), `summary` (one sentence on what changed), `commit` (full SHA), `files`, `changedLines`, `commands[]` (`command`, `exitCode`, `outputTail`), and `result`. Keep each verifier's `FAIL` row as written. Leave the fixed findings out of `humanSummary.blockers`, and state in `userNarrative.validationExplanation` that the lead fixed them in this run.
|
|
67
|
+
|
|
68
|
+
**Persistence after a passing fix.** The executor's `### Stage Carry Evidence` was emitted before the fix and names the pre-fix head. Before writing the carry sidecar, set `stageCommitRange.head` to the post-fix HEAD (`git rev-parse HEAD` in the stage worktree), add the fixed files to `filesChanged`, and append one `notes` line naming the lead commit and the finding ids. `stepResults` stays as the executor wrote it: a lead fix is not a plan step. Append the `done` row after the fix commit, so its `head_commit` is the post-fix HEAD, and capture `diffSummary` against the same head.
|
|
69
|
+
|
|
70
|
+
**Enforced:** `validators/validate-run.py` `_validate_lead_fixes` (blocking) fails an entry recorded `pass` whose command exited non-zero or that has no command, an entry over 30 changed lines, and a settled stage holding a `result: fail` entry; it reports (advisory) a settled stage whose `FAIL` row no passing entry names in `verifiers` — advisory because validate-run runs after the `done` row is appended, so blocking would undo nothing. `_validate_lead_fix_carry_head` (advisory) reports a carry sidecar whose head is not a passing lead fix commit. Eligibility rules 1, 3, 4 and 5 and the match between findings and `findingIds` are instructions only: the report carries no structured finding id or planned-path list to compare against.
|
|
71
|
+
|
|
44
72
|
## Self-review pass before finalising the report (the Okstra lead runs this; do not delegate it)
|
|
45
73
|
|
|
46
74
|
1. **Plan coverage** — for a selected-direction plan, every step in the approved `plan-ready` stage must point to a commit (or an explicit `Skipped: <reason>` entry), and the diff must preserve the selected snapshot's mechanism and invariants. For a legacy plan, every step in the effective implementation option (explicit frontmatter value or Recommended Option fallback) must point to a commit or an explicit skip. List gaps. A `RED:` step and its `GREEN:` step pointing to the same merged commit SHA is NOT a coverage gap — one SHA may be shared by both.
|
|
@@ -62,9 +90,9 @@ are collected and convergence finished. Phase 1-5 do not need it.
|
|
|
62
90
|
|
|
63
91
|
- Parse the executor's `### Stage Carry Evidence` JSON block. If absent or unparsable, end with status `contract-violated` and route to a follow-up `error-analysis`.
|
|
64
92
|
- The `### Stage Carry Evidence` JSON may include `designPrepEvidence[]`. Emit a row only when this stage produced concrete evidence that refines an effective PREP item: `itemId`, the injected `assessmentFingerprint`, `resolution`, and non-empty `evidence[]` are required; `overrides` is optional and only records observed, non-authoritative refinements. Carry evidence never represents user approval. Downstream resolution accepts it only from transitive dependency stages with the matching fingerprint.
|
|
65
|
-
- **A `FAIL` synthesised verdict withholds the two writes below.** They are what marks the stage `done`, so performing them on a stage whose verifier found a blocking defect stacks the next stage on a confirmed regression. When the synthesised verdict is `FAIL`: write NO carry sidecar, and append a `status:"failed"` row in place of the `done` row — same `okstra_ctl.consumers.append_consumer` call, carrying `report_path` and the SHA of HEAD. That row is terminal *without* completion: dependent stages stay blocked because this stage is not done, while its worktree-registry occupancy is released so a fix run can re-enter the same stage number — `--stage <N>` reuses the preserved worktree and branch instead of provisioning a new one. State the reason in the report's `Stage sidecar evidence` section as `withheld`. **Enforced:** `validators/validate-run.py` `_validate_stage_carry_sidecar_exists` accepts a missing carry file only when that field is non-empty, so silently skipping the sidecar still fails the run.
|
|
93
|
+
- **A `FAIL` synthesised verdict withholds the two writes below.** A verifier `FAIL` that a passing lead minor fix answered does not make the verdict `FAIL` (§"Lead minor fix"); persist with the post-fix head it names. They are what marks the stage `done`, so performing them on a stage whose verifier found a blocking defect stacks the next stage on a confirmed regression. When the synthesised verdict is `FAIL`: write NO carry sidecar, and append a `status:"failed"` row in place of the `done` row — same `okstra_ctl.consumers.append_consumer` call, carrying `report_path` and the SHA of HEAD. That row is terminal *without* completion: dependent stages stay blocked because this stage is not done, while its worktree-registry occupancy is released so a fix run can re-enter the same stage number — `--stage <N>` reuses the preserved worktree and branch instead of provisioning a new one. State the reason in the report's `Stage sidecar evidence` section as `withheld`. **Enforced:** `validators/validate-run.py` `_validate_stage_carry_sidecar_exists` accepts a missing carry file only when that field is non-empty, so silently skipping the sidecar still fails the run.
|
|
66
94
|
- **A run that ends with no verifier result still closes.** When the stage ends before any verifier returned a verdict — the executor stopped without a change and the user chose not to verify, or every dispatched verifier ended without a result — append the `failed` row as above, record each roster verifier `not-run` with its reason (`okstra worker-state transition --team-state <path> --worker <worker-id> --status not-run --reason <text>`; a worker the dispatch skipped already carries one), and write the convergence state with `okstra convergence skip --run-manifest <path> --reason <text>` before dispatching the report writer. Without that state the report-writer packet refuses with `convergence … required source is missing`. The command refuses while any analyser, verifier, or critic attempt returned a result or is still running; such a run converges normally. In the report, write each verifier's `verifierResults[]` row as `verdict: not-run`, and state the reason under `Stage sidecar evidence` as `withheld`.
|
|
67
|
-
- On a non-`FAIL` verdict, for this run's single stage: write its JSON verbatim to `runs/<impl-task-key>/carry/stage-<N>.json
|
|
95
|
+
- On a non-`FAIL` verdict, for this run's single stage: write its JSON verbatim to `runs/<impl-task-key>/carry/stage-<N>.json` — after a passing lead minor fix, with the head, files, and note that §"Lead minor fix" adds. Refuse to overwrite an existing file (one stage = one sidecar; a fix run re-entering after a `failed` row writes the first one, because a withheld stage never wrote it). A stage reopened after a `done` row (see "Reopening a settled stage" below) still has the sidecar of the run that first settled it, naming a head the stage has since moved past: move it unchanged to `runs/<impl-task-key>/carry/superseded/stage-<N>-from-<that run's task-type>-<seq>.json` before writing the new one, and name both paths in `Stage sidecar evidence`. Do not rename it inside `carry/`: the readers glob `carry/stage-*.json` (`okstra_ctl.stage_map`, `okstra_ctl.consumers.backfill_done_from_carry`), so a renamed file there is read again as a live sidecar. A reopened stage whose run ends `FAIL` writes nothing and leaves the old file where it is.
|
|
68
96
|
- On a non-`FAIL` verdict, for this run's single stage: append a `status:"done"` row to `runs/<plan-task-key>/consumers.jsonl` with `completed_at`, `carry_path`, `report_path` (this run's final-report path relative to the run root), and the SHA of HEAD. Append it with `okstra_ctl.consumers.append_consumer` (NOT a raw filesystem write) — that call honours the consumers lock AND releases this stage's worktree-registry occupancy, so later runs stop seeing a finished stage as a concurrent run. `report_path` lets `final-verification` cite each stage's originating report when assembling its Source Implementation Report list.
|
|
69
97
|
- **Reopening a settled stage.** Appending a `status:"failed"` row after a `done` one withdraws that stage: `--stage N` accepts it again, its dependents stop resolving a base from the withdrawn head, and whole-task final-verification blocks until it is settled again. Use the same `okstra_ctl.consumers.append_consumer` call with a `reason`, and re-run the stage rather than repairing the tree outside okstra — a stage reverted outside the ledger leaves the recorded `head_commit` and the carry sidecar naming a tree that no longer exists, and the next stage branches from it.
|
|
70
98
|
- The verifier round, Phase 5.5 convergence, and this Phase 6 report run **once per run** over this stage's diff — NOT per step.
|
|
@@ -3,6 +3,7 @@
|
|
|
3
3
|
- Purpose: realise the approved `implementation-planning` deliverable as actual source changes, with cross-model verification, while keeping the run reversible
|
|
4
4
|
- **Run-level fixed cost:** the verifier set, Phase 5.5 convergence, and the Phase 6 report-writer run exactly once per implementation run, over this run's single stage diff — never once per step.
|
|
5
5
|
- **Fix run (profile carries a "Fix-Run Carry" block):** the executor's scope is the carried blocking findings plus the previous routing recommendation — it MUST NOT re-execute plan steps the previous run completed. Verifiers apply the "Fix-run incremental scope" section of `_implementation-verifier.md`. The full validation-command re-run is NOT reduced.
|
|
6
|
+
- **Lead minor fix:** a verifier FAIL whose findings are small and local is fixed by the lead inside this run, after convergence and before the report-writer dispatch, under `_implementation-deliverable.md` "Lead minor fix"; a passing fix settles the stage without a fix run.
|
|
6
7
|
- **Executor binding (resolved at run-prep time, fixed for this run):**
|
|
7
8
|
- Executor display name: `{{EXECUTOR_DISPLAY_NAME}}`
|
|
8
9
|
- Executor worker ID: `{{EXECUTOR_WORKER_ID}}`
|
package/runtime/python/okstra_ctl/phases/implementation/report_assets/implementation.template.html
CHANGED
|
@@ -26,6 +26,12 @@
|
|
|
26
26
|
{% for row in implementation.verifierResults %}<article class="summary-card tone-{{ row.verdict | lower }}"><p class="eyebrow">{{ row.verifier | inline_code }} · <span class="status status-{{ row.verdict | lower }}">{{ row.verdict }}</span></p>{% for label, text in ((t('tasks.implementation.independent-rerun'), row.get("independentValidationRerun")), (t('tasks.implementation.command-log'), row.get("readOnlyCommandLog"))) if text %}<details><summary><strong>{{ label }}</strong> {{ t('tasks.implementation.entry-count') | replace('{count}', text.strip().splitlines() | select | list | length | string) }}</summary><pre>{{ text }}</pre></details>{% endfor %}{% if row.get("discrepancy") %}<p class="evidence-refs">{{ t('tasks.implementation.discrepancy') }} {{ row.discrepancy | inline_code }}</p>{% endif %}{% if row.get("declinedFixRecommendations") %}<p>{{ row.declinedFixRecommendations | inline_code }}</p>{% endif %}</article>{% endfor %}
|
|
27
27
|
</section>
|
|
28
28
|
|
|
29
|
+
{% if implementation.get("leadFixes") %}<section data-report-section="lead-fixes" data-report-field="implementation.leadFixes">
|
|
30
|
+
<h2>{{ t('tasks.implementation.lead-fixes') }}</h2>
|
|
31
|
+
<p>{{ t('tasks.implementation.lead-fixes-intro') }}</p>
|
|
32
|
+
{% for row in implementation.leadFixes %}<article class="summary-card tone-{{ 'neutral' if row.result == 'pass' else 'important' }}"><p class="eyebrow">{{ row.findingIds | join(', ') | inline_code }} · <span class="status status-{{ row.result }}">{{ t('tasks.implementation.lead-fix-result-' ~ row.result) }}</span></p><p>{{ row.summary | inline_code }}</p><table><tbody><tr><th>{{ t('tasks.implementation.commit') }}</th><td><code>{{ row.commit }}</code></td></tr><tr><th>{{ t('tasks.implementation.lead-fix-verifiers') }}</th><td>{{ row.verifiers | join(', ') | inline_code }}</td></tr><tr><th>{{ t('tasks.implementation.files') }}</th><td>{% for path in row.files %}<code>{{ path }}</code>{% if not loop.last %}, {% endif %}{% endfor %}</td></tr><tr><th>{{ t('tasks.implementation.lead-fix-changed-lines') }}</th><td class="figure">{{ row.changedLines }}</td></tr></tbody></table>{% for command in row.commands %}<details><summary><code title="{{ command.command }}">{{ command.command }}</code> · {{ t('tasks.implementation.exit-code') }} {{ command.exitCode }}</summary><pre>{{ command.outputTail }}</pre></details>{% endfor %}</article>{% endfor %}
|
|
33
|
+
</section>{% endif %}
|
|
34
|
+
|
|
29
35
|
<section data-report-section="remaining-issues">
|
|
30
36
|
<h2>{{ t('tasks.implementation.what-is-left') }}</h2>
|
|
31
37
|
{{ render_narrative(narrative.remainingIssues, "implementation.userNarrative.remainingIssues") }}
|
|
@@ -148,7 +148,7 @@ flowchart TD
|
|
|
148
148
|
Verdict --> Report[Final report preserves dissent]
|
|
149
149
|
```
|
|
150
150
|
|
|
151
|
-
Only the executor may mutate project files. The verifier independently re-runs the diff and validation command read-only in the same worktree. Even a verifier with the same provider as the executor runs again in a separate fresh CLI session. This is to prevent a structure where the same session approves a diff the same session wrote.
|
|
151
|
+
Only the executor may mutate project files, with one exception: after the verifier round and convergence, the lead may fix small, local verifier FAIL findings itself (at most 30 changed lines, no test-expectation, dependency, schema, migration or config-of-record change), re-run the checks that found them, and settle the stage when they pass, instead of opening a fix run (`instructions/_implementation-deliverable.md` "Lead minor fix"). The verifier independently re-runs the diff and validation command read-only in the same worktree. Even a verifier with the same provider as the executor runs again in a separate fresh CLI session. This is to prevent a structure where the same session approves a diff the same session wrote.
|
|
152
152
|
|
|
153
153
|
## 5. stage and consumers
|
|
154
154
|
|
|
@@ -195,6 +195,7 @@ The final report requires at least the following.
|
|
|
195
195
|
- actual stdout/stderr and exit code of the plan validation command
|
|
196
196
|
- TDD failing-then-passing evidence
|
|
197
197
|
- per-verifier independent validation rerun result
|
|
198
|
+
- lead minor fixes, when any: finding ids, commit, files, changed lines, and re-run checks (`implementation.leadFixes`)
|
|
198
199
|
- `carry/stage-<N>.json` evidence sidecar and `consumers.jsonl` started/done row
|
|
199
200
|
- rollback verification
|
|
200
201
|
- `implementation.manualUserTest` finalized against the actual diff and whether it is executable
|
|
@@ -175,6 +175,97 @@ def _validate_verifier_command_log_is_read_only(
|
|
|
175
175
|
)
|
|
176
176
|
|
|
177
177
|
|
|
178
|
+
LEAD_FIX_LINE_LIMIT = 30
|
|
179
|
+
_LEAD_FIX_BLOCKING = "lead-fix BLOCKING:"
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def _stage_settled(implementation: dict) -> bool:
|
|
183
|
+
"""carry 를 쓰고 `done` 행을 남긴 run — `withheld` 가 비어 있다."""
|
|
184
|
+
evidence = implementation.get("stageSidecarEvidence")
|
|
185
|
+
return isinstance(evidence, dict) and not str(evidence.get("withheld") or "").strip()
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _lead_fix_rows(implementation: dict) -> list[dict]:
|
|
189
|
+
return [row for row in implementation.get("leadFixes") or [] if isinstance(row, dict)]
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def lead_fixed_verifiers(implementation: dict) -> frozenset[str]:
|
|
193
|
+
"""정착된 stage 에서 통과한 리드 수정이 답한 FAIL 검증자 이름.
|
|
194
|
+
|
|
195
|
+
stage 가 `withheld` 이면 리드 수정은 아무것도 해소하지 않는다 — 다음 run 은
|
|
196
|
+
그 FAIL 행을 그대로 fix-run carry 로 받아야 한다.
|
|
197
|
+
"""
|
|
198
|
+
if not _stage_settled(implementation):
|
|
199
|
+
return frozenset()
|
|
200
|
+
return frozenset(
|
|
201
|
+
str(name)
|
|
202
|
+
for row in _lead_fix_rows(implementation)
|
|
203
|
+
if row.get("result") == "pass"
|
|
204
|
+
for name in row.get("verifiers") or []
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _validate_lead_fixes(data: dict, failures: list[str]) -> None:
|
|
209
|
+
"""리드 수정 기록과, FAIL 이 남은 채 정착된 stage 를 검사한다.
|
|
210
|
+
|
|
211
|
+
`_implementation-deliverable.md` "Lead minor fix": 검증자 FAIL 이 있는 stage 를
|
|
212
|
+
`done` 으로 정착시킬 수 있는 길은 재실행 검사가 전부 통과한 리드 수정뿐이다.
|
|
213
|
+
finding 단위 대응은 검사하지 않는다 — `verifierResults[]` 에 구조화된 finding
|
|
214
|
+
id 가 없어 계산할 수 없다. 검증자 행 단위(`verifiers`)까지만 본다.
|
|
215
|
+
"""
|
|
216
|
+
implementation = data.get("implementation")
|
|
217
|
+
if not isinstance(implementation, dict):
|
|
218
|
+
return
|
|
219
|
+
fixes = _lead_fix_rows(implementation)
|
|
220
|
+
for index, row in enumerate(fixes):
|
|
221
|
+
ids = ", ".join(str(item) for item in row.get("findingIds") or []) or "no finding id"
|
|
222
|
+
label = f"implementation.leadFixes[{index}] ({ids})"
|
|
223
|
+
commands = [cmd for cmd in row.get("commands") or [] if isinstance(cmd, dict)]
|
|
224
|
+
failed = [cmd for cmd in commands if cmd.get("exitCode") != 0]
|
|
225
|
+
if row.get("result") == "pass" and (failed or not commands):
|
|
226
|
+
evidence = (
|
|
227
|
+
f"`{str(failed[0].get('command'))[:120]}` exited {failed[0].get('exitCode')}"
|
|
228
|
+
if failed else "no re-run command is recorded"
|
|
229
|
+
)
|
|
230
|
+
failures.append(
|
|
231
|
+
f"{_LEAD_FIX_BLOCKING} {label} records `result: pass` but {evidence}. "
|
|
232
|
+
"A lead fix passes only when every re-run check exits 0; otherwise "
|
|
233
|
+
"record `result: fail` and withhold the stage."
|
|
234
|
+
)
|
|
235
|
+
lines = row.get("changedLines")
|
|
236
|
+
if isinstance(lines, int) and lines > LEAD_FIX_LINE_LIMIT:
|
|
237
|
+
failures.append(
|
|
238
|
+
f"{_LEAD_FIX_BLOCKING} {label} changed {lines} lines; a lead minor fix "
|
|
239
|
+
f"is limited to {LEAD_FIX_LINE_LIMIT}. A larger change belongs to a fix "
|
|
240
|
+
"run, where the verifiers review it."
|
|
241
|
+
)
|
|
242
|
+
if not _stage_settled(implementation):
|
|
243
|
+
return
|
|
244
|
+
unsettled = [index for index, row in enumerate(fixes) if row.get("result") != "pass"]
|
|
245
|
+
if unsettled:
|
|
246
|
+
failures.append(
|
|
247
|
+
f"{_LEAD_FIX_BLOCKING} implementation.leadFixes{unsettled} recorded "
|
|
248
|
+
"`result: fail` but the stage was settled (no `stageSidecarEvidence.withheld`). "
|
|
249
|
+
"A failed lead fix leaves the verdict FAIL: withhold the carry sidecar and "
|
|
250
|
+
"append a `failed` consumers row."
|
|
251
|
+
)
|
|
252
|
+
rejected = {
|
|
253
|
+
who for who, row in _verifier_rows(data)
|
|
254
|
+
if str(row.get("verdict") or "").strip() == "FAIL"
|
|
255
|
+
}
|
|
256
|
+
uncovered = sorted(rejected - lead_fixed_verifiers(implementation))
|
|
257
|
+
if uncovered:
|
|
258
|
+
# 권고로 둔다 — 정착 여부는 종합 판정이 정하고, 수렴에서 반박된 한 검증자의
|
|
259
|
+
# FAIL 위에 PASS 로 정착한 run 은 계약대로다. 막으면 그런 run 이 전부 멈춘다.
|
|
260
|
+
failures.append(
|
|
261
|
+
f"lead-fix advisory: verifier(s) {uncovered} recorded `verdict: FAIL` but "
|
|
262
|
+
"the stage was settled (no `stageSidecarEvidence.withheld`) and no "
|
|
263
|
+
"`implementation.leadFixes[]` entry with `result: pass` names them in "
|
|
264
|
+
"`verifiers`. If convergence did not refute that FAIL, the stage should have "
|
|
265
|
+
"been withheld or answered by a passing lead minor fix."
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
|
|
178
269
|
def _validate_verifier_fail_blocks_verdict(data: dict, failures: list[str]) -> None:
|
|
179
270
|
"""A verifier FAIL cannot be dropped during synthesis.
|
|
180
271
|
|
|
@@ -47,15 +47,15 @@ def _resolve_planning_input_path(path_value: str, project_root: Path) -> Path |
|
|
|
47
47
|
return None
|
|
48
48
|
|
|
49
49
|
|
|
50
|
-
def
|
|
51
|
-
path_value: str, project_root: Path,
|
|
50
|
+
def is_existing_planning_report(
|
|
51
|
+
path_value: str, project_root: Path, task_group: str, task_id: str,
|
|
52
52
|
) -> bool:
|
|
53
53
|
from okstra_ctl.paths import task_dir
|
|
54
54
|
|
|
55
55
|
try:
|
|
56
56
|
report = lexical_absolute_path(Path(path_value))
|
|
57
57
|
task_root = lexical_absolute_path(
|
|
58
|
-
task_dir(project_root,
|
|
58
|
+
task_dir(project_root, task_group, task_id)
|
|
59
59
|
)
|
|
60
60
|
relative = report.relative_to(task_root)
|
|
61
61
|
except ValueError:
|
|
@@ -95,8 +95,8 @@ def _validate_planning_entry_inputs(project_root: Path, inp: PlanningInputs) ->
|
|
|
95
95
|
if inp.selected_direction_path:
|
|
96
96
|
return
|
|
97
97
|
if inp.clarification_response_path:
|
|
98
|
-
if not
|
|
99
|
-
inp.clarification_response_path, project_root, inp
|
|
98
|
+
if not is_existing_planning_report(
|
|
99
|
+
inp.clarification_response_path, project_root, inp.task_group, inp.task_id
|
|
100
100
|
):
|
|
101
101
|
raise PrepareError(
|
|
102
102
|
"implementation-planning rerun --clarification-response must point "
|
|
@@ -15,6 +15,7 @@ from dataclasses import dataclass
|
|
|
15
15
|
from pathlib import Path
|
|
16
16
|
|
|
17
17
|
from .json_boundary import JsonBoundaryError, load_owned_object
|
|
18
|
+
from .phases.implementation.validation import lead_fixed_verifiers
|
|
18
19
|
|
|
19
20
|
_REPORT_RE = re.compile(r"final-report-implementation-(\d+)\.data\.json$")
|
|
20
21
|
|
|
@@ -68,9 +69,12 @@ def derive_stage_fix_carry(
|
|
|
68
69
|
impl = data.get("implementation")
|
|
69
70
|
if not isinstance(impl, dict):
|
|
70
71
|
return None
|
|
72
|
+
# 통과한 리드 수정이 답한 FAIL 은 그 run 안에서 해소됐다 — 다시 fix run 으로 넘기지 않는다.
|
|
73
|
+
fixed = lead_fixed_verifiers(impl)
|
|
71
74
|
failed = [
|
|
72
75
|
row for row in impl.get("verifierResults") or []
|
|
73
76
|
if isinstance(row, dict) and row.get("verdict") == "FAIL"
|
|
77
|
+
and str(row.get("verifier") or "") not in fixed
|
|
74
78
|
]
|
|
75
79
|
if not failed:
|
|
76
80
|
return None
|
|
@@ -588,10 +588,12 @@ STEPS: list[Step] = [
|
|
|
588
588
|
and s.confirmed is None),
|
|
589
589
|
build=_build_confirm, submit=_submit_confirm,
|
|
590
590
|
owns=("confirmed", "edit_target", "confirmation_prompt", "confirmation_stages", "confirmation_scope", "user_authorization")),
|
|
591
|
+
# 되감기는 이 step 뒤에서 일어나 answered 에 남는다. 반복 가능이 아니면 두 번째
|
|
592
|
+
# 수정에서 건너뛰어져 위저드가 미확정인 채 done 을 냈다(dev-11126 planning).
|
|
591
593
|
Step(S_EDIT_TARGET,
|
|
592
594
|
applies=lambda s: s.confirmed is False and not s.edit_target,
|
|
593
595
|
build=_build_edit_target, submit=_submit_edit_target,
|
|
594
|
-
owns=()),
|
|
596
|
+
owns=(), repeatable=True),
|
|
595
597
|
]
|
|
596
598
|
|
|
597
599
|
|
|
@@ -10,6 +10,7 @@ from okstra_ctl.clarification_items import user_response_sidecars
|
|
|
10
10
|
from okstra_ctl.manager_paths import MANAGER_CONTEXT_DIRECTIVE_PREFIX
|
|
11
11
|
from okstra_ctl.json_boundary import JsonBoundaryError, load_owned_object, write_owned_object_atomic
|
|
12
12
|
from okstra_ctl.report_language import is_language_tag
|
|
13
|
+
from okstra_ctl.phases.implementation_planning.entry import is_existing_planning_report
|
|
13
14
|
from okstra_ctl.pr_template import PrTemplateError, resolve_pr_template_path
|
|
14
15
|
from okstra_project.dirs import project_json_path
|
|
15
16
|
from okstra_project.state import StateError, list_project_tasks
|
|
@@ -358,6 +359,16 @@ def _submit_clarification(state: WizardState, value: str) -> Optional[str]:
|
|
|
358
359
|
state.clarification_response_path = ""
|
|
359
360
|
return "clarification: (none)"
|
|
360
361
|
p = _require_file(val, Path(state.project_root), "clarification-response")
|
|
362
|
+
# prepare 가 거절할 경로를 여기서 받으면 재실행이 조용히 새 계획 흐름
|
|
363
|
+
# (selected_direction_pick)으로 바뀌었다(dev-11126: implementation 리포트를 넣음).
|
|
364
|
+
if state.task_type == "implementation-planning" and not is_existing_planning_report(
|
|
365
|
+
str(p), Path(state.project_root), state.task_group, state.task_id,
|
|
366
|
+
):
|
|
367
|
+
raise WizardError(
|
|
368
|
+
"an implementation-planning rerun takes an implementation-planning report "
|
|
369
|
+
"of this task (runs/implementation-planning/reports/"
|
|
370
|
+
f"final-report-implementation-planning-<seq>.data.json); got {p}"
|
|
371
|
+
)
|
|
361
372
|
state.clarification_response_path = str(p)
|
|
362
373
|
return f"clarification: {p}"
|
|
363
374
|
|
|
@@ -1347,6 +1347,13 @@
|
|
|
1347
1347
|
"$ref": "#/$defs/VerifierResultBlock"
|
|
1348
1348
|
}
|
|
1349
1349
|
},
|
|
1350
|
+
"leadFixes": {
|
|
1351
|
+
"type": "array",
|
|
1352
|
+
"description": "Lead minor fixes applied in this run after the verifier round (`_implementation-deliverable.md` \"Lead minor fix\"). Optional so reports of tasks already in flight keep validating.",
|
|
1353
|
+
"items": {
|
|
1354
|
+
"$ref": "#/$defs/LeadFixRow"
|
|
1355
|
+
}
|
|
1356
|
+
},
|
|
1350
1357
|
"rollbackVerification": {
|
|
1351
1358
|
"type": "array",
|
|
1352
1359
|
"minItems": 1,
|
|
@@ -10063,6 +10070,92 @@
|
|
|
10063
10070
|
}
|
|
10064
10071
|
}
|
|
10065
10072
|
},
|
|
10073
|
+
"LeadFixRow": {
|
|
10074
|
+
"type": "object",
|
|
10075
|
+
"description": "One lead minor fix: the verifier findings it answers, the commit that carries it, and the re-run checks that decide whether the run may settle the stage. `changedLines` has no schema maximum on purpose: an over-limit fix must still be recordable, and `_validate_lead_fixes` reports it.",
|
|
10076
|
+
"required": [
|
|
10077
|
+
"findingIds",
|
|
10078
|
+
"verifiers",
|
|
10079
|
+
"summary",
|
|
10080
|
+
"commit",
|
|
10081
|
+
"files",
|
|
10082
|
+
"changedLines",
|
|
10083
|
+
"commands",
|
|
10084
|
+
"result"
|
|
10085
|
+
],
|
|
10086
|
+
"additionalProperties": false,
|
|
10087
|
+
"properties": {
|
|
10088
|
+
"findingIds": {
|
|
10089
|
+
"type": "array",
|
|
10090
|
+
"minItems": 1,
|
|
10091
|
+
"items": {
|
|
10092
|
+
"type": "string",
|
|
10093
|
+
"minLength": 1
|
|
10094
|
+
}
|
|
10095
|
+
},
|
|
10096
|
+
"verifiers": {
|
|
10097
|
+
"description": "The `verifierResults[].verifier` names whose FAIL this fix answers.",
|
|
10098
|
+
"type": "array",
|
|
10099
|
+
"minItems": 1,
|
|
10100
|
+
"items": {
|
|
10101
|
+
"type": "string",
|
|
10102
|
+
"minLength": 1
|
|
10103
|
+
}
|
|
10104
|
+
},
|
|
10105
|
+
"summary": {
|
|
10106
|
+
"type": "string",
|
|
10107
|
+
"minLength": 1
|
|
10108
|
+
},
|
|
10109
|
+
"commit": {
|
|
10110
|
+
"type": "string",
|
|
10111
|
+
"pattern": "^[0-9a-f]{7,40}$"
|
|
10112
|
+
},
|
|
10113
|
+
"files": {
|
|
10114
|
+
"type": "array",
|
|
10115
|
+
"minItems": 1,
|
|
10116
|
+
"items": {
|
|
10117
|
+
"type": "string",
|
|
10118
|
+
"minLength": 1
|
|
10119
|
+
}
|
|
10120
|
+
},
|
|
10121
|
+
"changedLines": {
|
|
10122
|
+
"description": "Added plus deleted lines of the lead commit (`git diff --numstat`).",
|
|
10123
|
+
"type": "integer",
|
|
10124
|
+
"minimum": 1
|
|
10125
|
+
},
|
|
10126
|
+
"commands": {
|
|
10127
|
+
"type": "array",
|
|
10128
|
+
"minItems": 1,
|
|
10129
|
+
"items": {
|
|
10130
|
+
"type": "object",
|
|
10131
|
+
"required": [
|
|
10132
|
+
"command",
|
|
10133
|
+
"exitCode",
|
|
10134
|
+
"outputTail"
|
|
10135
|
+
],
|
|
10136
|
+
"additionalProperties": false,
|
|
10137
|
+
"properties": {
|
|
10138
|
+
"command": {
|
|
10139
|
+
"type": "string",
|
|
10140
|
+
"minLength": 1
|
|
10141
|
+
},
|
|
10142
|
+
"exitCode": {
|
|
10143
|
+
"type": "integer"
|
|
10144
|
+
},
|
|
10145
|
+
"outputTail": {
|
|
10146
|
+
"type": "string"
|
|
10147
|
+
}
|
|
10148
|
+
}
|
|
10149
|
+
}
|
|
10150
|
+
},
|
|
10151
|
+
"result": {
|
|
10152
|
+
"enum": [
|
|
10153
|
+
"pass",
|
|
10154
|
+
"fail"
|
|
10155
|
+
]
|
|
10156
|
+
}
|
|
10157
|
+
}
|
|
10158
|
+
},
|
|
10066
10159
|
"RollbackVerificationRow": {
|
|
10067
10160
|
"type": "object",
|
|
10068
10161
|
"required": [
|
|
@@ -428,7 +428,7 @@ If the anchor (`implementation_base_commit`) is reported unresolvable, run the s
|
|
|
428
428
|
Because of the dependency closure, the chain queue **may include a stage that another implementation run has occupied as started/reserved.** That stage's `render-bundle` is rejected with `--stage N already in progress or reserved by another run` (StageTargetError). This is **not** an exception gate needing human judgment but a "next stage not yet ready" situation. On this rejection, **terminate the chain normally** and report the remaining queue to the user (e.g. `remaining queue: stage 4, 5 — resume with okstra-run after occupancy is released`). This is a different branch from the exception gate below (data corruption·concurrent-occupancy conflict confirmation).
|
|
429
429
|
|
|
430
430
|
### Stage ended FAIL — stop the queue and report (not an exception gate)
|
|
431
|
-
When a stage's synthesised verdict is `FAIL`, Phase 6 writes no carry sidecar and appends a `status:"failed"` row in place of `done` (`scripts/okstra_ctl/phases/implementation/instructions/_implementation-deliverable.md` "Lead post-stage persistence"). **Stop the queue at that stage** and report the failed stage, its report path, and the remaining queue (e.g. `stage 1 FAIL — remaining queue: stage 2, 3, 5; re-enter with okstra-run --stage 1 after the fix`). Do **not** continue to the next stage even when that stage is dependency-independent: an unattended chain that keeps building past a confirmed regression stacks later work on top of it. The `failed` row releases the stage's occupancy, so `--stage <N>` re-enters the same stage on its preserved worktree and branch — there is nothing to unblock by hand.
|
|
431
|
+
A verifier FAIL that a passing lead minor fix answered does not make the verdict `FAIL` (`_implementation-deliverable.md` "Lead minor fix"); that stage settles `done` and the queue continues. When a stage's synthesised verdict is `FAIL`, Phase 6 writes no carry sidecar and appends a `status:"failed"` row in place of `done` (`scripts/okstra_ctl/phases/implementation/instructions/_implementation-deliverable.md` "Lead post-stage persistence"). **Stop the queue at that stage** and report the failed stage, its report path, and the remaining queue (e.g. `stage 1 FAIL — remaining queue: stage 2, 3, 5; re-enter with okstra-run --stage 1 after the fix`). Do **not** continue to the next stage even when that stage is dependency-independent: an unattended chain that keeps building past a confirmed regression stacks later work on top of it. The `failed` row releases the stage's occupancy, so `--stage <N>` re-enters the same stage on its preserved worktree and branch — there is nothing to unblock by hand.
|
|
432
432
|
|
|
433
433
|
### Exception gate during chaining
|
|
434
434
|
If `render-bundle` raises Step 5's concurrent-run conflict detection (concurrent-run branch) or git stale-SHA reconciliation (git-reconcile branch), **stop the chain at that stage** and present the gate to the user exactly as Step 5 prescribes. Once the user resolves the gate, resume the chain in place (continue with the remaining queue). Data corruption·concurrent-occupancy conflicts are confirmed by a human — this is the safety boundary of unattended chaining. (Unlike the "not ready" rejection above, these two branches do not discard the queue; they wait for user resolution.)
|
|
@@ -673,6 +673,12 @@
|
|
|
673
673
|
"action-deleted": "Deleted",
|
|
674
674
|
"what-it-does": "What it does",
|
|
675
675
|
"advisory": "advisory",
|
|
676
|
+
"lead-fixes": "Fixed by the lead after verification",
|
|
677
|
+
"lead-fixes-intro": "A verifier rejected the change for a small, local defect. The lead fixed it in this run and re-ran the checks that found it, instead of starting a fix run.",
|
|
678
|
+
"lead-fix-verifiers": "Answers the FAIL of",
|
|
679
|
+
"lead-fix-changed-lines": "Changed lines",
|
|
680
|
+
"lead-fix-result-pass": "checks pass",
|
|
681
|
+
"lead-fix-result-fail": "checks still fail",
|
|
676
682
|
"entry-count": "{count} entries",
|
|
677
683
|
"manual-user-test": "Manual user test",
|
|
678
684
|
"environment": "Environment:",
|
|
@@ -673,6 +673,12 @@
|
|
|
673
673
|
"action-deleted": "삭제",
|
|
674
674
|
"what-it-does": "하는 일",
|
|
675
675
|
"advisory": "권고",
|
|
676
|
+
"lead-fixes": "검증 뒤 리드가 고친 것",
|
|
677
|
+
"lead-fixes-intro": "검증자가 작고 국소적인 결함으로 변경을 거부했습니다. 리드가 fix run 을 새로 여는 대신 이 run 안에서 고치고, 그 결함을 찾은 검사를 다시 돌렸습니다.",
|
|
678
|
+
"lead-fix-verifiers": "FAIL 을 낸 검증자",
|
|
679
|
+
"lead-fix-changed-lines": "바뀐 줄 수",
|
|
680
|
+
"lead-fix-result-pass": "검사 통과",
|
|
681
|
+
"lead-fix-result-fail": "검사 여전히 실패",
|
|
676
682
|
"entry-count": "{count}개 항목",
|
|
677
683
|
"manual-user-test": "수동 사용자 테스트",
|
|
678
684
|
"environment": "환경:",
|
|
@@ -40,6 +40,7 @@ from okstra_ctl.phases.implementation.validation import (
|
|
|
40
40
|
_validate_verifier_reran_independently,
|
|
41
41
|
_validate_verifier_discrepancy_is_not_passed,
|
|
42
42
|
_validate_verifier_fail_blocks_verdict,
|
|
43
|
+
_validate_lead_fixes,
|
|
43
44
|
_validate_verifier_command_log_is_read_only as validate_verifier_command_log_is_read_only,
|
|
44
45
|
_validate_verifier_discrepancy_names_checklist_phase as validate_verifier_discrepancy_names_checklist_phase,
|
|
45
46
|
_warn_out_of_plan_edits_not_in_diff as warn_out_of_plan_edits_not_in_diff,
|
|
@@ -194,6 +195,7 @@ from okstra_ctl.dispatch_state import ( # noqa: E402
|
|
|
194
195
|
v2_worker_state_key,
|
|
195
196
|
)
|
|
196
197
|
from okstra_ctl.paths import RunRef, okstra_home, project_rel # noqa: E402
|
|
198
|
+
from okstra_ctl.consumers import carry_head_commit # noqa: E402
|
|
197
199
|
from okstra_ctl.tdd_bypass import ( # noqa: E402
|
|
198
200
|
bypass_file as tdd_bypass_file,
|
|
199
201
|
granted_stages,
|
|
@@ -3182,6 +3184,8 @@ def validate_final_report_data(
|
|
|
3182
3184
|
)
|
|
3183
3185
|
elif task_type == "implementation":
|
|
3184
3186
|
_validate_stage_carry_sidecar_exists(data, report_path, failures)
|
|
3187
|
+
_validate_lead_fixes(data, failures)
|
|
3188
|
+
_validate_lead_fix_carry_head(data, report_path, failures)
|
|
3185
3189
|
_validate_lead_authored_report(data, report_path, failures)
|
|
3186
3190
|
if task_type == "error-analysis":
|
|
3187
3191
|
_validate_error_analysis_consistency(data, failures)
|
|
@@ -4090,6 +4094,53 @@ def _validate_stage_carry_sidecar_exists(
|
|
|
4090
4094
|
)
|
|
4091
4095
|
|
|
4092
4096
|
|
|
4097
|
+
def _validate_lead_fix_carry_head(
|
|
4098
|
+
data: dict,
|
|
4099
|
+
report_path: Path,
|
|
4100
|
+
failures: list[str],
|
|
4101
|
+
) -> None:
|
|
4102
|
+
"""A settled stage with a passing lead fix must carry the post-fix head.
|
|
4103
|
+
|
|
4104
|
+
The executor emits the carry before the verifier round, so its
|
|
4105
|
+
`stageCommitRange.head` names the pre-fix commit. Persisted verbatim, the
|
|
4106
|
+
next dependent stage branches from a head without the fix
|
|
4107
|
+
(`consumers.backfill_done_from_carry` reads the same field). Advisory: the
|
|
4108
|
+
carry file is never overwritten, so a blocking failure here would leave
|
|
4109
|
+
the run no way forward.
|
|
4110
|
+
"""
|
|
4111
|
+
implementation = data.get("implementation")
|
|
4112
|
+
if not isinstance(implementation, dict):
|
|
4113
|
+
return
|
|
4114
|
+
evidence = implementation.get("stageSidecarEvidence")
|
|
4115
|
+
if not isinstance(evidence, dict) or str(evidence.get("withheld") or "").strip():
|
|
4116
|
+
return
|
|
4117
|
+
stage = evidence.get("stageNumber")
|
|
4118
|
+
commits = {
|
|
4119
|
+
str(row.get("commit"))
|
|
4120
|
+
for row in implementation.get("leadFixes") or []
|
|
4121
|
+
if isinstance(row, dict) and row.get("result") == "pass" and row.get("commit")
|
|
4122
|
+
}
|
|
4123
|
+
if not isinstance(stage, int) or not commits:
|
|
4124
|
+
return
|
|
4125
|
+
carry_path = RunRef.from_report_path(report_path).carry(stage)
|
|
4126
|
+
if not carry_path.is_file():
|
|
4127
|
+
return
|
|
4128
|
+
try:
|
|
4129
|
+
carry = json.loads(carry_path.read_text(encoding="utf-8"))
|
|
4130
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
4131
|
+
failures.append(f"lead-fix carry: cannot read {carry_path.name} to compare its head: {exc}")
|
|
4132
|
+
return
|
|
4133
|
+
head = carry_head_commit(carry) if isinstance(carry, dict) else ""
|
|
4134
|
+
if not any(head and (head.startswith(c) or c.startswith(head)) for c in commits):
|
|
4135
|
+
failures.append(
|
|
4136
|
+
f"lead-fix carry: {carry_path.name} names head `{head or '<none>'}`, not "
|
|
4137
|
+
f"a passing lead fix commit {sorted(commits)}. Dependent stages branch "
|
|
4138
|
+
"from that head and miss the fix; set `stageCommitRange.head` to the "
|
|
4139
|
+
"post-fix HEAD before persisting the carry "
|
|
4140
|
+
'(_implementation-deliverable.md "Lead minor fix").'
|
|
4141
|
+
)
|
|
4142
|
+
|
|
4143
|
+
|
|
4093
4144
|
def _normalize_report_contracts(raw_contracts: object) -> set[str]:
|
|
4094
4145
|
if isinstance(raw_contracts, str):
|
|
4095
4146
|
candidates = [raw_contracts]
|