okstra 0.176.0 → 0.177.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. package/docs/task-process/final-verification.md +5 -3
  2. package/package.json +1 -1
  3. package/runtime/BUILD.json +2 -2
  4. package/runtime/agents/workers/report-writer-worker.md +1 -1
  5. package/runtime/bin/okstra-provider-exec.py +2 -7
  6. package/runtime/prompts/coding-preflight/overview.md +2 -1
  7. package/runtime/prompts/lead/convergence.md +11 -6
  8. package/runtime/prompts/lead/okstra-lead-contract.md +2 -2
  9. package/runtime/prompts/lead/plan-body-verification.md +4 -2
  10. package/runtime/prompts/lead/report-writer.md +8 -4
  11. package/runtime/prompts/profiles/_common-contract.md +1 -1
  12. package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
  13. package/runtime/prompts/profiles/final-verification.md +11 -9
  14. package/runtime/prompts/profiles/improvement-discovery.md +1 -1
  15. package/runtime/prompts/profiles/release-handoff.md +7 -6
  16. package/runtime/prompts/wizard/prompts.ko.json +2 -2
  17. package/runtime/python/okstra_ctl/agent_prompt_cli.py +2 -2
  18. package/runtime/python/okstra_ctl/dispatch_core.py +50 -6
  19. package/runtime/python/okstra_ctl/dispatch_state.py +28 -4
  20. package/runtime/python/okstra_ctl/execution_mutation_audit.py +49 -3
  21. package/runtime/python/okstra_ctl/handoff.py +27 -14
  22. package/runtime/python/okstra_ctl/release_gate.py +56 -0
  23. package/runtime/python/okstra_ctl/report_contract.py +1 -0
  24. package/runtime/python/okstra_ctl/report_finalize.py +54 -0
  25. package/runtime/python/okstra_ctl/report_html/view_models/final_verification.py +4 -1
  26. package/runtime/python/okstra_ctl/run.py +31 -16
  27. package/runtime/python/okstra_ctl/stage_targets.py +73 -1
  28. package/runtime/python/okstra_ctl/wizard.py +19 -10
  29. package/runtime/python/okstra_ctl/worker_liveness.py +3 -1
  30. package/runtime/python/okstra_ctl/wrapper_status.py +15 -0
  31. package/runtime/schemas/final-report-v2.0.schema.json +55 -18
  32. package/runtime/templates/reports/html/i18n/en.json +4 -0
  33. package/runtime/templates/reports/html/i18n/ko.json +4 -0
  34. package/runtime/templates/reports/html/tasks/final-verification.template.html +7 -1
  35. package/runtime/validators/validate-run.py +98 -12
  36. package/runtime/validators/validate_analysis_report.py +2 -5
@@ -107,7 +107,11 @@ from .worker_prompt_headers import (
107
107
  resolve_errors_log_path,
108
108
  )
109
109
  from .worker_artifact_paths import audit_sidecar_rel
110
- from .wrapper_status import read_wrapper_status, status_path_for_prompt
110
+ from .wrapper_status import (
111
+ log_path_for_prompt,
112
+ read_wrapper_status,
113
+ status_path_for_prompt,
114
+ )
111
115
  from .worker_request import verifier_extra_dirs
112
116
  from .write_policy import (
113
117
  WriteEnforcement,
@@ -394,11 +398,38 @@ def dispatch_plan(plan: DispatchPlan, *, wait: bool = True) -> int:
394
398
  if result != 0:
395
399
  return result
396
400
  return 0
397
- handles = [_spawn_job(plan, job, 1) for job in plan.jobs]
401
+ round_artifact_paths = _round_artifact_paths(plan)
402
+ handles = [
403
+ _spawn_job(plan, job, 1, batch_artifact_paths=round_artifact_paths)
404
+ for job in plan.jobs
405
+ ]
398
406
  _record_dispatch_facts(plan.team_state_path, _mode_from_handles(handles))
399
407
  return 0
400
408
 
401
409
 
410
+ def _round_artifact_paths(plan: DispatchPlan) -> tuple[Path, ...]:
411
+ """Every artifact this round's workers are entitled to write.
412
+
413
+ These jobs run concurrently into one shared artifact root, so each worker's
414
+ own result, audit sidecar, status sidecar and live log land inside every
415
+ sibling's audit window. Judged against one worker's policy alone, the
416
+ siblings' writes read as unauthorized artifact-root changes and failed a
417
+ whole round of otherwise clean verifiers.
418
+
419
+ They travel as orchestrator paths rather than as a widened policy union
420
+ because the snapshot carries orchestrator paths and the audit excuses them
421
+ without consulting a policy — while `_validate_snapshot_authority` compares
422
+ the recorded policy digests exactly, so a snapshot taken under a union can
423
+ no longer be closed by the per-job policy the awaiting process rebuilds.
424
+ Same effect on the verdict; no coupling between two processes' plans.
425
+ """
426
+ return tuple(
427
+ path
428
+ for job in plan.jobs
429
+ for path in _worker_artifact_paths(plan, job)
430
+ )
431
+
432
+
402
433
  def dispatch_cli_wrapper_plan(plan: DispatchPlan) -> int:
403
434
  """Start one dependency-free CLI batch before collecting any worker."""
404
435
  if any(job.backend != BACKEND_CLI_WRAPPER for job in plan.jobs):
@@ -1263,7 +1294,13 @@ def _jobs_from_file(
1263
1294
  )
1264
1295
 
1265
1296
 
1266
- def _spawn_job(plan: DispatchPlan, job: WorkerJob, attempt: int) -> WorkerHandle:
1297
+ def _spawn_job(
1298
+ plan: DispatchPlan,
1299
+ job: WorkerJob,
1300
+ attempt: int,
1301
+ *,
1302
+ batch_artifact_paths: Sequence[Path] = (),
1303
+ ) -> WorkerHandle:
1267
1304
  job = _prepare_job_attempt(plan, job, attempt)
1268
1305
  contract = _persisted_write_contract(plan, job)
1269
1306
  policies = (contract[0],) if contract else ()
@@ -1275,7 +1312,12 @@ def _spawn_job(plan: DispatchPlan, job: WorkerJob, attempt: int) -> WorkerHandle
1275
1312
  _load_json_object(snapshot_path, "mutation audit snapshot")
1276
1313
  )
1277
1314
  else:
1278
- snapshot = _mutation_snapshot(plan, policies, (job,))
1315
+ snapshot = _mutation_snapshot(
1316
+ plan,
1317
+ policies,
1318
+ (job,),
1319
+ round_artifact_paths=batch_artifact_paths,
1320
+ )
1279
1321
  handle = replace(
1280
1322
  _start_job(plan, job),
1281
1323
  mutation_snapshot=snapshot,
@@ -1373,11 +1415,13 @@ def _mutation_snapshot(
1373
1415
  plan: DispatchPlan,
1374
1416
  policies: Sequence[WritePolicy],
1375
1417
  jobs: Sequence[WorkerJob],
1418
+ round_artifact_paths: Sequence[Path] = (),
1376
1419
  ) -> MutationSnapshot:
1377
1420
  sidecars = tuple(_mutation_snapshot_path(job) for job in jobs)
1378
1421
  snapshot = ExecutionMutationAudit().snapshot(
1379
1422
  policies,
1380
1423
  orchestrator_paths=(
1424
+ *round_artifact_paths,
1381
1425
  plan.manifest_path,
1382
1426
  plan.team_state_path,
1383
1427
  Path(f"{plan.team_state_path}.lock"),
@@ -1478,7 +1522,7 @@ def _worker_artifact_paths(plan: DispatchPlan, job: WorkerJob) -> tuple[Path, ..
1478
1522
  *job.completion_paths,
1479
1523
  Path(audit_sidecar_rel(str(job.worker_result_path))),
1480
1524
  status_path_for_prompt(job.prompt_path),
1481
- job.prompt_path.with_suffix(job.prompt_path.suffix + ".log"),
1525
+ log_path_for_prompt(job.prompt_path),
1482
1526
  }
1483
1527
  error_logs = active_context.get("errorLogs")
1484
1528
  if isinstance(error_logs, Mapping):
@@ -2171,7 +2215,7 @@ def _wrapper_log_tail(job: WorkerJob) -> str | None:
2171
2215
  some process exited 1. Capped well under the writer's own excerpt limit
2172
2216
  because a whole record must stay inside one atomic `PIPE_BUF` append.
2173
2217
  """
2174
- log_path = job.prompt_path.with_suffix(job.prompt_path.suffix + ".log")
2218
+ log_path = log_path_for_prompt(job.prompt_path)
2175
2219
  try:
2176
2220
  with log_path.open("rb") as handle:
2177
2221
  handle.seek(0, 2)
@@ -57,7 +57,7 @@ from .worker_prompt_contract import (
57
57
  from .worker_runner import LIVE, QUIET
58
58
  from .worker_request import verifier_extra_dirs
59
59
  from .worker_artifact_paths import audit_sidecar_rel
60
- from .wrapper_status import status_path_for_prompt
60
+ from .wrapper_status import log_path_for_prompt, status_path_for_prompt
61
61
  from .write_policy import (
62
62
  build_invocation_write_contract,
63
63
  planned_paths_from_run_manifest,
@@ -176,7 +176,7 @@ class WorkerJob:
176
176
  self.model_execution_value,
177
177
  str(self.prompt_path),
178
178
  self.worktree_path,
179
- self.role,
179
+ self.wrapper_role,
180
180
  str(self.idle_timeout_seconds),
181
181
  "--presentation",
182
182
  self._presentation(),
@@ -190,6 +190,30 @@ class WorkerJob:
190
190
  argv += ["--session-id", self.session_id]
191
191
  return argv
192
192
 
193
+ @property
194
+ def wrapper_role(self) -> str:
195
+ """The canonical role the entrypoint's role positional is read as.
196
+
197
+ `self.role` is the roster label team-state carries (`Antigravity
198
+ worker`) so the report's execution-status row can quote it. The
199
+ entrypoint reads that same position through `normalize_role`, which
200
+ knows only the eleven canonical ids, and compares it against
201
+ `role_for_duty(dutyId)` from the invocation metadata. Handing it the
202
+ label failed that check before the status sidecar was written, so a
203
+ pane worker exited 64 leaving no `.log` and no `.status.json` while
204
+ team-state still read `in-progress`. It also silently denied the
205
+ verifier the toolchain dirs `verifier_extra_dirs` grants by role.
206
+
207
+ A legacy v1 job carries no duty id; the entrypoint skips the metadata
208
+ role check for it, so its existing value is preserved.
209
+ """
210
+ if not self.duty_id:
211
+ return self.role
212
+ try:
213
+ return role_for_duty(self.duty_id)
214
+ except RoleCatalogError:
215
+ return self.role
216
+
193
217
  def _presentation(self) -> str:
194
218
  # A pane is a screen a person watches; a cli-wrapper dispatch's stdout is
195
219
  # a subagent's context window. Progress belongs in the first and not the
@@ -790,7 +814,7 @@ def _manifest_worker_write_paths(
790
814
  result,
791
815
  Path(audit_sidecar_rel(str(result))),
792
816
  status_path_for_prompt(prompt_path),
793
- prompt_path.with_suffix(prompt_path.suffix + ".log"),
817
+ log_path_for_prompt(prompt_path),
794
818
  }
795
819
  active_value = authority.get("activeRunContextPath")
796
820
  active = (
@@ -847,7 +871,7 @@ def _prompt_write_paths(
847
871
  _project_or_absolute(project_root, value) for value in artifact_values
848
872
  }
849
873
  artifacts.add(status_path_for_prompt(prompt_path))
850
- artifacts.add(prompt_path.with_suffix(prompt_path.suffix + ".log"))
874
+ artifacts.add(log_path_for_prompt(prompt_path))
851
875
  worktree_value = values.get("Worktree")
852
876
  worktree = (
853
877
  _project_or_absolute(project_root, worktree_value)
@@ -171,7 +171,10 @@ class ExecutionMutationAudit:
171
171
  before.scratch_digests, after.scratch_digests
172
172
  )
173
173
  source_changes = _source_changes(changed, rows, before)
174
- git_changed = before.git_projection != after.git_projection
174
+ # Reported and, through `_retry_allowed`, load-bearing: a stat-cache
175
+ # refresh must not read as "this worker touched Git" and must not
176
+ # withhold the retry a worker that failed for another reason is owed.
177
+ git_changed = _stable_git_projection(before) != _stable_git_projection(after)
175
178
  violations = _policy_violations(
176
179
  before,
177
180
  after,
@@ -250,6 +253,22 @@ def _maximum_precision(policy: WritePolicy) -> str:
250
253
  return policy.maximum_boundary_precision
251
254
 
252
255
 
256
+ _INSTALLED_DEPENDENCY_DIRS = frozenset({"node_modules"})
257
+ """Trees a package manager installs, which no worker authored and no policy can
258
+ enumerate.
259
+
260
+ Excluded for the same reason `.git` is: the audit asks whether the worker
261
+ changed *source*, and these hold neither source nor run artifacts. What forced
262
+ the entry is that the tools inside them write to themselves — a `final-verification`
263
+ verifier running the Tier 1 / Tier 2 suites its own profile mandates had Vitest
264
+ persist `node_modules/.vite/vitest/<hash>/results.json`, and that single cache
265
+ file failed both acceptance verifiers of an otherwise clean stage.
266
+
267
+ One ecosystem, because one is what has been observed. A second name belongs here
268
+ when a run produces the same evidence for it, not before.
269
+ """
270
+
271
+
253
272
  def _content_snapshot(root: Path, generated: frozenset[str]) -> dict[str, str]:
254
273
  rows: dict[str, str] = {}
255
274
  for current, directories, filenames in os.walk(root, followlinks=False):
@@ -258,7 +277,11 @@ def _content_snapshot(root: Path, generated: frozenset[str]) -> dict[str, str]:
258
277
  for name in sorted(directories):
259
278
  path = current_path / name
260
279
  relative_path = path.relative_to(root)
261
- if name == ".git" or _excluded(relative_path, generated):
280
+ if (
281
+ name == ".git"
282
+ or name in _INSTALLED_DEPENDENCY_DIRS
283
+ or _excluded(relative_path, generated)
284
+ ):
262
285
  continue
263
286
  if path.is_symlink():
264
287
  rows[relative_path.as_posix()] = _path_digest(path)
@@ -473,7 +496,7 @@ def _policy_violations(
473
496
  ) -> list[str]:
474
497
  if all(policy.source_mode == "source-readonly" for policy in policies):
475
498
  failures = ["readonly source changed"] if source_changes else []
476
- if before.git_projection != after.git_projection:
499
+ if _stable_git_projection(before) != _stable_git_projection(after):
477
500
  failures.append("gitPolicy disabled but Git projection changed")
478
501
  return failures
479
502
  policy = next(row for row in policies if row.source_mode == "project-mutation")
@@ -482,6 +505,29 @@ def _policy_violations(
482
505
  return failures
483
506
 
484
507
 
508
+ def _stable_git_projection(snapshot: MutationSnapshot) -> dict[str, Any]:
509
+ """The projection minus the one field a read-only reader moves on its own.
510
+
511
+ `indexDigest` is Git's stat cache, and Git rewrites it whenever a plain
512
+ read refreshes a stale entry — `git status --short`, which the
513
+ `final-verification` profile requires of every verifier, is enough. Nothing
514
+ about the worktree's content changed when it moves, so comparing it made a
515
+ read-only worker fail for doing what its own phase told it to do (observed:
516
+ both stage-1 acceptance verifiers, where `indexDigest` was the only key that
517
+ differed and `stagedPaths` was empty on both sides).
518
+
519
+ Every field that does witness a mutation stays compared: `head` and
520
+ `branchCommit` for a moved ref, `stagedPaths` for content added to the
521
+ index, `worktreeRegistration` for a re-registered worktree, `reflogDigest`
522
+ for a ref rewrite, and both Git directories for a redirected repository.
523
+ """
524
+ return {
525
+ key: value
526
+ for key, value in snapshot.git_projection.items()
527
+ if key != "indexDigest"
528
+ }
529
+
530
+
485
531
  def _source_policy_failures(
486
532
  policy: WritePolicy,
487
533
  changed: set[str],
@@ -15,6 +15,11 @@ from typing import Any, Callable, Dict, List, Optional, Tuple
15
15
  from . import consumers, stage_targets, worktree_registry
16
16
  from .final_report_paths import final_report_markdown_path
17
17
  from .paths import RunRef
18
+ from .release_gate import (
19
+ blocking_condition_ids,
20
+ release_handoff_allowed,
21
+ verdict_token,
22
+ )
18
23
  from .stage_map import StageMapError, parse_stage_map_file, stage_map_records
19
24
  from .worktree import (compute_branch_name, compute_worktree_path,
20
25
  main_worktree_path, is_dirty_excluding_okstra,
@@ -52,12 +57,14 @@ def compute_eligibility(stage_map: List[Dict[str, Any]],
52
57
  ).handoff_eligibility()
53
58
 
54
59
 
55
- def latest_whole_task_fv_accepted(project_root, project_id: str,
56
- task_group: str, task_id: str) -> str:
57
- """accepted whole-task final-verification 보고서 경로. 없으면 ''.
60
+ def latest_whole_task_fv_release_ready(project_root, project_id: str,
61
+ task_group: str, task_id: str) -> str:
62
+ """release-handoff 로 넘어가도 되는 whole-task 검증 보고서 경로. 없으면 ''.
58
63
 
59
64
  판정의 SSOT 는 final-report 의 data.json 이다 (record_verified 와 동일
60
- 필드: header.taskType / verificationScope / finalVerdict.verdictToken).
65
+ 필드: header.taskType / verificationScope / finalVerdict). 통과 조건은
66
+ `release_gate.release_handoff_allowed` 한 곳이 소유한다 — `accepted`,
67
+ 또는 모든 조건이 릴리스를 막지 않는다고 선언한 `conditional-accept`.
61
68
  whole-task run 산출물만 평면 reports/ 에 남으므로 stage-* 는 걸리지 않는다."""
62
69
  from okstra_project.state import find_task_root
63
70
  root = find_task_root(Path(project_root),
@@ -71,11 +78,9 @@ def latest_whole_task_fv_accepted(project_root, project_id: str,
71
78
  data = json.loads(dj.read_text(encoding="utf-8"))
72
79
  except (OSError, json.JSONDecodeError):
73
80
  continue
74
- token = ((data.get("finalVerdict") or {}).get("verdictToken")
75
- or "").strip().lower()
76
81
  if ((data.get("header") or {}).get("taskType") == "final-verification"
77
82
  and data.get("verificationScope") == "whole-task"
78
- and token == "accepted"):
83
+ and release_handoff_allowed(data)):
79
84
  md = final_report_markdown_path(dj)
80
85
  return str(md if md.is_file() else dj)
81
86
  return ""
@@ -363,8 +368,10 @@ def _impl_task_key_for_any(rows: List[Dict[str, Any]], stages: List[int]) -> str
363
368
 
364
369
  def record_verified(*, plan_run_root, stage: int, report_path: str,
365
370
  data_json) -> Dict[str, Any]:
366
- """단독-stage accepted 만 기록. data.json 의 taskType/scope/verdict 를 검증해
367
- lead 가 임의 보고서를 verified 로 올리는 것을 막는다."""
371
+ """릴리스로 넘어갈 수 있는 단독-stage 판정만 기록. data.json 의
372
+ taskType/scope/verdict 를 검증해 lead 가 임의 보고서를 verified 로 올리는 것을
373
+ 막는다. 통과 조건은 whole-task 와 같고(`release_gate`), 기록되는 verdict 는
374
+ 보고서가 실제로 실은 토큰이다."""
368
375
  try:
369
376
  data = json.loads(Path(data_json).read_text(encoding="utf-8"))
370
377
  except (OSError, json.JSONDecodeError) as exc:
@@ -376,14 +383,20 @@ def record_verified(*, plan_run_root, stage: int, report_path: str,
376
383
  raise HandoffError(
377
384
  f"record-verified requires verificationScope single-stage, "
378
385
  f"got {scope!r}")
379
- token = ((data.get("finalVerdict") or {}).get("verdictToken")
380
- or "").strip().lower()
381
- if token != "accepted":
382
- raise HandoffError(f"verdict token must be `accepted`, got {token!r}")
386
+ token = verdict_token(data)
387
+ if not release_handoff_allowed(data):
388
+ blocking = blocking_condition_ids(data)
389
+ detail = (
390
+ f"condition(s) {blocking} declare `blocksReleaseHandoff: true`"
391
+ if blocking else f"got {token!r}"
392
+ )
393
+ raise HandoffError(
394
+ "verdict must be `accepted`, or `conditional-accept` whose every "
395
+ f"condition declares `blocksReleaseHandoff: false` — {detail}")
383
396
  rows = consumers.read_consumers(Path(plan_run_root))
384
397
  key = _impl_task_key_for(rows, stage)
385
398
  consumers.append_verified(Path(plan_run_root), impl_task_key=key,
386
- stage=stage, verdict="accepted",
399
+ stage=stage, verdict=token,
387
400
  report_path=report_path)
388
401
  return {"ok": True, "stage": stage, "report_path": report_path}
389
402
 
@@ -0,0 +1,56 @@
1
+ """final-verification 판정이 release-handoff 진입을 허용하는지 판정한다.
2
+
3
+ 검증기(`validators/validate-run.py`)와 핸드오프(`handoff.py`), 그리고 HTML 리포트가
4
+ 모두 같은 답을 내야 하므로 규칙은 여기 한 번만 산다.
5
+
6
+ `accepted` 는 그대로 통과한다. `conditional-accept` 는 모든 조건이 스스로
7
+ `blocksReleaseHandoff: false` 라고 선언했을 때만 통과한다 — 릴리스를 막는다고 적힌
8
+ 조건이 하나라도 있으면 막힌다. 조건 목록이 비어 있는 `conditional-accept` 는
9
+ 그 자체가 계약 위반이므로(조건을 빠짐없이 적어야 한다) 여기서도 막는다.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ from collections.abc import Mapping
14
+ from typing import Any
15
+
16
+ RELEASE_HANDOFF_TARGETS = frozenset({"release-handoff", "release-handoff(stage-group)"})
17
+
18
+
19
+ def verdict_token(data: Mapping[str, Any]) -> str:
20
+ """리포트의 유일한 판정 토큰 자리에서 읽은 값(소문자, 공백 제거)."""
21
+ final_verdict = data.get("finalVerdict")
22
+ if not isinstance(final_verdict, Mapping):
23
+ return ""
24
+ return str(final_verdict.get("verdictToken") or "").strip().lower()
25
+
26
+
27
+ def _conditions(data: Mapping[str, Any]) -> list[Mapping[str, Any]]:
28
+ final_verdict = data.get("finalVerdict")
29
+ if not isinstance(final_verdict, Mapping):
30
+ return []
31
+ rows = final_verdict.get("conditionalAcceptanceConditions")
32
+ if not isinstance(rows, list):
33
+ return []
34
+ return [row for row in rows if isinstance(row, Mapping)]
35
+
36
+
37
+ def blocking_condition_ids(data: Mapping[str, Any]) -> list[str]:
38
+ """릴리스를 막는다고 선언된 조건의 id. 선언이 없거나 참이면 막는 것으로 읽는다."""
39
+ return [
40
+ str(row.get("id") or "<id 없음>")
41
+ for row in _conditions(data)
42
+ if row.get("blocksReleaseHandoff") is not False
43
+ ]
44
+
45
+
46
+ def release_handoff_allowed(data: Mapping[str, Any]) -> bool:
47
+ """이 final-verification 리포트가 release-handoff 로 넘어가도 되는가."""
48
+ token = verdict_token(data)
49
+ if token == "accepted":
50
+ return True
51
+ if token != "conditional-accept":
52
+ return False
53
+ conditions = _conditions(data)
54
+ if not conditions:
55
+ return False
56
+ return not blocking_condition_ids(data)
@@ -111,6 +111,7 @@ TASK_TYPE_REQUIRED_HUMAN_FIELDS = {
111
111
  ),
112
112
  "final-verification": (
113
113
  "finalVerification.validationEvidence",
114
+ "finalVerification.addedSurfaceAudit",
114
115
  "finalVerification.acceptanceBlockers",
115
116
  "finalVerification.residualRisk",
116
117
  "finalVerification.manualUserTest",
@@ -42,6 +42,12 @@ from .report_contract import apply_execution_roles
42
42
  from .dispatch_state import DispatchError, link_agent_dispatch_result
43
43
  from .final_report_paths import final_report_data_path, final_report_markdown_path
44
44
  from .paths import task_dir, task_manifest_file
45
+ from .release_gate import release_handoff_allowed
46
+ from .stage_integrate import IntegrateError
47
+ from .stage_targets import (
48
+ StageTargetError,
49
+ integrate_and_teardown_whole_task,
50
+ )
45
51
  from .session import observe_lead_session
46
52
 
47
53
 
@@ -51,6 +57,7 @@ STEP_TOKEN_USAGE = "token-usage"
51
57
  STEP_RENDER_VIEWS = "render-views"
52
58
  STEP_SPAWN_FOLLOWUPS = "spawn-followups"
53
59
  STEP_VALIDATE_RUN = "validate-run"
60
+ STEP_TEARDOWN_STAGES = "teardown-stages"
54
61
 
55
62
  STEP_ORDER = (
56
63
  STEP_PROJECT_ACTIVITY,
@@ -62,6 +69,9 @@ STEP_ORDER = (
62
69
  STEP_RENDER_VIEWS,
63
70
  STEP_SPAWN_FOLLOWUPS,
64
71
  STEP_VALIDATE_RUN,
72
+ # Last, and only after the run validated: it removes the stage worktrees a
73
+ # blocked verdict would send the user straight back to.
74
+ STEP_TEARDOWN_STAGES,
65
75
  )
66
76
 
67
77
 
@@ -213,6 +223,7 @@ class FinalizeContext:
213
223
  task_key: str
214
224
  task_type: str
215
225
  task_group: str
226
+ task_id: str
216
227
  seq: str
217
228
  final_status_path: Path | None
218
229
 
@@ -241,6 +252,7 @@ class FinalizeContext:
241
252
  task_key=require_string(manifest, "taskKey"),
242
253
  task_type=require_string(manifest, "taskType"),
243
254
  task_group=task_group(manifest),
255
+ task_id=task_id(manifest),
244
256
  seq=report_seq(manifest),
245
257
  final_status_path=resolve_optional_path(
246
258
  project_root, manifest.get("finalStatusPath")
@@ -327,9 +339,49 @@ def build_commands(ctx: FinalizeContext) -> list[tuple[str, list[str]]]:
327
339
  ],
328
340
  ),
329
341
  (STEP_VALIDATE_RUN, _validate_run_command(ctx, markdown_path)),
342
+ (
343
+ STEP_TEARDOWN_STAGES,
344
+ ["<in-process>", "teardown-stages", str(ctx.data_path)],
345
+ ),
330
346
  ]
331
347
 
332
348
 
349
+ def _teardown_stage_worktrees(
350
+ ctx: FinalizeContext,
351
+ command: list[str],
352
+ ) -> subprocess.CompletedProcess:
353
+ """판정이 릴리스로 향할 때만 stage worktree 와 registry 키를 정리한다.
354
+
355
+ whole-task 진입이 통합만 하고 정리를 남겨두므로(`stage_targets`), 정리는 판정이
356
+ 나온 뒤인 여기서 한다. `blocked` 이거나 릴리스를 막는 조건이 남은 판정에서는
357
+ stage 작업물을 그대로 둬서 재작업이 바로 이어지게 한다. 이미 정리된 뒤 재실행돼도
358
+ 같은 결과를 낸다 — 병합은 `already_merged` 로, 없는 worktree 는 건너뛴다.
359
+ """
360
+ def _done(payload: Mapping[str, Any]) -> subprocess.CompletedProcess:
361
+ return subprocess.CompletedProcess(command, 0, json.dumps(payload), "")
362
+
363
+ if ctx.task_type != "final-verification":
364
+ return _done({"skipped": "not a final-verification run"})
365
+ try:
366
+ data = json.loads(Path(ctx.data_path).read_text(encoding="utf-8"))
367
+ except (OSError, json.JSONDecodeError) as exc:
368
+ return subprocess.CompletedProcess(
369
+ command, 1, "", f"cannot read final-report data.json: {exc}")
370
+ if data.get("verificationScope") != "whole-task":
371
+ return _done({"skipped": "single-stage verification owns no teardown"})
372
+ if not release_handoff_allowed(data):
373
+ return _done({"skipped": "verdict does not clear the work for release"})
374
+ try:
375
+ result = integrate_and_teardown_whole_task(
376
+ project_root=ctx.project_root,
377
+ task_group=ctx.task_group,
378
+ task_id=ctx.task_id,
379
+ )
380
+ except (IntegrateError, StageTargetError, OSError) as exc:
381
+ return subprocess.CompletedProcess(command, 1, "", str(exc))
382
+ return _done(result)
383
+
384
+
333
385
  def _validate_run_command(ctx: FinalizeContext, markdown_path: Path) -> list[str]:
334
386
  command = [
335
387
  sys.executable,
@@ -431,6 +483,8 @@ def run_finalize(
431
483
  )
432
484
  else:
433
485
  result = None
486
+ if name == STEP_TEARDOWN_STAGES:
487
+ result = _teardown_stage_worktrees(ctx, command)
434
488
  if name == STEP_VALIDATE_RUN:
435
489
  try:
436
490
  _link_lead_result_for_validation(ctx)
@@ -1,6 +1,7 @@
1
1
  """Human-first final-verification view model."""
2
2
  from __future__ import annotations
3
3
 
4
+ from ...release_gate import release_handoff_allowed
4
5
  from ..common import evidence_index
5
6
  from ..models import HumanReportView, VisualNode
6
7
  from ..visualizations import coverage_figure
@@ -25,13 +26,15 @@ def build_final_verification_view(data: dict) -> HumanReportView:
25
26
  figure = coverage_figure(
26
27
  rows=_coverage_nodes(final), title="Requirement verification coverage"
27
28
  )
29
+ verdict_token = data["finalVerdict"]["verdictToken"]
28
30
  context = {
29
31
  "humanSummary": data["humanSummary"],
30
32
  "verdict": data["verdictCard"],
33
+ "verdictToken": verdict_token,
31
34
  "final": final,
32
35
  "narrative": final["userNarrative"],
33
36
  "coverageFigure": figure,
34
- "releaseAllowed": data["verdictCard"]["verdictToken"] == "accepted",
37
+ "releaseAllowed": release_handoff_allowed(data),
35
38
  "evidenceIndex": evidence_index(data),
36
39
  }
37
40
  return HumanReportView(
@@ -1580,7 +1580,7 @@ def _materialize_release_handoff_input(
1580
1580
  반환: ctx 에 올릴 {"HANDOFF_MODE": ..., "HANDOFF_STAGES": ...}."""
1581
1581
  from .consumers import read_consumers
1582
1582
  from .handoff import (HandoffError, _require_eligible,
1583
- latest_whole_task_fv_accepted)
1583
+ latest_whole_task_fv_release_ready)
1584
1584
  from .paths import task_dir
1585
1585
  from .render import render_template_with_ctx
1586
1586
  from .run_context import _now_task_date
@@ -1622,7 +1622,7 @@ def _materialize_release_handoff_input(
1622
1622
  stages_csv = ",".join(str(n) for n in nums)
1623
1623
  report_rows = _collect_handoff_source_report_rows(rows, nums)
1624
1624
  else:
1625
- report = latest_whole_task_fv_accepted(
1625
+ report = latest_whole_task_fv_release_ready(
1626
1626
  project_root, inp.project_id, inp.task_group, inp.task_id)
1627
1627
  if not report:
1628
1628
  raise PrepareError(
@@ -1659,6 +1659,33 @@ def _materialize_release_handoff_input(
1659
1659
  return {"HANDOFF_MODE": mode, "HANDOFF_STAGES": stages_csv}
1660
1660
 
1661
1661
 
1662
+ QA_COMMAND_EXECUTING_TASK_TYPES = ("implementation", "final-verification")
1663
+
1664
+
1665
+ def validate_project_qa_commands(task_type: str, project_root: Path) -> None:
1666
+ """`qaCommands` 를 실행하는 phase 진입에서 변경성 토큰 선언을 막는다.
1667
+
1668
+ `implementation` 은 verifier 의 QA gate baseline 으로, `final-verification` 은
1669
+ 프로파일이 정의한 Tier 2 재실행 집합으로 같은 선언을 읽는다. 실행 직전에 리드가
1670
+ 스스로 걸러내게 두면 그 자기검사가 유일한 방어선이 되므로 진입에서 막는다.
1671
+ 나머지 task-type 은 이 선언을 읽지 않아 잘못된 값이 있어도 동작에 닿지 않는다.
1672
+ """
1673
+ if task_type not in QA_COMMAND_EXECUTING_TASK_TYPES:
1674
+ return
1675
+ project_json = project_json_path(project_root)
1676
+ if not project_json.is_file():
1677
+ return
1678
+ try:
1679
+ project_meta = json.loads(project_json.read_text())
1680
+ except (OSError, json.JSONDecodeError) as exc:
1681
+ raise PrepareError(
1682
+ f"project.json read failed at {project_json}: {exc}"
1683
+ ) from exc
1684
+ qa_errors = validate_qa_commands(project_meta.get("qaCommands"))
1685
+ if qa_errors:
1686
+ raise PrepareError(_format_qa_errors(qa_errors))
1687
+
1688
+
1662
1689
  def _apply_qa_waiver_if_requested(inp: "PrepareInputs", project_root: Path) -> None:
1663
1690
  """`--qa-waiver` 가 있으면 task-level 매니페스트 entry 의 waiver 를 채운다.
1664
1691
 
@@ -1745,21 +1772,9 @@ def _register_and_check_project(project_root: Path, inp: PrepareInputs) -> None:
1745
1772
  # is preserved by the `: {exc}` suffix and the `raise ... from exc`.
1746
1773
  raise PrepareError(f"project.json upsert failed for {project_root}: {exc}") from exc
1747
1774
 
1748
- # `qaCommands` 는 implementation phase verifier 의 QA gate baseline 으로만
1749
- # 쓰이므로 검증도 implementation 진입 시에만 수행한다. 다른 task-type 에서는
1750
- # 잘못된 선언이 있어도 동작에 영향이 없어 fail-fast 할 이유가 없다.
1775
+ validate_project_qa_commands(inp.task_type, project_root)
1776
+ # waiver 는 stage 단위 Tier 3 면제라 implementation 진입에서만 적용한다.
1751
1777
  if inp.task_type == "implementation":
1752
- project_json = project_json_path(project_root)
1753
- if project_json.is_file():
1754
- try:
1755
- project_meta = json.loads(project_json.read_text())
1756
- except (OSError, json.JSONDecodeError) as exc:
1757
- raise PrepareError(
1758
- f"project.json read failed at {project_json}: {exc}"
1759
- ) from exc
1760
- qa_errors = validate_qa_commands(project_meta.get("qaCommands"))
1761
- if qa_errors:
1762
- raise PrepareError(_format_qa_errors(qa_errors))
1763
1778
  _apply_qa_waiver_if_requested(inp, project_root)
1764
1779
 
1765
1780