okstra 0.176.0 → 0.177.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/task-process/final-verification.md +5 -3
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/workers/report-writer-worker.md +1 -1
- package/runtime/bin/okstra-provider-exec.py +2 -7
- package/runtime/prompts/coding-preflight/overview.md +2 -1
- package/runtime/prompts/lead/convergence.md +11 -6
- package/runtime/prompts/lead/okstra-lead-contract.md +2 -2
- package/runtime/prompts/lead/plan-body-verification.md +4 -2
- package/runtime/prompts/lead/report-writer.md +8 -4
- package/runtime/prompts/profiles/_common-contract.md +1 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
- package/runtime/prompts/profiles/final-verification.md +11 -9
- package/runtime/prompts/profiles/improvement-discovery.md +1 -1
- package/runtime/prompts/profiles/release-handoff.md +7 -6
- package/runtime/prompts/wizard/prompts.ko.json +2 -2
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +2 -2
- package/runtime/python/okstra_ctl/dispatch_core.py +50 -6
- package/runtime/python/okstra_ctl/dispatch_state.py +28 -4
- package/runtime/python/okstra_ctl/execution_mutation_audit.py +49 -3
- package/runtime/python/okstra_ctl/handoff.py +27 -14
- package/runtime/python/okstra_ctl/release_gate.py +56 -0
- package/runtime/python/okstra_ctl/report_contract.py +1 -0
- package/runtime/python/okstra_ctl/report_finalize.py +54 -0
- package/runtime/python/okstra_ctl/report_html/view_models/final_verification.py +4 -1
- package/runtime/python/okstra_ctl/run.py +31 -16
- package/runtime/python/okstra_ctl/stage_targets.py +73 -1
- package/runtime/python/okstra_ctl/wizard.py +19 -10
- package/runtime/python/okstra_ctl/worker_liveness.py +3 -1
- package/runtime/python/okstra_ctl/wrapper_status.py +15 -0
- package/runtime/schemas/final-report-v2.0.schema.json +55 -18
- package/runtime/templates/reports/html/i18n/en.json +4 -0
- package/runtime/templates/reports/html/i18n/ko.json +4 -0
- package/runtime/templates/reports/html/tasks/final-verification.template.html +7 -1
- package/runtime/validators/validate-run.py +98 -12
- package/runtime/validators/validate_analysis_report.py +2 -5
|
@@ -107,7 +107,11 @@ from .worker_prompt_headers import (
|
|
|
107
107
|
resolve_errors_log_path,
|
|
108
108
|
)
|
|
109
109
|
from .worker_artifact_paths import audit_sidecar_rel
|
|
110
|
-
from .wrapper_status import
|
|
110
|
+
from .wrapper_status import (
|
|
111
|
+
log_path_for_prompt,
|
|
112
|
+
read_wrapper_status,
|
|
113
|
+
status_path_for_prompt,
|
|
114
|
+
)
|
|
111
115
|
from .worker_request import verifier_extra_dirs
|
|
112
116
|
from .write_policy import (
|
|
113
117
|
WriteEnforcement,
|
|
@@ -394,11 +398,38 @@ def dispatch_plan(plan: DispatchPlan, *, wait: bool = True) -> int:
|
|
|
394
398
|
if result != 0:
|
|
395
399
|
return result
|
|
396
400
|
return 0
|
|
397
|
-
|
|
401
|
+
round_artifact_paths = _round_artifact_paths(plan)
|
|
402
|
+
handles = [
|
|
403
|
+
_spawn_job(plan, job, 1, batch_artifact_paths=round_artifact_paths)
|
|
404
|
+
for job in plan.jobs
|
|
405
|
+
]
|
|
398
406
|
_record_dispatch_facts(plan.team_state_path, _mode_from_handles(handles))
|
|
399
407
|
return 0
|
|
400
408
|
|
|
401
409
|
|
|
410
|
+
def _round_artifact_paths(plan: DispatchPlan) -> tuple[Path, ...]:
|
|
411
|
+
"""Every artifact this round's workers are entitled to write.
|
|
412
|
+
|
|
413
|
+
These jobs run concurrently into one shared artifact root, so each worker's
|
|
414
|
+
own result, audit sidecar, status sidecar and live log land inside every
|
|
415
|
+
sibling's audit window. Judged against one worker's policy alone, the
|
|
416
|
+
siblings' writes read as unauthorized artifact-root changes and failed a
|
|
417
|
+
whole round of otherwise clean verifiers.
|
|
418
|
+
|
|
419
|
+
They travel as orchestrator paths rather than as a widened policy union
|
|
420
|
+
because the snapshot carries orchestrator paths and the audit excuses them
|
|
421
|
+
without consulting a policy — while `_validate_snapshot_authority` compares
|
|
422
|
+
the recorded policy digests exactly, so a snapshot taken under a union can
|
|
423
|
+
no longer be closed by the per-job policy the awaiting process rebuilds.
|
|
424
|
+
Same effect on the verdict; no coupling between two processes' plans.
|
|
425
|
+
"""
|
|
426
|
+
return tuple(
|
|
427
|
+
path
|
|
428
|
+
for job in plan.jobs
|
|
429
|
+
for path in _worker_artifact_paths(plan, job)
|
|
430
|
+
)
|
|
431
|
+
|
|
432
|
+
|
|
402
433
|
def dispatch_cli_wrapper_plan(plan: DispatchPlan) -> int:
|
|
403
434
|
"""Start one dependency-free CLI batch before collecting any worker."""
|
|
404
435
|
if any(job.backend != BACKEND_CLI_WRAPPER for job in plan.jobs):
|
|
@@ -1263,7 +1294,13 @@ def _jobs_from_file(
|
|
|
1263
1294
|
)
|
|
1264
1295
|
|
|
1265
1296
|
|
|
1266
|
-
def _spawn_job(
|
|
1297
|
+
def _spawn_job(
|
|
1298
|
+
plan: DispatchPlan,
|
|
1299
|
+
job: WorkerJob,
|
|
1300
|
+
attempt: int,
|
|
1301
|
+
*,
|
|
1302
|
+
batch_artifact_paths: Sequence[Path] = (),
|
|
1303
|
+
) -> WorkerHandle:
|
|
1267
1304
|
job = _prepare_job_attempt(plan, job, attempt)
|
|
1268
1305
|
contract = _persisted_write_contract(plan, job)
|
|
1269
1306
|
policies = (contract[0],) if contract else ()
|
|
@@ -1275,7 +1312,12 @@ def _spawn_job(plan: DispatchPlan, job: WorkerJob, attempt: int) -> WorkerHandle
|
|
|
1275
1312
|
_load_json_object(snapshot_path, "mutation audit snapshot")
|
|
1276
1313
|
)
|
|
1277
1314
|
else:
|
|
1278
|
-
snapshot = _mutation_snapshot(
|
|
1315
|
+
snapshot = _mutation_snapshot(
|
|
1316
|
+
plan,
|
|
1317
|
+
policies,
|
|
1318
|
+
(job,),
|
|
1319
|
+
round_artifact_paths=batch_artifact_paths,
|
|
1320
|
+
)
|
|
1279
1321
|
handle = replace(
|
|
1280
1322
|
_start_job(plan, job),
|
|
1281
1323
|
mutation_snapshot=snapshot,
|
|
@@ -1373,11 +1415,13 @@ def _mutation_snapshot(
|
|
|
1373
1415
|
plan: DispatchPlan,
|
|
1374
1416
|
policies: Sequence[WritePolicy],
|
|
1375
1417
|
jobs: Sequence[WorkerJob],
|
|
1418
|
+
round_artifact_paths: Sequence[Path] = (),
|
|
1376
1419
|
) -> MutationSnapshot:
|
|
1377
1420
|
sidecars = tuple(_mutation_snapshot_path(job) for job in jobs)
|
|
1378
1421
|
snapshot = ExecutionMutationAudit().snapshot(
|
|
1379
1422
|
policies,
|
|
1380
1423
|
orchestrator_paths=(
|
|
1424
|
+
*round_artifact_paths,
|
|
1381
1425
|
plan.manifest_path,
|
|
1382
1426
|
plan.team_state_path,
|
|
1383
1427
|
Path(f"{plan.team_state_path}.lock"),
|
|
@@ -1478,7 +1522,7 @@ def _worker_artifact_paths(plan: DispatchPlan, job: WorkerJob) -> tuple[Path, ..
|
|
|
1478
1522
|
*job.completion_paths,
|
|
1479
1523
|
Path(audit_sidecar_rel(str(job.worker_result_path))),
|
|
1480
1524
|
status_path_for_prompt(job.prompt_path),
|
|
1481
|
-
|
|
1525
|
+
log_path_for_prompt(job.prompt_path),
|
|
1482
1526
|
}
|
|
1483
1527
|
error_logs = active_context.get("errorLogs")
|
|
1484
1528
|
if isinstance(error_logs, Mapping):
|
|
@@ -2171,7 +2215,7 @@ def _wrapper_log_tail(job: WorkerJob) -> str | None:
|
|
|
2171
2215
|
some process exited 1. Capped well under the writer's own excerpt limit
|
|
2172
2216
|
because a whole record must stay inside one atomic `PIPE_BUF` append.
|
|
2173
2217
|
"""
|
|
2174
|
-
log_path =
|
|
2218
|
+
log_path = log_path_for_prompt(job.prompt_path)
|
|
2175
2219
|
try:
|
|
2176
2220
|
with log_path.open("rb") as handle:
|
|
2177
2221
|
handle.seek(0, 2)
|
|
@@ -57,7 +57,7 @@ from .worker_prompt_contract import (
|
|
|
57
57
|
from .worker_runner import LIVE, QUIET
|
|
58
58
|
from .worker_request import verifier_extra_dirs
|
|
59
59
|
from .worker_artifact_paths import audit_sidecar_rel
|
|
60
|
-
from .wrapper_status import status_path_for_prompt
|
|
60
|
+
from .wrapper_status import log_path_for_prompt, status_path_for_prompt
|
|
61
61
|
from .write_policy import (
|
|
62
62
|
build_invocation_write_contract,
|
|
63
63
|
planned_paths_from_run_manifest,
|
|
@@ -176,7 +176,7 @@ class WorkerJob:
|
|
|
176
176
|
self.model_execution_value,
|
|
177
177
|
str(self.prompt_path),
|
|
178
178
|
self.worktree_path,
|
|
179
|
-
self.
|
|
179
|
+
self.wrapper_role,
|
|
180
180
|
str(self.idle_timeout_seconds),
|
|
181
181
|
"--presentation",
|
|
182
182
|
self._presentation(),
|
|
@@ -190,6 +190,30 @@ class WorkerJob:
|
|
|
190
190
|
argv += ["--session-id", self.session_id]
|
|
191
191
|
return argv
|
|
192
192
|
|
|
193
|
+
@property
|
|
194
|
+
def wrapper_role(self) -> str:
|
|
195
|
+
"""The canonical role the entrypoint's role positional is read as.
|
|
196
|
+
|
|
197
|
+
`self.role` is the roster label team-state carries (`Antigravity
|
|
198
|
+
worker`) so the report's execution-status row can quote it. The
|
|
199
|
+
entrypoint reads that same position through `normalize_role`, which
|
|
200
|
+
knows only the eleven canonical ids, and compares it against
|
|
201
|
+
`role_for_duty(dutyId)` from the invocation metadata. Handing it the
|
|
202
|
+
label failed that check before the status sidecar was written, so a
|
|
203
|
+
pane worker exited 64 leaving no `.log` and no `.status.json` while
|
|
204
|
+
team-state still read `in-progress`. It also silently denied the
|
|
205
|
+
verifier the toolchain dirs `verifier_extra_dirs` grants by role.
|
|
206
|
+
|
|
207
|
+
A legacy v1 job carries no duty id; the entrypoint skips the metadata
|
|
208
|
+
role check for it, so its existing value is preserved.
|
|
209
|
+
"""
|
|
210
|
+
if not self.duty_id:
|
|
211
|
+
return self.role
|
|
212
|
+
try:
|
|
213
|
+
return role_for_duty(self.duty_id)
|
|
214
|
+
except RoleCatalogError:
|
|
215
|
+
return self.role
|
|
216
|
+
|
|
193
217
|
def _presentation(self) -> str:
|
|
194
218
|
# A pane is a screen a person watches; a cli-wrapper dispatch's stdout is
|
|
195
219
|
# a subagent's context window. Progress belongs in the first and not the
|
|
@@ -790,7 +814,7 @@ def _manifest_worker_write_paths(
|
|
|
790
814
|
result,
|
|
791
815
|
Path(audit_sidecar_rel(str(result))),
|
|
792
816
|
status_path_for_prompt(prompt_path),
|
|
793
|
-
|
|
817
|
+
log_path_for_prompt(prompt_path),
|
|
794
818
|
}
|
|
795
819
|
active_value = authority.get("activeRunContextPath")
|
|
796
820
|
active = (
|
|
@@ -847,7 +871,7 @@ def _prompt_write_paths(
|
|
|
847
871
|
_project_or_absolute(project_root, value) for value in artifact_values
|
|
848
872
|
}
|
|
849
873
|
artifacts.add(status_path_for_prompt(prompt_path))
|
|
850
|
-
artifacts.add(
|
|
874
|
+
artifacts.add(log_path_for_prompt(prompt_path))
|
|
851
875
|
worktree_value = values.get("Worktree")
|
|
852
876
|
worktree = (
|
|
853
877
|
_project_or_absolute(project_root, worktree_value)
|
|
@@ -171,7 +171,10 @@ class ExecutionMutationAudit:
|
|
|
171
171
|
before.scratch_digests, after.scratch_digests
|
|
172
172
|
)
|
|
173
173
|
source_changes = _source_changes(changed, rows, before)
|
|
174
|
-
|
|
174
|
+
# Reported and, through `_retry_allowed`, load-bearing: a stat-cache
|
|
175
|
+
# refresh must not read as "this worker touched Git" and must not
|
|
176
|
+
# withhold the retry a worker that failed for another reason is owed.
|
|
177
|
+
git_changed = _stable_git_projection(before) != _stable_git_projection(after)
|
|
175
178
|
violations = _policy_violations(
|
|
176
179
|
before,
|
|
177
180
|
after,
|
|
@@ -250,6 +253,22 @@ def _maximum_precision(policy: WritePolicy) -> str:
|
|
|
250
253
|
return policy.maximum_boundary_precision
|
|
251
254
|
|
|
252
255
|
|
|
256
|
+
_INSTALLED_DEPENDENCY_DIRS = frozenset({"node_modules"})
|
|
257
|
+
"""Trees a package manager installs, which no worker authored and no policy can
|
|
258
|
+
enumerate.
|
|
259
|
+
|
|
260
|
+
Excluded for the same reason `.git` is: the audit asks whether the worker
|
|
261
|
+
changed *source*, and these hold neither source nor run artifacts. What forced
|
|
262
|
+
the entry is that the tools inside them write to themselves — a `final-verification`
|
|
263
|
+
verifier running the Tier 1 / Tier 2 suites its own profile mandates had Vitest
|
|
264
|
+
persist `node_modules/.vite/vitest/<hash>/results.json`, and that single cache
|
|
265
|
+
file failed both acceptance verifiers of an otherwise clean stage.
|
|
266
|
+
|
|
267
|
+
One ecosystem, because one is what has been observed. A second name belongs here
|
|
268
|
+
when a run produces the same evidence for it, not before.
|
|
269
|
+
"""
|
|
270
|
+
|
|
271
|
+
|
|
253
272
|
def _content_snapshot(root: Path, generated: frozenset[str]) -> dict[str, str]:
|
|
254
273
|
rows: dict[str, str] = {}
|
|
255
274
|
for current, directories, filenames in os.walk(root, followlinks=False):
|
|
@@ -258,7 +277,11 @@ def _content_snapshot(root: Path, generated: frozenset[str]) -> dict[str, str]:
|
|
|
258
277
|
for name in sorted(directories):
|
|
259
278
|
path = current_path / name
|
|
260
279
|
relative_path = path.relative_to(root)
|
|
261
|
-
if
|
|
280
|
+
if (
|
|
281
|
+
name == ".git"
|
|
282
|
+
or name in _INSTALLED_DEPENDENCY_DIRS
|
|
283
|
+
or _excluded(relative_path, generated)
|
|
284
|
+
):
|
|
262
285
|
continue
|
|
263
286
|
if path.is_symlink():
|
|
264
287
|
rows[relative_path.as_posix()] = _path_digest(path)
|
|
@@ -473,7 +496,7 @@ def _policy_violations(
|
|
|
473
496
|
) -> list[str]:
|
|
474
497
|
if all(policy.source_mode == "source-readonly" for policy in policies):
|
|
475
498
|
failures = ["readonly source changed"] if source_changes else []
|
|
476
|
-
if before
|
|
499
|
+
if _stable_git_projection(before) != _stable_git_projection(after):
|
|
477
500
|
failures.append("gitPolicy disabled but Git projection changed")
|
|
478
501
|
return failures
|
|
479
502
|
policy = next(row for row in policies if row.source_mode == "project-mutation")
|
|
@@ -482,6 +505,29 @@ def _policy_violations(
|
|
|
482
505
|
return failures
|
|
483
506
|
|
|
484
507
|
|
|
508
|
+
def _stable_git_projection(snapshot: MutationSnapshot) -> dict[str, Any]:
|
|
509
|
+
"""The projection minus the one field a read-only reader moves on its own.
|
|
510
|
+
|
|
511
|
+
`indexDigest` is Git's stat cache, and Git rewrites it whenever a plain
|
|
512
|
+
read refreshes a stale entry — `git status --short`, which the
|
|
513
|
+
`final-verification` profile requires of every verifier, is enough. Nothing
|
|
514
|
+
about the worktree's content changed when it moves, so comparing it made a
|
|
515
|
+
read-only worker fail for doing what its own phase told it to do (observed:
|
|
516
|
+
both stage-1 acceptance verifiers, where `indexDigest` was the only key that
|
|
517
|
+
differed and `stagedPaths` was empty on both sides).
|
|
518
|
+
|
|
519
|
+
Every field that does witness a mutation stays compared: `head` and
|
|
520
|
+
`branchCommit` for a moved ref, `stagedPaths` for content added to the
|
|
521
|
+
index, `worktreeRegistration` for a re-registered worktree, `reflogDigest`
|
|
522
|
+
for a ref rewrite, and both Git directories for a redirected repository.
|
|
523
|
+
"""
|
|
524
|
+
return {
|
|
525
|
+
key: value
|
|
526
|
+
for key, value in snapshot.git_projection.items()
|
|
527
|
+
if key != "indexDigest"
|
|
528
|
+
}
|
|
529
|
+
|
|
530
|
+
|
|
485
531
|
def _source_policy_failures(
|
|
486
532
|
policy: WritePolicy,
|
|
487
533
|
changed: set[str],
|
|
@@ -15,6 +15,11 @@ from typing import Any, Callable, Dict, List, Optional, Tuple
|
|
|
15
15
|
from . import consumers, stage_targets, worktree_registry
|
|
16
16
|
from .final_report_paths import final_report_markdown_path
|
|
17
17
|
from .paths import RunRef
|
|
18
|
+
from .release_gate import (
|
|
19
|
+
blocking_condition_ids,
|
|
20
|
+
release_handoff_allowed,
|
|
21
|
+
verdict_token,
|
|
22
|
+
)
|
|
18
23
|
from .stage_map import StageMapError, parse_stage_map_file, stage_map_records
|
|
19
24
|
from .worktree import (compute_branch_name, compute_worktree_path,
|
|
20
25
|
main_worktree_path, is_dirty_excluding_okstra,
|
|
@@ -52,12 +57,14 @@ def compute_eligibility(stage_map: List[Dict[str, Any]],
|
|
|
52
57
|
).handoff_eligibility()
|
|
53
58
|
|
|
54
59
|
|
|
55
|
-
def
|
|
56
|
-
|
|
57
|
-
"""
|
|
60
|
+
def latest_whole_task_fv_release_ready(project_root, project_id: str,
|
|
61
|
+
task_group: str, task_id: str) -> str:
|
|
62
|
+
"""release-handoff 로 넘어가도 되는 whole-task 검증 보고서 경로. 없으면 ''.
|
|
58
63
|
|
|
59
64
|
판정의 SSOT 는 final-report 의 data.json 이다 (record_verified 와 동일
|
|
60
|
-
필드: header.taskType / verificationScope / finalVerdict
|
|
65
|
+
필드: header.taskType / verificationScope / finalVerdict). 통과 조건은
|
|
66
|
+
`release_gate.release_handoff_allowed` 한 곳이 소유한다 — `accepted`,
|
|
67
|
+
또는 모든 조건이 릴리스를 막지 않는다고 선언한 `conditional-accept`.
|
|
61
68
|
whole-task run 산출물만 평면 reports/ 에 남으므로 stage-* 는 걸리지 않는다."""
|
|
62
69
|
from okstra_project.state import find_task_root
|
|
63
70
|
root = find_task_root(Path(project_root),
|
|
@@ -71,11 +78,9 @@ def latest_whole_task_fv_accepted(project_root, project_id: str,
|
|
|
71
78
|
data = json.loads(dj.read_text(encoding="utf-8"))
|
|
72
79
|
except (OSError, json.JSONDecodeError):
|
|
73
80
|
continue
|
|
74
|
-
token = ((data.get("finalVerdict") or {}).get("verdictToken")
|
|
75
|
-
or "").strip().lower()
|
|
76
81
|
if ((data.get("header") or {}).get("taskType") == "final-verification"
|
|
77
82
|
and data.get("verificationScope") == "whole-task"
|
|
78
|
-
and
|
|
83
|
+
and release_handoff_allowed(data)):
|
|
79
84
|
md = final_report_markdown_path(dj)
|
|
80
85
|
return str(md if md.is_file() else dj)
|
|
81
86
|
return ""
|
|
@@ -363,8 +368,10 @@ def _impl_task_key_for_any(rows: List[Dict[str, Any]], stages: List[int]) -> str
|
|
|
363
368
|
|
|
364
369
|
def record_verified(*, plan_run_root, stage: int, report_path: str,
|
|
365
370
|
data_json) -> Dict[str, Any]:
|
|
366
|
-
"""단독-stage
|
|
367
|
-
lead 가 임의 보고서를 verified 로 올리는 것을
|
|
371
|
+
"""릴리스로 넘어갈 수 있는 단독-stage 판정만 기록. data.json 의
|
|
372
|
+
taskType/scope/verdict 를 검증해 lead 가 임의 보고서를 verified 로 올리는 것을
|
|
373
|
+
막는다. 통과 조건은 whole-task 와 같고(`release_gate`), 기록되는 verdict 는
|
|
374
|
+
보고서가 실제로 실은 토큰이다."""
|
|
368
375
|
try:
|
|
369
376
|
data = json.loads(Path(data_json).read_text(encoding="utf-8"))
|
|
370
377
|
except (OSError, json.JSONDecodeError) as exc:
|
|
@@ -376,14 +383,20 @@ def record_verified(*, plan_run_root, stage: int, report_path: str,
|
|
|
376
383
|
raise HandoffError(
|
|
377
384
|
f"record-verified requires verificationScope single-stage, "
|
|
378
385
|
f"got {scope!r}")
|
|
379
|
-
token = (
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
386
|
+
token = verdict_token(data)
|
|
387
|
+
if not release_handoff_allowed(data):
|
|
388
|
+
blocking = blocking_condition_ids(data)
|
|
389
|
+
detail = (
|
|
390
|
+
f"condition(s) {blocking} declare `blocksReleaseHandoff: true`"
|
|
391
|
+
if blocking else f"got {token!r}"
|
|
392
|
+
)
|
|
393
|
+
raise HandoffError(
|
|
394
|
+
"verdict must be `accepted`, or `conditional-accept` whose every "
|
|
395
|
+
f"condition declares `blocksReleaseHandoff: false` — {detail}")
|
|
383
396
|
rows = consumers.read_consumers(Path(plan_run_root))
|
|
384
397
|
key = _impl_task_key_for(rows, stage)
|
|
385
398
|
consumers.append_verified(Path(plan_run_root), impl_task_key=key,
|
|
386
|
-
stage=stage, verdict=
|
|
399
|
+
stage=stage, verdict=token,
|
|
387
400
|
report_path=report_path)
|
|
388
401
|
return {"ok": True, "stage": stage, "report_path": report_path}
|
|
389
402
|
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""final-verification 판정이 release-handoff 진입을 허용하는지 판정한다.
|
|
2
|
+
|
|
3
|
+
검증기(`validators/validate-run.py`)와 핸드오프(`handoff.py`), 그리고 HTML 리포트가
|
|
4
|
+
모두 같은 답을 내야 하므로 규칙은 여기 한 번만 산다.
|
|
5
|
+
|
|
6
|
+
`accepted` 는 그대로 통과한다. `conditional-accept` 는 모든 조건이 스스로
|
|
7
|
+
`blocksReleaseHandoff: false` 라고 선언했을 때만 통과한다 — 릴리스를 막는다고 적힌
|
|
8
|
+
조건이 하나라도 있으면 막힌다. 조건 목록이 비어 있는 `conditional-accept` 는
|
|
9
|
+
그 자체가 계약 위반이므로(조건을 빠짐없이 적어야 한다) 여기서도 막는다.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from collections.abc import Mapping
|
|
14
|
+
from typing import Any
|
|
15
|
+
|
|
16
|
+
RELEASE_HANDOFF_TARGETS = frozenset({"release-handoff", "release-handoff(stage-group)"})
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def verdict_token(data: Mapping[str, Any]) -> str:
|
|
20
|
+
"""리포트의 유일한 판정 토큰 자리에서 읽은 값(소문자, 공백 제거)."""
|
|
21
|
+
final_verdict = data.get("finalVerdict")
|
|
22
|
+
if not isinstance(final_verdict, Mapping):
|
|
23
|
+
return ""
|
|
24
|
+
return str(final_verdict.get("verdictToken") or "").strip().lower()
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _conditions(data: Mapping[str, Any]) -> list[Mapping[str, Any]]:
|
|
28
|
+
final_verdict = data.get("finalVerdict")
|
|
29
|
+
if not isinstance(final_verdict, Mapping):
|
|
30
|
+
return []
|
|
31
|
+
rows = final_verdict.get("conditionalAcceptanceConditions")
|
|
32
|
+
if not isinstance(rows, list):
|
|
33
|
+
return []
|
|
34
|
+
return [row for row in rows if isinstance(row, Mapping)]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def blocking_condition_ids(data: Mapping[str, Any]) -> list[str]:
|
|
38
|
+
"""릴리스를 막는다고 선언된 조건의 id. 선언이 없거나 참이면 막는 것으로 읽는다."""
|
|
39
|
+
return [
|
|
40
|
+
str(row.get("id") or "<id 없음>")
|
|
41
|
+
for row in _conditions(data)
|
|
42
|
+
if row.get("blocksReleaseHandoff") is not False
|
|
43
|
+
]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def release_handoff_allowed(data: Mapping[str, Any]) -> bool:
|
|
47
|
+
"""이 final-verification 리포트가 release-handoff 로 넘어가도 되는가."""
|
|
48
|
+
token = verdict_token(data)
|
|
49
|
+
if token == "accepted":
|
|
50
|
+
return True
|
|
51
|
+
if token != "conditional-accept":
|
|
52
|
+
return False
|
|
53
|
+
conditions = _conditions(data)
|
|
54
|
+
if not conditions:
|
|
55
|
+
return False
|
|
56
|
+
return not blocking_condition_ids(data)
|
|
@@ -111,6 +111,7 @@ TASK_TYPE_REQUIRED_HUMAN_FIELDS = {
|
|
|
111
111
|
),
|
|
112
112
|
"final-verification": (
|
|
113
113
|
"finalVerification.validationEvidence",
|
|
114
|
+
"finalVerification.addedSurfaceAudit",
|
|
114
115
|
"finalVerification.acceptanceBlockers",
|
|
115
116
|
"finalVerification.residualRisk",
|
|
116
117
|
"finalVerification.manualUserTest",
|
|
@@ -42,6 +42,12 @@ from .report_contract import apply_execution_roles
|
|
|
42
42
|
from .dispatch_state import DispatchError, link_agent_dispatch_result
|
|
43
43
|
from .final_report_paths import final_report_data_path, final_report_markdown_path
|
|
44
44
|
from .paths import task_dir, task_manifest_file
|
|
45
|
+
from .release_gate import release_handoff_allowed
|
|
46
|
+
from .stage_integrate import IntegrateError
|
|
47
|
+
from .stage_targets import (
|
|
48
|
+
StageTargetError,
|
|
49
|
+
integrate_and_teardown_whole_task,
|
|
50
|
+
)
|
|
45
51
|
from .session import observe_lead_session
|
|
46
52
|
|
|
47
53
|
|
|
@@ -51,6 +57,7 @@ STEP_TOKEN_USAGE = "token-usage"
|
|
|
51
57
|
STEP_RENDER_VIEWS = "render-views"
|
|
52
58
|
STEP_SPAWN_FOLLOWUPS = "spawn-followups"
|
|
53
59
|
STEP_VALIDATE_RUN = "validate-run"
|
|
60
|
+
STEP_TEARDOWN_STAGES = "teardown-stages"
|
|
54
61
|
|
|
55
62
|
STEP_ORDER = (
|
|
56
63
|
STEP_PROJECT_ACTIVITY,
|
|
@@ -62,6 +69,9 @@ STEP_ORDER = (
|
|
|
62
69
|
STEP_RENDER_VIEWS,
|
|
63
70
|
STEP_SPAWN_FOLLOWUPS,
|
|
64
71
|
STEP_VALIDATE_RUN,
|
|
72
|
+
# Last, and only after the run validated: it removes the stage worktrees a
|
|
73
|
+
# blocked verdict would send the user straight back to.
|
|
74
|
+
STEP_TEARDOWN_STAGES,
|
|
65
75
|
)
|
|
66
76
|
|
|
67
77
|
|
|
@@ -213,6 +223,7 @@ class FinalizeContext:
|
|
|
213
223
|
task_key: str
|
|
214
224
|
task_type: str
|
|
215
225
|
task_group: str
|
|
226
|
+
task_id: str
|
|
216
227
|
seq: str
|
|
217
228
|
final_status_path: Path | None
|
|
218
229
|
|
|
@@ -241,6 +252,7 @@ class FinalizeContext:
|
|
|
241
252
|
task_key=require_string(manifest, "taskKey"),
|
|
242
253
|
task_type=require_string(manifest, "taskType"),
|
|
243
254
|
task_group=task_group(manifest),
|
|
255
|
+
task_id=task_id(manifest),
|
|
244
256
|
seq=report_seq(manifest),
|
|
245
257
|
final_status_path=resolve_optional_path(
|
|
246
258
|
project_root, manifest.get("finalStatusPath")
|
|
@@ -327,9 +339,49 @@ def build_commands(ctx: FinalizeContext) -> list[tuple[str, list[str]]]:
|
|
|
327
339
|
],
|
|
328
340
|
),
|
|
329
341
|
(STEP_VALIDATE_RUN, _validate_run_command(ctx, markdown_path)),
|
|
342
|
+
(
|
|
343
|
+
STEP_TEARDOWN_STAGES,
|
|
344
|
+
["<in-process>", "teardown-stages", str(ctx.data_path)],
|
|
345
|
+
),
|
|
330
346
|
]
|
|
331
347
|
|
|
332
348
|
|
|
349
|
+
def _teardown_stage_worktrees(
|
|
350
|
+
ctx: FinalizeContext,
|
|
351
|
+
command: list[str],
|
|
352
|
+
) -> subprocess.CompletedProcess:
|
|
353
|
+
"""판정이 릴리스로 향할 때만 stage worktree 와 registry 키를 정리한다.
|
|
354
|
+
|
|
355
|
+
whole-task 진입이 통합만 하고 정리를 남겨두므로(`stage_targets`), 정리는 판정이
|
|
356
|
+
나온 뒤인 여기서 한다. `blocked` 이거나 릴리스를 막는 조건이 남은 판정에서는
|
|
357
|
+
stage 작업물을 그대로 둬서 재작업이 바로 이어지게 한다. 이미 정리된 뒤 재실행돼도
|
|
358
|
+
같은 결과를 낸다 — 병합은 `already_merged` 로, 없는 worktree 는 건너뛴다.
|
|
359
|
+
"""
|
|
360
|
+
def _done(payload: Mapping[str, Any]) -> subprocess.CompletedProcess:
|
|
361
|
+
return subprocess.CompletedProcess(command, 0, json.dumps(payload), "")
|
|
362
|
+
|
|
363
|
+
if ctx.task_type != "final-verification":
|
|
364
|
+
return _done({"skipped": "not a final-verification run"})
|
|
365
|
+
try:
|
|
366
|
+
data = json.loads(Path(ctx.data_path).read_text(encoding="utf-8"))
|
|
367
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
368
|
+
return subprocess.CompletedProcess(
|
|
369
|
+
command, 1, "", f"cannot read final-report data.json: {exc}")
|
|
370
|
+
if data.get("verificationScope") != "whole-task":
|
|
371
|
+
return _done({"skipped": "single-stage verification owns no teardown"})
|
|
372
|
+
if not release_handoff_allowed(data):
|
|
373
|
+
return _done({"skipped": "verdict does not clear the work for release"})
|
|
374
|
+
try:
|
|
375
|
+
result = integrate_and_teardown_whole_task(
|
|
376
|
+
project_root=ctx.project_root,
|
|
377
|
+
task_group=ctx.task_group,
|
|
378
|
+
task_id=ctx.task_id,
|
|
379
|
+
)
|
|
380
|
+
except (IntegrateError, StageTargetError, OSError) as exc:
|
|
381
|
+
return subprocess.CompletedProcess(command, 1, "", str(exc))
|
|
382
|
+
return _done(result)
|
|
383
|
+
|
|
384
|
+
|
|
333
385
|
def _validate_run_command(ctx: FinalizeContext, markdown_path: Path) -> list[str]:
|
|
334
386
|
command = [
|
|
335
387
|
sys.executable,
|
|
@@ -431,6 +483,8 @@ def run_finalize(
|
|
|
431
483
|
)
|
|
432
484
|
else:
|
|
433
485
|
result = None
|
|
486
|
+
if name == STEP_TEARDOWN_STAGES:
|
|
487
|
+
result = _teardown_stage_worktrees(ctx, command)
|
|
434
488
|
if name == STEP_VALIDATE_RUN:
|
|
435
489
|
try:
|
|
436
490
|
_link_lead_result_for_validation(ctx)
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
"""Human-first final-verification view model."""
|
|
2
2
|
from __future__ import annotations
|
|
3
3
|
|
|
4
|
+
from ...release_gate import release_handoff_allowed
|
|
4
5
|
from ..common import evidence_index
|
|
5
6
|
from ..models import HumanReportView, VisualNode
|
|
6
7
|
from ..visualizations import coverage_figure
|
|
@@ -25,13 +26,15 @@ def build_final_verification_view(data: dict) -> HumanReportView:
|
|
|
25
26
|
figure = coverage_figure(
|
|
26
27
|
rows=_coverage_nodes(final), title="Requirement verification coverage"
|
|
27
28
|
)
|
|
29
|
+
verdict_token = data["finalVerdict"]["verdictToken"]
|
|
28
30
|
context = {
|
|
29
31
|
"humanSummary": data["humanSummary"],
|
|
30
32
|
"verdict": data["verdictCard"],
|
|
33
|
+
"verdictToken": verdict_token,
|
|
31
34
|
"final": final,
|
|
32
35
|
"narrative": final["userNarrative"],
|
|
33
36
|
"coverageFigure": figure,
|
|
34
|
-
"releaseAllowed": data
|
|
37
|
+
"releaseAllowed": release_handoff_allowed(data),
|
|
35
38
|
"evidenceIndex": evidence_index(data),
|
|
36
39
|
}
|
|
37
40
|
return HumanReportView(
|
|
@@ -1580,7 +1580,7 @@ def _materialize_release_handoff_input(
|
|
|
1580
1580
|
반환: ctx 에 올릴 {"HANDOFF_MODE": ..., "HANDOFF_STAGES": ...}."""
|
|
1581
1581
|
from .consumers import read_consumers
|
|
1582
1582
|
from .handoff import (HandoffError, _require_eligible,
|
|
1583
|
-
|
|
1583
|
+
latest_whole_task_fv_release_ready)
|
|
1584
1584
|
from .paths import task_dir
|
|
1585
1585
|
from .render import render_template_with_ctx
|
|
1586
1586
|
from .run_context import _now_task_date
|
|
@@ -1622,7 +1622,7 @@ def _materialize_release_handoff_input(
|
|
|
1622
1622
|
stages_csv = ",".join(str(n) for n in nums)
|
|
1623
1623
|
report_rows = _collect_handoff_source_report_rows(rows, nums)
|
|
1624
1624
|
else:
|
|
1625
|
-
report =
|
|
1625
|
+
report = latest_whole_task_fv_release_ready(
|
|
1626
1626
|
project_root, inp.project_id, inp.task_group, inp.task_id)
|
|
1627
1627
|
if not report:
|
|
1628
1628
|
raise PrepareError(
|
|
@@ -1659,6 +1659,33 @@ def _materialize_release_handoff_input(
|
|
|
1659
1659
|
return {"HANDOFF_MODE": mode, "HANDOFF_STAGES": stages_csv}
|
|
1660
1660
|
|
|
1661
1661
|
|
|
1662
|
+
QA_COMMAND_EXECUTING_TASK_TYPES = ("implementation", "final-verification")
|
|
1663
|
+
|
|
1664
|
+
|
|
1665
|
+
def validate_project_qa_commands(task_type: str, project_root: Path) -> None:
|
|
1666
|
+
"""`qaCommands` 를 실행하는 phase 진입에서 변경성 토큰 선언을 막는다.
|
|
1667
|
+
|
|
1668
|
+
`implementation` 은 verifier 의 QA gate baseline 으로, `final-verification` 은
|
|
1669
|
+
프로파일이 정의한 Tier 2 재실행 집합으로 같은 선언을 읽는다. 실행 직전에 리드가
|
|
1670
|
+
스스로 걸러내게 두면 그 자기검사가 유일한 방어선이 되므로 진입에서 막는다.
|
|
1671
|
+
나머지 task-type 은 이 선언을 읽지 않아 잘못된 값이 있어도 동작에 닿지 않는다.
|
|
1672
|
+
"""
|
|
1673
|
+
if task_type not in QA_COMMAND_EXECUTING_TASK_TYPES:
|
|
1674
|
+
return
|
|
1675
|
+
project_json = project_json_path(project_root)
|
|
1676
|
+
if not project_json.is_file():
|
|
1677
|
+
return
|
|
1678
|
+
try:
|
|
1679
|
+
project_meta = json.loads(project_json.read_text())
|
|
1680
|
+
except (OSError, json.JSONDecodeError) as exc:
|
|
1681
|
+
raise PrepareError(
|
|
1682
|
+
f"project.json read failed at {project_json}: {exc}"
|
|
1683
|
+
) from exc
|
|
1684
|
+
qa_errors = validate_qa_commands(project_meta.get("qaCommands"))
|
|
1685
|
+
if qa_errors:
|
|
1686
|
+
raise PrepareError(_format_qa_errors(qa_errors))
|
|
1687
|
+
|
|
1688
|
+
|
|
1662
1689
|
def _apply_qa_waiver_if_requested(inp: "PrepareInputs", project_root: Path) -> None:
|
|
1663
1690
|
"""`--qa-waiver` 가 있으면 task-level 매니페스트 entry 의 waiver 를 채운다.
|
|
1664
1691
|
|
|
@@ -1745,21 +1772,9 @@ def _register_and_check_project(project_root: Path, inp: PrepareInputs) -> None:
|
|
|
1745
1772
|
# is preserved by the `: {exc}` suffix and the `raise ... from exc`.
|
|
1746
1773
|
raise PrepareError(f"project.json upsert failed for {project_root}: {exc}") from exc
|
|
1747
1774
|
|
|
1748
|
-
|
|
1749
|
-
#
|
|
1750
|
-
# 잘못된 선언이 있어도 동작에 영향이 없어 fail-fast 할 이유가 없다.
|
|
1775
|
+
validate_project_qa_commands(inp.task_type, project_root)
|
|
1776
|
+
# waiver 는 stage 단위 Tier 3 면제라 implementation 진입에서만 적용한다.
|
|
1751
1777
|
if inp.task_type == "implementation":
|
|
1752
|
-
project_json = project_json_path(project_root)
|
|
1753
|
-
if project_json.is_file():
|
|
1754
|
-
try:
|
|
1755
|
-
project_meta = json.loads(project_json.read_text())
|
|
1756
|
-
except (OSError, json.JSONDecodeError) as exc:
|
|
1757
|
-
raise PrepareError(
|
|
1758
|
-
f"project.json read failed at {project_json}: {exc}"
|
|
1759
|
-
) from exc
|
|
1760
|
-
qa_errors = validate_qa_commands(project_meta.get("qaCommands"))
|
|
1761
|
-
if qa_errors:
|
|
1762
|
-
raise PrepareError(_format_qa_errors(qa_errors))
|
|
1763
1778
|
_apply_qa_waiver_if_requested(inp, project_root)
|
|
1764
1779
|
|
|
1765
1780
|
|