okstra 0.176.1 → 0.177.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/commands/execute/team.mjs +14 -4
- package/dist/commands/execute/team.mjs.map +1 -1
- package/dist/commands/lifecycle/install.mjs +0 -1
- package/dist/commands/lifecycle/install.mjs.map +1 -1
- package/docs/architecture.md +3 -3
- package/docs/cli.md +1 -1
- package/docs/project-structure-overview.md +2 -2
- package/docs/task-process/final-verification.md +5 -3
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/workers/report-writer-worker.md +2 -2
- package/runtime/bin/okstra-compact-reminder.sh +2 -2
- package/runtime/bin/okstra-provider-exec.py +2 -7
- package/runtime/bin/okstra-render-report-views.py +13 -10
- package/runtime/prompts/coding-preflight/overview.md +2 -1
- package/runtime/prompts/lead/convergence.md +11 -6
- package/runtime/prompts/lead/okstra-lead-contract.md +3 -3
- package/runtime/prompts/lead/plan-body-verification.md +4 -2
- package/runtime/prompts/lead/report-writer.md +8 -4
- package/runtime/prompts/profiles/_common-contract.md +2 -2
- package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
- package/runtime/prompts/profiles/final-verification.md +11 -9
- package/runtime/prompts/profiles/improvement-discovery.md +1 -1
- package/runtime/prompts/profiles/release-handoff.md +7 -6
- package/runtime/prompts/wizard/prompts.ko.json +2 -2
- package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +5 -6
- package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +23 -1
- package/runtime/python/okstra_ctl/agent_invocation.py +17 -0
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +15 -2
- package/runtime/python/okstra_ctl/dispatch_core.py +140 -17
- package/runtime/python/okstra_ctl/dispatch_state.py +60 -3
- package/runtime/python/okstra_ctl/execution_mutation_audit.py +49 -3
- package/runtime/python/okstra_ctl/handoff.py +27 -14
- package/runtime/python/okstra_ctl/model_cli.py +11 -2
- package/runtime/python/okstra_ctl/model_discovery.py +12 -0
- package/runtime/python/okstra_ctl/pane_reclaim.py +49 -43
- package/runtime/python/okstra_ctl/release_gate.py +56 -0
- package/runtime/python/okstra_ctl/render.py +53 -0
- package/runtime/python/okstra_ctl/report_contract.py +1 -0
- package/runtime/python/okstra_ctl/report_finalize.py +54 -0
- package/runtime/python/okstra_ctl/report_html/render.py +7 -4
- package/runtime/python/okstra_ctl/report_html/view_models/final_verification.py +4 -1
- package/runtime/python/okstra_ctl/run.py +119 -18
- package/runtime/python/okstra_ctl/stage_targets.py +73 -1
- package/runtime/python/okstra_ctl/team.py +84 -14
- package/runtime/python/okstra_ctl/tmux.py +2 -3
- package/runtime/python/okstra_ctl/wizard.py +19 -10
- package/runtime/python/okstra_ctl/worker_liveness.py +3 -1
- package/runtime/python/okstra_ctl/worker_runner.py +2 -2
- package/runtime/python/okstra_ctl/wrapper_status.py +15 -0
- package/runtime/python/okstra_ctl/write_policy.py +9 -1
- package/runtime/schemas/final-report-v2.0.schema.json +58 -19
- package/runtime/skills/okstra-run/SKILL.md +3 -3
- package/runtime/templates/reports/html/base.template.html +1 -2
- package/runtime/templates/reports/html/i18n/en.json +4 -0
- package/runtime/templates/reports/html/i18n/ko.json +4 -0
- package/runtime/templates/reports/html/tasks/final-verification.template.html +7 -1
- package/runtime/validators/validate-report-views.py +30 -17
- package/runtime/validators/validate-run.py +221 -90
- package/runtime/validators/validate_analysis_report.py +2 -5
- package/runtime/validators/validate_session_conformance.py +1 -1
- package/runtime/bin/okstra-trace-cleanup.sh +0 -185
|
@@ -65,6 +65,7 @@ from .domain.worker_runtime import (
|
|
|
65
65
|
from .ports.worker_runtime import WorkerRuntimePort
|
|
66
66
|
from .execution_identity import Attempt, Invocation, RoleExecution, model_spec_digest
|
|
67
67
|
from .execution_manifest import (
|
|
68
|
+
ExecutionManifestError,
|
|
68
69
|
finish_attempt_mutation,
|
|
69
70
|
read_execution_manifest,
|
|
70
71
|
record_invocation_attempt,
|
|
@@ -107,7 +108,11 @@ from .worker_prompt_headers import (
|
|
|
107
108
|
resolve_errors_log_path,
|
|
108
109
|
)
|
|
109
110
|
from .worker_artifact_paths import audit_sidecar_rel
|
|
110
|
-
from .wrapper_status import
|
|
111
|
+
from .wrapper_status import (
|
|
112
|
+
log_path_for_prompt,
|
|
113
|
+
read_wrapper_status,
|
|
114
|
+
status_path_for_prompt,
|
|
115
|
+
)
|
|
111
116
|
from .worker_request import verifier_extra_dirs
|
|
112
117
|
from .write_policy import (
|
|
113
118
|
WriteEnforcement,
|
|
@@ -204,11 +209,23 @@ def verify_served_model(
|
|
|
204
209
|
return attestation
|
|
205
210
|
if role_execution.model_ref is None or not attestation.normalized_model_ref:
|
|
206
211
|
raise DispatchError("served model differs from selected model")
|
|
212
|
+
# An unregistered ref and a genuine substitution are different failures.
|
|
213
|
+
# Folding both into "differs from selected" hid which one happened, and a
|
|
214
|
+
# catalog gap reads as a provider swapping the model out from under us.
|
|
207
215
|
try:
|
|
208
216
|
selected = pool.resolve(role_execution.model_ref)
|
|
217
|
+
except ValueError as exc:
|
|
218
|
+
raise DispatchError(
|
|
219
|
+
f"selected model is not in the catalog: {role_execution.model_ref}"
|
|
220
|
+
) from exc
|
|
221
|
+
try:
|
|
209
222
|
observed = pool.resolve(attestation.normalized_model_ref)
|
|
210
223
|
except ValueError as exc:
|
|
211
|
-
raise DispatchError(
|
|
224
|
+
raise DispatchError(
|
|
225
|
+
"served model is not in the catalog: "
|
|
226
|
+
f"{attestation.normalized_model_ref} "
|
|
227
|
+
f"(provider reported {attestation.observed_model!r})"
|
|
228
|
+
) from exc
|
|
212
229
|
expected_level = "channel" if observed.version_kind == "channel" else "exact"
|
|
213
230
|
if attestation.level != expected_level:
|
|
214
231
|
raise DispatchError("served model attestation level is inconsistent")
|
|
@@ -394,11 +411,38 @@ def dispatch_plan(plan: DispatchPlan, *, wait: bool = True) -> int:
|
|
|
394
411
|
if result != 0:
|
|
395
412
|
return result
|
|
396
413
|
return 0
|
|
397
|
-
|
|
414
|
+
round_artifact_paths = _round_artifact_paths(plan)
|
|
415
|
+
handles = [
|
|
416
|
+
_spawn_job(plan, job, 1, batch_artifact_paths=round_artifact_paths)
|
|
417
|
+
for job in plan.jobs
|
|
418
|
+
]
|
|
398
419
|
_record_dispatch_facts(plan.team_state_path, _mode_from_handles(handles))
|
|
399
420
|
return 0
|
|
400
421
|
|
|
401
422
|
|
|
423
|
+
def _round_artifact_paths(plan: DispatchPlan) -> tuple[Path, ...]:
|
|
424
|
+
"""Every artifact this round's workers are entitled to write.
|
|
425
|
+
|
|
426
|
+
These jobs run concurrently into one shared artifact root, so each worker's
|
|
427
|
+
own result, audit sidecar, status sidecar and live log land inside every
|
|
428
|
+
sibling's audit window. Judged against one worker's policy alone, the
|
|
429
|
+
siblings' writes read as unauthorized artifact-root changes and failed a
|
|
430
|
+
whole round of otherwise clean verifiers.
|
|
431
|
+
|
|
432
|
+
They travel as orchestrator paths rather than as a widened policy union
|
|
433
|
+
because the snapshot carries orchestrator paths and the audit excuses them
|
|
434
|
+
without consulting a policy — while `_validate_snapshot_authority` compares
|
|
435
|
+
the recorded policy digests exactly, so a snapshot taken under a union can
|
|
436
|
+
no longer be closed by the per-job policy the awaiting process rebuilds.
|
|
437
|
+
Same effect on the verdict; no coupling between two processes' plans.
|
|
438
|
+
"""
|
|
439
|
+
return tuple(
|
|
440
|
+
path
|
|
441
|
+
for job in plan.jobs
|
|
442
|
+
for path in _worker_artifact_paths(plan, job)
|
|
443
|
+
)
|
|
444
|
+
|
|
445
|
+
|
|
402
446
|
def dispatch_cli_wrapper_plan(plan: DispatchPlan) -> int:
|
|
403
447
|
"""Start one dependency-free CLI batch before collecting any worker."""
|
|
404
448
|
if any(job.backend != BACKEND_CLI_WRAPPER for job in plan.jobs):
|
|
@@ -1263,19 +1307,40 @@ def _jobs_from_file(
|
|
|
1263
1307
|
)
|
|
1264
1308
|
|
|
1265
1309
|
|
|
1266
|
-
def _spawn_job(
|
|
1310
|
+
def _spawn_job(
|
|
1311
|
+
plan: DispatchPlan,
|
|
1312
|
+
job: WorkerJob,
|
|
1313
|
+
attempt: int,
|
|
1314
|
+
*,
|
|
1315
|
+
batch_artifact_paths: Sequence[Path] = (),
|
|
1316
|
+
) -> WorkerHandle:
|
|
1267
1317
|
job = _prepare_job_attempt(plan, job, attempt)
|
|
1268
|
-
|
|
1269
|
-
|
|
1270
|
-
|
|
1271
|
-
|
|
1272
|
-
|
|
1273
|
-
|
|
1274
|
-
|
|
1275
|
-
|
|
1276
|
-
|
|
1277
|
-
|
|
1278
|
-
|
|
1318
|
+
# Everything between recording the attempt and starting the worker runs
|
|
1319
|
+
# before any worker process exists. A failure here used to leave the attempt
|
|
1320
|
+
# `started` forever: the manifest then refused attempt 1 again ("next
|
|
1321
|
+
# attempt must be 2") and refused attempt 2 as well, because the prompt
|
|
1322
|
+
# metadata still said attempt 1. The invocation had no way forward and the
|
|
1323
|
+
# lead had to mint a new invocation id and prompt path to escape.
|
|
1324
|
+
try:
|
|
1325
|
+
contract = _persisted_write_contract(plan, job)
|
|
1326
|
+
policies = (contract[0],) if contract else ()
|
|
1327
|
+
snapshot = None
|
|
1328
|
+
if contract is not None and contract[1].mutation_audit == "batch":
|
|
1329
|
+
snapshot_path = _mutation_snapshot_path(job)
|
|
1330
|
+
if snapshot_path.is_file():
|
|
1331
|
+
snapshot = MutationSnapshot.from_payload(
|
|
1332
|
+
_load_json_object(snapshot_path, "mutation audit snapshot")
|
|
1333
|
+
)
|
|
1334
|
+
else:
|
|
1335
|
+
snapshot = _mutation_snapshot(
|
|
1336
|
+
plan,
|
|
1337
|
+
policies,
|
|
1338
|
+
(job,),
|
|
1339
|
+
round_artifact_paths=batch_artifact_paths,
|
|
1340
|
+
)
|
|
1341
|
+
except Exception:
|
|
1342
|
+
_abandon_unstarted_attempt(plan, job, attempt)
|
|
1343
|
+
raise
|
|
1279
1344
|
handle = replace(
|
|
1280
1345
|
_start_job(plan, job),
|
|
1281
1346
|
mutation_snapshot=snapshot,
|
|
@@ -1373,11 +1438,14 @@ def _mutation_snapshot(
|
|
|
1373
1438
|
plan: DispatchPlan,
|
|
1374
1439
|
policies: Sequence[WritePolicy],
|
|
1375
1440
|
jobs: Sequence[WorkerJob],
|
|
1441
|
+
round_artifact_paths: Sequence[Path] = (),
|
|
1376
1442
|
) -> MutationSnapshot:
|
|
1377
1443
|
sidecars = tuple(_mutation_snapshot_path(job) for job in jobs)
|
|
1378
1444
|
snapshot = ExecutionMutationAudit().snapshot(
|
|
1379
1445
|
policies,
|
|
1380
1446
|
orchestrator_paths=(
|
|
1447
|
+
*round_artifact_paths,
|
|
1448
|
+
*_run_errors_log_path(plan),
|
|
1381
1449
|
plan.manifest_path,
|
|
1382
1450
|
plan.team_state_path,
|
|
1383
1451
|
Path(f"{plan.team_state_path}.lock"),
|
|
@@ -1478,7 +1546,13 @@ def _worker_artifact_paths(plan: DispatchPlan, job: WorkerJob) -> tuple[Path, ..
|
|
|
1478
1546
|
*job.completion_paths,
|
|
1479
1547
|
Path(audit_sidecar_rel(str(job.worker_result_path))),
|
|
1480
1548
|
status_path_for_prompt(job.prompt_path),
|
|
1481
|
-
|
|
1549
|
+
log_path_for_prompt(job.prompt_path),
|
|
1550
|
+
# The prompt's three derived files are written together and belong in
|
|
1551
|
+
# one list. The audit snapshot used to be listed only for the job whose
|
|
1552
|
+
# snapshot it was, so a sibling's snapshot — written by okstra as that
|
|
1553
|
+
# sibling started — landed inside this worker's window as an
|
|
1554
|
+
# unauthorized artifact-root change.
|
|
1555
|
+
_mutation_snapshot_path(job),
|
|
1482
1556
|
}
|
|
1483
1557
|
error_logs = active_context.get("errorLogs")
|
|
1484
1558
|
if isinstance(error_logs, Mapping):
|
|
@@ -1490,6 +1564,26 @@ def _worker_artifact_paths(plan: DispatchPlan, job: WorkerJob) -> tuple[Path, ..
|
|
|
1490
1564
|
return tuple(sorted(paths, key=str))
|
|
1491
1565
|
|
|
1492
1566
|
|
|
1567
|
+
def _run_errors_log_path(plan: DispatchPlan) -> tuple[Path, ...]:
|
|
1568
|
+
"""The run-level errors log, which the lead appends to while workers run.
|
|
1569
|
+
|
|
1570
|
+
The lead contract requires it to record an observed worker failure as soon
|
|
1571
|
+
as it happens, so a worker still running at that moment sees the write. It
|
|
1572
|
+
is a lead-owned run artifact like the manifest and the lead events log, and
|
|
1573
|
+
is listed for the same reason.
|
|
1574
|
+
"""
|
|
1575
|
+
active_context = _load_optional_json(
|
|
1576
|
+
plan.project_root, plan.manifest.get("activeRunContextPath")
|
|
1577
|
+
)
|
|
1578
|
+
error_logs = active_context.get("errorLogs")
|
|
1579
|
+
if not isinstance(error_logs, Mapping):
|
|
1580
|
+
return ()
|
|
1581
|
+
value = _string_value(error_logs.get("runErrorsLogPath"))
|
|
1582
|
+
if not value:
|
|
1583
|
+
return ()
|
|
1584
|
+
return (_resolve_project_path(plan.project_root, value),)
|
|
1585
|
+
|
|
1586
|
+
|
|
1493
1587
|
|
|
1494
1588
|
|
|
1495
1589
|
def _job_for_attempt(job: WorkerJob, attempt: int) -> WorkerJob:
|
|
@@ -1840,6 +1934,35 @@ def _out_of_plan_edit_paths(result_path: Path) -> tuple[str, ...]:
|
|
|
1840
1934
|
)
|
|
1841
1935
|
|
|
1842
1936
|
|
|
1937
|
+
def _abandon_unstarted_attempt(
|
|
1938
|
+
plan: DispatchPlan, job: WorkerJob, attempt: int
|
|
1939
|
+
) -> None:
|
|
1940
|
+
"""Close an attempt whose worker never started, so a retry can follow it.
|
|
1941
|
+
|
|
1942
|
+
`failed-no-mutation` is the truthful status: the dispatch died before the
|
|
1943
|
+
worker process existed, so nothing wrote anything. It is also the only
|
|
1944
|
+
terminal status the manifest lets another attempt follow.
|
|
1945
|
+
"""
|
|
1946
|
+
if not job.has_execution_identity:
|
|
1947
|
+
return
|
|
1948
|
+
try:
|
|
1949
|
+
finish_attempt_mutation(
|
|
1950
|
+
plan.manifest_path,
|
|
1951
|
+
invocation_ref=job.invocation_ref,
|
|
1952
|
+
attempt=attempt,
|
|
1953
|
+
finished_at=_utc_now(),
|
|
1954
|
+
status="failed-no-mutation",
|
|
1955
|
+
result_path=None,
|
|
1956
|
+
error_path=None,
|
|
1957
|
+
change_summary={},
|
|
1958
|
+
task_key=_require_string(plan.manifest, "taskKey"),
|
|
1959
|
+
)
|
|
1960
|
+
except (ExecutionManifestError, DispatchError, OSError):
|
|
1961
|
+
# The original dispatch failure is what the caller needs to see; a
|
|
1962
|
+
# manifest that cannot be closed here is reported by the next read.
|
|
1963
|
+
return
|
|
1964
|
+
|
|
1965
|
+
|
|
1843
1966
|
def _finish_manifest_attempt(
|
|
1844
1967
|
plan: DispatchPlan,
|
|
1845
1968
|
job: WorkerJob,
|
|
@@ -2171,7 +2294,7 @@ def _wrapper_log_tail(job: WorkerJob) -> str | None:
|
|
|
2171
2294
|
some process exited 1. Capped well under the writer's own excerpt limit
|
|
2172
2295
|
because a whole record must stay inside one atomic `PIPE_BUF` append.
|
|
2173
2296
|
"""
|
|
2174
|
-
log_path =
|
|
2297
|
+
log_path = log_path_for_prompt(job.prompt_path)
|
|
2175
2298
|
try:
|
|
2176
2299
|
with log_path.open("rb") as handle:
|
|
2177
2300
|
handle.seek(0, 2)
|
|
@@ -31,6 +31,8 @@ from typing import Any, Callable, Mapping, Sequence
|
|
|
31
31
|
from . import cmux
|
|
32
32
|
from .agent_invocation import (
|
|
33
33
|
AgentInvocationError,
|
|
34
|
+
AgentModelAssignment,
|
|
35
|
+
InvocationMetadataIdentity,
|
|
34
36
|
agent_model_assignment_from_payload,
|
|
35
37
|
invocation_metadata_identity,
|
|
36
38
|
v2_role_assignment_authority_errors,
|
|
@@ -57,7 +59,7 @@ from .worker_prompt_contract import (
|
|
|
57
59
|
from .worker_runner import LIVE, QUIET
|
|
58
60
|
from .worker_request import verifier_extra_dirs
|
|
59
61
|
from .worker_artifact_paths import audit_sidecar_rel
|
|
60
|
-
from .wrapper_status import status_path_for_prompt
|
|
62
|
+
from .wrapper_status import log_path_for_prompt, status_path_for_prompt
|
|
61
63
|
from .write_policy import (
|
|
62
64
|
build_invocation_write_contract,
|
|
63
65
|
planned_paths_from_run_manifest,
|
|
@@ -104,6 +106,14 @@ WORKTREE_TASK_TYPES = frozenset({"implementation", "final-verification"})
|
|
|
104
106
|
WORKER_STATUSES = frozenset(
|
|
105
107
|
{"in-progress", "completed", "timeout", "error", "not-run"}
|
|
106
108
|
)
|
|
109
|
+
# Which of those mean the dispatch is still expected to produce something. Two
|
|
110
|
+
# readers key off this split — `team reclaim` closes a finished dispatch's pane
|
|
111
|
+
# and must never touch a live one, and the compact-reminder hook calls a run
|
|
112
|
+
# in-flight when any dispatch is still here. Both restated the terminal four
|
|
113
|
+
# locally before, so a sixth status would have read as finished in one place and
|
|
114
|
+
# as live in the other.
|
|
115
|
+
NON_TERMINAL_WORKER_STATUSES = frozenset({"in-progress"})
|
|
116
|
+
TERMINAL_WORKER_STATUSES = WORKER_STATUSES - NON_TERMINAL_WORKER_STATUSES
|
|
107
117
|
REASON_REQUIRED_STATUSES = frozenset({"timeout", "error", "not-run"})
|
|
108
118
|
|
|
109
119
|
|
|
@@ -564,6 +574,9 @@ def record_verified_agent_dispatch(
|
|
|
564
574
|
if (
|
|
565
575
|
enforcement_mode == "host-native-spec-link-gate"
|
|
566
576
|
and assignment.runner != "native-session"
|
|
577
|
+
and not _is_current_session_lead(
|
|
578
|
+
run_manifest_path, execution_identity, assignment
|
|
579
|
+
)
|
|
567
580
|
):
|
|
568
581
|
raise DispatchError(
|
|
569
582
|
"host-native enforcement requires a native-session assignment"
|
|
@@ -718,6 +731,50 @@ def record_verified_agent_dispatch(
|
|
|
718
731
|
return record
|
|
719
732
|
|
|
720
733
|
|
|
734
|
+
def _is_current_session_lead(
|
|
735
|
+
manifest_path: Path,
|
|
736
|
+
execution_identity: InvocationMetadataIdentity | None,
|
|
737
|
+
assignment: AgentModelAssignment,
|
|
738
|
+
) -> bool:
|
|
739
|
+
"""Report whether this dispatch is the lead attesting its own session.
|
|
740
|
+
|
|
741
|
+
A current-session lead has no model binding, so the run manifest projects
|
|
742
|
+
its assignment as ``cli-wrapper`` while the execution manifest records the
|
|
743
|
+
participant as ``current-session``. It runs through no dispatch boundary at
|
|
744
|
+
all: the spec link is the lead associating the already-running session with
|
|
745
|
+
its verified invocation specification, which is what the host-native gate
|
|
746
|
+
records. Keyed on the execution manifest's own participant row rather than
|
|
747
|
+
on the projected runner string, so a genuine cli-wrapper worker never
|
|
748
|
+
reaches the native gate.
|
|
749
|
+
"""
|
|
750
|
+
if execution_identity is None or assignment.runner == "native-session":
|
|
751
|
+
return False
|
|
752
|
+
manifest = read_execution_manifest(manifest_path)
|
|
753
|
+
if manifest.legacy:
|
|
754
|
+
return False
|
|
755
|
+
participant = next(
|
|
756
|
+
(
|
|
757
|
+
row for row in manifest.participant_assignments
|
|
758
|
+
if row.participant_ref == execution_identity.participant_ref
|
|
759
|
+
),
|
|
760
|
+
None,
|
|
761
|
+
)
|
|
762
|
+
execution = next(
|
|
763
|
+
(
|
|
764
|
+
row for row in manifest.role_executions
|
|
765
|
+
if row.role_execution_ref == execution_identity.role_execution_ref
|
|
766
|
+
),
|
|
767
|
+
None,
|
|
768
|
+
)
|
|
769
|
+
return (
|
|
770
|
+
participant is not None
|
|
771
|
+
and execution is not None
|
|
772
|
+
and execution.role == "leader"
|
|
773
|
+
and participant.runner == "current-session"
|
|
774
|
+
and participant.entry_mode == "current-session"
|
|
775
|
+
)
|
|
776
|
+
|
|
777
|
+
|
|
721
778
|
def _agent_write_contract(
|
|
722
779
|
project_root: Path,
|
|
723
780
|
manifest_path: Path,
|
|
@@ -814,7 +871,7 @@ def _manifest_worker_write_paths(
|
|
|
814
871
|
result,
|
|
815
872
|
Path(audit_sidecar_rel(str(result))),
|
|
816
873
|
status_path_for_prompt(prompt_path),
|
|
817
|
-
|
|
874
|
+
log_path_for_prompt(prompt_path),
|
|
818
875
|
}
|
|
819
876
|
active_value = authority.get("activeRunContextPath")
|
|
820
877
|
active = (
|
|
@@ -871,7 +928,7 @@ def _prompt_write_paths(
|
|
|
871
928
|
_project_or_absolute(project_root, value) for value in artifact_values
|
|
872
929
|
}
|
|
873
930
|
artifacts.add(status_path_for_prompt(prompt_path))
|
|
874
|
-
artifacts.add(
|
|
931
|
+
artifacts.add(log_path_for_prompt(prompt_path))
|
|
875
932
|
worktree_value = values.get("Worktree")
|
|
876
933
|
worktree = (
|
|
877
934
|
_project_or_absolute(project_root, worktree_value)
|
|
@@ -171,7 +171,10 @@ class ExecutionMutationAudit:
|
|
|
171
171
|
before.scratch_digests, after.scratch_digests
|
|
172
172
|
)
|
|
173
173
|
source_changes = _source_changes(changed, rows, before)
|
|
174
|
-
|
|
174
|
+
# Reported and, through `_retry_allowed`, load-bearing: a stat-cache
|
|
175
|
+
# refresh must not read as "this worker touched Git" and must not
|
|
176
|
+
# withhold the retry a worker that failed for another reason is owed.
|
|
177
|
+
git_changed = _stable_git_projection(before) != _stable_git_projection(after)
|
|
175
178
|
violations = _policy_violations(
|
|
176
179
|
before,
|
|
177
180
|
after,
|
|
@@ -250,6 +253,22 @@ def _maximum_precision(policy: WritePolicy) -> str:
|
|
|
250
253
|
return policy.maximum_boundary_precision
|
|
251
254
|
|
|
252
255
|
|
|
256
|
+
_INSTALLED_DEPENDENCY_DIRS = frozenset({"node_modules"})
|
|
257
|
+
"""Trees a package manager installs, which no worker authored and no policy can
|
|
258
|
+
enumerate.
|
|
259
|
+
|
|
260
|
+
Excluded for the same reason `.git` is: the audit asks whether the worker
|
|
261
|
+
changed *source*, and these hold neither source nor run artifacts. What forced
|
|
262
|
+
the entry is that the tools inside them write to themselves — a `final-verification`
|
|
263
|
+
verifier running the Tier 1 / Tier 2 suites its own profile mandates had Vitest
|
|
264
|
+
persist `node_modules/.vite/vitest/<hash>/results.json`, and that single cache
|
|
265
|
+
file failed both acceptance verifiers of an otherwise clean stage.
|
|
266
|
+
|
|
267
|
+
One ecosystem, because one is what has been observed. A second name belongs here
|
|
268
|
+
when a run produces the same evidence for it, not before.
|
|
269
|
+
"""
|
|
270
|
+
|
|
271
|
+
|
|
253
272
|
def _content_snapshot(root: Path, generated: frozenset[str]) -> dict[str, str]:
|
|
254
273
|
rows: dict[str, str] = {}
|
|
255
274
|
for current, directories, filenames in os.walk(root, followlinks=False):
|
|
@@ -258,7 +277,11 @@ def _content_snapshot(root: Path, generated: frozenset[str]) -> dict[str, str]:
|
|
|
258
277
|
for name in sorted(directories):
|
|
259
278
|
path = current_path / name
|
|
260
279
|
relative_path = path.relative_to(root)
|
|
261
|
-
if
|
|
280
|
+
if (
|
|
281
|
+
name == ".git"
|
|
282
|
+
or name in _INSTALLED_DEPENDENCY_DIRS
|
|
283
|
+
or _excluded(relative_path, generated)
|
|
284
|
+
):
|
|
262
285
|
continue
|
|
263
286
|
if path.is_symlink():
|
|
264
287
|
rows[relative_path.as_posix()] = _path_digest(path)
|
|
@@ -473,7 +496,7 @@ def _policy_violations(
|
|
|
473
496
|
) -> list[str]:
|
|
474
497
|
if all(policy.source_mode == "source-readonly" for policy in policies):
|
|
475
498
|
failures = ["readonly source changed"] if source_changes else []
|
|
476
|
-
if before
|
|
499
|
+
if _stable_git_projection(before) != _stable_git_projection(after):
|
|
477
500
|
failures.append("gitPolicy disabled but Git projection changed")
|
|
478
501
|
return failures
|
|
479
502
|
policy = next(row for row in policies if row.source_mode == "project-mutation")
|
|
@@ -482,6 +505,29 @@ def _policy_violations(
|
|
|
482
505
|
return failures
|
|
483
506
|
|
|
484
507
|
|
|
508
|
+
def _stable_git_projection(snapshot: MutationSnapshot) -> dict[str, Any]:
|
|
509
|
+
"""The projection minus the one field a read-only reader moves on its own.
|
|
510
|
+
|
|
511
|
+
`indexDigest` is Git's stat cache, and Git rewrites it whenever a plain
|
|
512
|
+
read refreshes a stale entry — `git status --short`, which the
|
|
513
|
+
`final-verification` profile requires of every verifier, is enough. Nothing
|
|
514
|
+
about the worktree's content changed when it moves, so comparing it made a
|
|
515
|
+
read-only worker fail for doing what its own phase told it to do (observed:
|
|
516
|
+
both stage-1 acceptance verifiers, where `indexDigest` was the only key that
|
|
517
|
+
differed and `stagedPaths` was empty on both sides).
|
|
518
|
+
|
|
519
|
+
Every field that does witness a mutation stays compared: `head` and
|
|
520
|
+
`branchCommit` for a moved ref, `stagedPaths` for content added to the
|
|
521
|
+
index, `worktreeRegistration` for a re-registered worktree, `reflogDigest`
|
|
522
|
+
for a ref rewrite, and both Git directories for a redirected repository.
|
|
523
|
+
"""
|
|
524
|
+
return {
|
|
525
|
+
key: value
|
|
526
|
+
for key, value in snapshot.git_projection.items()
|
|
527
|
+
if key != "indexDigest"
|
|
528
|
+
}
|
|
529
|
+
|
|
530
|
+
|
|
485
531
|
def _source_policy_failures(
|
|
486
532
|
policy: WritePolicy,
|
|
487
533
|
changed: set[str],
|
|
@@ -15,6 +15,11 @@ from typing import Any, Callable, Dict, List, Optional, Tuple
|
|
|
15
15
|
from . import consumers, stage_targets, worktree_registry
|
|
16
16
|
from .final_report_paths import final_report_markdown_path
|
|
17
17
|
from .paths import RunRef
|
|
18
|
+
from .release_gate import (
|
|
19
|
+
blocking_condition_ids,
|
|
20
|
+
release_handoff_allowed,
|
|
21
|
+
verdict_token,
|
|
22
|
+
)
|
|
18
23
|
from .stage_map import StageMapError, parse_stage_map_file, stage_map_records
|
|
19
24
|
from .worktree import (compute_branch_name, compute_worktree_path,
|
|
20
25
|
main_worktree_path, is_dirty_excluding_okstra,
|
|
@@ -52,12 +57,14 @@ def compute_eligibility(stage_map: List[Dict[str, Any]],
|
|
|
52
57
|
).handoff_eligibility()
|
|
53
58
|
|
|
54
59
|
|
|
55
|
-
def
|
|
56
|
-
|
|
57
|
-
"""
|
|
60
|
+
def latest_whole_task_fv_release_ready(project_root, project_id: str,
|
|
61
|
+
task_group: str, task_id: str) -> str:
|
|
62
|
+
"""release-handoff 로 넘어가도 되는 whole-task 검증 보고서 경로. 없으면 ''.
|
|
58
63
|
|
|
59
64
|
판정의 SSOT 는 final-report 의 data.json 이다 (record_verified 와 동일
|
|
60
|
-
필드: header.taskType / verificationScope / finalVerdict
|
|
65
|
+
필드: header.taskType / verificationScope / finalVerdict). 통과 조건은
|
|
66
|
+
`release_gate.release_handoff_allowed` 한 곳이 소유한다 — `accepted`,
|
|
67
|
+
또는 모든 조건이 릴리스를 막지 않는다고 선언한 `conditional-accept`.
|
|
61
68
|
whole-task run 산출물만 평면 reports/ 에 남으므로 stage-* 는 걸리지 않는다."""
|
|
62
69
|
from okstra_project.state import find_task_root
|
|
63
70
|
root = find_task_root(Path(project_root),
|
|
@@ -71,11 +78,9 @@ def latest_whole_task_fv_accepted(project_root, project_id: str,
|
|
|
71
78
|
data = json.loads(dj.read_text(encoding="utf-8"))
|
|
72
79
|
except (OSError, json.JSONDecodeError):
|
|
73
80
|
continue
|
|
74
|
-
token = ((data.get("finalVerdict") or {}).get("verdictToken")
|
|
75
|
-
or "").strip().lower()
|
|
76
81
|
if ((data.get("header") or {}).get("taskType") == "final-verification"
|
|
77
82
|
and data.get("verificationScope") == "whole-task"
|
|
78
|
-
and
|
|
83
|
+
and release_handoff_allowed(data)):
|
|
79
84
|
md = final_report_markdown_path(dj)
|
|
80
85
|
return str(md if md.is_file() else dj)
|
|
81
86
|
return ""
|
|
@@ -363,8 +368,10 @@ def _impl_task_key_for_any(rows: List[Dict[str, Any]], stages: List[int]) -> str
|
|
|
363
368
|
|
|
364
369
|
def record_verified(*, plan_run_root, stage: int, report_path: str,
|
|
365
370
|
data_json) -> Dict[str, Any]:
|
|
366
|
-
"""단독-stage
|
|
367
|
-
lead 가 임의 보고서를 verified 로 올리는 것을
|
|
371
|
+
"""릴리스로 넘어갈 수 있는 단독-stage 판정만 기록. data.json 의
|
|
372
|
+
taskType/scope/verdict 를 검증해 lead 가 임의 보고서를 verified 로 올리는 것을
|
|
373
|
+
막는다. 통과 조건은 whole-task 와 같고(`release_gate`), 기록되는 verdict 는
|
|
374
|
+
보고서가 실제로 실은 토큰이다."""
|
|
368
375
|
try:
|
|
369
376
|
data = json.loads(Path(data_json).read_text(encoding="utf-8"))
|
|
370
377
|
except (OSError, json.JSONDecodeError) as exc:
|
|
@@ -376,14 +383,20 @@ def record_verified(*, plan_run_root, stage: int, report_path: str,
|
|
|
376
383
|
raise HandoffError(
|
|
377
384
|
f"record-verified requires verificationScope single-stage, "
|
|
378
385
|
f"got {scope!r}")
|
|
379
|
-
token = (
|
|
380
|
-
|
|
381
|
-
|
|
382
|
-
|
|
386
|
+
token = verdict_token(data)
|
|
387
|
+
if not release_handoff_allowed(data):
|
|
388
|
+
blocking = blocking_condition_ids(data)
|
|
389
|
+
detail = (
|
|
390
|
+
f"condition(s) {blocking} declare `blocksReleaseHandoff: true`"
|
|
391
|
+
if blocking else f"got {token!r}"
|
|
392
|
+
)
|
|
393
|
+
raise HandoffError(
|
|
394
|
+
"verdict must be `accepted`, or `conditional-accept` whose every "
|
|
395
|
+
f"condition declares `blocksReleaseHandoff: false` — {detail}")
|
|
383
396
|
rows = consumers.read_consumers(Path(plan_run_root))
|
|
384
397
|
key = _impl_task_key_for(rows, stage)
|
|
385
398
|
consumers.append_verified(Path(plan_run_root), impl_task_key=key,
|
|
386
|
-
stage=stage, verdict=
|
|
399
|
+
stage=stage, verdict=token,
|
|
387
400
|
report_path=report_path)
|
|
388
401
|
return {"ok": True, "stage": stage, "report_path": report_path}
|
|
389
402
|
|
|
@@ -158,8 +158,17 @@ def _catalog_row(model, adapter, role: str | None) -> dict[str, Any]:
|
|
|
158
158
|
)
|
|
159
159
|
)
|
|
160
160
|
except HostModelBindingError as exc:
|
|
161
|
-
|
|
162
|
-
|
|
161
|
+
# Not "unusable as a lead": the host cannot bind this model to a native
|
|
162
|
+
# session, so the lead runs through the provider CLI instead. Reporting
|
|
163
|
+
# it as unselectable made the listing disagree with what assignment
|
|
164
|
+
# resolution actually does — it accepts the model and falls back to
|
|
165
|
+
# `cli-wrapper` — leaving no way to find out why a lead was not native.
|
|
166
|
+
row["leaderRunner"] = "cli-wrapper"
|
|
167
|
+
row["reason"] = (
|
|
168
|
+
f"native-session unavailable ({exc}); the lead runs via cli-wrapper"
|
|
169
|
+
)
|
|
170
|
+
else:
|
|
171
|
+
row["leaderRunner"] = "native-session"
|
|
163
172
|
return row
|
|
164
173
|
|
|
165
174
|
|
|
@@ -51,6 +51,18 @@ def _dispatch_effort(execution: str, role: str) -> str:
|
|
|
51
51
|
return effort
|
|
52
52
|
|
|
53
53
|
|
|
54
|
+
def dispatch_tier_suffixes() -> frozenset[str]:
|
|
55
|
+
"""Every tier suffix dispatch can append to an agy execution value.
|
|
56
|
+
|
|
57
|
+
The provider serves back the suffixed identity it was given, so reading
|
|
58
|
+
that identity means undoing exactly this set — not guessing at whatever
|
|
59
|
+
trails the last hyphen. `low` is here because `_dispatch_effort` demotes
|
|
60
|
+
the untrusted high tier to it.
|
|
61
|
+
"""
|
|
62
|
+
efforts = {value.lower() for value in (*ROLE_EFFORT.values(), _DEFAULT_EFFORT)}
|
|
63
|
+
return frozenset(efforts | {"low"})
|
|
64
|
+
|
|
65
|
+
|
|
54
66
|
@lru_cache(maxsize=1)
|
|
55
67
|
def agy_models(agy_bin: str = "agy") -> tuple[str, ...]:
|
|
56
68
|
"""Live `agy models` ids, or () when agy is unavailable (non-blocking).
|