okstra 0.197.1 → 0.198.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/cli-registry.mjs +9 -0
- package/dist/cli-registry.mjs.map +1 -1
- package/docs/cli.md +4 -2
- package/docs/project-structure-overview.md +1 -0
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/lead/convergence.md +19 -4
- package/runtime/prompts/lead/report-writer.md +1 -1
- package/runtime/prompts/lead/team-contract.md +4 -3
- package/runtime/prompts/profiles/implementation-option-selection.md +4 -0
- package/runtime/prompts/wizard/prompts.ko.json +6 -2
- package/runtime/python/okstra_ctl/convergence.py +58 -3
- package/runtime/python/okstra_ctl/convergence_engine.py +71 -0
- package/runtime/python/okstra_ctl/convergence_reverify_prompt.py +14 -3
- package/runtime/python/okstra_ctl/convergence_store.py +36 -0
- package/runtime/python/okstra_ctl/dispatch_core.py +10 -3
- package/runtime/python/okstra_ctl/dispatch_state.py +30 -4
- package/runtime/python/okstra_ctl/implementation_options.py +123 -0
- package/runtime/python/okstra_ctl/option_votes.py +194 -0
- package/runtime/python/okstra_ctl/report_assembly.py +10 -0
- package/runtime/python/okstra_ctl/verdict_blocks.py +27 -0
- package/runtime/python/okstra_ctl/wizard/engine.py +4 -0
- package/runtime/python/okstra_ctl/wizard/ids.py +1 -0
- package/runtime/python/okstra_ctl/wizard/registry.py +8 -0
- package/runtime/python/okstra_ctl/wizard/steps_options.py +37 -1
- package/runtime/python/okstra_ctl/worker_audit_check.py +38 -16
- package/runtime/python/okstra_ctl/worker_liveness.py +48 -2
- package/runtime/python/okstra_ctl/workflow.py +1 -1
- package/runtime/skills/okstra-run/SKILL.md +1 -0
- package/runtime/validators/validate-run.py +44 -1
|
@@ -31,12 +31,16 @@ _CLI_EPILOG = r"""Usage:
|
|
|
31
31
|
--seq <nnn> [--worker <id>]
|
|
32
32
|
|
|
33
33
|
--run-dir the run directory; worker-results/ and prompts/ hang off it
|
|
34
|
+
--seq this run's seq; a bare number is zero-padded to three digits
|
|
34
35
|
--worker check only this worker (default: every worker in the run)
|
|
35
36
|
|
|
36
37
|
Emits one JSON object — `{ok, inspected, inspectedFiles[], failures[],
|
|
37
38
|
blocking[], advisory[], runImpact}` — and exits 2 when failures[] is
|
|
38
39
|
non-empty, 0 otherwise. `failures` is `blocking` followed by `advisory`.
|
|
39
40
|
|
|
41
|
+
A selector that matches no result file also exits 2, carrying `selectorError`
|
|
42
|
+
instead of failures: nothing was checked, which is not a pass.
|
|
43
|
+
|
|
40
44
|
Runs the Phase 7 audit-sidecar rules now, while the worker session is still
|
|
41
45
|
alive: that every result file carries no `## 0. Reading Confirmation` heading,
|
|
42
46
|
that its audit sidecar exists (blocking), and that every backticked `path:line`
|
|
@@ -49,6 +53,15 @@ row alone.
|
|
|
49
53
|
"""
|
|
50
54
|
|
|
51
55
|
|
|
56
|
+
# 결과 파일명은 seq 를 3자리로 적고, 선택은 그 문자열과의 동등 비교다
|
|
57
|
+
# (`worker_audit_ledger.worker_result_files`). `--seq 1` 은 `001` 과 같지 않아
|
|
58
|
+
# 전건이 걸러지고, 검사 0건은 실패 0건이므로 통과로 읽혔다 — 2026-09-10
|
|
59
|
+
# dev-10642-15 final-verification 001 실측.
|
|
60
|
+
def _normalized_seq(value: str) -> str:
|
|
61
|
+
text = value.strip()
|
|
62
|
+
return text.zfill(3) if text.isdigit() else text
|
|
63
|
+
|
|
64
|
+
|
|
52
65
|
def _parser() -> argparse.ArgumentParser:
|
|
53
66
|
parser = argparse.ArgumentParser(
|
|
54
67
|
epilog=_CLI_EPILOG,
|
|
@@ -59,8 +72,9 @@ def _parser() -> argparse.ArgumentParser:
|
|
|
59
72
|
parser.add_argument("--run-dir", type=Path, required=True,
|
|
60
73
|
help="runs/<task-type>/ for this run")
|
|
61
74
|
parser.add_argument("--task-type", required=True)
|
|
62
|
-
parser.add_argument("--seq", required=True,
|
|
63
|
-
help="this run's
|
|
75
|
+
parser.add_argument("--seq", required=True, type=_normalized_seq,
|
|
76
|
+
help="this run's seq; a bare number is zero-padded to "
|
|
77
|
+
"three digits")
|
|
64
78
|
parser.add_argument("--worker", default=None,
|
|
65
79
|
help="check only this worker id, with or without the "
|
|
66
80
|
"`-worker` suffix (default: every worker)")
|
|
@@ -83,20 +97,28 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
83
97
|
args.run_dir, args.task_type, args.seq, args.worker
|
|
84
98
|
)
|
|
85
99
|
]
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
+
# 검사한 파일이 없다는 것과 위반이 없다는 것은 다른 사실이고, `ok` 는 둘을
|
|
101
|
+
# 구분하지 못한다. 선택자가 아무것도 고르지 못했으면 판정 자체가 없었으므로
|
|
102
|
+
# 통과로 보고하지 않는다.
|
|
103
|
+
selector_error = "" if inspected else (
|
|
104
|
+
f"no worker-results file matched task-type={args.task_type} "
|
|
105
|
+
f"seq={args.seq}"
|
|
106
|
+
+ (f" worker={args.worker}" if args.worker else "")
|
|
107
|
+
+ f" under {args.run_dir}"
|
|
108
|
+
)
|
|
109
|
+
payload = {
|
|
110
|
+
"ok": bool(not failures and inspected),
|
|
111
|
+
"inspected": len(inspected),
|
|
112
|
+
"inspectedFiles": inspected,
|
|
113
|
+
"failures": failures,
|
|
114
|
+
"blocking": blocking,
|
|
115
|
+
"advisory": advisory,
|
|
116
|
+
"runImpact": RUN_IMPACT,
|
|
117
|
+
}
|
|
118
|
+
if selector_error:
|
|
119
|
+
payload["selectorError"] = selector_error
|
|
120
|
+
print(json.dumps(payload, ensure_ascii=False, indent=2))
|
|
121
|
+
return 2 if failures or selector_error else 0
|
|
100
122
|
|
|
101
123
|
|
|
102
124
|
if __name__ == "__main__":
|
|
@@ -32,10 +32,11 @@ import argparse
|
|
|
32
32
|
import json
|
|
33
33
|
import sys
|
|
34
34
|
import time
|
|
35
|
-
from collections.abc import Callable
|
|
35
|
+
from collections.abc import Callable, Mapping
|
|
36
36
|
from dataclasses import dataclass
|
|
37
37
|
from datetime import datetime, timezone
|
|
38
38
|
from pathlib import Path
|
|
39
|
+
from typing import Any
|
|
39
40
|
|
|
40
41
|
from .wrapper_status import log_path_for_prompt
|
|
41
42
|
|
|
@@ -353,7 +354,47 @@ _ARTIFACT_FIELD_BY_MODE = {
|
|
|
353
354
|
}
|
|
354
355
|
|
|
355
356
|
|
|
357
|
+
def _dispatch_worker_id(record: Mapping[str, Any]) -> str:
|
|
358
|
+
"""이 디스패치 행이 말하는 워커 id.
|
|
359
|
+
|
|
360
|
+
execution-identity v2 행에는 `workerId` 가 실리지 않는다
|
|
361
|
+
(`dispatch_core._dispatch_record`). 그 경우 `assignmentRef` 의 마지막 마디가
|
|
362
|
+
워커 id 이므로 — `initial/codex-verifier`, `critic/acceptance` — 거기서 읽는다.
|
|
363
|
+
"""
|
|
364
|
+
worker_id = record.get("workerId")
|
|
365
|
+
if isinstance(worker_id, str) and worker_id.strip():
|
|
366
|
+
return worker_id.strip()
|
|
367
|
+
assignment_ref = record.get("assignmentRef")
|
|
368
|
+
if isinstance(assignment_ref, str) and assignment_ref.strip():
|
|
369
|
+
return assignment_ref.strip().rsplit("/", 1)[-1]
|
|
370
|
+
return ""
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def _dispatch_fallback(state: Mapping[str, Any], worker_id: str) -> dict:
|
|
374
|
+
"""같은 워커의 디스패치 행 중 마지막 것.
|
|
375
|
+
|
|
376
|
+
로스터 행에 없는 키를 여기서 보충한다. 재디스패치는 같은 워커의 행을 뒤에
|
|
377
|
+
덧붙이므로 마지막 행이 이번 시도다.
|
|
378
|
+
"""
|
|
379
|
+
records = state.get("workerDispatches")
|
|
380
|
+
if not isinstance(records, list):
|
|
381
|
+
return {}
|
|
382
|
+
matches = [
|
|
383
|
+
row for row in records
|
|
384
|
+
if isinstance(row, Mapping) and _dispatch_worker_id(row) == worker_id
|
|
385
|
+
]
|
|
386
|
+
return dict(matches[-1]) if matches else {}
|
|
387
|
+
|
|
388
|
+
|
|
356
389
|
def _worker_row(team_state_path: Path, worker_id: str) -> dict:
|
|
390
|
+
"""이 워커의 프로브 입력 — 로스터 행에 디스패치 행을 덧댄 것.
|
|
391
|
+
|
|
392
|
+
`livenessMode` 는 디스패치 행에만 실린다(`dispatch_core._dispatch_record`);
|
|
393
|
+
로스터 행은 그 키를 갖지 않는다. 로스터만 읽으면 cmux 백엔드의 모든 run 에서
|
|
394
|
+
프로브가 전면 거부되고, 리드는 계약이 금지한 자체 폴링으로 밀려난다.
|
|
395
|
+
보충은 로스터에 없거나 빈 키에만 적용한다 — `startedAt` 처럼 로스터가 정본인
|
|
396
|
+
값을 디스패치 행이 덮어쓰면 grace 앵커가 이번 시도에서 어긋난다.
|
|
397
|
+
"""
|
|
357
398
|
state = load_json_object(team_state_path, "team-state")
|
|
358
399
|
workers = state.get("workers")
|
|
359
400
|
if not isinstance(workers, list):
|
|
@@ -367,7 +408,12 @@ def _worker_row(team_state_path: Path, worker_id: str) -> dict:
|
|
|
367
408
|
)
|
|
368
409
|
if worker is None:
|
|
369
410
|
raise DispatchError(f"team-state has no workerId={worker_id}: {team_state_path}")
|
|
370
|
-
|
|
411
|
+
merged = dict(worker)
|
|
412
|
+
for key, value in _dispatch_fallback(state, worker_id).items():
|
|
413
|
+
current = merged.get(key)
|
|
414
|
+
if key not in merged or (isinstance(current, str) and not current.strip()):
|
|
415
|
+
merged[key] = value
|
|
416
|
+
return merged
|
|
371
417
|
|
|
372
418
|
|
|
373
419
|
def probe_target(team_state_value: str, worker_id: str) -> ProbeTarget:
|
|
@@ -88,7 +88,7 @@ PHASE_RULES: dict[str, dict[str, str]] = {
|
|
|
88
88
|
"implementation-option-selection": {
|
|
89
89
|
"allowed": (
|
|
90
90
|
" - candidate-comparison evidence for up to three options per worker\n"
|
|
91
|
-
" - preselected-validation that re-
|
|
91
|
+
" - preselected-validation of exactly ONE candidate: every analyser returns its own feasibility verdict on that one direction, `candidateAudit` stays empty, and no analyser generates an alternative. It is not a re-evaluation of the merged set — `okstra_ctl.implementation_options._validate_mode` rejects a preselected-validation report carrying more than one option or a non-empty audit\n"
|
|
92
92
|
" - a ranked display of at most three options with requirement mappings\n"
|
|
93
93
|
" - an audit record for every rejected candidate\n"
|
|
94
94
|
" - one endStateCoverage row per brief end-state id (this phase authors no goal of its own)"
|
|
@@ -191,6 +191,7 @@ Repeat until `next.kind == "done"` (or `"aborted"` — terminal cancel, see "How
|
|
|
191
191
|
|
|
192
192
|
That is the entire interactive flow. The wizard handles:
|
|
193
193
|
|
|
194
|
+
- project report language before task selection: when `.okstra/project.json` has no `reportLanguage` value, ask for a language tag and save the answer to that file. An existing value skips this question. Prompt rendering and progress estimation do not save a language,
|
|
194
195
|
- new-vs-existing task split (remaining work — `workStatus != done` — top-3 newest recommendations + Enter directly), task-group / task-id slug validation (task-group offers the top-3 newest candidates combining recent task use + recent `.okstra/briefs/<group>/` creation activity + Enter directly; task-id offers the top-3 recent candidates from the same group + Enter directly),
|
|
195
196
|
- task-type pick (3 options + Enter directly; Enter directly is validated against the full task-type whitelist in a follow-up `text` step). **For a brand-new task the brief is asked first** and the three options are entry phases only (`requirements-discovery` / `improvement-discovery` / `project-analysis` / `feature-analysis` / `change-impact-analysis` / `error-analysis`) — a new task-key has no approved plan, no implementation and no commits, so `implementation`, `final-verification` and `release-handoff` cannot be entered there; the recommended slot comes from the selected brief's `Recommended next phase:` line and falls back to `requirements-discovery` when the brief carries none. For an existing task-key what fills those three depends on `workflow.nextRecommendedPhase` — an object `{phase, status, rationale}`, not a phase-name string. `status: ready` gives the classic trio: its `phase` marked recommended, re-run the current phase, the lifecycle's next step. Any other status contributes nothing at all, since both the recommended slot and the next-step slot derive from that phase — re-run the current phase becomes the first option and the remaining slots fill unlabelled from recently used task-types and then the whitelist. Never recover a phase name from a non-`ready` pointer and propose it yourself: prepare deliberately lowers the pointer to `pending` while its run is unfinished, and the missing recommendation is that signal,
|
|
196
197
|
- brief path — **for a new task it is asked right after task-group, before the task-type** (it is the only input that says what the task is, and the task-type recommendation reads it); for an existing task it is **asked only for entry task-types (requirements-discovery / error-analysis / improvement-discovery / project-analysis / feature-analysis / change-impact-analysis)** (same-group `.okstra/briefs/<task-group>/**/*.md` candidates first, sorted by the newer of file-created/modified time and latest task-catalog use; direct input last; `Keep / Change` for existing entry tasks). `project-analysis`, `feature-analysis`, and `change-impact-analysis` are brief entry task types. On an existing task a downstream lifecycle task-type auto carries in the manifest's brief, and when no registered brief exists a `brief_carry` 3-option prompt appears (recommend switching to entry / Enter directly / Abort). `release-handoff` has no brief step of its own — a new-task run answers the brief before the type is known, and the brief is dropped at render time, so prepare always generates the input document that cites the verification report,
|
|
@@ -115,6 +115,7 @@ from okstra_ctl.final_report_paths import ( # noqa: E402
|
|
|
115
115
|
)
|
|
116
116
|
from okstra_token_usage.report import _match_worker_index # noqa: E402
|
|
117
117
|
from okstra_ctl.implementation_options import ( # noqa: E402
|
|
118
|
+
validate_blocked_answer_channel,
|
|
118
119
|
validate_implementation_option_selection,
|
|
119
120
|
)
|
|
120
121
|
from okstra_ctl.implementation_direction import ( # noqa: E402
|
|
@@ -149,6 +150,7 @@ from okstra_ctl.agent.invocation import ( # noqa: E402
|
|
|
149
150
|
AgentInvocationError,
|
|
150
151
|
agent_model_assignment_from_payload,
|
|
151
152
|
invocation_execution_identity_from_manifest,
|
|
153
|
+
invocation_input_digest,
|
|
152
154
|
verify_agent_invocation,
|
|
153
155
|
)
|
|
154
156
|
from okstra_ctl.execution_identity import ExecutionManifestError # noqa: E402
|
|
@@ -253,6 +255,39 @@ def _result_link_attempt_status_failure(
|
|
|
253
255
|
)
|
|
254
256
|
|
|
255
257
|
|
|
258
|
+
def _dispatch_input_digest(
|
|
259
|
+
project_root: Path, row: Mapping[str, Any]
|
|
260
|
+
) -> tuple[str | None, str]:
|
|
261
|
+
"""이 디스패치의 예약 입력 해시, 예약이 계산한 것과 같은 방법으로.
|
|
262
|
+
|
|
263
|
+
`inputDigest` 와 `promptDigest` 는 서로 다른 대상이다. 전달 계약
|
|
264
|
+
`execution-identity-v1` 부터 예약은 **논리 작업**을 해시한다 — 시도별 전달값과
|
|
265
|
+
prompt history 경로를 뺀 본문(`agent_prompt_task_bytes`) — 반면 `promptDigest`
|
|
266
|
+
는 프롬프트 파일 전체 바이트다. 둘을 동등 비교하면 그 계약을 쓰는 run 은
|
|
267
|
+
통과할 수 있는 값이 하나도 없다(2026-09-08 f56ec08 이후 전부, 2026-09-10
|
|
268
|
+
dev-10642-15 final-verification 001 에서 7/7 디스패치가 이 규칙에 걸렸다).
|
|
269
|
+
|
|
270
|
+
그래서 재구현하지 않고 예약이 쓰는 함수를 그대로 부른다. 메타데이터 경로가
|
|
271
|
+
없는 구형 행만 `promptDigest` 로 돌아간다 — 그 계약에서는 두 값이 같다.
|
|
272
|
+
"""
|
|
273
|
+
metadata_value = row.get("promptMetadataPath")
|
|
274
|
+
if not isinstance(metadata_value, str) or not metadata_value.strip():
|
|
275
|
+
return row.get("promptDigest"), ""
|
|
276
|
+
try:
|
|
277
|
+
metadata = json.loads(
|
|
278
|
+
_resolve_prompt_record_path(project_root, metadata_value)
|
|
279
|
+
.read_text(encoding="utf-8")
|
|
280
|
+
)
|
|
281
|
+
except (OSError, json.JSONDecodeError):
|
|
282
|
+
return None, "prompt metadata is missing or invalid"
|
|
283
|
+
if not isinstance(metadata, Mapping):
|
|
284
|
+
return None, "prompt metadata is not an object"
|
|
285
|
+
try:
|
|
286
|
+
return invocation_input_digest(metadata, project_root), ""
|
|
287
|
+
except (AgentInvocationError, KeyError, OSError, ValueError) as exc:
|
|
288
|
+
return None, f"input digest cannot be recomputed: {exc}"
|
|
289
|
+
|
|
290
|
+
|
|
256
291
|
def _validate_agent_dispatch_contract(
|
|
257
292
|
*,
|
|
258
293
|
project_root: Path,
|
|
@@ -399,12 +434,16 @@ def _validate_agent_dispatch_contract(
|
|
|
399
434
|
)
|
|
400
435
|
continue
|
|
401
436
|
invocation = canonical_invocations.get(row.get("invocationRef"))
|
|
437
|
+
input_digest, digest_error = _dispatch_input_digest(project_root, row)
|
|
438
|
+
if digest_error:
|
|
439
|
+
failures.append(f"agent dispatch {dispatch_id}: {digest_error}")
|
|
440
|
+
continue
|
|
402
441
|
if not isinstance(invocation, Mapping) or any((
|
|
403
442
|
invocation.get("participantRef") != row.get("participantRef"),
|
|
404
443
|
invocation.get("roleExecutionRef") != row.get("roleExecutionRef"),
|
|
405
444
|
invocation.get("dutyId") != row.get("dutyId"),
|
|
406
445
|
invocation.get("dispatchKind") != dispatch_kind,
|
|
407
|
-
invocation.get("inputDigest") !=
|
|
446
|
+
invocation.get("inputDigest") != input_digest,
|
|
408
447
|
)):
|
|
409
448
|
failures.append(
|
|
410
449
|
f"agent dispatch {dispatch_id}: does not match canonical invocation"
|
|
@@ -3273,6 +3312,10 @@ def validate_final_report_data(
|
|
|
3273
3312
|
participating_analysers,
|
|
3274
3313
|
)
|
|
3275
3314
|
)
|
|
3315
|
+
failures.extend(
|
|
3316
|
+
f"implementation-option-selection: {error}"
|
|
3317
|
+
for error in validate_blocked_answer_channel(data)
|
|
3318
|
+
)
|
|
3276
3319
|
elif task_type == "implementation":
|
|
3277
3320
|
_validate_stage_carry_sidecar_exists(data, report_path, failures)
|
|
3278
3321
|
_validate_lead_authored_report(data, report_path, failures)
|