okstra 0.206.1 → 0.207.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/README.md +1 -1
  2. package/dist/cli-registry.mjs +7 -1
  3. package/dist/cli-registry.mjs.map +1 -1
  4. package/docs/architecture/storage-model.md +1 -0
  5. package/docs/architecture.md +28 -4
  6. package/docs/cli.md +13 -11
  7. package/docs/project-structure-overview.md +4 -2
  8. package/package.json +1 -1
  9. package/runtime/BUILD.json +2 -2
  10. package/runtime/agents/operations/code-review.json +1 -1
  11. package/runtime/bin/lib/okstra/usage.sh +3 -3
  12. package/runtime/bin/okstra-compact-reminder.sh +1 -1
  13. package/runtime/prompts/duties/direction-selection-worker.json +1 -1
  14. package/runtime/prompts/launch.template.md +1 -1
  15. package/runtime/prompts/lead/adapters/cmux.md +4 -3
  16. package/runtime/prompts/lead/convergence.md +41 -9
  17. package/runtime/prompts/lead/okstra-lead-contract.md +31 -19
  18. package/runtime/prompts/lead/report-writer.md +8 -6
  19. package/runtime/prompts/profiles/_clarification-recommendation.md +4 -4
  20. package/runtime/prompts/profiles/_common-contract.md +1 -1
  21. package/runtime/prompts/wizard/prompts.ko.json +2 -1
  22. package/runtime/python/okstra_ctl/adapters/hosts/antigravity/relay.md +1 -1
  23. package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +5 -5
  24. package/runtime/python/okstra_ctl/adapters/hosts/codex/relay.md +1 -1
  25. package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +3 -2
  26. package/runtime/python/okstra_ctl/adapters/hosts/grok/relay.md +1 -1
  27. package/runtime/python/okstra_ctl/adapters/hosts/kimi/relay.md +1 -1
  28. package/runtime/python/okstra_ctl/adapters/providers/codex/adapter.py +17 -26
  29. package/runtime/python/okstra_ctl/agent/prompt_cli/batch.py +183 -0
  30. package/runtime/python/okstra_ctl/agent/prompt_cli/cli.py +60 -10
  31. package/runtime/python/okstra_ctl/agent/prompt_cli/jobs.py +21 -4
  32. package/runtime/python/okstra_ctl/approval_decisions.py +32 -2
  33. package/runtime/python/okstra_ctl/assignment_resolver.py +8 -0
  34. package/runtime/python/okstra_ctl/blocking_checks.py +7 -0
  35. package/runtime/python/okstra_ctl/code_review_target.py +92 -6
  36. package/runtime/python/okstra_ctl/dispatch_checkpoints.py +121 -0
  37. package/runtime/python/okstra_ctl/dispatch_core.py +54 -32
  38. package/runtime/python/okstra_ctl/dispatch_state.py +12 -5
  39. package/runtime/python/okstra_ctl/domain/provider.py +5 -0
  40. package/runtime/python/okstra_ctl/domain/worker_presentation.py +21 -2
  41. package/runtime/python/okstra_ctl/domain/write_policy.py +2 -1
  42. package/runtime/python/okstra_ctl/execution_mutation_audit.py +19 -8
  43. package/runtime/python/okstra_ctl/initial_prompt_materialization.py +5 -0
  44. package/runtime/python/okstra_ctl/lead_progress.py +33 -1
  45. package/runtime/python/okstra_ctl/manager_view.py +26 -19
  46. package/runtime/python/okstra_ctl/model_io/lines.py +21 -4
  47. package/runtime/python/okstra_ctl/models.py +4 -1
  48. package/runtime/python/okstra_ctl/operation_invocation.py +11 -2
  49. package/runtime/python/okstra_ctl/phases/final_verification/profile.md +1 -1
  50. package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-executor.md +1 -1
  51. package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-verifier.md +14 -3
  52. package/runtime/python/okstra_ctl/phases/implementation_option_selection/profile.md +1 -1
  53. package/runtime/python/okstra_ctl/phases/implementation_planning/profile.md +1 -1
  54. package/runtime/python/okstra_ctl/phases/technical_verification/profile.md +1 -1
  55. package/runtime/python/okstra_ctl/process_group.py +118 -0
  56. package/runtime/python/okstra_ctl/render.py +6 -2
  57. package/runtime/python/okstra_ctl/report_assembly.py +17 -2
  58. package/runtime/python/okstra_ctl/report_finalize.py +106 -2
  59. package/runtime/python/okstra_ctl/run.py +1 -1
  60. package/runtime/python/okstra_ctl/run_artifact_prune.py +200 -0
  61. package/runtime/python/okstra_ctl/team.py +108 -9
  62. package/runtime/python/okstra_ctl/wizard/steps_options.py +8 -0
  63. package/runtime/python/okstra_ctl/worker_dispatch.py +44 -3
  64. package/runtime/python/okstra_ctl/worker_prompt_policy.py +19 -0
  65. package/runtime/python/okstra_ctl/worker_runner.py +21 -3
  66. package/runtime/python/okstra_ctl/write_policy.py +57 -7
  67. package/runtime/python/okstra_project/dirs.py +14 -0
  68. package/runtime/python/okstra_project/resolver.py +2 -1
  69. package/runtime/schemas/execution-manifest-v2.schema.json +2 -1
  70. package/runtime/skills/okstra-code-review/SKILL.md +70 -32
  71. package/runtime/skills/okstra-code-review/references/review-calibration.md +26 -6
  72. package/runtime/skills/okstra-run/SKILL.md +2 -2
  73. package/runtime/templates/manager/view.template.html +18 -1
@@ -10,12 +10,18 @@ from typing import Any, Mapping, Sequence
10
10
  from . import dispatch_core
11
11
  from .adapters.dispatch.cli_wrapper import CliWrapperDispatchPort
12
12
  from .application.dispatch_assignments import dispatch_assignments
13
- from .dispatch_state import DispatchError
13
+ from .dispatch_checkpoints import (
14
+ record_collect_checkpoints,
15
+ record_dispatch_checkpoints,
16
+ settled_initial_dispatches,
17
+ )
18
+ from .dispatch_state import DispatchError, load_json_object
14
19
  from .models import provider_wrappers
15
20
  from .ports.worker_dispatch import WorkerDispatchRequest
16
21
 
17
22
 
18
23
  SUPPORTED_CLI_WORKERS = provider_wrappers("analyser")
24
+ _SOURCE = "okstra worker-dispatch"
19
25
 
20
26
 
21
27
  def main(argv: Sequence[str] | None = None) -> int:
@@ -38,14 +44,42 @@ def main(argv: Sequence[str] | None = None) -> int:
38
44
  if args.dry_run:
39
45
  _print_json(_backend_payload(backend_plan, dry_run=True))
40
46
  return 0
41
- code = dispatch_core.dispatch_cli_wrapper_plan(backend_plan)
42
- _print_json({**_backend_payload(backend_plan, dry_run=False), "exitCode": code})
47
+ code, lines = _dispatch_recording_checkpoints(backend_plan)
48
+ _print_json({
49
+ **_backend_payload(backend_plan, dry_run=False),
50
+ "exitCode": code,
51
+ "progressLines": lines,
52
+ })
43
53
  return code
44
54
  except DispatchError as exc:
45
55
  print(f"error: {exc}", file=sys.stderr)
46
56
  return 2
47
57
 
48
58
 
59
+ def _dispatch_recording_checkpoints(backend_plan) -> tuple[int, list[str]]:
60
+ if not backend_plan.jobs:
61
+ return dispatch_core.dispatch_cli_wrapper_plan(backend_plan), []
62
+ before = load_json_object(backend_plan.team_state_path, "team-state")
63
+ lines: list[str] = []
64
+ code = dispatch_core.dispatch_cli_wrapper_plan(
65
+ backend_plan,
66
+ before_start=lambda prepared: lines.extend(record_dispatch_checkpoints(
67
+ prepared.project_root, prepared.manifest_path,
68
+ load_json_object(prepared.team_state_path, "team-state"),
69
+ prepared.jobs, source=_SOURCE,
70
+ )),
71
+ )
72
+ lines.extend(record_collect_checkpoints(
73
+ backend_plan.project_root, backend_plan.manifest_path,
74
+ settled_initial_dispatches(before),
75
+ settled_initial_dispatches(
76
+ load_json_object(backend_plan.team_state_path, "team-state")
77
+ ),
78
+ source=_SOURCE,
79
+ ))
80
+ return code, lines
81
+
82
+
49
83
  def _dispatch_request(args: argparse.Namespace) -> WorkerDispatchRequest:
50
84
  return WorkerDispatchRequest(
51
85
  project_root=Path(args.project_root),
@@ -75,6 +109,13 @@ whose persisted runner is 'cli-wrapper'. It verifies each invocation contract
75
109
  before execution and uses the persisted provider, model, and registered CLI.
76
110
  Native-session assignments remain owned by the active host session.
77
111
 
112
+ It records the lead checkpoints it owns and prints them as `progressLines`:
113
+ `phase-3-team-create` when it writes the implicit-team marker,
114
+ `phase-4-dispatch` for each `initial` job, `phase-6-synthesis` on the first
115
+ report-writer dispatch, and `phase-5-collect` for each `initial` dispatch it
116
+ settled. The lead emits those lines instead of calling
117
+ `okstra lead-progress append` for them.
118
+
78
119
  Missing worker prompt files are generated automatically from the immutable run snapshot.
79
120
  Existing prompt and metadata files are never overwritten.
80
121
  When report-writer completes, this command also runs the idempotent post-report
@@ -35,6 +35,10 @@ ERRORS_PATH_HEADERS = (
35
35
  # itself: the lead's launch prompt carries them, and no worker reads that.
36
36
  APPROVED_PLAN_HEADER = "**Approved plan:**"
37
37
  IMPLEMENTATION_STAGE_HEADER = "**Stage for this implementation run:**"
38
+ # A stage's verifiers share one worktree and run as one batch. The Tier 3 script
39
+ # (builds, fixed ports) and the self-mock mutation probe (edits sources in place)
40
+ # must each run once, by the verifier named here.
41
+ STAGE_QA_OWNER_HEADER = "**Stage QA owner:**"
38
42
  IMPLEMENTATION_HEADERS = (
39
43
  "**Worktree:**",
40
44
  APPROVED_PLAN_HEADER,
@@ -295,6 +299,21 @@ def worker_assignment(
295
299
  return None
296
300
 
297
301
 
302
+ def stage_qa_owner(manifest: Mapping[str, Any]) -> str:
303
+ """The first verifier in roster order, or '' when the roster has none.
304
+
305
+ Derived from the manifest alone: prompts are published once and reused, and
306
+ the lead may materialize or re-dispatch one verifier at a time, so an owner
307
+ taken from the dispatch batch could disagree with a published prompt.
308
+ """
309
+ roster = manifest.get("recommendedWorkers")
310
+ for worker_id in roster if isinstance(roster, list) else []:
311
+ row = worker_assignment(manifest, worker_id) if isinstance(worker_id, str) else None
312
+ if row is not None and row.get("role") == "verifier":
313
+ return worker_id
314
+ return ""
315
+
316
+
298
317
  def _plan(
299
318
  audience: PromptAudience,
300
319
  *,
@@ -29,6 +29,7 @@ from .domain.worker_exec import (
29
29
  WorkerExecRequest,
30
30
  )
31
31
  from .domain.worker_presentation import JsonEvents, Presentation
32
+ from .process_group import MemoryWatch, kill_group, memory_cap_bytes
32
33
  from .session_transcript import SessionTranscript
33
34
  from .json_boundary import JsonBoundaryError, write_owned_object_atomic
34
35
 
@@ -42,6 +43,8 @@ _TERM_GRACE_SECONDS = 5
42
43
  # this only has to cover a drain, never a producer.
43
44
  _DRAIN_AFTER_EXIT_SECONDS = 2
44
45
  _TIMEOUT_EXIT_CODE = 124
46
+ # sysexits 의 EX_OSERR. 워커 그룹이 메모리 상한을 넘어 러너가 죽인 run.
47
+ MEMORY_CAP_EXIT_CODE = 71
45
48
  _READ_SIZE = 8192
46
49
  _WRITE_SIZE = 8192
47
50
  _NO_STATUS_EXTRA: Mapping[str, Any] = {}
@@ -66,6 +69,7 @@ def run_worker(
66
69
  _validate_status_extra(status_extra)
67
70
  _validate_request_write_contract(request, status_extra)
68
71
  command = strategy.build_command(request)
72
+ memory_watch = MemoryWatch(memory_cap_bytes(request.project_root))
69
73
  started_monotonic = time.monotonic()
70
74
  status = _started_status(status_extra, log_path)
71
75
  _write_status(status_path, status)
@@ -73,11 +77,12 @@ def run_worker(
73
77
 
74
78
  try:
75
79
  with guard:
76
- exit_code, timed_out, idle_seconds, raw_model, usage = _launch(
80
+ exit_code, timed_out, idle_seconds, raw_model, usage, reaped = _launch(
77
81
  command,
78
82
  log_path,
79
83
  presentation=presentation,
80
84
  idle_timeout_seconds=request.idle_timeout_seconds,
85
+ memory_watch=memory_watch,
81
86
  on_spawn=guard.watch,
82
87
  )
83
88
  at_exit = getattr(command.presentation, "served_model_at_exit", None)
@@ -107,6 +112,11 @@ def run_worker(
107
112
  if failure:
108
113
  status["failure"] = failure
109
114
  exit_code = SERVED_MODEL_MISMATCH_EXIT_CODE
115
+ if reaped:
116
+ status["leftoverProcessesKilled"] = True
117
+ if memory_watch.exceeded_bytes:
118
+ exit_code = MEMORY_CAP_EXIT_CODE
119
+ status.update(failure=memory_watch.failure(), terminated_by="memory-cap")
110
120
  status["exit_code"] = exit_code
111
121
  if timed_out:
112
122
  status.update(
@@ -176,8 +186,9 @@ def _launch(
176
186
  *,
177
187
  presentation: str,
178
188
  idle_timeout_seconds: int,
189
+ memory_watch: MemoryWatch,
179
190
  on_spawn: Callable[[subprocess.Popen[bytes]], None],
180
- ) -> tuple[int, bool, int, str | None, Mapping[str, Any] | None]:
191
+ ) -> tuple[int, bool, int, str | None, Mapping[str, Any] | None, bool]:
181
192
  live = presentation == LIVE
182
193
  transcript = SessionTranscript(log_path, live=live)
183
194
  observation = _ServedModelObservation()
@@ -211,14 +222,18 @@ def _launch(
211
222
  strategy=strategy,
212
223
  presentation=presentation,
213
224
  idle_timeout_seconds=idle_timeout_seconds,
225
+ memory_watch=memory_watch,
214
226
  stdin_text=command.stdin_text,
215
227
  )
228
+ # 워커가 띄운 빌드 워커·서버는 워커가 끝나도 그룹에 남아 계속 자란다.
229
+ reaped = kill_group(process.pid)
216
230
  return (
217
231
  exit_code,
218
232
  timed_out,
219
233
  idle_seconds,
220
234
  observation.raw_model,
221
235
  usage_observation.usage,
236
+ reaped,
222
237
  )
223
238
  finally:
224
239
  transcript.close()
@@ -386,6 +401,7 @@ def _pump(
386
401
  strategy: Presentation,
387
402
  presentation: str,
388
403
  idle_timeout_seconds: int,
404
+ memory_watch: MemoryWatch,
389
405
  stdin_text: str | None = None,
390
406
  ) -> tuple[int, bool, int]:
391
407
  selector = selectors.DefaultSelector()
@@ -418,6 +434,7 @@ def _pump(
418
434
  ):
419
435
  timed_out = True
420
436
  _terminate(process)
437
+ memory_watch.check(process.pid, time.monotonic())
421
438
  drain_deadline = _drain_deadline(process, drain_deadline)
422
439
  if drain_deadline is not None and time.monotonic() >= drain_deadline:
423
440
  _stop_reading(selector)
@@ -457,7 +474,8 @@ def _drain_deadline(
457
474
  holding the write end of the pipe open for as long as it lives. The shell
458
475
  wrappers returned as soon as `wait <pid>` did, so blocking on a grandchild is
459
476
  a regression rather than a policy, and the exit code the caller gets stays
460
- the worker's own.
477
+ the worker's own. Whatever is still in the group afterwards is killed
478
+ (`_launch`).
461
479
 
462
480
  The deadline is set once and never pushed back: output arriving after the
463
481
  worker exited is the grandchild's, and letting it extend the wait would
@@ -26,6 +26,7 @@ from .domain.write_policy import ( # noqa: F401 — 재노출(값 코어는 도
26
26
  write_policy_digest,
27
27
  write_policy_from_payload,
28
28
  )
29
+ from .conformance import conformance_result_file
29
30
  from .final_report_paths import final_report_data_path
30
31
  from .json_boundary import JsonBoundaryError, load_owned_object
31
32
  from .path_hints import hydrate_active_run_context
@@ -65,11 +66,15 @@ def build_write_policy(
65
66
  ),
66
67
  "externalRoots": [str(path) for path in external_roots],
67
68
  }
69
+ artifact_policy: dict[str, Any] = {
70
+ "allowedRoot": str(project_root),
71
+ "allowedPaths": list(artifact_paths),
72
+ }
73
+ preserved_paths = _relative_paths(invocation.get("preservedPaths", ()))
74
+ if preserved_paths:
75
+ artifact_policy["preservedPaths"] = list(preserved_paths)
68
76
  policy = WritePolicy(
69
- artifact_policy={
70
- "allowedRoot": str(project_root),
71
- "allowedPaths": list(artifact_paths),
72
- },
77
+ artifact_policy=artifact_policy,
73
78
  source_policy=source_policy,
74
79
  git_policy=git_policy,
75
80
  auxiliary_policy=auxiliary_policy,
@@ -94,9 +99,12 @@ def _role_qa_artifact_paths(role: str, task_root: Path | None) -> tuple[Path, ..
94
99
  conformance script, its `tsconfig.json` and any real-IO qa spec under
95
100
  `qa/scripts/`, plus the manifest entry naming them
96
101
  (`_implementation-executor.md` §"Stage conformance script", §"Real-IO test
97
- isolation"). Everything else under `qa/` — the self-mock sidecar and its
98
- diff, the conformance run's `result-*.json` — is written while the verifier
99
- runs its own gates. Keeping the executor's grant to those two entries is
102
+ isolation"), plus `qa/output/` for whatever those scripts write when the
103
+ executor runs them against its own work — a baseline capture, build logs
104
+ (§"Verifier gates are not yours to run" lets it run them). Everything else
105
+ under `qa/` — the self-mock sidecar and its diff, the conformance run's
106
+ `result-*.json` — is written while the verifier runs its own gates.
107
+ Keeping the executor's grant to those three entries is
100
108
  what leaves the audit able to catch an executor that runs the verifier's
101
109
  self-mock gate (`_implementation-executor.md` §"Verifier gates are not
102
110
  yours to run"); widening it to `qa/` would silence that.
@@ -114,6 +122,7 @@ def _role_qa_artifact_paths(role: str, task_root: Path | None) -> tuple[Path, ..
114
122
  if role == "implementer":
115
123
  return (
116
124
  task_qa_dir(task_root) / "scripts",
125
+ task_qa_dir(task_root) / "output",
117
126
  task_conformance_manifest_file(task_root),
118
127
  )
119
128
  if role == "verifier":
@@ -121,6 +130,42 @@ def _role_qa_artifact_paths(role: str, task_root: Path | None) -> tuple[Path, ..
121
130
  return ()
122
131
 
123
132
 
133
+ def _inherited_qa_evidence(
134
+ role: str, task_type: str, task_root: Path | None,
135
+ ) -> tuple[Path, ...]:
136
+ """Existing qa files a final-verification verifier must leave byte-identical.
137
+
138
+ Its grant is the whole `qa/` tree, so re-running a plan step that captures a
139
+ baseline overwrote the implementation's RED evidence and still passed the
140
+ audit (observed 2026-09-26, dev-11054 `qa/baseline/base.json`). New files and
141
+ the conformance results it re-runs (`qa/result-<stageKey>.json`, the only qa
142
+ write `final_verification/boundary.json` names) stay writable.
143
+ """
144
+ if role != "verifier" or task_type != "final-verification" or task_root is None:
145
+ return ()
146
+ qa_dir = task_qa_dir(task_root)
147
+ if not qa_dir.is_dir():
148
+ return ()
149
+ rerun = {
150
+ conformance_result_file(qa_dir, key)
151
+ for key in _conformance_stage_keys(task_root)
152
+ }
153
+ return tuple(sorted(
154
+ path for path in qa_dir.rglob("*")
155
+ if path.is_file() and not path.is_symlink() and path not in rerun
156
+ ))
157
+
158
+
159
+ def _conformance_stage_keys(task_root: Path) -> tuple[str, ...]:
160
+ path = task_conformance_manifest_file(task_root)
161
+ if not path.is_file():
162
+ return ()
163
+ entries = _read_json(path, "conformance manifest").get("entries")
164
+ return tuple(
165
+ entry["stageKey"] for entry in entries or ()
166
+ if isinstance(entry, dict) and isinstance(entry.get("stageKey"), str)
167
+ )
168
+
124
169
 
125
170
  def _technical_experiment_paths(
126
171
  artifact_paths: Sequence[Path], task_root: Path | None,
@@ -145,6 +190,7 @@ def _technical_experiment_paths(
145
190
  def build_invocation_write_contract(
146
191
  *,
147
192
  role: str,
193
+ task_type: str,
148
194
  project_root: Path,
149
195
  worktree: Path | None,
150
196
  artifact_paths: Sequence[Path],
@@ -180,6 +226,10 @@ def build_invocation_write_contract(
180
226
  "plannedPaths": list(source_paths),
181
227
  "plannedPathsDeclared": planned_paths_declared,
182
228
  "protectedPaths": [".okstra", ".git"],
229
+ "preservedPaths": [
230
+ _relative_to_root(path, root, "preserved")
231
+ for path in _inherited_qa_evidence(role, task_type, task_root)
232
+ ],
183
233
  "generatedPaths": list(generated_paths),
184
234
  "scratchRoots": [str(path) for path in scratch_roots],
185
235
  "auxiliaryRoots": [str(path) for path in auxiliary_roots],
@@ -87,6 +87,20 @@ def okstra_root(project_root: Path) -> Path:
87
87
  return Path(project_root) / OKSTRA_RELATIVE
88
88
 
89
89
 
90
+ def ensure_okstra_gitignore(project_root: Path) -> Path:
91
+ """`.okstra/.gitignore` 에 `*` 를 둔다. 이미 있으면 건드리지 않는다.
92
+
93
+ 전역 gitignore 로만 `.okstra` 를 빼면 그것을 읽지 않는 도구가 산출물을 훑는다.
94
+ Tailwind v4(4.3.2) 스캐너는 전역 설정은 무시하고 중첩 `.gitignore` 는 따른다
95
+ (2026-09-27 실측) — 실험 사이트 복사본을 읽어 빌드가 수십 GB 로 커졌다.
96
+ """
97
+ path = okstra_root(project_root) / ".gitignore"
98
+ if not path.exists():
99
+ path.parent.mkdir(parents=True, exist_ok=True)
100
+ path.write_text("*\n", encoding="utf-8")
101
+ return path
102
+
103
+
90
104
  def project_json_path(project_root: Path) -> Path:
91
105
  """`<project_root>/.okstra/project.json` 절대 path."""
92
106
  return Path(project_root) / PROJECT_JSON_RELATIVE
@@ -11,6 +11,7 @@ from typing import Optional
11
11
  from .dirs import (
12
12
  OKSTRA_DIR_NAME,
13
13
  PROJECT_JSON_RELATIVE,
14
+ ensure_okstra_gitignore,
14
15
  project_json_path,
15
16
  )
16
17
 
@@ -154,7 +155,7 @@ def upsert_project_json(project_root: Path, project_id: str, *,
154
155
  if not project_id:
155
156
  raise ResolverError("project_id is required for upsert_project_json")
156
157
  target = project_json_path(project_root)
157
- target.parent.mkdir(parents=True, exist_ok=True)
158
+ ensure_okstra_gitignore(project_root)
158
159
  when = now or _now_iso()
159
160
  abs_root = str(Path(project_root).resolve())
160
161
  if target.is_file():
@@ -137,7 +137,8 @@
137
137
  "required": ["allowedRoot", "allowedPaths"],
138
138
  "properties": {
139
139
  "allowedRoot": {"type": "string", "pattern": "^/"},
140
- "allowedPaths": {"type": "array", "items": {"type": "string", "pattern": "\\S"}, "minItems": 1, "uniqueItems": true}
140
+ "allowedPaths": {"type": "array", "items": {"type": "string", "pattern": "\\S"}, "minItems": 1, "uniqueItems": true},
141
+ "preservedPaths": {"type": "array", "items": {"type": "string", "pattern": "\\S"}, "uniqueItems": true}
141
142
  },
142
143
  "additionalProperties": false
143
144
  },
@@ -6,9 +6,10 @@ description: >-
6
6
  unrelated to an okstra run. The tell is a review request over a diff:
7
7
  "review this stage", "review my branch", "code review", "leave the review
8
8
  in a file". The orchestrator censuses the diff into an explicit worklist,
9
- four parallel reviewers return a verdict for every cell against this
10
- project's coding-preflight rules, and a coverage audit re-dispatches any
11
- gap. NOT for writing a PR body (okstra-pr-gen), starting a run
9
+ one or two reviewers (the user picks) each return a verdict for every cell
10
+ against this project's coding-preflight rules, a coverage audit
11
+ re-dispatches any gap, and the orchestrator settles what the reviewers
12
+ disagree on. NOT for writing a PR body (okstra-pr-gen), starting a run
12
13
  (okstra-run), or inspecting a finished task (okstra-inspect).
13
14
  ---
14
15
 
@@ -79,7 +80,7 @@ okstra code-review target --task-key <taskKey> --stage <N> --project-root <proje
79
80
  okstra code-review target --branch <name> [--base <ref>] --project-root <projectRoot> --text
80
81
  ```
81
82
 
82
- `Status: ready` carries `Project root`, `Mode`, `Worktree path`, `Branch`, `Base commit`, `Head commit`, `Review path`, and `Round`; stage mode adds `Task key`, `Task root`, and `Stage`. Carry every field verbatim into the later steps — none of them is recomputed anywhere below.
83
+ `Status: ready` carries `Project root`, `Mode`, `Worktree path`, `Branch`, `Base commit`, `Head commit`, `Review path`, `Round`, and `Report language`; stage mode adds `Task key`, `Task root`, and `Stage`. Carry every field verbatim into the later steps — none of them is recomputed anywhere below.
83
84
 
84
85
  `Status: error` carries `Failure stage` and `Failure reason`. Report both and stop, unless the Exceptions table names that case.
85
86
 
@@ -95,7 +96,7 @@ pass task manifests, target-CLI JSON, or arbitrary JSON fields to a reviewer.
95
96
 
96
97
  **You never derive the base.** Pass `--base <ref>` only when the user named one; otherwise the CLI resolves it. `baseCommit` is a **ref, not necessarily a commit id** — a caller-supplied `--base` passes through verbatim — so use it as given in `git diff <baseCommit>..<headCommit>` and never present it as "commit `<sha>`".
97
98
 
98
- **Where to run git.** Use `worktreePath` when it is non-empty; otherwise run git in `projectRoot` and read the stage's `branch` ref. An empty `worktreePath` does **not** mean the stage is gone: a completed stage's registry row is `released`, so the field is empty even when the directory is still on disk. Either way the commits are on the branch.
99
+ **Where to run git.** Use `worktreePath` when it is non-empty; otherwise run git in `projectRoot` and read the stage's `branch` ref. An empty `worktreePath` does **not** mean the stage is gone: a completed stage's registry row is `released`, so the field is empty even when the directory is still on disk. Either way the commits are on the branch. With no worktree there is no checked-out tree to read whole files from, so every brief tells the reviewer to read a file at the reviewed commit with `git -C <projectRoot> show <headCommit>:<path>` — never from the project root's working tree, which holds a different commit.
99
100
 
100
101
  **Then show the base and confirm it — branch mode and stage mode alike.** The CLI's answer is a recommendation the user has not seen yet, and a base nobody looked at is how unrelated commits slip into a review unnoticed. Print `baseCommit` verbatim next to `git -C <workdir> log -1 --oneline <baseCommit>` so the commit it names is legible, then ask with a 3-option picker:
101
102
 
@@ -117,43 +118,49 @@ pass task manifests, target-CLI JSON, or arbitrary JSON fields to a reviewer.
117
118
 
118
119
  A large census is never truncated. Report the cell count and confirm before dispatching — a silent cut is a false "I looked at everything" signal.
119
120
 
120
- ## Step 3 — Materialize and dispatch four reviewers in parallel
121
+ ## Step 3 — Materialize and dispatch the reviewers in parallel
121
122
 
122
- Ask the runtime what this operation runs — do not choose the role, the providers, or the reviewer count
123
- here:
123
+ **Ask how many reviewers** with a 3-option picker:
124
+
125
+ 1. `2 reviewers` — **the recommendation**: two different models each review the whole census, and you settle only where they disagree.
126
+ 2. `1 reviewer` — cheaper; you adjudicate every finding it returns.
127
+ 3. `Enter directly` — always last; accept only `1` or `2`.
128
+
129
+ Then ask the runtime what those reviewers run — do not choose the role or the providers here:
124
130
 
125
131
  ```
126
- okstra agent-prompt resolve-operation --operation code-review
132
+ okstra agent-prompt resolve-operation --operation code-review --count <1|2>
127
133
  ```
128
134
 
129
135
  It prints the duty, the role, the reviewer count, and one `slot` line per reviewer carrying that slot's
130
- provider and model. The contract owns those values (`agents/operations/code-review.json`), so a machine
131
- with too few distinct models fails here rather than quietly running fewer reviewers. Dispatch exactly the
132
- slots it prints.
136
+ provider and model. The contract owns those values (`agents/operations/code-review.json`, whose `count` is
137
+ the ceiling), so a machine with too few distinct models fails here rather than quietly running fewer
138
+ reviewers. Dispatch exactly the slots it prints.
133
139
 
134
140
  Every reviewer and later gap-fill is a separate auditable standalone invocation. For each slot, create
135
141
  `.okstra/agent-invocations/code-review/<invocation-id>.instructions.md` from that reviewer's brief, then run
136
- `okstra agent-prompt materialize` with `--purpose code-review`, `--audience <dutyId>`, that slot's
137
- `--provider` and `--model <modelRef>`, and the canonical `.prompt.md` path beside it; the returned
138
- assignment is authoritative. Run `okstra agent-prompt verify` against the returned `metadataPath` before
142
+ `okstra agent-prompt materialize --project-root <projectRoot> --invocation-id <invocation-id> --host-runtime <runtime> --provider <slot provider> --model <slot model> --model-role <role> --audience <duty> --purpose code-review --instruction <instructions-path> --prompt <prompt-path>`,
143
+ copying `role`, `duty`, and the slot's `provider` and `model` verbatim from `resolve-operation` (`--model`
144
+ takes the `provider/model` value the slot line prints). The prompt path is the canonical `.prompt.md` beside
145
+ the instructions file; the returned assignment is authoritative. Run `okstra agent-prompt verify` against the returned `metadataPath` before
139
146
  invoking any model.
140
147
 
141
148
  For a native host call, pass the verified prompt body and `hostModelValue`. For a deterministic provider
142
149
  process, run the provider wrapper `~/.okstra/bin/okstra-<provider>-exec.sh <projectRoot> <modelExecutionValue> <prompt-path>`
143
150
  with the verified prompt path (`okstra worker-dispatch` dispatches only a run manifest's assignments, not a
144
151
  standalone prompt). The wrapper records the provider's output in the prompt path with `.md` replaced by
145
- `.log`. Never substitute one model value for the other. Dispatch the four verified calls in parallel when the host supports
146
- it. Every brief carries:
152
+ `.log`. Never substitute one model value for the other. Dispatch the verified calls in parallel when the host supports
153
+ it. **Every reviewer receives the same brief** — two reviewers are worth their cost only when both look at every cell. Every brief carries:
147
154
 
148
- - the diff, plus the work directory path so the reviewer can read whole files for context
155
+ - the diff, plus how to read whole files at the reviewed commit: the `worktreePath` when it is non-empty, otherwise `git -C <projectRoot> show <headCommit>:<path>`
149
156
  - the project layout in one or two lines (where source, tests, and — if the routing found one — domain / ports / adapters live)
150
- - **its own axis's cell list**, verbatim from the census
151
- - the absolute paths of the packs its axis reads (from step 2's routing)
157
+ - **the whole census** — every cell of all four axes, verbatim
158
+ - the absolute paths of every applied pack (from step 2's routing)
152
159
  - the calibration path, written out in full as Step 2 fixed it — `~/.agents/skills/okstra-code-review/references/review-calibration.md`. The verdict format, the severity points, and the rules for a legitimate `clean` are defined there, not in the brief; a reviewer that cannot open this file cannot return a usable verdict, so never hand it a relative path or a "next to the skill" hint
153
160
 
154
161
  Each axis is **one rule group**, so a cell is `target × <axis>` — never `target × <individual rule>`. The reviewer names the specific rule it found violated inside the verdict's `rule` field, and one cell may carry findings from several rules of its group.
155
162
 
156
- Axis scope — the `Reads` column names that group's rules, and a brief never restates them; the bodies are in the packs:
163
+ Axis scope — each reviewer works all four axes; the `Reads` column names each group's rules, and a brief never restates them; the bodies are in the packs:
157
164
 
158
165
  | Axis | Cell | Reads |
159
166
  |---|---|---|
@@ -172,18 +179,47 @@ review result and cannot contribute a verdict.
172
179
 
173
180
  ## Step 3.5 — Audit the coverage
174
181
 
175
- Diff each reviewer's verified `returnedBody` cells against the slice you handed it. Any cell without a verdict
176
- → dispatch **one gap-fill invocation per axis**, carrying only the missing cells and the same brief. Each
182
+ Diff each reviewer's verified `returnedBody` cells against the census. Any cell without a verdict
183
+ → dispatch **one gap-fill invocation per reviewer**, on that reviewer's slot, carrying only its missing cells and the same brief. Each
177
184
  gap-fill uses a new invocation ID and repeats the full materialize → verify → dispatch → materialize-result →
178
- complete → verify-completion boundary from Step 3. Repeat until every cell of every axis has a verdict.
185
+ complete → verify-completion boundary from Step 3. Repeat until every reviewer has a verdict for every cell.
186
+
187
+ A missing verdict is unfinished work, never an implicit `clean`. Do not start Step 3.6 while a single cell is unaccounted for.
188
+
189
+ ## Step 3.6 — Check every citation
190
+
191
+ Reviewers miscount lines. Pass every finding's `path:line` to one call:
192
+
193
+ ```bash
194
+ okstra code-review check-lines --project-root <projectRoot> --base <baseCommit> --head <headCommit> --cite <path:line> [--cite <path:line> ...]
195
+ ```
196
+
197
+ `ok` keeps the citation. For `not-changed` or `not-in-diff`, find the finding's `snippet` in `git -C <projectRoot> show <headCommit>:<path>`; if it sits on one of the changed lines the output lists, replace the line number with that line and say so in the finding. A finding whose snippet is on no changed line goes to `Rejected` as "cites no changed line". Re-run `check-lines` until every kept citation is `ok`.
198
+
199
+ ## Step 3.7 — Settle the findings
200
+
201
+ Reviewers are wrong in a way a census cannot catch: a claim about code they did not open. What you settle depends on the reviewer count. A `must-fix` or `should-fix` with no `failure` or no `evidence` is rejected as "no failure scenario" in either case — the calibration requires both.
202
+
203
+ **Two reviewers.** Pair the findings first: two findings are the same finding when they cite the same `path:line` and describe the same defect. Then:
204
+
205
+ - **Agreed** — both reviewers report it with the same severity and timing. It stands as reported; you do not adjudicate it.
206
+ - **Graded differently** — both report it, with a different severity or timing. Pick one of the two values after reading the code, and record both values and your reason in the finding. Never pick a third value.
207
+ - **One-sided** — one reviewer reports it and the other returned `clean` for that cell or reported something else. Adjudicate it as below.
208
+
209
+ **One reviewer.** Adjudicate every finding it returns.
210
+
211
+ **Adjudicating** a finding: open each `evidence` location at `headCommit`, plus the definition of every symbol its `failure` depends on, and check the `failure` against that code.
212
+
213
+ - **Confirmed** — the code produces the stated failure. The finding keeps its severity and timing.
214
+ - **Rejected** — the code contradicts the claim (the called function never reads `this`, the value cannot be null there, the branch is unreachable). Move it to `Rejected` with the contradicting `path:line` and one sentence on what it shows.
179
215
 
180
- A missing verdict is unfinished work, never an implicit `clean`. Do not start Step 4 while a single cell is unaccounted for.
216
+ Rejecting is a statement about the code, so it always carries a `path:line`. Never reject for being unconvinced. Apart from choosing between two reviewers' values, never promote, demote, or re-time a finding. Confirmed and agreed `park` findings go to `Parked`, unscored.
181
217
 
182
218
  ## Step 4 — Merge and write
183
219
 
184
220
  1. **Read `references/review-calibration.md`** (next to this file) and follow its report section — you are the one writing the file, and it fixes the Coverage sentence, the per-finding line format, the Score table columns, and the total row. The reviewers were given it for their verdicts; the report obeys it too.
185
- 2. **Dedupe across axes.** The same defect surfaced by two axes stays once, under the rule that explains it best.
186
- 3. **Severity is the reviewer's.** The merge concatenates and dedupes; it never re-grades. If a verdict looks wrong, the brief was wrong — improve the brief for next run and ship this one as returned.
221
+ 2. **Dedupe across axes and reviewers.** The same defect surfaced by two axes or two reviewers stays once, under the rule that explains it best, and each finding says how it was settled: `agreed`, `graded by orchestrator`, or `confirmed by orchestrator`.
222
+ 3. **Severity is the reviewers'; truth is yours.** The merge never re-grades or re-times a finding except by choosing between two reviewers' values in Step 3.7.
187
223
  4. **Write the report to `reviewPath`** with the Write tool (it creates the parent directories). Frontmatter fields, in this order:
188
224
 
189
225
  ```yaml
@@ -192,14 +228,15 @@ taskKey: <task-key> # stage mode
192
228
  branch: <branch> # branch mode
193
229
  stage: <N> # stage mode only
194
230
  round: <round>
231
+ reviewers: [<provider/model>, ...] # the slots Step 3 dispatched
195
232
  baseCommit: <exactly as the CLI returned it>
196
233
  headCommit: <headCommit>
197
234
  packs: [<applied coding-preflight pack paths>]
198
235
  generatedAt: <YYYY-MM-DD HH:MM>
199
236
  ```
200
237
 
201
- Body sections, in this order: `## Coverage`, `## Must-fix`, `## Should-fix`, `## Nits`, `## Score`. Empty severity sections are omitted; `Coverage` and `Score` are always present, and a review with no findings still emits the Score table with a total of 0.
202
- 5. **In the session, print only** the `reviewPath`, the count per severity, and the score total. The file is the deliverable — do not replay the findings in chat.
238
+ Body sections, in this order: `## Coverage`, `## Must-fix`, `## Should-fix`, `## Nits`, `## Parked`, `## Rejected`, `## Score`. Empty sections are omitted; `Coverage` and `Score` are always present, and a review with no confirmed `now` findings still emits the Score table with a total of 0. The score counts confirmed `now` findings only. Write the prose in the `Report language` from Step 1.
239
+ 5. **In the session, print only** the `reviewPath`, the count per severity, the parked and rejected counts, and the score total. The file is the deliverable — do not replay the findings in chat.
203
240
 
204
241
  ## Exceptions
205
242
 
@@ -207,7 +244,7 @@ generatedAt: <YYYY-MM-DD HH:MM>
207
244
  |---|---|
208
245
  | `preflight` reports `Okstra preflight: failed` | retry with `--cwd <dir>`; if that also fails, tell the user to run `/okstra-setup` first and stop |
209
246
  | `unknown command: code-review` | the `okstra` binary predates this skill — tell the user to update it (`npm i -g okstra@latest`) and stop |
210
- | `worktreePath` is empty | not a hard stop and not a missing stage: run git in `projectRoot` against the stage's `branch` ref and continue |
247
+ | `worktreePath` is empty | not a hard stop and not a missing stage: run git in `projectRoot` against the stage's `branch` ref, and have reviewers read whole files with `git -C <projectRoot> show <headCommit>:<path>` |
211
248
  | `baseCommit` is not an ancestor of `headCommit` — `git -C <workdir> rev-list --count <headCommit>..<baseCommit>` returns a **non-zero** count, meaning a rebase or squash rewrote the history the stage was recorded against, and `<baseCommit>..<headCommit>` would drag predecessor work in backwards | code-review is read-only, so do not force a reconcile. Offer two options: review against the branch's current tip, or run `okstra git-reconcile` first and retry |
212
249
  | the diff is empty | dispatch no reviewers; write the "no changes" report to `reviewPath` and stop. It is a normal report, not a free-form note: the same frontmatter, `## Coverage` reading "0 changed files → 0 cells on every axis, 0 files excluded" plus the applied packs, every severity section omitted, and `## Score` carrying the table with its single total row reading 0 |
213
250
  | the census is large | never truncate — report the cell count and confirm before dispatching |
@@ -215,9 +252,10 @@ generatedAt: <YYYY-MM-DD HH:MM>
215
252
 
216
253
  ## Principles
217
254
 
218
- - **Stay in the diff.** Every finding cites a line this diff changed. A cell whose only wart sits on untouched lines verdicts `clean` — pre-existing issues are not this change's problem.
255
+ - **Stay in the diff.** Every finding cites a line this diff changed, and Step 3.6 checks it. A cell whose only wart sits on untouched lines verdicts `clean` — pre-existing issues are not this change's problem.
219
256
  - **Don't manufacture findings.** A census fully verdicted `clean` is a valid, useful result.
220
257
  - **No finding without a fix.** Readability findings carry a pseudocode sketch; naming findings carry a concrete alternative name.
221
258
  - **The census is law.** A reviewer that rebuilds its own worklist reintroduces exactly the run-to-run variance this skill exists to kill.
222
259
  - **Every cell gets a verdict.** `clean` is a result, not an omission; the audit treats a gap as unfinished work.
223
- - **The report prose is Korean.** Paths, identifiers, rule names, and quoted code stay verbatim.
260
+ - **Only a checked claim scores.** A finding reaches the score after its citation is on a changed line and its failure holds against the code; everything else is `Parked` or `Rejected`, with its reason.
261
+ - **The report prose follows `Report language`.** Paths, identifiers, rule names, and quoted code stay verbatim.