okstra 0.206.1 → 0.207.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/cli-registry.mjs +7 -1
- package/dist/cli-registry.mjs.map +1 -1
- package/docs/architecture/storage-model.md +1 -0
- package/docs/architecture.md +28 -4
- package/docs/cli.md +13 -11
- package/docs/project-structure-overview.md +4 -2
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/operations/code-review.json +1 -1
- package/runtime/bin/lib/okstra/usage.sh +3 -3
- package/runtime/bin/okstra-compact-reminder.sh +1 -1
- package/runtime/prompts/duties/direction-selection-worker.json +1 -1
- package/runtime/prompts/launch.template.md +1 -1
- package/runtime/prompts/lead/adapters/cmux.md +4 -3
- package/runtime/prompts/lead/convergence.md +41 -9
- package/runtime/prompts/lead/okstra-lead-contract.md +31 -19
- package/runtime/prompts/lead/report-writer.md +8 -6
- package/runtime/prompts/profiles/_clarification-recommendation.md +4 -4
- package/runtime/prompts/profiles/_common-contract.md +1 -1
- package/runtime/prompts/wizard/prompts.ko.json +2 -1
- package/runtime/python/okstra_ctl/adapters/hosts/antigravity/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +5 -5
- package/runtime/python/okstra_ctl/adapters/hosts/codex/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +3 -2
- package/runtime/python/okstra_ctl/adapters/hosts/grok/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/kimi/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/providers/codex/adapter.py +17 -26
- package/runtime/python/okstra_ctl/agent/prompt_cli/batch.py +183 -0
- package/runtime/python/okstra_ctl/agent/prompt_cli/cli.py +60 -10
- package/runtime/python/okstra_ctl/agent/prompt_cli/jobs.py +21 -4
- package/runtime/python/okstra_ctl/approval_decisions.py +32 -2
- package/runtime/python/okstra_ctl/assignment_resolver.py +8 -0
- package/runtime/python/okstra_ctl/blocking_checks.py +7 -0
- package/runtime/python/okstra_ctl/code_review_target.py +92 -6
- package/runtime/python/okstra_ctl/dispatch_checkpoints.py +121 -0
- package/runtime/python/okstra_ctl/dispatch_core.py +54 -32
- package/runtime/python/okstra_ctl/dispatch_state.py +12 -5
- package/runtime/python/okstra_ctl/domain/provider.py +5 -0
- package/runtime/python/okstra_ctl/domain/worker_presentation.py +21 -2
- package/runtime/python/okstra_ctl/domain/write_policy.py +2 -1
- package/runtime/python/okstra_ctl/execution_mutation_audit.py +19 -8
- package/runtime/python/okstra_ctl/initial_prompt_materialization.py +5 -0
- package/runtime/python/okstra_ctl/lead_progress.py +33 -1
- package/runtime/python/okstra_ctl/manager_view.py +26 -19
- package/runtime/python/okstra_ctl/model_io/lines.py +21 -4
- package/runtime/python/okstra_ctl/models.py +4 -1
- package/runtime/python/okstra_ctl/operation_invocation.py +11 -2
- package/runtime/python/okstra_ctl/phases/final_verification/profile.md +1 -1
- package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-executor.md +1 -1
- package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-verifier.md +14 -3
- package/runtime/python/okstra_ctl/phases/implementation_option_selection/profile.md +1 -1
- package/runtime/python/okstra_ctl/phases/implementation_planning/profile.md +1 -1
- package/runtime/python/okstra_ctl/phases/technical_verification/profile.md +1 -1
- package/runtime/python/okstra_ctl/process_group.py +118 -0
- package/runtime/python/okstra_ctl/render.py +6 -2
- package/runtime/python/okstra_ctl/report_assembly.py +17 -2
- package/runtime/python/okstra_ctl/report_finalize.py +106 -2
- package/runtime/python/okstra_ctl/run.py +1 -1
- package/runtime/python/okstra_ctl/run_artifact_prune.py +200 -0
- package/runtime/python/okstra_ctl/team.py +108 -9
- package/runtime/python/okstra_ctl/wizard/steps_options.py +8 -0
- package/runtime/python/okstra_ctl/worker_dispatch.py +44 -3
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +19 -0
- package/runtime/python/okstra_ctl/worker_runner.py +21 -3
- package/runtime/python/okstra_ctl/write_policy.py +57 -7
- package/runtime/python/okstra_project/dirs.py +14 -0
- package/runtime/python/okstra_project/resolver.py +2 -1
- package/runtime/schemas/execution-manifest-v2.schema.json +2 -1
- package/runtime/skills/okstra-code-review/SKILL.md +70 -32
- package/runtime/skills/okstra-code-review/references/review-calibration.md +26 -6
- package/runtime/skills/okstra-run/SKILL.md +2 -2
- package/runtime/templates/manager/view.template.html +18 -1
|
@@ -10,12 +10,18 @@ from typing import Any, Mapping, Sequence
|
|
|
10
10
|
from . import dispatch_core
|
|
11
11
|
from .adapters.dispatch.cli_wrapper import CliWrapperDispatchPort
|
|
12
12
|
from .application.dispatch_assignments import dispatch_assignments
|
|
13
|
-
from .
|
|
13
|
+
from .dispatch_checkpoints import (
|
|
14
|
+
record_collect_checkpoints,
|
|
15
|
+
record_dispatch_checkpoints,
|
|
16
|
+
settled_initial_dispatches,
|
|
17
|
+
)
|
|
18
|
+
from .dispatch_state import DispatchError, load_json_object
|
|
14
19
|
from .models import provider_wrappers
|
|
15
20
|
from .ports.worker_dispatch import WorkerDispatchRequest
|
|
16
21
|
|
|
17
22
|
|
|
18
23
|
SUPPORTED_CLI_WORKERS = provider_wrappers("analyser")
|
|
24
|
+
_SOURCE = "okstra worker-dispatch"
|
|
19
25
|
|
|
20
26
|
|
|
21
27
|
def main(argv: Sequence[str] | None = None) -> int:
|
|
@@ -38,14 +44,42 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|
|
38
44
|
if args.dry_run:
|
|
39
45
|
_print_json(_backend_payload(backend_plan, dry_run=True))
|
|
40
46
|
return 0
|
|
41
|
-
code =
|
|
42
|
-
_print_json({
|
|
47
|
+
code, lines = _dispatch_recording_checkpoints(backend_plan)
|
|
48
|
+
_print_json({
|
|
49
|
+
**_backend_payload(backend_plan, dry_run=False),
|
|
50
|
+
"exitCode": code,
|
|
51
|
+
"progressLines": lines,
|
|
52
|
+
})
|
|
43
53
|
return code
|
|
44
54
|
except DispatchError as exc:
|
|
45
55
|
print(f"error: {exc}", file=sys.stderr)
|
|
46
56
|
return 2
|
|
47
57
|
|
|
48
58
|
|
|
59
|
+
def _dispatch_recording_checkpoints(backend_plan) -> tuple[int, list[str]]:
|
|
60
|
+
if not backend_plan.jobs:
|
|
61
|
+
return dispatch_core.dispatch_cli_wrapper_plan(backend_plan), []
|
|
62
|
+
before = load_json_object(backend_plan.team_state_path, "team-state")
|
|
63
|
+
lines: list[str] = []
|
|
64
|
+
code = dispatch_core.dispatch_cli_wrapper_plan(
|
|
65
|
+
backend_plan,
|
|
66
|
+
before_start=lambda prepared: lines.extend(record_dispatch_checkpoints(
|
|
67
|
+
prepared.project_root, prepared.manifest_path,
|
|
68
|
+
load_json_object(prepared.team_state_path, "team-state"),
|
|
69
|
+
prepared.jobs, source=_SOURCE,
|
|
70
|
+
)),
|
|
71
|
+
)
|
|
72
|
+
lines.extend(record_collect_checkpoints(
|
|
73
|
+
backend_plan.project_root, backend_plan.manifest_path,
|
|
74
|
+
settled_initial_dispatches(before),
|
|
75
|
+
settled_initial_dispatches(
|
|
76
|
+
load_json_object(backend_plan.team_state_path, "team-state")
|
|
77
|
+
),
|
|
78
|
+
source=_SOURCE,
|
|
79
|
+
))
|
|
80
|
+
return code, lines
|
|
81
|
+
|
|
82
|
+
|
|
49
83
|
def _dispatch_request(args: argparse.Namespace) -> WorkerDispatchRequest:
|
|
50
84
|
return WorkerDispatchRequest(
|
|
51
85
|
project_root=Path(args.project_root),
|
|
@@ -75,6 +109,13 @@ whose persisted runner is 'cli-wrapper'. It verifies each invocation contract
|
|
|
75
109
|
before execution and uses the persisted provider, model, and registered CLI.
|
|
76
110
|
Native-session assignments remain owned by the active host session.
|
|
77
111
|
|
|
112
|
+
It records the lead checkpoints it owns and prints them as `progressLines`:
|
|
113
|
+
`phase-3-team-create` when it writes the implicit-team marker,
|
|
114
|
+
`phase-4-dispatch` for each `initial` job, `phase-6-synthesis` on the first
|
|
115
|
+
report-writer dispatch, and `phase-5-collect` for each `initial` dispatch it
|
|
116
|
+
settled. The lead emits those lines instead of calling
|
|
117
|
+
`okstra lead-progress append` for them.
|
|
118
|
+
|
|
78
119
|
Missing worker prompt files are generated automatically from the immutable run snapshot.
|
|
79
120
|
Existing prompt and metadata files are never overwritten.
|
|
80
121
|
When report-writer completes, this command also runs the idempotent post-report
|
|
@@ -35,6 +35,10 @@ ERRORS_PATH_HEADERS = (
|
|
|
35
35
|
# itself: the lead's launch prompt carries them, and no worker reads that.
|
|
36
36
|
APPROVED_PLAN_HEADER = "**Approved plan:**"
|
|
37
37
|
IMPLEMENTATION_STAGE_HEADER = "**Stage for this implementation run:**"
|
|
38
|
+
# A stage's verifiers share one worktree and run as one batch. The Tier 3 script
|
|
39
|
+
# (builds, fixed ports) and the self-mock mutation probe (edits sources in place)
|
|
40
|
+
# must each run once, by the verifier named here.
|
|
41
|
+
STAGE_QA_OWNER_HEADER = "**Stage QA owner:**"
|
|
38
42
|
IMPLEMENTATION_HEADERS = (
|
|
39
43
|
"**Worktree:**",
|
|
40
44
|
APPROVED_PLAN_HEADER,
|
|
@@ -295,6 +299,21 @@ def worker_assignment(
|
|
|
295
299
|
return None
|
|
296
300
|
|
|
297
301
|
|
|
302
|
+
def stage_qa_owner(manifest: Mapping[str, Any]) -> str:
|
|
303
|
+
"""The first verifier in roster order, or '' when the roster has none.
|
|
304
|
+
|
|
305
|
+
Derived from the manifest alone: prompts are published once and reused, and
|
|
306
|
+
the lead may materialize or re-dispatch one verifier at a time, so an owner
|
|
307
|
+
taken from the dispatch batch could disagree with a published prompt.
|
|
308
|
+
"""
|
|
309
|
+
roster = manifest.get("recommendedWorkers")
|
|
310
|
+
for worker_id in roster if isinstance(roster, list) else []:
|
|
311
|
+
row = worker_assignment(manifest, worker_id) if isinstance(worker_id, str) else None
|
|
312
|
+
if row is not None and row.get("role") == "verifier":
|
|
313
|
+
return worker_id
|
|
314
|
+
return ""
|
|
315
|
+
|
|
316
|
+
|
|
298
317
|
def _plan(
|
|
299
318
|
audience: PromptAudience,
|
|
300
319
|
*,
|
|
@@ -29,6 +29,7 @@ from .domain.worker_exec import (
|
|
|
29
29
|
WorkerExecRequest,
|
|
30
30
|
)
|
|
31
31
|
from .domain.worker_presentation import JsonEvents, Presentation
|
|
32
|
+
from .process_group import MemoryWatch, kill_group, memory_cap_bytes
|
|
32
33
|
from .session_transcript import SessionTranscript
|
|
33
34
|
from .json_boundary import JsonBoundaryError, write_owned_object_atomic
|
|
34
35
|
|
|
@@ -42,6 +43,8 @@ _TERM_GRACE_SECONDS = 5
|
|
|
42
43
|
# this only has to cover a drain, never a producer.
|
|
43
44
|
_DRAIN_AFTER_EXIT_SECONDS = 2
|
|
44
45
|
_TIMEOUT_EXIT_CODE = 124
|
|
46
|
+
# sysexits 의 EX_OSERR. 워커 그룹이 메모리 상한을 넘어 러너가 죽인 run.
|
|
47
|
+
MEMORY_CAP_EXIT_CODE = 71
|
|
45
48
|
_READ_SIZE = 8192
|
|
46
49
|
_WRITE_SIZE = 8192
|
|
47
50
|
_NO_STATUS_EXTRA: Mapping[str, Any] = {}
|
|
@@ -66,6 +69,7 @@ def run_worker(
|
|
|
66
69
|
_validate_status_extra(status_extra)
|
|
67
70
|
_validate_request_write_contract(request, status_extra)
|
|
68
71
|
command = strategy.build_command(request)
|
|
72
|
+
memory_watch = MemoryWatch(memory_cap_bytes(request.project_root))
|
|
69
73
|
started_monotonic = time.monotonic()
|
|
70
74
|
status = _started_status(status_extra, log_path)
|
|
71
75
|
_write_status(status_path, status)
|
|
@@ -73,11 +77,12 @@ def run_worker(
|
|
|
73
77
|
|
|
74
78
|
try:
|
|
75
79
|
with guard:
|
|
76
|
-
exit_code, timed_out, idle_seconds, raw_model, usage = _launch(
|
|
80
|
+
exit_code, timed_out, idle_seconds, raw_model, usage, reaped = _launch(
|
|
77
81
|
command,
|
|
78
82
|
log_path,
|
|
79
83
|
presentation=presentation,
|
|
80
84
|
idle_timeout_seconds=request.idle_timeout_seconds,
|
|
85
|
+
memory_watch=memory_watch,
|
|
81
86
|
on_spawn=guard.watch,
|
|
82
87
|
)
|
|
83
88
|
at_exit = getattr(command.presentation, "served_model_at_exit", None)
|
|
@@ -107,6 +112,11 @@ def run_worker(
|
|
|
107
112
|
if failure:
|
|
108
113
|
status["failure"] = failure
|
|
109
114
|
exit_code = SERVED_MODEL_MISMATCH_EXIT_CODE
|
|
115
|
+
if reaped:
|
|
116
|
+
status["leftoverProcessesKilled"] = True
|
|
117
|
+
if memory_watch.exceeded_bytes:
|
|
118
|
+
exit_code = MEMORY_CAP_EXIT_CODE
|
|
119
|
+
status.update(failure=memory_watch.failure(), terminated_by="memory-cap")
|
|
110
120
|
status["exit_code"] = exit_code
|
|
111
121
|
if timed_out:
|
|
112
122
|
status.update(
|
|
@@ -176,8 +186,9 @@ def _launch(
|
|
|
176
186
|
*,
|
|
177
187
|
presentation: str,
|
|
178
188
|
idle_timeout_seconds: int,
|
|
189
|
+
memory_watch: MemoryWatch,
|
|
179
190
|
on_spawn: Callable[[subprocess.Popen[bytes]], None],
|
|
180
|
-
) -> tuple[int, bool, int, str | None, Mapping[str, Any] | None]:
|
|
191
|
+
) -> tuple[int, bool, int, str | None, Mapping[str, Any] | None, bool]:
|
|
181
192
|
live = presentation == LIVE
|
|
182
193
|
transcript = SessionTranscript(log_path, live=live)
|
|
183
194
|
observation = _ServedModelObservation()
|
|
@@ -211,14 +222,18 @@ def _launch(
|
|
|
211
222
|
strategy=strategy,
|
|
212
223
|
presentation=presentation,
|
|
213
224
|
idle_timeout_seconds=idle_timeout_seconds,
|
|
225
|
+
memory_watch=memory_watch,
|
|
214
226
|
stdin_text=command.stdin_text,
|
|
215
227
|
)
|
|
228
|
+
# 워커가 띄운 빌드 워커·서버는 워커가 끝나도 그룹에 남아 계속 자란다.
|
|
229
|
+
reaped = kill_group(process.pid)
|
|
216
230
|
return (
|
|
217
231
|
exit_code,
|
|
218
232
|
timed_out,
|
|
219
233
|
idle_seconds,
|
|
220
234
|
observation.raw_model,
|
|
221
235
|
usage_observation.usage,
|
|
236
|
+
reaped,
|
|
222
237
|
)
|
|
223
238
|
finally:
|
|
224
239
|
transcript.close()
|
|
@@ -386,6 +401,7 @@ def _pump(
|
|
|
386
401
|
strategy: Presentation,
|
|
387
402
|
presentation: str,
|
|
388
403
|
idle_timeout_seconds: int,
|
|
404
|
+
memory_watch: MemoryWatch,
|
|
389
405
|
stdin_text: str | None = None,
|
|
390
406
|
) -> tuple[int, bool, int]:
|
|
391
407
|
selector = selectors.DefaultSelector()
|
|
@@ -418,6 +434,7 @@ def _pump(
|
|
|
418
434
|
):
|
|
419
435
|
timed_out = True
|
|
420
436
|
_terminate(process)
|
|
437
|
+
memory_watch.check(process.pid, time.monotonic())
|
|
421
438
|
drain_deadline = _drain_deadline(process, drain_deadline)
|
|
422
439
|
if drain_deadline is not None and time.monotonic() >= drain_deadline:
|
|
423
440
|
_stop_reading(selector)
|
|
@@ -457,7 +474,8 @@ def _drain_deadline(
|
|
|
457
474
|
holding the write end of the pipe open for as long as it lives. The shell
|
|
458
475
|
wrappers returned as soon as `wait <pid>` did, so blocking on a grandchild is
|
|
459
476
|
a regression rather than a policy, and the exit code the caller gets stays
|
|
460
|
-
the worker's own.
|
|
477
|
+
the worker's own. Whatever is still in the group afterwards is killed
|
|
478
|
+
(`_launch`).
|
|
461
479
|
|
|
462
480
|
The deadline is set once and never pushed back: output arriving after the
|
|
463
481
|
worker exited is the grandchild's, and letting it extend the wait would
|
|
@@ -26,6 +26,7 @@ from .domain.write_policy import ( # noqa: F401 — 재노출(값 코어는 도
|
|
|
26
26
|
write_policy_digest,
|
|
27
27
|
write_policy_from_payload,
|
|
28
28
|
)
|
|
29
|
+
from .conformance import conformance_result_file
|
|
29
30
|
from .final_report_paths import final_report_data_path
|
|
30
31
|
from .json_boundary import JsonBoundaryError, load_owned_object
|
|
31
32
|
from .path_hints import hydrate_active_run_context
|
|
@@ -65,11 +66,15 @@ def build_write_policy(
|
|
|
65
66
|
),
|
|
66
67
|
"externalRoots": [str(path) for path in external_roots],
|
|
67
68
|
}
|
|
69
|
+
artifact_policy: dict[str, Any] = {
|
|
70
|
+
"allowedRoot": str(project_root),
|
|
71
|
+
"allowedPaths": list(artifact_paths),
|
|
72
|
+
}
|
|
73
|
+
preserved_paths = _relative_paths(invocation.get("preservedPaths", ()))
|
|
74
|
+
if preserved_paths:
|
|
75
|
+
artifact_policy["preservedPaths"] = list(preserved_paths)
|
|
68
76
|
policy = WritePolicy(
|
|
69
|
-
artifact_policy=
|
|
70
|
-
"allowedRoot": str(project_root),
|
|
71
|
-
"allowedPaths": list(artifact_paths),
|
|
72
|
-
},
|
|
77
|
+
artifact_policy=artifact_policy,
|
|
73
78
|
source_policy=source_policy,
|
|
74
79
|
git_policy=git_policy,
|
|
75
80
|
auxiliary_policy=auxiliary_policy,
|
|
@@ -94,9 +99,12 @@ def _role_qa_artifact_paths(role: str, task_root: Path | None) -> tuple[Path, ..
|
|
|
94
99
|
conformance script, its `tsconfig.json` and any real-IO qa spec under
|
|
95
100
|
`qa/scripts/`, plus the manifest entry naming them
|
|
96
101
|
(`_implementation-executor.md` §"Stage conformance script", §"Real-IO test
|
|
97
|
-
isolation")
|
|
98
|
-
|
|
99
|
-
|
|
102
|
+
isolation"), plus `qa/output/` for whatever those scripts write when the
|
|
103
|
+
executor runs them against its own work — a baseline capture, build logs
|
|
104
|
+
(§"Verifier gates are not yours to run" lets it run them). Everything else
|
|
105
|
+
under `qa/` — the self-mock sidecar and its diff, the conformance run's
|
|
106
|
+
`result-*.json` — is written while the verifier runs its own gates.
|
|
107
|
+
Keeping the executor's grant to those three entries is
|
|
100
108
|
what leaves the audit able to catch an executor that runs the verifier's
|
|
101
109
|
self-mock gate (`_implementation-executor.md` §"Verifier gates are not
|
|
102
110
|
yours to run"); widening it to `qa/` would silence that.
|
|
@@ -114,6 +122,7 @@ def _role_qa_artifact_paths(role: str, task_root: Path | None) -> tuple[Path, ..
|
|
|
114
122
|
if role == "implementer":
|
|
115
123
|
return (
|
|
116
124
|
task_qa_dir(task_root) / "scripts",
|
|
125
|
+
task_qa_dir(task_root) / "output",
|
|
117
126
|
task_conformance_manifest_file(task_root),
|
|
118
127
|
)
|
|
119
128
|
if role == "verifier":
|
|
@@ -121,6 +130,42 @@ def _role_qa_artifact_paths(role: str, task_root: Path | None) -> tuple[Path, ..
|
|
|
121
130
|
return ()
|
|
122
131
|
|
|
123
132
|
|
|
133
|
+
def _inherited_qa_evidence(
|
|
134
|
+
role: str, task_type: str, task_root: Path | None,
|
|
135
|
+
) -> tuple[Path, ...]:
|
|
136
|
+
"""Existing qa files a final-verification verifier must leave byte-identical.
|
|
137
|
+
|
|
138
|
+
Its grant is the whole `qa/` tree, so re-running a plan step that captures a
|
|
139
|
+
baseline overwrote the implementation's RED evidence and still passed the
|
|
140
|
+
audit (observed 2026-09-26, dev-11054 `qa/baseline/base.json`). New files and
|
|
141
|
+
the conformance results it re-runs (`qa/result-<stageKey>.json`, the only qa
|
|
142
|
+
write `final_verification/boundary.json` names) stay writable.
|
|
143
|
+
"""
|
|
144
|
+
if role != "verifier" or task_type != "final-verification" or task_root is None:
|
|
145
|
+
return ()
|
|
146
|
+
qa_dir = task_qa_dir(task_root)
|
|
147
|
+
if not qa_dir.is_dir():
|
|
148
|
+
return ()
|
|
149
|
+
rerun = {
|
|
150
|
+
conformance_result_file(qa_dir, key)
|
|
151
|
+
for key in _conformance_stage_keys(task_root)
|
|
152
|
+
}
|
|
153
|
+
return tuple(sorted(
|
|
154
|
+
path for path in qa_dir.rglob("*")
|
|
155
|
+
if path.is_file() and not path.is_symlink() and path not in rerun
|
|
156
|
+
))
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _conformance_stage_keys(task_root: Path) -> tuple[str, ...]:
|
|
160
|
+
path = task_conformance_manifest_file(task_root)
|
|
161
|
+
if not path.is_file():
|
|
162
|
+
return ()
|
|
163
|
+
entries = _read_json(path, "conformance manifest").get("entries")
|
|
164
|
+
return tuple(
|
|
165
|
+
entry["stageKey"] for entry in entries or ()
|
|
166
|
+
if isinstance(entry, dict) and isinstance(entry.get("stageKey"), str)
|
|
167
|
+
)
|
|
168
|
+
|
|
124
169
|
|
|
125
170
|
def _technical_experiment_paths(
|
|
126
171
|
artifact_paths: Sequence[Path], task_root: Path | None,
|
|
@@ -145,6 +190,7 @@ def _technical_experiment_paths(
|
|
|
145
190
|
def build_invocation_write_contract(
|
|
146
191
|
*,
|
|
147
192
|
role: str,
|
|
193
|
+
task_type: str,
|
|
148
194
|
project_root: Path,
|
|
149
195
|
worktree: Path | None,
|
|
150
196
|
artifact_paths: Sequence[Path],
|
|
@@ -180,6 +226,10 @@ def build_invocation_write_contract(
|
|
|
180
226
|
"plannedPaths": list(source_paths),
|
|
181
227
|
"plannedPathsDeclared": planned_paths_declared,
|
|
182
228
|
"protectedPaths": [".okstra", ".git"],
|
|
229
|
+
"preservedPaths": [
|
|
230
|
+
_relative_to_root(path, root, "preserved")
|
|
231
|
+
for path in _inherited_qa_evidence(role, task_type, task_root)
|
|
232
|
+
],
|
|
183
233
|
"generatedPaths": list(generated_paths),
|
|
184
234
|
"scratchRoots": [str(path) for path in scratch_roots],
|
|
185
235
|
"auxiliaryRoots": [str(path) for path in auxiliary_roots],
|
|
@@ -87,6 +87,20 @@ def okstra_root(project_root: Path) -> Path:
|
|
|
87
87
|
return Path(project_root) / OKSTRA_RELATIVE
|
|
88
88
|
|
|
89
89
|
|
|
90
|
+
def ensure_okstra_gitignore(project_root: Path) -> Path:
|
|
91
|
+
"""`.okstra/.gitignore` 에 `*` 를 둔다. 이미 있으면 건드리지 않는다.
|
|
92
|
+
|
|
93
|
+
전역 gitignore 로만 `.okstra` 를 빼면 그것을 읽지 않는 도구가 산출물을 훑는다.
|
|
94
|
+
Tailwind v4(4.3.2) 스캐너는 전역 설정은 무시하고 중첩 `.gitignore` 는 따른다
|
|
95
|
+
(2026-09-27 실측) — 실험 사이트 복사본을 읽어 빌드가 수십 GB 로 커졌다.
|
|
96
|
+
"""
|
|
97
|
+
path = okstra_root(project_root) / ".gitignore"
|
|
98
|
+
if not path.exists():
|
|
99
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
100
|
+
path.write_text("*\n", encoding="utf-8")
|
|
101
|
+
return path
|
|
102
|
+
|
|
103
|
+
|
|
90
104
|
def project_json_path(project_root: Path) -> Path:
|
|
91
105
|
"""`<project_root>/.okstra/project.json` 절대 path."""
|
|
92
106
|
return Path(project_root) / PROJECT_JSON_RELATIVE
|
|
@@ -11,6 +11,7 @@ from typing import Optional
|
|
|
11
11
|
from .dirs import (
|
|
12
12
|
OKSTRA_DIR_NAME,
|
|
13
13
|
PROJECT_JSON_RELATIVE,
|
|
14
|
+
ensure_okstra_gitignore,
|
|
14
15
|
project_json_path,
|
|
15
16
|
)
|
|
16
17
|
|
|
@@ -154,7 +155,7 @@ def upsert_project_json(project_root: Path, project_id: str, *,
|
|
|
154
155
|
if not project_id:
|
|
155
156
|
raise ResolverError("project_id is required for upsert_project_json")
|
|
156
157
|
target = project_json_path(project_root)
|
|
157
|
-
|
|
158
|
+
ensure_okstra_gitignore(project_root)
|
|
158
159
|
when = now or _now_iso()
|
|
159
160
|
abs_root = str(Path(project_root).resolve())
|
|
160
161
|
if target.is_file():
|
|
@@ -137,7 +137,8 @@
|
|
|
137
137
|
"required": ["allowedRoot", "allowedPaths"],
|
|
138
138
|
"properties": {
|
|
139
139
|
"allowedRoot": {"type": "string", "pattern": "^/"},
|
|
140
|
-
"allowedPaths": {"type": "array", "items": {"type": "string", "pattern": "\\S"}, "minItems": 1, "uniqueItems": true}
|
|
140
|
+
"allowedPaths": {"type": "array", "items": {"type": "string", "pattern": "\\S"}, "minItems": 1, "uniqueItems": true},
|
|
141
|
+
"preservedPaths": {"type": "array", "items": {"type": "string", "pattern": "\\S"}, "uniqueItems": true}
|
|
141
142
|
},
|
|
142
143
|
"additionalProperties": false
|
|
143
144
|
},
|
|
@@ -6,9 +6,10 @@ description: >-
|
|
|
6
6
|
unrelated to an okstra run. The tell is a review request over a diff:
|
|
7
7
|
"review this stage", "review my branch", "code review", "leave the review
|
|
8
8
|
in a file". The orchestrator censuses the diff into an explicit worklist,
|
|
9
|
-
|
|
10
|
-
project's coding-preflight rules,
|
|
11
|
-
|
|
9
|
+
one or two reviewers (the user picks) each return a verdict for every cell
|
|
10
|
+
against this project's coding-preflight rules, a coverage audit
|
|
11
|
+
re-dispatches any gap, and the orchestrator settles what the reviewers
|
|
12
|
+
disagree on. NOT for writing a PR body (okstra-pr-gen), starting a run
|
|
12
13
|
(okstra-run), or inspecting a finished task (okstra-inspect).
|
|
13
14
|
---
|
|
14
15
|
|
|
@@ -79,7 +80,7 @@ okstra code-review target --task-key <taskKey> --stage <N> --project-root <proje
|
|
|
79
80
|
okstra code-review target --branch <name> [--base <ref>] --project-root <projectRoot> --text
|
|
80
81
|
```
|
|
81
82
|
|
|
82
|
-
`Status: ready` carries `Project root`, `Mode`, `Worktree path`, `Branch`, `Base commit`, `Head commit`, `Review path`, and `
|
|
83
|
+
`Status: ready` carries `Project root`, `Mode`, `Worktree path`, `Branch`, `Base commit`, `Head commit`, `Review path`, `Round`, and `Report language`; stage mode adds `Task key`, `Task root`, and `Stage`. Carry every field verbatim into the later steps — none of them is recomputed anywhere below.
|
|
83
84
|
|
|
84
85
|
`Status: error` carries `Failure stage` and `Failure reason`. Report both and stop, unless the Exceptions table names that case.
|
|
85
86
|
|
|
@@ -95,7 +96,7 @@ pass task manifests, target-CLI JSON, or arbitrary JSON fields to a reviewer.
|
|
|
95
96
|
|
|
96
97
|
**You never derive the base.** Pass `--base <ref>` only when the user named one; otherwise the CLI resolves it. `baseCommit` is a **ref, not necessarily a commit id** — a caller-supplied `--base` passes through verbatim — so use it as given in `git diff <baseCommit>..<headCommit>` and never present it as "commit `<sha>`".
|
|
97
98
|
|
|
98
|
-
**Where to run git.** Use `worktreePath` when it is non-empty; otherwise run git in `projectRoot` and read the stage's `branch` ref. An empty `worktreePath` does **not** mean the stage is gone: a completed stage's registry row is `released`, so the field is empty even when the directory is still on disk. Either way the commits are on the branch.
|
|
99
|
+
**Where to run git.** Use `worktreePath` when it is non-empty; otherwise run git in `projectRoot` and read the stage's `branch` ref. An empty `worktreePath` does **not** mean the stage is gone: a completed stage's registry row is `released`, so the field is empty even when the directory is still on disk. Either way the commits are on the branch. With no worktree there is no checked-out tree to read whole files from, so every brief tells the reviewer to read a file at the reviewed commit with `git -C <projectRoot> show <headCommit>:<path>` — never from the project root's working tree, which holds a different commit.
|
|
99
100
|
|
|
100
101
|
**Then show the base and confirm it — branch mode and stage mode alike.** The CLI's answer is a recommendation the user has not seen yet, and a base nobody looked at is how unrelated commits slip into a review unnoticed. Print `baseCommit` verbatim next to `git -C <workdir> log -1 --oneline <baseCommit>` so the commit it names is legible, then ask with a 3-option picker:
|
|
101
102
|
|
|
@@ -117,43 +118,49 @@ pass task manifests, target-CLI JSON, or arbitrary JSON fields to a reviewer.
|
|
|
117
118
|
|
|
118
119
|
A large census is never truncated. Report the cell count and confirm before dispatching — a silent cut is a false "I looked at everything" signal.
|
|
119
120
|
|
|
120
|
-
## Step 3 — Materialize and dispatch
|
|
121
|
+
## Step 3 — Materialize and dispatch the reviewers in parallel
|
|
121
122
|
|
|
122
|
-
Ask
|
|
123
|
-
|
|
123
|
+
**Ask how many reviewers** with a 3-option picker:
|
|
124
|
+
|
|
125
|
+
1. `2 reviewers` — **the recommendation**: two different models each review the whole census, and you settle only where they disagree.
|
|
126
|
+
2. `1 reviewer` — cheaper; you adjudicate every finding it returns.
|
|
127
|
+
3. `Enter directly` — always last; accept only `1` or `2`.
|
|
128
|
+
|
|
129
|
+
Then ask the runtime what those reviewers run — do not choose the role or the providers here:
|
|
124
130
|
|
|
125
131
|
```
|
|
126
|
-
okstra agent-prompt resolve-operation --operation code-review
|
|
132
|
+
okstra agent-prompt resolve-operation --operation code-review --count <1|2>
|
|
127
133
|
```
|
|
128
134
|
|
|
129
135
|
It prints the duty, the role, the reviewer count, and one `slot` line per reviewer carrying that slot's
|
|
130
|
-
provider and model. The contract owns those values (`agents/operations/code-review.json
|
|
131
|
-
with too few distinct models fails here rather than quietly running fewer
|
|
132
|
-
slots it prints.
|
|
136
|
+
provider and model. The contract owns those values (`agents/operations/code-review.json`, whose `count` is
|
|
137
|
+
the ceiling), so a machine with too few distinct models fails here rather than quietly running fewer
|
|
138
|
+
reviewers. Dispatch exactly the slots it prints.
|
|
133
139
|
|
|
134
140
|
Every reviewer and later gap-fill is a separate auditable standalone invocation. For each slot, create
|
|
135
141
|
`.okstra/agent-invocations/code-review/<invocation-id>.instructions.md` from that reviewer's brief, then run
|
|
136
|
-
`okstra agent-prompt materialize
|
|
137
|
-
|
|
138
|
-
|
|
142
|
+
`okstra agent-prompt materialize --project-root <projectRoot> --invocation-id <invocation-id> --host-runtime <runtime> --provider <slot provider> --model <slot model> --model-role <role> --audience <duty> --purpose code-review --instruction <instructions-path> --prompt <prompt-path>`,
|
|
143
|
+
copying `role`, `duty`, and the slot's `provider` and `model` verbatim from `resolve-operation` (`--model`
|
|
144
|
+
takes the `provider/model` value the slot line prints). The prompt path is the canonical `.prompt.md` beside
|
|
145
|
+
the instructions file; the returned assignment is authoritative. Run `okstra agent-prompt verify` against the returned `metadataPath` before
|
|
139
146
|
invoking any model.
|
|
140
147
|
|
|
141
148
|
For a native host call, pass the verified prompt body and `hostModelValue`. For a deterministic provider
|
|
142
149
|
process, run the provider wrapper `~/.okstra/bin/okstra-<provider>-exec.sh <projectRoot> <modelExecutionValue> <prompt-path>`
|
|
143
150
|
with the verified prompt path (`okstra worker-dispatch` dispatches only a run manifest's assignments, not a
|
|
144
151
|
standalone prompt). The wrapper records the provider's output in the prompt path with `.md` replaced by
|
|
145
|
-
`.log`. Never substitute one model value for the other. Dispatch the
|
|
146
|
-
it. Every brief carries:
|
|
152
|
+
`.log`. Never substitute one model value for the other. Dispatch the verified calls in parallel when the host supports
|
|
153
|
+
it. **Every reviewer receives the same brief** — two reviewers are worth their cost only when both look at every cell. Every brief carries:
|
|
147
154
|
|
|
148
|
-
- the diff, plus
|
|
155
|
+
- the diff, plus how to read whole files at the reviewed commit: the `worktreePath` when it is non-empty, otherwise `git -C <projectRoot> show <headCommit>:<path>`
|
|
149
156
|
- the project layout in one or two lines (where source, tests, and — if the routing found one — domain / ports / adapters live)
|
|
150
|
-
- **
|
|
151
|
-
- the absolute paths of
|
|
157
|
+
- **the whole census** — every cell of all four axes, verbatim
|
|
158
|
+
- the absolute paths of every applied pack (from step 2's routing)
|
|
152
159
|
- the calibration path, written out in full as Step 2 fixed it — `~/.agents/skills/okstra-code-review/references/review-calibration.md`. The verdict format, the severity points, and the rules for a legitimate `clean` are defined there, not in the brief; a reviewer that cannot open this file cannot return a usable verdict, so never hand it a relative path or a "next to the skill" hint
|
|
153
160
|
|
|
154
161
|
Each axis is **one rule group**, so a cell is `target × <axis>` — never `target × <individual rule>`. The reviewer names the specific rule it found violated inside the verdict's `rule` field, and one cell may carry findings from several rules of its group.
|
|
155
162
|
|
|
156
|
-
Axis scope — the `Reads` column names
|
|
163
|
+
Axis scope — each reviewer works all four axes; the `Reads` column names each group's rules, and a brief never restates them; the bodies are in the packs:
|
|
157
164
|
|
|
158
165
|
| Axis | Cell | Reads |
|
|
159
166
|
|---|---|---|
|
|
@@ -172,18 +179,47 @@ review result and cannot contribute a verdict.
|
|
|
172
179
|
|
|
173
180
|
## Step 3.5 — Audit the coverage
|
|
174
181
|
|
|
175
|
-
Diff each reviewer's verified `returnedBody` cells against the
|
|
176
|
-
→ dispatch **one gap-fill invocation per
|
|
182
|
+
Diff each reviewer's verified `returnedBody` cells against the census. Any cell without a verdict
|
|
183
|
+
→ dispatch **one gap-fill invocation per reviewer**, on that reviewer's slot, carrying only its missing cells and the same brief. Each
|
|
177
184
|
gap-fill uses a new invocation ID and repeats the full materialize → verify → dispatch → materialize-result →
|
|
178
|
-
complete → verify-completion boundary from Step 3. Repeat until every
|
|
185
|
+
complete → verify-completion boundary from Step 3. Repeat until every reviewer has a verdict for every cell.
|
|
186
|
+
|
|
187
|
+
A missing verdict is unfinished work, never an implicit `clean`. Do not start Step 3.6 while a single cell is unaccounted for.
|
|
188
|
+
|
|
189
|
+
## Step 3.6 — Check every citation
|
|
190
|
+
|
|
191
|
+
Reviewers miscount lines. Pass every finding's `path:line` to one call:
|
|
192
|
+
|
|
193
|
+
```bash
|
|
194
|
+
okstra code-review check-lines --project-root <projectRoot> --base <baseCommit> --head <headCommit> --cite <path:line> [--cite <path:line> ...]
|
|
195
|
+
```
|
|
196
|
+
|
|
197
|
+
`ok` keeps the citation. For `not-changed` or `not-in-diff`, find the finding's `snippet` in `git -C <projectRoot> show <headCommit>:<path>`; if it sits on one of the changed lines the output lists, replace the line number with that line and say so in the finding. A finding whose snippet is on no changed line goes to `Rejected` as "cites no changed line". Re-run `check-lines` until every kept citation is `ok`.
|
|
198
|
+
|
|
199
|
+
## Step 3.7 — Settle the findings
|
|
200
|
+
|
|
201
|
+
Reviewers are wrong in a way a census cannot catch: a claim about code they did not open. What you settle depends on the reviewer count. A `must-fix` or `should-fix` with no `failure` or no `evidence` is rejected as "no failure scenario" in either case — the calibration requires both.
|
|
202
|
+
|
|
203
|
+
**Two reviewers.** Pair the findings first: two findings are the same finding when they cite the same `path:line` and describe the same defect. Then:
|
|
204
|
+
|
|
205
|
+
- **Agreed** — both reviewers report it with the same severity and timing. It stands as reported; you do not adjudicate it.
|
|
206
|
+
- **Graded differently** — both report it, with a different severity or timing. Pick one of the two values after reading the code, and record both values and your reason in the finding. Never pick a third value.
|
|
207
|
+
- **One-sided** — one reviewer reports it and the other returned `clean` for that cell or reported something else. Adjudicate it as below.
|
|
208
|
+
|
|
209
|
+
**One reviewer.** Adjudicate every finding it returns.
|
|
210
|
+
|
|
211
|
+
**Adjudicating** a finding: open each `evidence` location at `headCommit`, plus the definition of every symbol its `failure` depends on, and check the `failure` against that code.
|
|
212
|
+
|
|
213
|
+
- **Confirmed** — the code produces the stated failure. The finding keeps its severity and timing.
|
|
214
|
+
- **Rejected** — the code contradicts the claim (the called function never reads `this`, the value cannot be null there, the branch is unreachable). Move it to `Rejected` with the contradicting `path:line` and one sentence on what it shows.
|
|
179
215
|
|
|
180
|
-
|
|
216
|
+
Rejecting is a statement about the code, so it always carries a `path:line`. Never reject for being unconvinced. Apart from choosing between two reviewers' values, never promote, demote, or re-time a finding. Confirmed and agreed `park` findings go to `Parked`, unscored.
|
|
181
217
|
|
|
182
218
|
## Step 4 — Merge and write
|
|
183
219
|
|
|
184
220
|
1. **Read `references/review-calibration.md`** (next to this file) and follow its report section — you are the one writing the file, and it fixes the Coverage sentence, the per-finding line format, the Score table columns, and the total row. The reviewers were given it for their verdicts; the report obeys it too.
|
|
185
|
-
2. **Dedupe across axes.** The same defect surfaced by two axes stays once, under the rule that explains it best
|
|
186
|
-
3. **Severity is the
|
|
221
|
+
2. **Dedupe across axes and reviewers.** The same defect surfaced by two axes or two reviewers stays once, under the rule that explains it best, and each finding says how it was settled: `agreed`, `graded by orchestrator`, or `confirmed by orchestrator`.
|
|
222
|
+
3. **Severity is the reviewers'; truth is yours.** The merge never re-grades or re-times a finding except by choosing between two reviewers' values in Step 3.7.
|
|
187
223
|
4. **Write the report to `reviewPath`** with the Write tool (it creates the parent directories). Frontmatter fields, in this order:
|
|
188
224
|
|
|
189
225
|
```yaml
|
|
@@ -192,14 +228,15 @@ taskKey: <task-key> # stage mode
|
|
|
192
228
|
branch: <branch> # branch mode
|
|
193
229
|
stage: <N> # stage mode only
|
|
194
230
|
round: <round>
|
|
231
|
+
reviewers: [<provider/model>, ...] # the slots Step 3 dispatched
|
|
195
232
|
baseCommit: <exactly as the CLI returned it>
|
|
196
233
|
headCommit: <headCommit>
|
|
197
234
|
packs: [<applied coding-preflight pack paths>]
|
|
198
235
|
generatedAt: <YYYY-MM-DD HH:MM>
|
|
199
236
|
```
|
|
200
237
|
|
|
201
|
-
Body sections, in this order: `## Coverage`, `## Must-fix`, `## Should-fix`, `## Nits`, `## Score`. Empty
|
|
202
|
-
5. **In the session, print only** the `reviewPath`, the count per severity, and the score total. The file is the deliverable — do not replay the findings in chat.
|
|
238
|
+
Body sections, in this order: `## Coverage`, `## Must-fix`, `## Should-fix`, `## Nits`, `## Parked`, `## Rejected`, `## Score`. Empty sections are omitted; `Coverage` and `Score` are always present, and a review with no confirmed `now` findings still emits the Score table with a total of 0. The score counts confirmed `now` findings only. Write the prose in the `Report language` from Step 1.
|
|
239
|
+
5. **In the session, print only** the `reviewPath`, the count per severity, the parked and rejected counts, and the score total. The file is the deliverable — do not replay the findings in chat.
|
|
203
240
|
|
|
204
241
|
## Exceptions
|
|
205
242
|
|
|
@@ -207,7 +244,7 @@ generatedAt: <YYYY-MM-DD HH:MM>
|
|
|
207
244
|
|---|---|
|
|
208
245
|
| `preflight` reports `Okstra preflight: failed` | retry with `--cwd <dir>`; if that also fails, tell the user to run `/okstra-setup` first and stop |
|
|
209
246
|
| `unknown command: code-review` | the `okstra` binary predates this skill — tell the user to update it (`npm i -g okstra@latest`) and stop |
|
|
210
|
-
| `worktreePath` is empty | not a hard stop and not a missing stage: run git in `projectRoot` against the stage's `branch` ref and
|
|
247
|
+
| `worktreePath` is empty | not a hard stop and not a missing stage: run git in `projectRoot` against the stage's `branch` ref, and have reviewers read whole files with `git -C <projectRoot> show <headCommit>:<path>` |
|
|
211
248
|
| `baseCommit` is not an ancestor of `headCommit` — `git -C <workdir> rev-list --count <headCommit>..<baseCommit>` returns a **non-zero** count, meaning a rebase or squash rewrote the history the stage was recorded against, and `<baseCommit>..<headCommit>` would drag predecessor work in backwards | code-review is read-only, so do not force a reconcile. Offer two options: review against the branch's current tip, or run `okstra git-reconcile` first and retry |
|
|
212
249
|
| the diff is empty | dispatch no reviewers; write the "no changes" report to `reviewPath` and stop. It is a normal report, not a free-form note: the same frontmatter, `## Coverage` reading "0 changed files → 0 cells on every axis, 0 files excluded" plus the applied packs, every severity section omitted, and `## Score` carrying the table with its single total row reading 0 |
|
|
213
250
|
| the census is large | never truncate — report the cell count and confirm before dispatching |
|
|
@@ -215,9 +252,10 @@ generatedAt: <YYYY-MM-DD HH:MM>
|
|
|
215
252
|
|
|
216
253
|
## Principles
|
|
217
254
|
|
|
218
|
-
- **Stay in the diff.** Every finding cites a line this diff changed. A cell whose only wart sits on untouched lines verdicts `clean` — pre-existing issues are not this change's problem.
|
|
255
|
+
- **Stay in the diff.** Every finding cites a line this diff changed, and Step 3.6 checks it. A cell whose only wart sits on untouched lines verdicts `clean` — pre-existing issues are not this change's problem.
|
|
219
256
|
- **Don't manufacture findings.** A census fully verdicted `clean` is a valid, useful result.
|
|
220
257
|
- **No finding without a fix.** Readability findings carry a pseudocode sketch; naming findings carry a concrete alternative name.
|
|
221
258
|
- **The census is law.** A reviewer that rebuilds its own worklist reintroduces exactly the run-to-run variance this skill exists to kill.
|
|
222
259
|
- **Every cell gets a verdict.** `clean` is a result, not an omission; the audit treats a gap as unfinished work.
|
|
223
|
-
- **
|
|
260
|
+
- **Only a checked claim scores.** A finding reaches the score after its citation is on a changed line and its failure holds against the code; everything else is `Parked` or `Rejected`, with its reason.
|
|
261
|
+
- **The report prose follows `Report language`.** Paths, identifiers, rule names, and quoted code stay verbatim.
|