okstra 0.169.1 → 0.170.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/architecture.md +17 -1
- package/docs/cli.md +12 -2
- package/docs/for-ai/skills/okstra-setup.md +8 -0
- package/docs/project-structure-overview.md +3 -1
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/duties/acceptance-critic.md +25 -5
- package/runtime/prompts/duties/acceptance-verifier.md +25 -5
- package/runtime/prompts/duties/analysis-worker.md +25 -5
- package/runtime/prompts/duties/code-reviewer.md +25 -5
- package/runtime/prompts/duties/common.md +15 -11
- package/runtime/prompts/duties/diagnosis-worker.md +44 -0
- package/runtime/prompts/duties/discovery-worker.md +44 -0
- package/runtime/prompts/duties/implementation-executor.md +25 -5
- package/runtime/prompts/duties/implementation-verifier.md +25 -5
- package/runtime/prompts/duties/lead.md +25 -5
- package/runtime/prompts/duties/planning-worker.md +44 -0
- package/runtime/prompts/duties/report-writer.md +25 -5
- package/runtime/prompts/duties/reverification-worker.md +25 -5
- package/runtime/prompts/duties/schedule-verifier.md +25 -5
- package/runtime/prompts/duties/scope-critic.md +25 -5
- package/runtime/prompts/duties/translator.md +25 -5
- package/runtime/prompts/lead/plan-body-verification.md +7 -2
- package/runtime/prompts/lead/report-writer.md +1 -1
- package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
- package/runtime/prompts/profiles/final-verification.md +1 -1
- package/runtime/prompts/profiles/implementation-planning.md +2 -2
- package/runtime/python/okstra_ctl/agent_invocation.py +92 -3
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +30 -0
- package/runtime/python/okstra_ctl/cmux.py +36 -19
- package/runtime/python/okstra_ctl/dispatch_core.py +92 -22
- package/runtime/python/okstra_ctl/dispatch_state.py +143 -9
- package/runtime/python/okstra_ctl/doctor.py +31 -0
- package/runtime/python/okstra_ctl/plan_derivations.py +94 -0
- package/runtime/python/okstra_ctl/plan_items_cli.py +131 -4
- package/runtime/python/okstra_ctl/run.py +7 -1
- package/runtime/python/okstra_ctl/schema_excerpt.py +34 -0
- package/runtime/python/okstra_ctl/verdict_blocks.py +17 -0
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +12 -1
- package/runtime/python/okstra_project/resolver.py +34 -0
- package/runtime/schemas/final-report-v2.0.schema.json +5 -0
- package/runtime/skills/okstra-setup/references/project-config.md +38 -0
- package/runtime/validators/lib/fixtures.sh +9 -1
- package/runtime/validators/validate-run.py +108 -2
|
@@ -108,6 +108,40 @@ def resolve_architecture(project_root: Path | str) -> str:
|
|
|
108
108
|
return style if style in _ARCHITECTURE_STYLES else "none"
|
|
109
109
|
|
|
110
110
|
|
|
111
|
+
def resolve_review_rule_packs(project_root: Path | str) -> tuple[str, ...]:
|
|
112
|
+
"""Return the project's declared review rule pack paths, else ``()``.
|
|
113
|
+
|
|
114
|
+
A review rule pack is a project's own review standard — the file a phase
|
|
115
|
+
reads before judging a diff. It used to reach a run only when the task
|
|
116
|
+
brief cited its exact path, so a team standard applied or not depending on
|
|
117
|
+
who wrote the brief. Declaring it here makes it apply to every run in the
|
|
118
|
+
project; the brief citation still works and the two are a union.
|
|
119
|
+
|
|
120
|
+
Mirrors resolve_architecture's failure posture: any read/parse problem or a
|
|
121
|
+
non-list value degrades to "none declared" rather than raising inside a run.
|
|
122
|
+
An entry that is not an absolute path is dropped, because workers run with
|
|
123
|
+
a worktree as cwd — a relative path would resolve against a tree that does
|
|
124
|
+
not contain the pack, and would mean a different file per worker.
|
|
125
|
+
"""
|
|
126
|
+
try:
|
|
127
|
+
payload = json.loads(project_json_path(Path(project_root)).read_text(encoding="utf-8"))
|
|
128
|
+
except (OSError, ValueError):
|
|
129
|
+
return ()
|
|
130
|
+
if not isinstance(payload, dict):
|
|
131
|
+
return ()
|
|
132
|
+
declared = payload.get("reviewRulePacks")
|
|
133
|
+
if not isinstance(declared, list):
|
|
134
|
+
return ()
|
|
135
|
+
packs = []
|
|
136
|
+
for entry in declared:
|
|
137
|
+
if not isinstance(entry, str) or not entry.strip():
|
|
138
|
+
continue
|
|
139
|
+
candidate = Path(entry.strip()).expanduser()
|
|
140
|
+
if candidate.is_absolute():
|
|
141
|
+
packs.append(str(candidate))
|
|
142
|
+
return tuple(dict.fromkeys(packs))
|
|
143
|
+
|
|
144
|
+
|
|
111
145
|
def upsert_project_json(project_root: Path, project_id: str, *,
|
|
112
146
|
now: Optional[str] = None) -> dict:
|
|
113
147
|
"""project.json 을 읽거나 새로 만든다.
|
|
@@ -7746,6 +7746,11 @@
|
|
|
7746
7746
|
},
|
|
7747
7747
|
"note": {
|
|
7748
7748
|
"type": "string"
|
|
7749
|
+
},
|
|
7750
|
+
"round": {
|
|
7751
|
+
"description": "The plan-body verification round this verdict was cast in. A self-fix round rewrites the plan after a verification round, so a verdict whose round is at or before `selfFixRoundsApplied` judged text that has since changed. Without it there is no way to tell how many rewrites a surviving verdict predates, and a gate can pass on judgements two generations stale.",
|
|
7752
|
+
"type": "integer",
|
|
7753
|
+
"minimum": 1
|
|
7749
7754
|
}
|
|
7750
7755
|
}
|
|
7751
7756
|
}
|
|
@@ -252,3 +252,41 @@ rules from advisory to a binding planning + verification constraint:
|
|
|
252
252
|
|
|
253
253
|
The full two-layer model lives in `docs/architecture.md` § Project
|
|
254
254
|
self-registration.
|
|
255
|
+
|
|
256
|
+
## G. Project review rule packs (`reviewRulePacks`)
|
|
257
|
+
|
|
258
|
+
`reviewRulePacks` declares the review standards this project's phases read
|
|
259
|
+
before they judge a plan or a diff — a team PR-review skill's `SKILL.md`, for
|
|
260
|
+
example. Without it a pack reaches a run only when the task brief cites its
|
|
261
|
+
exact path, so whether the team standard applied came down to who wrote the
|
|
262
|
+
brief. A declared pack applies to every run in the project; the brief citation
|
|
263
|
+
keeps working, and the two channels are a union.
|
|
264
|
+
|
|
265
|
+
`okstra setup` never writes it. Hand-add it to `project.json` and the upsert
|
|
266
|
+
preserves it:
|
|
267
|
+
|
|
268
|
+
```json
|
|
269
|
+
{
|
|
270
|
+
"reviewRulePacks": [
|
|
271
|
+
"/Users/me/.claude/skills/team-pr-reviewer/SKILL.md"
|
|
272
|
+
]
|
|
273
|
+
}
|
|
274
|
+
```
|
|
275
|
+
|
|
276
|
+
- **Absolute paths only.** A worker runs with a task worktree as its cwd, so a
|
|
277
|
+
relative path names a different file there — relative entries are dropped by
|
|
278
|
+
`scripts/okstra_project/resolver.py::resolve_review_rule_packs`, which also
|
|
279
|
+
falls back to "none declared" on an unreadable or malformed `project.json`.
|
|
280
|
+
A leading `~` is expanded.
|
|
281
|
+
- **Who reads it.** `implementation-planning` (to plan away the findings before
|
|
282
|
+
the code exists), the implementation executor's coding-conventions preflight,
|
|
283
|
+
and the static review passes of the implementation verifier and
|
|
284
|
+
`final-verification`. Each reads only the declared file and the
|
|
285
|
+
`references/*.md` files it directly names — never a parent directory or a
|
|
286
|
+
host skill catalog.
|
|
287
|
+
- **What is machine-checked.** `okstra doctor --phase <phase>` fails its
|
|
288
|
+
`review rule packs` check when a declared path is not a readable file,
|
|
289
|
+
because a stale path otherwise costs the whole pack in silence: the phase
|
|
290
|
+
records that it read no rules and the run still passes. Whether a pack that
|
|
291
|
+
does resolve was actually read and applied stays the phase's own
|
|
292
|
+
`project-review-rules:` record — no machine check reads a worker's reasoning.
|
|
@@ -544,6 +544,14 @@ if WORKSPACE_ROOT:
|
|
|
544
544
|
link_agent_dispatch_result,
|
|
545
545
|
record_verified_agent_dispatch,
|
|
546
546
|
)
|
|
547
|
+
# The analysis duty depends on the task type, exactly as it does in a real
|
|
548
|
+
# run — a fixture that assumed one audience for every phase would be
|
|
549
|
+
# rejected by the manifest's own allowlist.
|
|
550
|
+
from okstra_ctl.worker_prompt_policy import ANALYSIS_DUTY_BY_TASK_TYPE
|
|
551
|
+
|
|
552
|
+
analysis_audience = ANALYSIS_DUTY_BY_TASK_TYPE.get(
|
|
553
|
+
str(run_manifest.get("taskType") or ""), "analysis-worker"
|
|
554
|
+
)
|
|
547
555
|
|
|
548
556
|
lead_record = record_verified_agent_dispatch(
|
|
549
557
|
project_root=project_root,
|
|
@@ -580,7 +588,7 @@ if WORKSPACE_ROOT:
|
|
|
580
588
|
worker_id=worker_id,
|
|
581
589
|
audience=(
|
|
582
590
|
"report-writer" if worker_id == "report-writer"
|
|
583
|
-
else
|
|
591
|
+
else analysis_audience
|
|
584
592
|
),
|
|
585
593
|
assignment_ref=assignment_ref,
|
|
586
594
|
purpose=None,
|
|
@@ -5351,6 +5351,76 @@ def _validate_round_recorded_verdicts(data: dict, failures: list[str]) -> None:
|
|
|
5351
5351
|
)
|
|
5352
5352
|
|
|
5353
5353
|
|
|
5354
|
+
def _validate_verdict_rounds_outlive_self_fix(
|
|
5355
|
+
data: dict,
|
|
5356
|
+
failures: list[str],
|
|
5357
|
+
) -> None:
|
|
5358
|
+
"""A verdict must judge the plan the gate is about to pass.
|
|
5359
|
+
|
|
5360
|
+
Rounds interleave with rewrites: round 1, self-fix 1, round 2, self-fix 2 …
|
|
5361
|
+
so a verdict cast in round R judged the text as it stood after self-fix
|
|
5362
|
+
R-1. If any self-fix ran afterwards — `selfFixRoundsApplied >= R` — that
|
|
5363
|
+
text has changed and the verdict is stale by construction. No semantic
|
|
5364
|
+
analysis is needed to know that; the arithmetic settles it.
|
|
5365
|
+
|
|
5366
|
+
The sibling `_validate_verdicts_match_current_subjects` cannot see this. It
|
|
5367
|
+
compares each row's own recorded `subject`, which catches a positional shift
|
|
5368
|
+
but not the case that matters here: an item whose own wording never changed
|
|
5369
|
+
while the stage it points at was rewritten under it. Observed on a real run
|
|
5370
|
+
— the gate read `passed-with-dissent` with zero blockers, and re-running one
|
|
5371
|
+
round flipped 3 of 27 items to `majority-disagree`, all correctness-critical,
|
|
5372
|
+
because their surviving verdicts predated two self-fix rounds.
|
|
5373
|
+
|
|
5374
|
+
Scoped to items this run verified: a `carriedForwardFromSeq` row belongs to
|
|
5375
|
+
the prior run's record and is judged by that run's seq, not this one's
|
|
5376
|
+
rounds.
|
|
5377
|
+
"""
|
|
5378
|
+
ip = data.get("implementationPlanning")
|
|
5379
|
+
if not isinstance(ip, dict):
|
|
5380
|
+
return
|
|
5381
|
+
pbv = ip.get("planBodyVerification")
|
|
5382
|
+
if not isinstance(pbv, dict):
|
|
5383
|
+
return
|
|
5384
|
+
applied = pbv.get("selfFixRoundsApplied")
|
|
5385
|
+
if not isinstance(applied, int) or applied < 1:
|
|
5386
|
+
# With no rewrite after any round there is nothing a verdict can be
|
|
5387
|
+
# stale against, and an unstamped row is then simply unremarkable.
|
|
5388
|
+
return
|
|
5389
|
+
|
|
5390
|
+
stale: list[str] = []
|
|
5391
|
+
unstamped: list[str] = []
|
|
5392
|
+
for item in pbv.get("planItems") or []:
|
|
5393
|
+
if not isinstance(item, dict) or item.get("carriedForwardFromSeq"):
|
|
5394
|
+
continue
|
|
5395
|
+
item_id = str(item.get("id") or "").strip()
|
|
5396
|
+
for verdict in item.get("verdicts") or []:
|
|
5397
|
+
if not isinstance(verdict, dict):
|
|
5398
|
+
continue
|
|
5399
|
+
round_number = verdict.get("round")
|
|
5400
|
+
if not isinstance(round_number, int) or isinstance(round_number, bool):
|
|
5401
|
+
unstamped.append(item_id)
|
|
5402
|
+
elif round_number <= applied:
|
|
5403
|
+
stale.append(item_id)
|
|
5404
|
+
if unstamped:
|
|
5405
|
+
failures.append(
|
|
5406
|
+
f"final-report data.json: plan item(s) {sorted(set(unstamped))} carry "
|
|
5407
|
+
f"a verdict with no `round`, and {applied} self-fix round(s) rewrote "
|
|
5408
|
+
"the plan. Without the round there is no way to tell whether the "
|
|
5409
|
+
"verdict judged the current text or a version two rewrites old. "
|
|
5410
|
+
"Re-record the round's votes with `okstra plan-items apply-verdicts "
|
|
5411
|
+
"--round <N>`."
|
|
5412
|
+
)
|
|
5413
|
+
if stale:
|
|
5414
|
+
failures.append(
|
|
5415
|
+
f"final-report data.json: plan item(s) {sorted(set(stale))} carry a "
|
|
5416
|
+
f"verdict from a round at or before self-fix round {applied}, so the "
|
|
5417
|
+
"text they judged has since been rewritten. The gate is computed "
|
|
5418
|
+
"from these votes, so passing on them declares a plan verified that "
|
|
5419
|
+
"nobody verified. Re-verify those items in a round after the last "
|
|
5420
|
+
'self-fix (plan-body-verification.md §"Round protocol" step 7).'
|
|
5421
|
+
)
|
|
5422
|
+
|
|
5423
|
+
|
|
5354
5424
|
def _validate_verdicts_match_current_subjects(
|
|
5355
5425
|
data: dict,
|
|
5356
5426
|
failures: list[str],
|
|
@@ -5470,6 +5540,27 @@ def _plan_verify_result_workers(report_path: Path, task_type: str) -> set[str] |
|
|
|
5470
5540
|
return workers
|
|
5471
5541
|
|
|
5472
5542
|
|
|
5543
|
+
def _plan_verify_seq_near_misses(report_path: Path, task_type: str) -> list[str]:
|
|
5544
|
+
"""Plan-verify results in the directory that this run's seq filter excluded.
|
|
5545
|
+
|
|
5546
|
+
The seq comes from the report filename, and a run whose `reports` and
|
|
5547
|
+
`workerResults` sequences differ makes the other one look equally plausible
|
|
5548
|
+
to a lead naming the file by hand. Naming a near miss is the difference
|
|
5549
|
+
between "the file is missing" and "the file is there under a different seq"
|
|
5550
|
+
— the first sends a lead to re-dispatch two workers that already ran.
|
|
5551
|
+
"""
|
|
5552
|
+
seq = _report_run_seq(report_path)
|
|
5553
|
+
if not seq:
|
|
5554
|
+
return []
|
|
5555
|
+
directory = report_path.parent.parent / "worker-results"
|
|
5556
|
+
matched = set(directory.glob(f"*-plan-verify-r*-{task_type}-{seq}.md"))
|
|
5557
|
+
return sorted(
|
|
5558
|
+
path.name
|
|
5559
|
+
for path in directory.glob(f"*-plan-verify-r*-{task_type}-*.md")
|
|
5560
|
+
if path not in matched
|
|
5561
|
+
)
|
|
5562
|
+
|
|
5563
|
+
|
|
5473
5564
|
def _validate_plan_body_verdict_provenance(
|
|
5474
5565
|
data: dict,
|
|
5475
5566
|
report_path: Path,
|
|
@@ -5514,14 +5605,28 @@ def _validate_plan_body_verdict_provenance(
|
|
|
5514
5605
|
"final-report data.json: planBodyVerification records verdicts from "
|
|
5515
5606
|
f"{unbacked} but no matching plan-body reverify result file exists "
|
|
5516
5607
|
f"under `runs/{task_type}/worker-results/` "
|
|
5517
|
-
f"(
|
|
5518
|
-
"
|
|
5608
|
+
f"(globbed `*-plan-verify-r*-{task_type}-"
|
|
5609
|
+
f"{_report_run_seq(report_path) or '*'}.md`, where the seq is the "
|
|
5610
|
+
f"report's own — `final-report-{task_type}-<seq>` — not this run's "
|
|
5611
|
+
f"`workerResults` sequence)"
|
|
5612
|
+
+ _near_miss_clause(report_path, task_type)
|
|
5613
|
+
+ ". A vote the gate is computed from MUST trace back to a dispatch "
|
|
5519
5614
|
"that actually returned — otherwise the round can be skipped and "
|
|
5520
5615
|
'the gate still read `passed` (plan-body-verification.md §"Round '
|
|
5521
5616
|
'protocol" step 3).'
|
|
5522
5617
|
)
|
|
5523
5618
|
|
|
5524
5619
|
|
|
5620
|
+
def _near_miss_clause(report_path: Path, task_type: str) -> str:
|
|
5621
|
+
near = _plan_verify_seq_near_misses(report_path, task_type)
|
|
5622
|
+
if not near:
|
|
5623
|
+
return ""
|
|
5624
|
+
return (
|
|
5625
|
+
f"; the directory does hold {near}, which the seq filter excluded — "
|
|
5626
|
+
f"rename to this run's report seq rather than re-dispatching"
|
|
5627
|
+
)
|
|
5628
|
+
|
|
5629
|
+
|
|
5525
5630
|
_UNIFORM_VERIFIER_MIN_ITEMS = 5
|
|
5526
5631
|
|
|
5527
5632
|
|
|
@@ -6332,6 +6437,7 @@ def validate_plan_body_section(
|
|
|
6332
6437
|
_validate_aborted_gate_has_clarification(data, failures)
|
|
6333
6438
|
_validate_round_recorded_verdicts(data, failures)
|
|
6334
6439
|
_validate_verdicts_match_current_subjects(data, failures)
|
|
6440
|
+
_validate_verdict_rounds_outlive_self_fix(data, failures)
|
|
6335
6441
|
_validate_plan_item_extraction_completeness(data, failures)
|
|
6336
6442
|
_validate_plan_item_subject_substance(data, failures)
|
|
6337
6443
|
_validate_plan_body_clarification_matching(data, failures)
|