okstra 0.169.1 → 0.170.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/docs/architecture.md +17 -1
  2. package/docs/cli.md +12 -2
  3. package/docs/for-ai/skills/okstra-setup.md +8 -0
  4. package/docs/project-structure-overview.md +3 -1
  5. package/package.json +1 -1
  6. package/runtime/BUILD.json +2 -2
  7. package/runtime/prompts/duties/acceptance-critic.md +25 -5
  8. package/runtime/prompts/duties/acceptance-verifier.md +25 -5
  9. package/runtime/prompts/duties/analysis-worker.md +25 -5
  10. package/runtime/prompts/duties/code-reviewer.md +25 -5
  11. package/runtime/prompts/duties/common.md +15 -11
  12. package/runtime/prompts/duties/diagnosis-worker.md +44 -0
  13. package/runtime/prompts/duties/discovery-worker.md +44 -0
  14. package/runtime/prompts/duties/implementation-executor.md +25 -5
  15. package/runtime/prompts/duties/implementation-verifier.md +25 -5
  16. package/runtime/prompts/duties/lead.md +25 -5
  17. package/runtime/prompts/duties/planning-worker.md +44 -0
  18. package/runtime/prompts/duties/report-writer.md +25 -5
  19. package/runtime/prompts/duties/reverification-worker.md +25 -5
  20. package/runtime/prompts/duties/schedule-verifier.md +25 -5
  21. package/runtime/prompts/duties/scope-critic.md +25 -5
  22. package/runtime/prompts/duties/translator.md +25 -5
  23. package/runtime/prompts/lead/plan-body-verification.md +7 -2
  24. package/runtime/prompts/lead/report-writer.md +1 -1
  25. package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -1
  26. package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
  27. package/runtime/prompts/profiles/final-verification.md +1 -1
  28. package/runtime/prompts/profiles/implementation-planning.md +2 -2
  29. package/runtime/python/okstra_ctl/agent_invocation.py +92 -3
  30. package/runtime/python/okstra_ctl/agent_prompt_cli.py +30 -0
  31. package/runtime/python/okstra_ctl/cmux.py +36 -19
  32. package/runtime/python/okstra_ctl/dispatch_core.py +92 -22
  33. package/runtime/python/okstra_ctl/dispatch_state.py +143 -9
  34. package/runtime/python/okstra_ctl/doctor.py +31 -0
  35. package/runtime/python/okstra_ctl/plan_derivations.py +94 -0
  36. package/runtime/python/okstra_ctl/plan_items_cli.py +131 -4
  37. package/runtime/python/okstra_ctl/run.py +7 -1
  38. package/runtime/python/okstra_ctl/schema_excerpt.py +34 -0
  39. package/runtime/python/okstra_ctl/verdict_blocks.py +17 -0
  40. package/runtime/python/okstra_ctl/worker_prompt_policy.py +12 -1
  41. package/runtime/python/okstra_project/resolver.py +34 -0
  42. package/runtime/schemas/final-report-v2.0.schema.json +5 -0
  43. package/runtime/skills/okstra-setup/references/project-config.md +38 -0
  44. package/runtime/validators/lib/fixtures.sh +9 -1
  45. package/runtime/validators/validate-run.py +108 -2
@@ -108,6 +108,40 @@ def resolve_architecture(project_root: Path | str) -> str:
108
108
  return style if style in _ARCHITECTURE_STYLES else "none"
109
109
 
110
110
 
111
+ def resolve_review_rule_packs(project_root: Path | str) -> tuple[str, ...]:
112
+ """Return the project's declared review rule pack paths, else ``()``.
113
+
114
+ A review rule pack is a project's own review standard — the file a phase
115
+ reads before judging a diff. It used to reach a run only when the task
116
+ brief cited its exact path, so a team standard applied or not depending on
117
+ who wrote the brief. Declaring it here makes it apply to every run in the
118
+ project; the brief citation still works and the two are a union.
119
+
120
+ Mirrors resolve_architecture's failure posture: any read/parse problem or a
121
+ non-list value degrades to "none declared" rather than raising inside a run.
122
+ An entry that is not an absolute path is dropped, because workers run with
123
+ a worktree as cwd — a relative path would resolve against a tree that does
124
+ not contain the pack, and would mean a different file per worker.
125
+ """
126
+ try:
127
+ payload = json.loads(project_json_path(Path(project_root)).read_text(encoding="utf-8"))
128
+ except (OSError, ValueError):
129
+ return ()
130
+ if not isinstance(payload, dict):
131
+ return ()
132
+ declared = payload.get("reviewRulePacks")
133
+ if not isinstance(declared, list):
134
+ return ()
135
+ packs = []
136
+ for entry in declared:
137
+ if not isinstance(entry, str) or not entry.strip():
138
+ continue
139
+ candidate = Path(entry.strip()).expanduser()
140
+ if candidate.is_absolute():
141
+ packs.append(str(candidate))
142
+ return tuple(dict.fromkeys(packs))
143
+
144
+
111
145
  def upsert_project_json(project_root: Path, project_id: str, *,
112
146
  now: Optional[str] = None) -> dict:
113
147
  """project.json 을 읽거나 새로 만든다.
@@ -7746,6 +7746,11 @@
7746
7746
  },
7747
7747
  "note": {
7748
7748
  "type": "string"
7749
+ },
7750
+ "round": {
7751
+ "description": "The plan-body verification round this verdict was cast in. A self-fix round rewrites the plan after a verification round, so a verdict whose round is at or before `selfFixRoundsApplied` judged text that has since changed. Without it there is no way to tell how many rewrites a surviving verdict predates, and a gate can pass on judgements two generations stale.",
7752
+ "type": "integer",
7753
+ "minimum": 1
7749
7754
  }
7750
7755
  }
7751
7756
  }
@@ -252,3 +252,41 @@ rules from advisory to a binding planning + verification constraint:
252
252
 
253
253
  The full two-layer model lives in `docs/architecture.md` § Project
254
254
  self-registration.
255
+
256
+ ## G. Project review rule packs (`reviewRulePacks`)
257
+
258
+ `reviewRulePacks` declares the review standards this project's phases read
259
+ before they judge a plan or a diff — a team PR-review skill's `SKILL.md`, for
260
+ example. Without it a pack reaches a run only when the task brief cites its
261
+ exact path, so whether the team standard applied came down to who wrote the
262
+ brief. A declared pack applies to every run in the project; the brief citation
263
+ keeps working, and the two channels are a union.
264
+
265
+ `okstra setup` never writes it. Hand-add it to `project.json` and the upsert
266
+ preserves it:
267
+
268
+ ```json
269
+ {
270
+ "reviewRulePacks": [
271
+ "/Users/me/.claude/skills/team-pr-reviewer/SKILL.md"
272
+ ]
273
+ }
274
+ ```
275
+
276
+ - **Absolute paths only.** A worker runs with a task worktree as its cwd, so a
277
+ relative path names a different file there — relative entries are dropped by
278
+ `scripts/okstra_project/resolver.py::resolve_review_rule_packs`, which also
279
+ falls back to "none declared" on an unreadable or malformed `project.json`.
280
+ A leading `~` is expanded.
281
+ - **Who reads it.** `implementation-planning` (to plan away the findings before
282
+ the code exists), the implementation executor's coding-conventions preflight,
283
+ and the static review passes of the implementation verifier and
284
+ `final-verification`. Each reads only the declared file and the
285
+ `references/*.md` files it directly names — never a parent directory or a
286
+ host skill catalog.
287
+ - **What is machine-checked.** `okstra doctor --phase <phase>` fails its
288
+ `review rule packs` check when a declared path is not a readable file,
289
+ because a stale path otherwise costs the whole pack in silence: the phase
290
+ records that it read no rules and the run still passes. Whether a pack that
291
+ does resolve was actually read and applied stays the phase's own
292
+ `project-review-rules:` record — no machine check reads a worker's reasoning.
@@ -544,6 +544,14 @@ if WORKSPACE_ROOT:
544
544
  link_agent_dispatch_result,
545
545
  record_verified_agent_dispatch,
546
546
  )
547
+ # The analysis duty depends on the task type, exactly as it does in a real
548
+ # run — a fixture that assumed one audience for every phase would be
549
+ # rejected by the manifest's own allowlist.
550
+ from okstra_ctl.worker_prompt_policy import ANALYSIS_DUTY_BY_TASK_TYPE
551
+
552
+ analysis_audience = ANALYSIS_DUTY_BY_TASK_TYPE.get(
553
+ str(run_manifest.get("taskType") or ""), "analysis-worker"
554
+ )
547
555
 
548
556
  lead_record = record_verified_agent_dispatch(
549
557
  project_root=project_root,
@@ -580,7 +588,7 @@ if WORKSPACE_ROOT:
580
588
  worker_id=worker_id,
581
589
  audience=(
582
590
  "report-writer" if worker_id == "report-writer"
583
- else "analysis-worker"
591
+ else analysis_audience
584
592
  ),
585
593
  assignment_ref=assignment_ref,
586
594
  purpose=None,
@@ -5351,6 +5351,76 @@ def _validate_round_recorded_verdicts(data: dict, failures: list[str]) -> None:
5351
5351
  )
5352
5352
 
5353
5353
 
5354
+ def _validate_verdict_rounds_outlive_self_fix(
5355
+ data: dict,
5356
+ failures: list[str],
5357
+ ) -> None:
5358
+ """A verdict must judge the plan the gate is about to pass.
5359
+
5360
+ Rounds interleave with rewrites: round 1, self-fix 1, round 2, self-fix 2 …
5361
+ so a verdict cast in round R judged the text as it stood after self-fix
5362
+ R-1. If any self-fix ran afterwards — `selfFixRoundsApplied >= R` — that
5363
+ text has changed and the verdict is stale by construction. No semantic
5364
+ analysis is needed to know that; the arithmetic settles it.
5365
+
5366
+ The sibling `_validate_verdicts_match_current_subjects` cannot see this. It
5367
+ compares each row's own recorded `subject`, which catches a positional shift
5368
+ but not the case that matters here: an item whose own wording never changed
5369
+ while the stage it points at was rewritten under it. Observed on a real run
5370
+ — the gate read `passed-with-dissent` with zero blockers, and re-running one
5371
+ round flipped 3 of 27 items to `majority-disagree`, all correctness-critical,
5372
+ because their surviving verdicts predated two self-fix rounds.
5373
+
5374
+ Scoped to items this run verified: a `carriedForwardFromSeq` row belongs to
5375
+ the prior run's record and is judged by that run's seq, not this one's
5376
+ rounds.
5377
+ """
5378
+ ip = data.get("implementationPlanning")
5379
+ if not isinstance(ip, dict):
5380
+ return
5381
+ pbv = ip.get("planBodyVerification")
5382
+ if not isinstance(pbv, dict):
5383
+ return
5384
+ applied = pbv.get("selfFixRoundsApplied")
5385
+ if not isinstance(applied, int) or applied < 1:
5386
+ # With no rewrite after any round there is nothing a verdict can be
5387
+ # stale against, and an unstamped row is then simply unremarkable.
5388
+ return
5389
+
5390
+ stale: list[str] = []
5391
+ unstamped: list[str] = []
5392
+ for item in pbv.get("planItems") or []:
5393
+ if not isinstance(item, dict) or item.get("carriedForwardFromSeq"):
5394
+ continue
5395
+ item_id = str(item.get("id") or "").strip()
5396
+ for verdict in item.get("verdicts") or []:
5397
+ if not isinstance(verdict, dict):
5398
+ continue
5399
+ round_number = verdict.get("round")
5400
+ if not isinstance(round_number, int) or isinstance(round_number, bool):
5401
+ unstamped.append(item_id)
5402
+ elif round_number <= applied:
5403
+ stale.append(item_id)
5404
+ if unstamped:
5405
+ failures.append(
5406
+ f"final-report data.json: plan item(s) {sorted(set(unstamped))} carry "
5407
+ f"a verdict with no `round`, and {applied} self-fix round(s) rewrote "
5408
+ "the plan. Without the round there is no way to tell whether the "
5409
+ "verdict judged the current text or a version two rewrites old. "
5410
+ "Re-record the round's votes with `okstra plan-items apply-verdicts "
5411
+ "--round <N>`."
5412
+ )
5413
+ if stale:
5414
+ failures.append(
5415
+ f"final-report data.json: plan item(s) {sorted(set(stale))} carry a "
5416
+ f"verdict from a round at or before self-fix round {applied}, so the "
5417
+ "text they judged has since been rewritten. The gate is computed "
5418
+ "from these votes, so passing on them declares a plan verified that "
5419
+ "nobody verified. Re-verify those items in a round after the last "
5420
+ 'self-fix (plan-body-verification.md §"Round protocol" step 7).'
5421
+ )
5422
+
5423
+
5354
5424
  def _validate_verdicts_match_current_subjects(
5355
5425
  data: dict,
5356
5426
  failures: list[str],
@@ -5470,6 +5540,27 @@ def _plan_verify_result_workers(report_path: Path, task_type: str) -> set[str] |
5470
5540
  return workers
5471
5541
 
5472
5542
 
5543
+ def _plan_verify_seq_near_misses(report_path: Path, task_type: str) -> list[str]:
5544
+ """Plan-verify results in the directory that this run's seq filter excluded.
5545
+
5546
+ The seq comes from the report filename, and a run whose `reports` and
5547
+ `workerResults` sequences differ makes the other one look equally plausible
5548
+ to a lead naming the file by hand. Naming a near miss is the difference
5549
+ between "the file is missing" and "the file is there under a different seq"
5550
+ — the first sends a lead to re-dispatch two workers that already ran.
5551
+ """
5552
+ seq = _report_run_seq(report_path)
5553
+ if not seq:
5554
+ return []
5555
+ directory = report_path.parent.parent / "worker-results"
5556
+ matched = set(directory.glob(f"*-plan-verify-r*-{task_type}-{seq}.md"))
5557
+ return sorted(
5558
+ path.name
5559
+ for path in directory.glob(f"*-plan-verify-r*-{task_type}-*.md")
5560
+ if path not in matched
5561
+ )
5562
+
5563
+
5473
5564
  def _validate_plan_body_verdict_provenance(
5474
5565
  data: dict,
5475
5566
  report_path: Path,
@@ -5514,14 +5605,28 @@ def _validate_plan_body_verdict_provenance(
5514
5605
  "final-report data.json: planBodyVerification records verdicts from "
5515
5606
  f"{unbacked} but no matching plan-body reverify result file exists "
5516
5607
  f"under `runs/{task_type}/worker-results/` "
5517
- f"(expected `<worker>-plan-verify-r<N>-{task_type}-<seq>.md`). "
5518
- "A vote the gate is computed from MUST trace back to a dispatch "
5608
+ f"(globbed `*-plan-verify-r*-{task_type}-"
5609
+ f"{_report_run_seq(report_path) or '*'}.md`, where the seq is the "
5610
+ f"report's own — `final-report-{task_type}-<seq>` — not this run's "
5611
+ f"`workerResults` sequence)"
5612
+ + _near_miss_clause(report_path, task_type)
5613
+ + ". A vote the gate is computed from MUST trace back to a dispatch "
5519
5614
  "that actually returned — otherwise the round can be skipped and "
5520
5615
  'the gate still read `passed` (plan-body-verification.md §"Round '
5521
5616
  'protocol" step 3).'
5522
5617
  )
5523
5618
 
5524
5619
 
5620
+ def _near_miss_clause(report_path: Path, task_type: str) -> str:
5621
+ near = _plan_verify_seq_near_misses(report_path, task_type)
5622
+ if not near:
5623
+ return ""
5624
+ return (
5625
+ f"; the directory does hold {near}, which the seq filter excluded — "
5626
+ f"rename to this run's report seq rather than re-dispatching"
5627
+ )
5628
+
5629
+
5525
5630
  _UNIFORM_VERIFIER_MIN_ITEMS = 5
5526
5631
 
5527
5632
 
@@ -6332,6 +6437,7 @@ def validate_plan_body_section(
6332
6437
  _validate_aborted_gate_has_clarification(data, failures)
6333
6438
  _validate_round_recorded_verdicts(data, failures)
6334
6439
  _validate_verdicts_match_current_subjects(data, failures)
6440
+ _validate_verdict_rounds_outlive_self_fix(data, failures)
6335
6441
  _validate_plan_item_extraction_completeness(data, failures)
6336
6442
  _validate_plan_item_subject_substance(data, failures)
6337
6443
  _validate_plan_body_clarification_matching(data, failures)