okstra 0.159.0 → 0.161.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/README.md +1 -1
  2. package/docs/architecture/storage-model.md +2 -0
  3. package/docs/architecture.md +2 -1
  4. package/docs/cli.md +8 -3
  5. package/docs/for-ai/README.md +2 -2
  6. package/docs/for-ai/skills/okstra-inspect.md +3 -0
  7. package/docs/for-ai/skills/okstra-run.md +2 -1
  8. package/docs/for-ai/skills/okstra-user-response.md +5 -5
  9. package/docs/project-structure-overview.md +5 -1
  10. package/docs/task-process/implementation.md +28 -0
  11. package/package.json +1 -1
  12. package/runtime/BUILD.json +2 -2
  13. package/runtime/bin/okstra-claude-exec.sh +4 -1
  14. package/runtime/prompts/host-orchestration/README.md +18 -0
  15. package/runtime/prompts/host-orchestration/implementation.md +57 -0
  16. package/runtime/prompts/launch.template.md +10 -1
  17. package/runtime/prompts/lead/adapters/claude-code.md +1 -1
  18. package/runtime/prompts/lead/adapters/cmux.md +67 -0
  19. package/runtime/prompts/lead/context-loader.md +5 -2
  20. package/runtime/prompts/lead/convergence.md +3 -1
  21. package/runtime/prompts/lead/plan-body-verification.md +21 -2
  22. package/runtime/prompts/lead/team-contract.md +2 -1
  23. package/runtime/prompts/profiles/_clarification-recommendation.md +11 -1
  24. package/runtime/prompts/profiles/_common-contract.md +3 -1
  25. package/runtime/prompts/profiles/implementation-planning.md +2 -0
  26. package/runtime/prompts/profiles/requirements-discovery.md +1 -1
  27. package/runtime/prompts/wizard/prompts.ko.json +3 -0
  28. package/runtime/python/okstra_ctl/clarification_items.py +9 -0
  29. package/runtime/python/okstra_ctl/cmux.py +531 -0
  30. package/runtime/python/okstra_ctl/codex_dispatch.py +6 -6
  31. package/runtime/python/okstra_ctl/convergence.py +168 -11
  32. package/runtime/python/okstra_ctl/dispatch_core.py +76 -7
  33. package/runtime/python/okstra_ctl/dispatch_state.py +16 -0
  34. package/runtime/python/okstra_ctl/error_issue.py +640 -0
  35. package/runtime/python/okstra_ctl/error_report.py +56 -0
  36. package/runtime/python/okstra_ctl/error_zip.py +23 -10
  37. package/runtime/python/okstra_ctl/incremental_scope.py +159 -19
  38. package/runtime/python/okstra_ctl/initial_prompt_materialization.py +18 -5
  39. package/runtime/python/okstra_ctl/issue_signals.py +186 -0
  40. package/runtime/python/okstra_ctl/lead_runtime.py +30 -2
  41. package/runtime/python/okstra_ctl/paths.py +38 -0
  42. package/runtime/python/okstra_ctl/plan_items_cli.py +167 -3
  43. package/runtime/python/okstra_ctl/profile_show.py +134 -0
  44. package/runtime/python/okstra_ctl/recap.py +63 -0
  45. package/runtime/python/okstra_ctl/render.py +7 -2
  46. package/runtime/python/okstra_ctl/render_final_report.py +7 -22
  47. package/runtime/python/okstra_ctl/report_translation.py +4 -0
  48. package/runtime/python/okstra_ctl/report_views.py +7 -3
  49. package/runtime/python/okstra_ctl/run.py +54 -3
  50. package/runtime/python/okstra_ctl/run_audit.py +477 -0
  51. package/runtime/python/okstra_ctl/team.py +50 -11
  52. package/runtime/python/okstra_ctl/user_response.py +25 -10
  53. package/runtime/python/okstra_ctl/verdict_blocks.py +183 -0
  54. package/runtime/python/okstra_ctl/wizard.py +64 -10
  55. package/runtime/python/okstra_ctl/worker_audit_check.py +44 -0
  56. package/runtime/python/okstra_ctl/worker_audit_ledger.py +207 -0
  57. package/runtime/python/okstra_ctl/worker_heartbeat.py +9 -3
  58. package/runtime/python/okstra_ctl/worker_liveness.py +81 -9
  59. package/runtime/schemas/final-report-v1.0.schema.json +14 -0
  60. package/runtime/schemas/final-report-v2.0.schema.json +51 -1
  61. package/runtime/skills/okstra-inspect/SKILL.md +3 -1
  62. package/runtime/skills/okstra-inspect/facets/error-issue.md +77 -0
  63. package/runtime/skills/okstra-inspect/facets/run-audit.md +34 -0
  64. package/runtime/skills/okstra-run/SKILL.md +28 -10
  65. package/runtime/skills/okstra-user-response/SKILL.md +18 -18
  66. package/runtime/templates/reports/final-report.template.md +4 -0
  67. package/runtime/templates/reports/html/i18n/en.json +5 -1
  68. package/runtime/templates/reports/html/i18n/ko.json +5 -1
  69. package/runtime/templates/reports/html/macros/forms.html +15 -0
  70. package/runtime/templates/reports/html/tasks/implementation-planning.template.html +1 -0
  71. package/runtime/templates/reports/i18n/en.json +2 -0
  72. package/runtime/validators/validate-run.py +267 -208
  73. package/runtime/validators/validate-workflow.sh +6 -0
  74. package/runtime/validators/validate_session_conformance.py +135 -31
  75. package/src/cli-registry.mjs +34 -0
  76. package/src/commands/execute/incremental-scope.mjs +10 -0
  77. package/src/commands/execute/worker-audit-check.mjs +35 -0
  78. package/src/commands/inspect/error-issue.mjs +27 -0
  79. package/src/commands/inspect/profile-show.mjs +29 -0
  80. package/src/commands/inspect/run-audit.mjs +26 -0
@@ -1,4 +1,9 @@
1
- """Neutral okstra team CLI for external tmux-pane worker dispatch."""
1
+ """Neutral okstra team CLI for pane-backed worker dispatch.
2
+
3
+ Outside cmux this is the external lead's door onto tmux panes. Under cmux it
4
+ is every lead's door onto cmux surfaces, because okstra owns the panes there
5
+ rather than the host. Which backend a run uses is read from its run manifest.
6
+ """
2
7
  from __future__ import annotations
3
8
 
4
9
  import argparse
@@ -7,8 +12,10 @@ import sys
7
12
  from pathlib import Path
8
13
  from typing import Any, Mapping, Sequence
9
14
 
15
+ from . import cmux
10
16
  from . import tmux
11
17
  from .dispatch_core import (
18
+ BACKEND_CMUX_PANE,
12
19
  BACKEND_TMUX_PANE,
13
20
  DispatchError,
14
21
  DispatchPlan,
@@ -51,7 +58,7 @@ def _parser() -> argparse.ArgumentParser:
51
58
 
52
59
 
53
60
  def _add_dispatch_parser(sub) -> None:
54
- parser = sub.add_parser("dispatch", help="dispatch tmux-pane workers")
61
+ parser = sub.add_parser("dispatch", help="dispatch pane-backed workers")
55
62
  _add_run_args(parser)
56
63
  parser.add_argument("--workers", default="")
57
64
  parser.add_argument("--jobs-file", default="")
@@ -61,7 +68,7 @@ def _add_dispatch_parser(sub) -> None:
61
68
 
62
69
 
63
70
  def _add_await_parser(sub) -> None:
64
- parser = sub.add_parser("await", help="wait for tmux-pane workers")
71
+ parser = sub.add_parser("await", help="wait for pane-backed workers")
65
72
  _add_run_args(parser)
66
73
  parser.add_argument("--poll-interval-seconds", type=int, default=5)
67
74
  parser.add_argument("--timeout-seconds", type=int, default=None)
@@ -70,7 +77,7 @@ def _add_await_parser(sub) -> None:
70
77
 
71
78
 
72
79
  def _add_teardown_parser(sub) -> None:
73
- parser = sub.add_parser("teardown", help="kill tmux-pane workers")
80
+ parser = sub.add_parser("teardown", help="reclaim this run's worker panes")
74
81
  _add_run_args(parser)
75
82
  parser.add_argument("--dry-run", action="store_true")
76
83
  parser.add_argument("--json", action="store_true")
@@ -94,8 +101,8 @@ def _dispatch(args) -> int:
94
101
  okstra_bin=Path(args.okstra_bin),
95
102
  requested_workers=requested,
96
103
  idle_timeout_seconds=args.idle_timeout_seconds,
97
- required_lead_runtime="external",
98
- default_backend=BACKEND_TMUX_PANE,
104
+ required_lead_runtime=None if _is_cmux_run(manifest) else "external",
105
+ default_backend=_manifest_backend(manifest),
99
106
  supported_worker_wrappers=_SUPPORTED_WRAPPERS,
100
107
  unsupported_worker_label="external lead",
101
108
  dispatch_kind=args.dispatch_kind,
@@ -133,13 +140,13 @@ def _teardown(args) -> int:
133
140
  team_state_path = _resolve_project_path(project_root, _require_string(manifest, "teamStatePath"))
134
141
  team_state = _load_json(team_state_path, "team-state")
135
142
  run_dir = _resolve_project_path(project_root, _require_string(manifest, "runDirectoryPath"))
136
- lead_pane = tmux.resolve_caller_pane()
137
- panes = _teardown_panes(team_state, run_dir, lead_pane)
143
+ panes = _reclaimable_panes(manifest, team_state, run_dir)
138
144
  if args.dry_run:
139
145
  _emit_teardown(args.json, panes)
140
146
  return 0
147
+ reclaim = cmux.close_surface if _is_cmux_run(manifest) else tmux.kill_pane
141
148
  for pane in panes:
142
- tmux.kill_pane(pane["paneId"])
149
+ reclaim(pane["paneId"])
143
150
  _mark_teardown_errors(team_state_path)
144
151
  _emit_teardown(args.json, panes)
145
152
  return 0
@@ -158,7 +165,7 @@ def _plan_for_existing(
158
165
  lead_events_path=_resolve_project_path(root, _require_string(manifest, "leadEventsPath")),
159
166
  manifest=manifest,
160
167
  jobs=(),
161
- default_backend=BACKEND_TMUX_PANE,
168
+ default_backend=_manifest_backend(manifest),
162
169
  )
163
170
 
164
171
 
@@ -174,12 +181,25 @@ def _await_payload(plan: DispatchPlan, completed: bool) -> dict[str, Any]:
174
181
  }
175
182
 
176
183
 
177
- def _teardown_panes(team_state: Mapping[str, Any], run_dir: Path, lead_pane: str) -> list[dict[str, str]]:
184
+ def _reclaimable_panes(
185
+ manifest: Mapping[str, Any], team_state: Mapping[str, Any], run_dir: Path
186
+ ) -> list[dict[str, str]]:
187
+ """Everything this run owns and may close.
188
+
189
+ Under cmux the recorded ids are the whole list. There is no per-pane tag API
190
+ to sweep with, and scanning by title would be worse than nothing: cmux labels
191
+ its own agent surfaces with the same glyph okstra's tmux cleanup treats as a
192
+ teammate marker, so a sweep could close the lead. Only surfaces okstra
193
+ created are recorded, so only those can be closed.
194
+ """
178
195
  seen: set[str] = set()
179
196
  panes: list[dict[str, str]] = []
180
197
  for record in team_state.get("workerDispatches", []):
181
198
  if isinstance(record, dict):
182
199
  _append_pane(panes, seen, str(record.get("paneId", "")), "worker")
200
+ if _is_cmux_run(manifest):
201
+ return panes
202
+ lead_pane = tmux.resolve_caller_pane()
183
203
  for pane in tmux.list_run_panes(run_dir, lead_pane=lead_pane):
184
204
  _append_pane(panes, seen, pane.pane_id, pane.kind)
185
205
  return panes
@@ -208,10 +228,29 @@ def _emit_teardown(as_json: bool, panes: list[dict[str, str]]) -> None:
208
228
  print(f"{pane['paneId']}\t{pane['kind']}")
209
229
 
210
230
 
231
+ def _manifest_backend(manifest: Mapping[str, Any]) -> str:
232
+ """The backend prepare recorded for this run.
233
+
234
+ Deliberately not a flag on this command: the manifest already answers it,
235
+ and a flag would be a second answer free to disagree. A manifest written
236
+ before the field existed reads as tmux, which is what those runs used.
237
+ """
238
+ return str(manifest.get("terminalBackend") or "") or BACKEND_TMUX_PANE
239
+
240
+
241
+ def _is_cmux_run(manifest: Mapping[str, Any]) -> bool:
242
+ return _manifest_backend(manifest) == BACKEND_CMUX_PANE
243
+
244
+
211
245
  def _validate_external_manifest(manifest: Mapping[str, Any]) -> None:
212
246
  runtime = manifest.get("leadRuntime")
213
247
  if runtime == "external":
214
248
  return
249
+ if _is_cmux_run(manifest):
250
+ # Under cmux okstra owns the panes for every lead, so this command is no
251
+ # longer the external lead's private door and the advice below no longer
252
+ # applies — there is nowhere else for a codex or Claude Code lead to go.
253
+ return
215
254
  if runtime == "codex":
216
255
  raise DispatchError("use okstra codex-dispatch for leadRuntime=codex")
217
256
  if runtime == "claude-code":
@@ -20,6 +20,7 @@ from typing import Optional
20
20
 
21
21
  from okstra_ctl.report_views import (
22
22
  serialize_user_response, UserResponseEntry, UserResponseApproval, infer_run_meta,
23
+ parse_expected_form_options,
23
24
  )
24
25
  from okstra_ctl.report_view_artifacts import user_responses_dir_for_report
25
26
  from okstra_ctl.listing import list_runs, absolute_final_report_path
@@ -473,18 +474,31 @@ def list_awaiting_tasks(home: Path, project_id: str, limit: int) -> list[dict]:
473
474
  # match. `§x.y` and `path.ext:line` are the other two ref shapes.
474
475
  _SECTION_REF_RE = re.compile(r"§[\d.]+|[A-Z]{1,4}-\d+|[\w./-]+\.\w+:\d+")
475
476
  _ID_TOKEN_RE = re.compile(r"^[A-Z]{1,4}-\d+$")
476
- _ALTERNATIVES_CUE = "Alternatives:"
477
477
  _DEFINITION_SNIPPET_CAP = 200
478
478
  _SNIPPET_NOISE_RE = re.compile(r'<a id="[^"]*"></a>|`|\*\*')
479
+ _OPTION_LETTER_LABEL_RE = re.compile(r"^\([a-z]\)\s*")
479
480
 
480
481
 
481
- def _alternatives_from_expected(expected_form: str) -> list[str]:
482
- idx = expected_form.find(_ALTERNATIVES_CUE)
483
- if idx < 0:
484
- return []
485
- tail = expected_form[idx + len(_ALTERNATIVES_CUE):]
486
- parts = re.split(r"[/;]", tail)
487
- return [p.strip(" .,;—-") for p in parts if p.strip(" .,;—-")]
482
+ def _options_from_expected_form(expected_form: str) -> list[dict]:
483
+ """Rebuild a schema-v1 row's options from its ``Expected form`` cell.
484
+
485
+ v1 keeps the choices as one string and has nowhere to record their impact,
486
+ so those fields come back empty and the picker reports them as unstated
487
+ rather than inventing them. Splitting goes through the canonical
488
+ ``parse_expected_form_options`` — the parser the HTML view already uses and
489
+ the only one under test.
490
+ """
491
+ return [
492
+ {
493
+ "role": "recommended" if value == "recommended" else "alternative",
494
+ "answer": _OPTION_LETTER_LABEL_RE.sub("", label).strip(),
495
+ "rationale": "",
496
+ "scopeImpact": [],
497
+ "addedWork": "",
498
+ "directionChange": "",
499
+ }
500
+ for value, label in parse_expected_form_options(expected_form)
501
+ ]
488
502
 
489
503
 
490
504
  def _clean_snippet(line: str) -> str:
@@ -558,8 +572,9 @@ def show_open_rows(report_path: Path) -> dict:
558
572
  refs = sorted(set(_SECTION_REF_RE.findall(statement + " " + expected)))
559
573
  rows.append({"id": it.row_id, "kind": it.kind, "blocks": it.blocks,
560
574
  "status": it.status, "statement": statement,
561
- "recommended": expected, # raw cell keeps answer + rationale
562
- "alternatives": _alternatives_from_expected(expected),
575
+ "expectedForm": expected,
576
+ # v2 authors the choices; v1 only ever had the string.
577
+ "options": r["options"] or _options_from_expected_form(expected),
563
578
  "contextRefs": refs,
564
579
  "resolvedRefs": resolve_refs(text, refs)})
565
580
  return {"reportPath": str(report_path), "rows": rows}
@@ -0,0 +1,183 @@
1
+ """Parser for the worker verdict block that plan-body and convergence share.
2
+
3
+ The response shape is fixed by contract (`prompts/lead/plan-body-verification.md`
4
+ §"Response format"), so the parser belongs here rather than in each lead. A
5
+ per-round ad-hoc regex makes the round's fidelity depend on whoever wrote it,
6
+ and its failure mode is silence: dev-10400 lost 19 of 37 assigned items to a
7
+ no-match that nothing reported. Every shape this module cannot read is an error.
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import re
12
+ from dataclasses import dataclass
13
+
14
+ VERDICT_TOKENS = frozenset({
15
+ "AGREE", "DISAGREE", "SUPPLEMENT", "UNVERIFIABLE", "VERIFICATION-ERROR",
16
+ })
17
+ FIXABILITY_VALUES = frozenset({"planner-fixable", "needs-user-input"})
18
+
19
+ # The convergence reverify prompts use their own vocabularies. Collaborative and
20
+ # full-reanalysis rounds speak the schema's own words; the adversarial round asks
21
+ # the verifier to break the finding, so it answers in break/survive terms that
22
+ # have to be translated back (`prompts/lead/convergence.md`
23
+ # §"Adversarial Re-verification Prompt").
24
+ COLLABORATIVE_VERDICTS = {
25
+ "AGREE": "agree",
26
+ "DISAGREE": "disagree",
27
+ "SUPPLEMENT": "supplement",
28
+ "UNVERIFIABLE": "unverifiable",
29
+ "VERIFICATION-ERROR": "verification-error",
30
+ }
31
+ ADVERSARIAL_VERDICTS = {
32
+ "SURVIVES": "agree",
33
+ "SURVIVES-WITH-CAVEAT": "supplement",
34
+ "REFUTED": "disagree",
35
+ # A verifier that looked and could not check is not a verifier that failed.
36
+ # `verification-error` drops the vote from the participating count, which
37
+ # shrinks the roster without saying so.
38
+ "UNVERIFIABLE": "unverifiable",
39
+ "VERIFICATION-ERROR": "verification-error",
40
+ }
41
+ DISAGREE_BASES = frozenset({"counter-evidence", "burden-not-met"})
42
+
43
+ _ITEM_RE = re.compile(r"^###[ \t]+(?P<id>[^\s:]+)[ \t]*:?.*$", re.MULTILINE)
44
+ # The contract writes some labels with a parenthetical qualifier —
45
+ # `**Fixability** (only when DISAGREE):` — so the colon may trail a `(...)`.
46
+ _FIELD_RE = re.compile(
47
+ r"^\*\*(?P<key>Verdict|Fixability|Note|Explanation|Prior dissent|Basis"
48
+ r"|Your evidence)\*\*"
49
+ r"(?:[ \t]*\([^)]*\))?[ \t]*:[ \t]*(?P<value>.*)$",
50
+ re.MULTILINE,
51
+ )
52
+ _VERDICT_RE = re.compile(r"^(?P<token>[A-Z-]+)(?:\((?P<kind>[a-f])\))?$")
53
+
54
+
55
+ class VerdictBlockError(ValueError):
56
+ """Raised when a worker response does not match the contract shape."""
57
+
58
+
59
+ @dataclass(frozen=True)
60
+ class FindingVote:
61
+ """One worker's vote on one convergence finding, in schema vocabulary."""
62
+
63
+ finding_id: str
64
+ verdict: str
65
+ disagree_basis: str | None
66
+ explanation: str
67
+
68
+
69
+ def _scan_blocks(text: str) -> dict[str, dict[str, str]]:
70
+ """Every `### <id>` block in *text*, as id → field map, in document order."""
71
+ matches = list(_ITEM_RE.finditer(text))
72
+ blocks: dict[str, dict[str, str]] = {}
73
+ for index, match in enumerate(matches):
74
+ end = matches[index + 1].start() if index + 1 < len(matches) else len(text)
75
+ item_id = match.group("id")
76
+ if item_id in blocks:
77
+ raise VerdictBlockError(f"item `{item_id}` appears twice in one response")
78
+ body = text[match.end():end]
79
+ blocks[item_id] = {
80
+ m.group("key"): m.group("value").strip() for m in _FIELD_RE.finditer(body)
81
+ }
82
+ return blocks
83
+
84
+
85
+ def parse_finding_votes(text: str, *, adversarial: bool) -> dict[str, FindingVote]:
86
+ """Convergence reverify votes in *text*, keyed by finding id.
87
+
88
+ *adversarial* selects the vocabulary; guessing it from the content would make
89
+ an unfamiliar token silently read as a different verdict.
90
+ """
91
+ vocabulary = ADVERSARIAL_VERDICTS if adversarial else COLLABORATIVE_VERDICTS
92
+ return {
93
+ finding_id: _finding_vote(finding_id, fields, vocabulary)
94
+ for finding_id, fields in _scan_blocks(text).items()
95
+ }
96
+
97
+
98
+ def _finding_vote(
99
+ finding_id: str, fields: dict[str, str], vocabulary: dict[str, str]
100
+ ) -> FindingVote:
101
+ raw = fields.get("Verdict", "")
102
+ token = raw.strip().strip("`").strip().upper()
103
+ if token not in vocabulary:
104
+ raise VerdictBlockError(
105
+ f"finding `{finding_id}` has an unknown verdict: {raw or '(missing)'} "
106
+ f"— expected one of {sorted(vocabulary)}"
107
+ )
108
+ verdict = vocabulary[token]
109
+ explanation = fields.get("Explanation", "")
110
+ if not explanation:
111
+ raise VerdictBlockError(
112
+ f"finding `{finding_id}` has no `**Explanation**:` line — every vote "
113
+ f"the classifier reads carries one"
114
+ )
115
+ basis = fields.get("Basis", "").strip() or None
116
+ if verdict != "disagree":
117
+ basis = None
118
+ elif basis is not None and basis not in DISAGREE_BASES:
119
+ raise VerdictBlockError(
120
+ f"finding `{finding_id}` has basis `{basis}` — expected one of "
121
+ f"{sorted(DISAGREE_BASES)}"
122
+ )
123
+ return FindingVote(finding_id, verdict, basis, explanation)
124
+
125
+
126
+ @dataclass(frozen=True)
127
+ class VerdictBlock:
128
+ """One worker's verdict on one item."""
129
+
130
+ item_id: str
131
+ verdict: str
132
+ breakage_kind: str
133
+ fixability: str
134
+ note: str
135
+ explanation: str
136
+ prior_dissent: str
137
+
138
+
139
+ def parse_verdict_blocks(text: str) -> dict[str, VerdictBlock]:
140
+ """Every `### <item-id>` plan-body verdict block in *text*, keyed by item id."""
141
+ return {
142
+ item_id: _block(item_id, fields)
143
+ for item_id, fields in _scan_blocks(text).items()
144
+ }
145
+
146
+
147
+ def _verdict_token(item_id: str, raw: str) -> tuple[str, str]:
148
+ parsed = _VERDICT_RE.match(raw.strip().strip("`").strip())
149
+ if parsed is None or parsed.group("token") not in VERDICT_TOKENS:
150
+ raise VerdictBlockError(
151
+ f"item `{item_id}` has an unknown verdict: {raw or '(missing)'}"
152
+ )
153
+ return parsed.group("token"), parsed.group("kind") or ""
154
+
155
+
156
+ def _block(item_id: str, fields: dict[str, str]) -> VerdictBlock:
157
+ raw = fields.get("Verdict", "")
158
+ if not raw:
159
+ raise VerdictBlockError(f"item `{item_id}` has no `**Verdict**:` line")
160
+ token, kind = _verdict_token(item_id, raw)
161
+ fixability = fields.get("Fixability", "")
162
+ if token == "DISAGREE":
163
+ if not kind:
164
+ raise VerdictBlockError(
165
+ f"item `{item_id}` is DISAGREE with no breakage kind — the gate "
166
+ f"reads the kind to decide whether one vote blocks, so a bare "
167
+ f"`DISAGREE` cannot be scored"
168
+ )
169
+ if fixability not in FIXABILITY_VALUES:
170
+ raise VerdictBlockError(
171
+ f"item `{item_id}` is DISAGREE with fixability "
172
+ f"`{fixability or '(missing)'}` — expected one of "
173
+ f"{sorted(FIXABILITY_VALUES)}"
174
+ )
175
+ return VerdictBlock(
176
+ item_id=item_id,
177
+ verdict=token,
178
+ breakage_kind=kind,
179
+ fixability=fixability,
180
+ note=fields.get("Note", ""),
181
+ explanation=fields.get("Explanation", ""),
182
+ prior_dissent=fields.get("Prior dissent", ""),
183
+ )
@@ -53,8 +53,10 @@ from okstra_ctl.lead_runtime import lead_runtime_info
53
53
  from okstra_ctl.runner_resolution import native_provider_for_host
54
54
  from okstra_ctl.clarification_items import (
55
55
  scan_approval_gate,
56
+ sidecar_answers,
56
57
  user_response_sidecars,
57
58
  )
59
+ from okstra_ctl.incremental_scope import preview_link_availability_for_report
58
60
  from okstra_ctl.design_prep import (
59
61
  DesignPrepError,
60
62
  load_design_prep_items,
@@ -94,6 +96,7 @@ from okstra_ctl.workers import (
94
96
  from okstra_ctl.workflow import PHASE_SEQUENCE
95
97
  from okstra_ctl.wizard_stage_intent import (
96
98
  WHOLE_TASK_STAGE,
99
+ WizardStageIntent,
97
100
  WizardStageIntentError,
98
101
  resolve_wizard_stage_intent,
99
102
  wizard_stage_confirmation_label,
@@ -4570,6 +4573,23 @@ def submit(state: WizardState, value: str) -> dict[str, Any]:
4570
4573
  return {"echo": echo or "", "next": prompt_payload(state, nxt)}
4571
4574
 
4572
4575
 
4576
+ def _stage_intent(state: WizardState) -> WizardStageIntent:
4577
+ """This state's stage selection, resolved once for every consumer.
4578
+
4579
+ `render_args` needs the single stage this run prepares; `wizard_outcome`
4580
+ needs the chain the skill drives. Deriving them separately would let the two
4581
+ disagree about which stage the run is for.
4582
+ """
4583
+ try:
4584
+ return resolve_wizard_stage_intent(
4585
+ task_type=state.task_type,
4586
+ selected_stage=state.selected_stage,
4587
+ selected_stages=state.selected_stages,
4588
+ )
4589
+ except WizardStageIntentError as exc:
4590
+ raise WizardError(str(exc)) from exc
4591
+
4592
+
4573
4593
  def render_args(state: WizardState) -> dict[str, str]:
4574
4594
  """Convert finalized state into ``okstra render-bundle`` argument map."""
4575
4595
  if state.aborted:
@@ -4584,14 +4604,7 @@ def render_args(state: WizardState) -> dict[str, str]:
4584
4604
  if state.reuse_worktree or state.task_type == "final-verification"
4585
4605
  else state.base_ref
4586
4606
  )
4587
- try:
4588
- stage_intent = resolve_wizard_stage_intent(
4589
- task_type=state.task_type,
4590
- selected_stage=state.selected_stage,
4591
- selected_stages=state.selected_stages,
4592
- )
4593
- except WizardStageIntentError as exc:
4594
- raise WizardError(str(exc)) from exc
4607
+ stage_intent = _stage_intent(state)
4595
4608
  pr_template = (
4596
4609
  state.pr_template_path
4597
4610
  if state.task_type == "release-handoff"
@@ -4624,7 +4637,6 @@ def render_args(state: WizardState) -> dict[str, str]:
4624
4637
  "approved-plan": state.approved_plan_path,
4625
4638
  "stage": stage_intent.stage,
4626
4639
  "stages": state.handoff_stages,
4627
- "chain-stages": stage_intent.chain_stages,
4628
4640
  "base-ref": base_ref,
4629
4641
  "workers": workers,
4630
4642
  "directive": state.directive,
@@ -4643,6 +4655,37 @@ def render_args(state: WizardState) -> dict[str, str]:
4643
4655
  }
4644
4656
 
4645
4657
 
4658
+ def _reverify_scope_line(state: WizardState) -> Optional[str]:
4659
+ """이번 clarification 재실행이 좁혀질지 — 확인 단계에서 미리 보여주는 줄.
4660
+
4661
+ 이 판정의 절반(답변된 id 가 stage 로 되짚어지는지)은 base SHA 없이 결정되고
4662
+ 직전 리포트만 있으면 이미 정해져 있다. 그런데 지금까지는 `okstra recap
4663
+ assemble` 을 따로 돌려야만 보였고, run 이 시작된 뒤에 full 로 밝혀지면 두
4664
+ 시간을 물린 뒤였다. 확인 단계는 그 전에 되돌릴 수 있는 마지막 지점이다.
4665
+ """
4666
+ if state.task_type != "implementation-planning":
4667
+ return None
4668
+ if not state.clarification_response_path or not state.project_root:
4669
+ return None
4670
+ report = _resolve_path(
4671
+ state.clarification_response_path, Path(state.project_root)
4672
+ )
4673
+ if not report.is_file():
4674
+ return None
4675
+ preview = preview_link_availability_for_report(
4676
+ report, set(sidecar_answers(report))
4677
+ )
4678
+ if not preview["wouldForceFull"]:
4679
+ return _msg(state.workspace_root, "confirmation",
4680
+ "reverify_scope_incremental")
4681
+ if preview["unlinkedIds"]:
4682
+ return _msg(state.workspace_root, "confirmation",
4683
+ "reverify_scope_unlinked",
4684
+ ids=", ".join(preview["unlinkedIds"]))
4685
+ return _msg(state.workspace_root, "confirmation", "reverify_scope_full",
4686
+ reason=preview["reason"])
4687
+
4688
+
4646
4689
  def confirmation_block(state: WizardState) -> str:
4647
4690
  """Human-readable echo of the resolved selections (for the Confirm step)."""
4648
4691
  header = _msg(state.workspace_root, "confirmation", "header")
@@ -4727,6 +4770,9 @@ def confirmation_block(state: WizardState) -> str:
4727
4770
  lines.append(f" stage : {stage}")
4728
4771
  if state.clarification_response_path:
4729
4772
  lines.append(f" clarification : {state.clarification_response_path}")
4773
+ reverify_line = _reverify_scope_line(state)
4774
+ if reverify_line is not None:
4775
+ lines.append(reverify_line)
4730
4776
  if state.task_type == "release-handoff" and state.handoff_mode:
4731
4777
  scope = (
4732
4778
  _msg(state.workspace_root, "confirmation",
@@ -4760,13 +4806,21 @@ def _wizard_persist_actions(state: WizardState) -> list[dict[str, str]]:
4760
4806
 
4761
4807
 
4762
4808
  def wizard_outcome(state: WizardState) -> dict[str, Any]:
4763
- """Public outcome for callers that need launch data and follow-up writes."""
4809
+ """Public outcome for callers that need launch data and follow-up writes.
4810
+
4811
+ `renderArgs` carries only what `okstra render-bundle` accepts, so a caller
4812
+ can pass every entry through unfiltered — which is exactly what the
4813
+ okstra-run skill is told to do. Signals the skill consumes itself, like the
4814
+ unattended stage chain, live under `orchestration`; mixing them into
4815
+ `renderArgs` made the renderer reject the wizard's own output.
4816
+ """
4764
4817
  if state.aborted:
4765
4818
  raise WizardError("wizard was aborted by the user — outcome is unavailable")
4766
4819
  if state.confirmed is not True:
4767
4820
  raise WizardError("wizard is not complete — outcome is unavailable")
4768
4821
  return {
4769
4822
  "renderArgs": render_args(state),
4823
+ "orchestration": {"chainStages": _stage_intent(state).chain_stages},
4770
4824
  "persistActions": _wizard_persist_actions(state),
4771
4825
  "confirmationText": confirmation_block(state),
4772
4826
  }
@@ -0,0 +1,44 @@
1
+ """CLI adapter for the worker audit-sidecar contract (`okstra worker-audit-check`).
2
+
3
+ Phase 7 runs the same rules through `validate-run.py`, but by then the worker
4
+ session is gone and the only remedies left are a retroactive edit — which breaks
5
+ the audit chain — or a failed run. Called the moment a worker returns, the same
6
+ rules cost one message to a worker that is still listening.
7
+ """
8
+ from __future__ import annotations
9
+
10
+ import argparse
11
+ import json
12
+ import sys
13
+ from pathlib import Path
14
+
15
+ from okstra_ctl.worker_audit_ledger import check_worker_results_audit
16
+
17
+
18
+ def _parser() -> argparse.ArgumentParser:
19
+ parser = argparse.ArgumentParser(
20
+ prog="okstra worker-audit-check",
21
+ description="Check one run's worker audit sidecars (read-only).",
22
+ )
23
+ parser.add_argument("--run-dir", type=Path, required=True,
24
+ help="runs/<task-type>/ for this run")
25
+ parser.add_argument("--task-type", required=True)
26
+ parser.add_argument("--seq", required=True,
27
+ help="this run's 3-digit seq")
28
+ parser.add_argument("--worker", default=None,
29
+ help="check only this worker id (default: every worker)")
30
+ return parser
31
+
32
+
33
+ def main(argv: list[str] | None = None) -> int:
34
+ args = _parser().parse_args(argv)
35
+ failures = check_worker_results_audit(
36
+ args.run_dir, args.task_type, args.seq, worker=args.worker
37
+ )
38
+ print(json.dumps({"ok": not failures, "failures": failures},
39
+ ensure_ascii=False, indent=2))
40
+ return 2 if failures else 0
41
+
42
+
43
+ if __name__ == "__main__":
44
+ raise SystemExit(main(sys.argv[1:]))