okstra 0.159.0 → 0.161.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (80) hide show
  1. package/README.md +1 -1
  2. package/docs/architecture/storage-model.md +2 -0
  3. package/docs/architecture.md +2 -1
  4. package/docs/cli.md +8 -3
  5. package/docs/for-ai/README.md +2 -2
  6. package/docs/for-ai/skills/okstra-inspect.md +3 -0
  7. package/docs/for-ai/skills/okstra-run.md +2 -1
  8. package/docs/for-ai/skills/okstra-user-response.md +5 -5
  9. package/docs/project-structure-overview.md +5 -1
  10. package/docs/task-process/implementation.md +28 -0
  11. package/package.json +1 -1
  12. package/runtime/BUILD.json +2 -2
  13. package/runtime/bin/okstra-claude-exec.sh +4 -1
  14. package/runtime/prompts/host-orchestration/README.md +18 -0
  15. package/runtime/prompts/host-orchestration/implementation.md +57 -0
  16. package/runtime/prompts/launch.template.md +10 -1
  17. package/runtime/prompts/lead/adapters/claude-code.md +1 -1
  18. package/runtime/prompts/lead/adapters/cmux.md +67 -0
  19. package/runtime/prompts/lead/context-loader.md +5 -2
  20. package/runtime/prompts/lead/convergence.md +3 -1
  21. package/runtime/prompts/lead/plan-body-verification.md +21 -2
  22. package/runtime/prompts/lead/team-contract.md +2 -1
  23. package/runtime/prompts/profiles/_clarification-recommendation.md +11 -1
  24. package/runtime/prompts/profiles/_common-contract.md +3 -1
  25. package/runtime/prompts/profiles/implementation-planning.md +2 -0
  26. package/runtime/prompts/profiles/requirements-discovery.md +1 -1
  27. package/runtime/prompts/wizard/prompts.ko.json +3 -0
  28. package/runtime/python/okstra_ctl/clarification_items.py +9 -0
  29. package/runtime/python/okstra_ctl/cmux.py +531 -0
  30. package/runtime/python/okstra_ctl/codex_dispatch.py +6 -6
  31. package/runtime/python/okstra_ctl/convergence.py +168 -11
  32. package/runtime/python/okstra_ctl/dispatch_core.py +76 -7
  33. package/runtime/python/okstra_ctl/dispatch_state.py +16 -0
  34. package/runtime/python/okstra_ctl/error_issue.py +640 -0
  35. package/runtime/python/okstra_ctl/error_report.py +56 -0
  36. package/runtime/python/okstra_ctl/error_zip.py +23 -10
  37. package/runtime/python/okstra_ctl/incremental_scope.py +159 -19
  38. package/runtime/python/okstra_ctl/initial_prompt_materialization.py +18 -5
  39. package/runtime/python/okstra_ctl/issue_signals.py +186 -0
  40. package/runtime/python/okstra_ctl/lead_runtime.py +30 -2
  41. package/runtime/python/okstra_ctl/paths.py +38 -0
  42. package/runtime/python/okstra_ctl/plan_items_cli.py +167 -3
  43. package/runtime/python/okstra_ctl/profile_show.py +134 -0
  44. package/runtime/python/okstra_ctl/recap.py +63 -0
  45. package/runtime/python/okstra_ctl/render.py +7 -2
  46. package/runtime/python/okstra_ctl/render_final_report.py +7 -22
  47. package/runtime/python/okstra_ctl/report_translation.py +4 -0
  48. package/runtime/python/okstra_ctl/report_views.py +7 -3
  49. package/runtime/python/okstra_ctl/run.py +54 -3
  50. package/runtime/python/okstra_ctl/run_audit.py +477 -0
  51. package/runtime/python/okstra_ctl/team.py +50 -11
  52. package/runtime/python/okstra_ctl/user_response.py +25 -10
  53. package/runtime/python/okstra_ctl/verdict_blocks.py +183 -0
  54. package/runtime/python/okstra_ctl/wizard.py +64 -10
  55. package/runtime/python/okstra_ctl/worker_audit_check.py +44 -0
  56. package/runtime/python/okstra_ctl/worker_audit_ledger.py +207 -0
  57. package/runtime/python/okstra_ctl/worker_heartbeat.py +9 -3
  58. package/runtime/python/okstra_ctl/worker_liveness.py +81 -9
  59. package/runtime/schemas/final-report-v1.0.schema.json +14 -0
  60. package/runtime/schemas/final-report-v2.0.schema.json +51 -1
  61. package/runtime/skills/okstra-inspect/SKILL.md +3 -1
  62. package/runtime/skills/okstra-inspect/facets/error-issue.md +77 -0
  63. package/runtime/skills/okstra-inspect/facets/run-audit.md +34 -0
  64. package/runtime/skills/okstra-run/SKILL.md +28 -10
  65. package/runtime/skills/okstra-user-response/SKILL.md +18 -18
  66. package/runtime/templates/reports/final-report.template.md +4 -0
  67. package/runtime/templates/reports/html/i18n/en.json +5 -1
  68. package/runtime/templates/reports/html/i18n/ko.json +5 -1
  69. package/runtime/templates/reports/html/macros/forms.html +15 -0
  70. package/runtime/templates/reports/html/tasks/implementation-planning.template.html +1 -0
  71. package/runtime/templates/reports/i18n/en.json +2 -0
  72. package/runtime/validators/validate-run.py +267 -208
  73. package/runtime/validators/validate-workflow.sh +6 -0
  74. package/runtime/validators/validate_session_conformance.py +135 -31
  75. package/src/cli-registry.mjs +34 -0
  76. package/src/commands/execute/incremental-scope.mjs +10 -0
  77. package/src/commands/execute/worker-audit-check.mjs +35 -0
  78. package/src/commands/inspect/error-issue.mjs +27 -0
  79. package/src/commands/inspect/profile-show.mjs +29 -0
  80. package/src/commands/inspect/run-audit.mjs +26 -0
@@ -0,0 +1,207 @@
1
+ """The worker audit-sidecar contract, shared by Phase 7 and the mid-run check.
2
+
3
+ Phase 7 has always enforced this post-hoc, but by then the worker session is
4
+ gone and the only remedies left are editing the result after the fact — which
5
+ breaks the audit chain — or failing the run. `okstra worker-audit-check` runs
6
+ the same rules the moment a worker returns, while the worker is still listening
7
+ and can fix its own citation. Both consumers must agree on what a violation is,
8
+ so the rules live here rather than inside either one — the same split
9
+ `worker_heartbeat` makes for the heartbeat cadence.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import re
14
+ from pathlib import Path
15
+
16
+ from okstra_ctl.worker_prompt_headers import EVIDENCE_LEDGER_HEADER
17
+
18
+ # Worker-results filename pattern: `<worker-role>-<task-type>-<seq>.md`.
19
+ # Every analysis-worker role name ends in `-worker` (`claude-worker`,
20
+ # `codex-worker`, `antigravity-worker`, `report-writer-worker`), so anchor the
21
+ # split on that suffix — otherwise `antigravity-worker-error-analysis-001.md`
22
+ # ambiguously parses as `worker=antigravity, task=worker-error-analysis`.
23
+ # Audit sidecars (`*-audit-*`) and errors sidecars (`.json`) are not matched here.
24
+ _WORKER_RESULT_BASENAME_RE = re.compile(
25
+ r"^(?P<worker>[a-z][a-z0-9-]*-worker)-(?P<task_type>[a-z][a-z-]*?)-(?P<seq>\d{3})\.md$"
26
+ )
27
+
28
+ READING_CONFIRMATION_HEADING_RE = re.compile(
29
+ r"^##[ \t]+0\.[ \t]+Reading Confirmation\b", re.MULTILINE
30
+ )
31
+
32
+ _EVIDENCE_READ_RE = re.compile(
33
+ r"^- Evidence read: `(?P<path>[^`\n]+)`\s*$",
34
+ re.MULTILINE,
35
+ )
36
+ _FILE_LINE_CITATION_RE = re.compile(
37
+ r"`(?P<path>(?!https?://)[^`\n]+?):(?P<line>\d+(?:-\d+)?)`"
38
+ )
39
+ _EXTENSIONLESS_SOURCE_FILENAMES = frozenset(
40
+ {"Dockerfile", "Justfile", "Makefile", "Procfile", "Rakefile"}
41
+ )
42
+
43
+
44
+ def _looks_like_file_path(path: str) -> bool:
45
+ if (
46
+ not path
47
+ or path.startswith(("-", "$"))
48
+ or any(char.isspace() for char in path)
49
+ ):
50
+ return False
51
+ if re.fullmatch(r"[0-9a-fA-F]{7,64}", path):
52
+ return False
53
+ return (
54
+ "/" in path
55
+ or "." in Path(path).name
56
+ or Path(path).name in _EXTENSIONLESS_SOURCE_FILENAMES
57
+ )
58
+
59
+
60
+ def _cited_file_paths(content: str) -> set[str]:
61
+ paths: set[str] = set()
62
+ for match in _FILE_LINE_CITATION_RE.finditer(content):
63
+ path = match.group("path")
64
+ if _looks_like_file_path(path):
65
+ paths.add(path)
66
+ return paths
67
+
68
+
69
+ def _audit_evidence_read_paths(content: str) -> set[str]:
70
+ return {
71
+ match.group("path")
72
+ for match in _EVIDENCE_READ_RE.finditer(content)
73
+ }
74
+
75
+
76
+ def _worker_prompt_path(
77
+ run_dir: Path,
78
+ worker_role: str,
79
+ task_type: str,
80
+ seq: str,
81
+ ) -> Path:
82
+ return run_dir / "prompts" / f"{worker_role}-prompt-{task_type}-{seq}.md"
83
+
84
+
85
+ def _evidence_read_ledger_failures(
86
+ *,
87
+ run_dir: Path,
88
+ worker_role: str,
89
+ task_type: str,
90
+ seq: str,
91
+ result_name: str,
92
+ result_content: str,
93
+ audit_path: Path,
94
+ ) -> list[str]:
95
+ if worker_role == "report-writer-worker":
96
+ return []
97
+ prompt_path = _worker_prompt_path(run_dir, worker_role, task_type, seq)
98
+ try:
99
+ prompt_content = prompt_path.read_text(encoding="utf-8")
100
+ except OSError:
101
+ return []
102
+ if EVIDENCE_LEDGER_HEADER not in prompt_content.splitlines():
103
+ return []
104
+ try:
105
+ audit_content = audit_path.read_text(encoding="utf-8")
106
+ except OSError as exc:
107
+ return [f"worker audit sidecar unreadable: {audit_path.name} ({exc})"]
108
+
109
+ missing_paths = sorted(
110
+ _cited_file_paths(result_content) - _audit_evidence_read_paths(audit_content)
111
+ )
112
+ return [
113
+ f"worker `{worker_role}` result `{result_name}` cites "
114
+ f"`{missing_path}:line` without an Evidence read row for "
115
+ f"`{missing_path}` in `{audit_path.name}`"
116
+ for missing_path in missing_paths
117
+ ]
118
+
119
+
120
+ def _result_files(run_dir: Path, task_type: str, seq: str | None, worker: str | None):
121
+ """Every worker-results file in *run_dir* this check owns, in name order."""
122
+ for path in sorted((run_dir / "worker-results").glob("*.md")):
123
+ if "-audit-" in path.name:
124
+ continue
125
+ match = _WORKER_RESULT_BASENAME_RE.match(path.name)
126
+ if match is None:
127
+ # Files that don't match the canonical pattern (e.g. ad-hoc notes
128
+ # left by the operator) are out of contract scope.
129
+ continue
130
+ if match.group("task_type") != task_type:
131
+ # Cross-phase artifacts shouldn't appear here; skip rather than
132
+ # fail to keep the check focused on the current phase.
133
+ continue
134
+ if seq is not None and match.group("seq") != seq:
135
+ # A prior run's artifact. Its contract was judged when it ran.
136
+ continue
137
+ if worker is not None and match.group("worker") != worker:
138
+ continue
139
+ yield path, match.group("worker"), match.group("seq")
140
+
141
+
142
+ def check_worker_results_audit(
143
+ run_dir: Path,
144
+ task_type: str,
145
+ seq: str | None,
146
+ *,
147
+ worker: str | None = None,
148
+ ) -> list[str]:
149
+ """Every audit-sidecar contract failure among this run's worker results.
150
+
151
+ *run_dir* is `runs/<task-type>/`; `worker-results/` and `prompts/` hang off
152
+ it. *seq* scopes the check to one run — `worker-results/` accumulates every
153
+ run's artifacts, so scanning the whole directory judged a run by files it
154
+ did not produce. Pass ``None`` only when the seq is genuinely unknown, which
155
+ falls back to not filtering rather than silently checking nothing.
156
+
157
+ For each result file this checks that it carries no `## 0. Reading
158
+ Confirmation` heading (that block moved to the sidecar), that the matching
159
+ sidecar exists, and — for prompts carrying the required-v1 evidence-ledger
160
+ marker — that every backticked `path:line` citation has an Evidence read row.
161
+ """
162
+ failures: list[str] = []
163
+ if not (run_dir / "worker-results").is_dir():
164
+ # No worker-results directory means no analysis workers ran (e.g.
165
+ # `release-handoff`, which is single-lead). Nothing to enforce.
166
+ return failures
167
+
168
+ for path, worker_role, result_seq in _result_files(run_dir, task_type, seq, worker):
169
+ rel = path.name
170
+ try:
171
+ content = path.read_text()
172
+ except OSError as exc:
173
+ failures.append(f"worker-results file unreadable: {rel} ({exc})")
174
+ continue
175
+
176
+ if READING_CONFIRMATION_HEADING_RE.search(content) is not None:
177
+ failures.append(
178
+ f"worker-results file `{rel}` contains a `## 0. Reading "
179
+ f"Confirmation` heading — that block moved to the audit "
180
+ f"sidecar (`{worker_role}-audit-{task_type}-{result_seq}.md`). "
181
+ f"Remove the §0 heading + body from the main file and "
182
+ f"write a fresh sidecar."
183
+ )
184
+
185
+ audit_path = (
186
+ run_dir / "worker-results"
187
+ / f"{worker_role}-audit-{task_type}-{result_seq}.md"
188
+ )
189
+ if not audit_path.exists():
190
+ failures.append(
191
+ f"worker `{worker_role}` produced `{rel}` but no audit sidecar "
192
+ f"at `{audit_path.name}` — the sidecar must carry the Reading "
193
+ f"Confirmation block (one short line per input file). Workers "
194
+ f"write this in the same step as the main worker-results file."
195
+ )
196
+ continue
197
+
198
+ failures.extend(_evidence_read_ledger_failures(
199
+ run_dir=run_dir,
200
+ worker_role=worker_role,
201
+ task_type=task_type,
202
+ seq=result_seq,
203
+ result_name=rel,
204
+ result_content=content,
205
+ audit_path=audit_path,
206
+ ))
207
+ return failures
@@ -18,6 +18,10 @@ HEARTBEAT_LINE_RE = re.compile(
18
18
  r"^-[ \t]*PROGRESS:[ \t]*(?P<stage>\S+)[ \t]+(?P<ts>\S+)[ \t]*$", re.MULTILINE
19
19
  )
20
20
 
21
+ # 한 단계가 cadence 보다 길어질 때 워커가 append 하는 진행 라인의 접두사
22
+ # (claude-worker.md "Heartbeat"). 뒤에 붙는 stage 이름은 원래 단계 그대로다.
23
+ IN_STAGE_PREFIX = "in-stage:"
24
+
21
25
  # 계약상 cadence 는 5분. append 직전 측정한 시각과 실제 쓰기 사이 지연을 흡수하는
22
26
  # 고정 grace 60초를 더한다.
23
27
  HEARTBEAT_MAX_GAP_SECONDS = 5 * 60 + 60
@@ -48,10 +52,12 @@ def max_gap_seconds_after(stage: str) -> int:
48
52
  """*stage* 를 알린 뒤 다음 하트비트까지 허용되는 최대 공백(초).
49
53
 
50
54
  간격이 재는 것은 직전에 선언된 단계의 작업 시간이므로, 예산은 언제나
51
- 구간을 *여는* 단계에서 고른다."""
52
- if stage in SINGLE_WRITE_STAGES:
55
+ 구간을 *여는* 단계에서 고른다. `in-stage:<X>` 는 아직 X 안에 있다는 뜻이라
56
+ 그 라인이 여는 구간도 여전히 X 의 일부다 — 접두사를 떼고 X 의 예산을 쓴다."""
57
+ opener = stage.removeprefix(IN_STAGE_PREFIX)
58
+ if opener in SINGLE_WRITE_STAGES:
53
59
  return SINGLE_WRITE_MAX_GAP_SECONDS
54
- if stage in SYNTHESIS_STAGES:
60
+ if opener in SYNTHESIS_STAGES:
55
61
  return SYNTHESIS_MAX_GAP_SECONDS
56
62
  return HEARTBEAT_MAX_GAP_SECONDS
57
63
 
@@ -54,6 +54,14 @@ DEFAULT_LAUNCH_GRACE_SECONDS = 60
54
54
  DEFAULT_POLL_INTERVAL_SECONDS = 20.0
55
55
  DEFAULT_WAIT_TIMEOUT_SECONDS = 2400.0
56
56
 
57
+ # A budget breach is one observation, and "slow" and "dead" are only
58
+ # distinguishable across two. On a breach the probe re-reads the sidecar this
59
+ # far into the future — a fraction of the stage's own budget, so a stage with a
60
+ # longer budget also gets a longer confirmation. Measured false positives this
61
+ # absorbs (dev-10400): `analysis` 386s against a 360s budget,
62
+ # `data-json-write-start` 1602s against 1260s.
63
+ DEFAULT_STALL_CONFIRM_RATIO = 0.5
64
+
57
65
 
58
66
  def _utc_now() -> datetime:
59
67
  return datetime.now(timezone.utc)
@@ -183,13 +191,58 @@ class ProbeTarget:
183
191
  result_path: Path | None = None
184
192
 
185
193
 
186
- def probe_one(target: ProbeTarget, *, now: datetime, max_idle: float,
187
- launch_grace: float) -> dict:
188
- if target.liveness_mode == LIVENESS_AUDIT_HEARTBEAT:
189
- return probe_heartbeat(
190
- target.artifact, target.dispatched_at, now, max_idle, launch_grace
191
- )
192
- return probe_launch(target.artifact, target.dispatched_at, now, launch_grace)
194
+ def _confirm_window(probe: dict, stall_confirm: float | None) -> float:
195
+ """Seconds to wait before a budget breach becomes a verdict.
196
+
197
+ Only a breach carries ``budgetSeconds``. The other stalled shapes — a
198
+ sidecar with no heartbeat at all, a newest beat that predates this dispatch
199
+ — are not "the worker is mid-tool-call", so waiting tells us nothing new
200
+ about them.
201
+ """
202
+ if "budgetSeconds" not in probe:
203
+ return 0.0
204
+ if stall_confirm is not None:
205
+ return max(0.0, float(stall_confirm))
206
+ return probe["budgetSeconds"] * DEFAULT_STALL_CONFIRM_RATIO
207
+
208
+
209
+ def probe_one(
210
+ target: ProbeTarget,
211
+ *,
212
+ now: datetime,
213
+ max_idle: float,
214
+ launch_grace: float,
215
+ stall_confirm: float | None = None,
216
+ sleep: Callable[[float], None] = time.sleep,
217
+ clock: Callable[[], datetime] = _utc_now,
218
+ ) -> dict:
219
+ """One worker's verdict, with a budget breach confirmed before it stands.
220
+
221
+ The wait this costs is bounded by the confirmation window; what the probe
222
+ exists to avoid is paying ``DEFAULT_WAIT_TIMEOUT_SECONDS`` for a worker that
223
+ died early, and that is still never paid.
224
+ """
225
+ if target.liveness_mode != LIVENESS_AUDIT_HEARTBEAT:
226
+ return probe_launch(target.artifact, target.dispatched_at, now, launch_grace)
227
+ probe = probe_heartbeat(
228
+ target.artifact, target.dispatched_at, now, max_idle, launch_grace
229
+ )
230
+ if probe["state"] != "stalled":
231
+ return probe
232
+ window = _confirm_window(probe, stall_confirm)
233
+ if window <= 0:
234
+ return probe
235
+ sleep(window)
236
+ confirmed = probe_heartbeat(
237
+ target.artifact, target.dispatched_at, clock(), max_idle, launch_grace
238
+ )
239
+ if confirmed.get("lastHeartbeat") == probe.get("lastHeartbeat"):
240
+ return confirmed
241
+ # The question the window asks is whether the heartbeat moved, not whether
242
+ # the re-read is healthy on its own terms. A beat opening a stage with a
243
+ # smaller budget than the window we just slept reads as stale the instant it
244
+ # lands, which would call a worker dead for proving it is alive.
245
+ return {**confirmed, "state": "live", "reason": ""}
193
246
 
194
247
 
195
248
  def probe_all(
@@ -198,9 +251,15 @@ def probe_all(
198
251
  now: datetime,
199
252
  max_idle: float,
200
253
  launch_grace: float,
254
+ stall_confirm: float | None = None,
255
+ sleep: Callable[[float], None] = time.sleep,
256
+ clock: Callable[[], datetime] = _utc_now,
201
257
  ) -> dict:
202
258
  probes = [
203
- probe_one(t, now=now, max_idle=max_idle, launch_grace=launch_grace)
259
+ probe_one(
260
+ t, now=now, max_idle=max_idle, launch_grace=launch_grace,
261
+ stall_confirm=stall_confirm, sleep=sleep, clock=clock,
262
+ )
204
263
  for t in targets
205
264
  ]
206
265
  unhealthy = [p for p in probes if p["state"] in ("stalled", "did-not-launch")]
@@ -221,6 +280,7 @@ def wait_for_results(
221
280
  launch_grace: float,
222
281
  interval: float,
223
282
  timeout: float,
283
+ stall_confirm: float | None = None,
224
284
  clock: Callable[[], datetime] = _utc_now,
225
285
  sleep: Callable[[float], None] = time.sleep,
226
286
  ) -> dict:
@@ -239,7 +299,8 @@ def wait_for_results(
239
299
  while True:
240
300
  now = clock()
241
301
  result = probe_all(
242
- targets, now=now, max_idle=max_idle, launch_grace=launch_grace
302
+ targets, now=now, max_idle=max_idle, launch_grace=launch_grace,
303
+ stall_confirm=stall_confirm, sleep=sleep, clock=clock,
243
304
  )
244
305
  pending = [
245
306
  str(t.result_path) for t in targets if not result_ready(t)
@@ -367,6 +428,15 @@ def main(argv: list[str] | None = None) -> int:
367
428
  help="--wait poll interval in seconds")
368
429
  parser.add_argument("--timeout", type=float, default=DEFAULT_WAIT_TIMEOUT_SECONDS,
369
430
  help="--wait deadline in seconds")
431
+ parser.add_argument(
432
+ "--stall-confirm", type=float, default=None,
433
+ help=(
434
+ "seconds to re-check a heartbeat budget breach before calling it "
435
+ "stalled (default: half that stage's budget; 0 disables). A slow "
436
+ "worker appends its next heartbeat inside this window; a dead one "
437
+ "does not."
438
+ ),
439
+ )
370
440
  args = parser.parse_args(argv)
371
441
 
372
442
  if len(args.team_state) != len(args.worker):
@@ -388,6 +458,7 @@ def main(argv: list[str] | None = None) -> int:
388
458
  now=_utc_now(),
389
459
  max_idle=args.max_idle,
390
460
  launch_grace=args.launch_grace,
461
+ stall_confirm=args.stall_confirm,
391
462
  )
392
463
  print(json.dumps(result, ensure_ascii=False, indent=2))
393
464
  # Non-zero on an unhealthy worker so a poll loop can branch on the exit
@@ -410,6 +481,7 @@ def main(argv: list[str] | None = None) -> int:
410
481
  launch_grace=args.launch_grace,
411
482
  interval=args.interval,
412
483
  timeout=args.timeout,
484
+ stall_confirm=args.stall_confirm,
413
485
  )
414
486
  print(json.dumps(result, ensure_ascii=False, indent=2))
415
487
  return {"completed": 0, "unhealthy": 1}.get(result["outcome"], 2)
@@ -2959,6 +2959,20 @@
2959
2959
  "required": ["roundCount", "gateResult", "planItems", "dissentLog"],
2960
2960
  "additionalProperties": false,
2961
2961
  "properties": {
2962
+ "uniformVerifiers": {
2963
+ "type": "array",
2964
+ "description": "Verifiers whose every vote this round was one verdict. Advisory: a unanimous round is legitimate, but the gate reads as a three-way cross-check unless this sits beside it.",
2965
+ "items": {
2966
+ "type": "object",
2967
+ "additionalProperties": false,
2968
+ "required": ["worker", "verdict", "itemCount"],
2969
+ "properties": {
2970
+ "worker": { "type": "string", "minLength": 1 },
2971
+ "verdict": { "type": "string", "minLength": 1 },
2972
+ "itemCount": { "type": "integer", "minimum": 1 }
2973
+ }
2974
+ }
2975
+ },
2962
2976
  "roundCount": { "type": "integer", "minimum": 0 },
2963
2977
  "gateResult": {
2964
2978
  "enum": [
@@ -1703,6 +1703,28 @@
1703
1703
  "enum": ["open", "answered", "resolved", "obsolete"]
1704
1704
  },
1705
1705
 
1706
+ "ClarificationScopeToken": {
1707
+ "enum": ["in-repo", "cross-repo", "new-schema", "deferrable"]
1708
+ },
1709
+
1710
+ "ClarificationOption": {
1711
+ "type": "object",
1712
+ "required": ["role", "answer", "rationale", "scopeImpact", "addedWork", "directionChange"],
1713
+ "additionalProperties": false,
1714
+ "properties": {
1715
+ "role": { "enum": ["recommended", "alternative"] },
1716
+ "answer": { "type": "string", "minLength": 1 },
1717
+ "rationale": { "type": "string", "minLength": 1 },
1718
+ "scopeImpact": {
1719
+ "type": "array",
1720
+ "minItems": 1,
1721
+ "items": { "$ref": "#/$defs/ClarificationScopeToken" }
1722
+ },
1723
+ "addedWork": { "type": "string", "minLength": 1 },
1724
+ "directionChange": { "type": "string", "minLength": 1 }
1725
+ }
1726
+ },
1727
+
1706
1728
  "WorkerStatus": {
1707
1729
  "enum": ["completed", "error", "timeout", "not-run", "synthesis-only"]
1708
1730
  },
@@ -3740,6 +3762,20 @@
3740
3762
  "required": ["roundCount", "gateResult", "planItems", "dissentLog"],
3741
3763
  "additionalProperties": false,
3742
3764
  "properties": {
3765
+ "uniformVerifiers": {
3766
+ "type": "array",
3767
+ "description": "Verifiers whose every vote this round was one verdict. Advisory: a unanimous round is legitimate, but the gate reads as a three-way cross-check unless this sits beside it.",
3768
+ "items": {
3769
+ "type": "object",
3770
+ "additionalProperties": false,
3771
+ "required": ["worker", "verdict", "itemCount"],
3772
+ "properties": {
3773
+ "worker": { "type": "string", "minLength": 1 },
3774
+ "verdict": { "type": "string", "minLength": 1 },
3775
+ "itemCount": { "type": "integer", "minimum": 1 }
3776
+ }
3777
+ }
3778
+ },
3743
3779
  "roundCount": { "type": "integer", "minimum": 0 },
3744
3780
  "gateResult": {
3745
3781
  "enum": [
@@ -4078,6 +4114,15 @@
4078
4114
  "type": "object",
4079
4115
  "required": ["id", "ticketId", "kind", "statement", "expectedForm", "blocks", "status"],
4080
4116
  "additionalProperties": false,
4117
+ "allOf": [
4118
+ {
4119
+ "if": {
4120
+ "properties": { "kind": { "const": "decision" } },
4121
+ "required": ["kind"]
4122
+ },
4123
+ "then": { "required": ["options"] }
4124
+ }
4125
+ ],
4081
4126
  "properties": {
4082
4127
  "id": { "type": "string", "pattern": "^C-\\d{3,}$" },
4083
4128
  "ticketId": { "$ref": "#/$defs/TicketId" },
@@ -4086,7 +4131,12 @@
4086
4131
  "expectedForm": { "type": "string", "minLength": 1 },
4087
4132
  "blocks": { "$ref": "#/$defs/ClarificationBlocks" },
4088
4133
  "status": { "$ref": "#/$defs/ClarificationStatus" },
4089
- "userInput": { "type": "string" }
4134
+ "userInput": { "type": "string" },
4135
+ "options": {
4136
+ "type": "array",
4137
+ "minItems": 2,
4138
+ "items": { "$ref": "#/$defs/ClarificationOption" }
4139
+ }
4090
4140
  }
4091
4141
  },
4092
4142
 
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  name: okstra-inspect
3
3
  description: >-
4
- Use this for everything that happens AFTER a single okstra task has already run — inspecting it or light bookkeeping on it, never launching new work. The tell is usually a named task id (PROD-1623, dev-9184), often dropped without the word "okstra." Reach for it when the user wants one task's: status, current/next phase, blockers, or approval gate; its final report — where it is or whether it passed (verdict / pass); its elapsed time or context/read cost; its run history, re-run, or resume; to mark it done / in-progress / blocked / todo; or a failed run's error logs gathered into a report (error report). Also bundles cross-project okstra errors into an anonymized feedback zip (error feedback). Use it even for a bare "mark it done" or "where's the report." NOT for starting a run (okstra-run), rollups/schedules (okstra-rollup / okstra-schedule-gen), a brief (okstra-brief-gen), setup (okstra-setup), or cross-project management (okstra-manager).
4
+ Use this for everything that happens AFTER a single okstra task has already run — inspecting it or light bookkeeping on it, never launching new work. The tell is usually a named task id (PROD-1623, dev-9184), often dropped without the word "okstra." Reach for it when the user wants one task's: status, current/next phase, blockers, or approval gate; its final report — where it is or whether it passed (verdict / pass); its elapsed time or context/read cost; its run history, re-run, or resume; to mark it done / in-progress / blocked / todo; or a failed run's error logs gathered into a report (error report). Also turns cross-project okstra errors into an anonymized zip (error feedback) or GitHub issues (file okstra issues), and audits run health across tasks (run audit). NOT for starting a run (okstra-run), rollups/schedules (okstra-rollup / okstra-schedule-gen), a brief (okstra-brief-gen), setup (okstra-setup), or cross-project management (okstra-manager).
5
5
  ---
6
6
 
7
7
  # OKSTRA Inspect
@@ -18,6 +18,8 @@ Single read-side entry point for okstra runtime inspection plus the one status m
18
18
  | `cost` | `facets/cost.md` | Estimate file/read context cost for a task bundle. |
19
19
  | `errors` | `facets/errors.md` | Aggregate okstra-run error logs for a task into a timestamped markdown report; print a summary. |
20
20
  | `error-zip` | `facets/error-zip.md` | Collect cross-project okstra error logs into an anonymized zip (report + raw) and summarize clusters. |
21
+ | `error-issue` | `facets/error-issue.md` | Turn cross-project okstra anomalies into GitHub issue candidates; file the approved ones after explicit user approval. |
22
+ | `run-audit` | `facets/run-audit.md` | Check every run's artifacts against progress invariants; report what went wrong without an error ever being logged. |
21
23
  | `recap` | `facets/recap.md` | Summarize a task's run-to-run phase transitions, then answer free-form questions over its `.okstra` artifacts. Appends each summary/Q&A to `recap/recap-log.jsonl`, and writes agent-authored notes to `notes/` for feeding into later runs. |
22
24
 
23
25
  ## Step 0: Preflight (shared)
@@ -0,0 +1,77 @@
1
+ # okstra-inspect facet — error-issue
2
+
3
+ Loaded lazily by the dispatch table in `SKILL.md` (core). Shared rules — Step 0 preflight, the standard task-key resolution rule (0/1/N), the no-task fallback, and Output Rules — live in the core file and still apply here.
4
+
5
+ ## error-issue
6
+
7
+ Trigger phrases: "okstra error-issue", "issue candidates", "file okstra issues", "report okstra defects".
8
+
9
+ Turn okstra run anomalies into GitHub issue candidates, then file the approved ones. Read-only over every target's `.okstra/`. Apart from the plan file you edit between the two commands, the only write is the GitHub issue — and that only after the user approves.
10
+
11
+ **Never run `submit` without an explicit user approval in this session.** `plan` is free to run unattended; `submit` is not.
12
+
13
+ ### error-issue.1 — Build the plan
14
+
15
+ ```bash
16
+ okstra error-issue plan --out ~/.okstra/error-issue-plan.json
17
+ ```
18
+
19
+ Parse the stdout JSON: `candidateCount`, `belowGate`, `notOkstraDefect`, `conflicted`, `unreachableRuns`.
20
+
21
+ `candidateCount` includes candidates whose action is already `skip`, so it is not the number of issues that will be filed. Report the filed count from the create/comment rows of the board below.
22
+
23
+ ### error-issue.2 — Review each candidate's classification
24
+
25
+ Read the plan file. For each candidate, verify the recorded `classification` against its `signals` list. The classification was computed by `issue_signals.classify`; your job is to confirm it reads correctly against the raw evidence, not to invent a new one.
26
+
27
+ If a candidate's signals do not support its classification, drop it from the plan file rather than arguing with it in prose. If deciding requires reading the raw `stderrExcerpt`, add an evidence entry and quote the line — an evidence entry always carries all three of `signal`, `value`, `source`, so write `{"signal": "<the signal it backs>", "value": "<the quoted line>", "source": "message excerpt"}`. `submit` rejects an entry missing any of the three; a two-key entry is not a lighter form of evidence, it is an unreadable one.
28
+
29
+ ### error-issue.3 — Present the approval board (Korean)
30
+
31
+ First state the destination on its own line — `plan.json` 의 `repo` 값을 그대로 읽어 `등록 대상: <repo>` 로 적는다. 이 값은 `~/.okstra/error-issue.json` 이 덮어쓸 수 있으므로 okstra 레포라고 단정하지 않는다. CLI 가 `repo` 없는 plan 을 거부하는 이유가 바로 이것 — 목적지는 승인 화면에 반드시 있어야 할 값이다.
32
+
33
+ Then show two tables, never merged into one.
34
+
35
+ **등록 예정** — candidates whose action is `create` or `comment`. `title` 열은 실제로 올라갈 제목 그대로 싣는다. 그것이 없으면 "승인 화면에서 본 것과 올라가는 것이 같다"가 성립하지 않는다:
36
+
37
+ | # | action | title | fingerprint | errorType / phase | 발생 | 지지 신호(`evidence`) | 기존 이슈 |
38
+ |---|---|---|---|---|---:|---|---|
39
+
40
+ **보류** — candidates whose action is already `skip`. They are NOT filed; they are shown so the human sees what was set aside. `skip` has two distinct origins and the row must say which:
41
+
42
+ - **반대 신호** — `conflictingSignals` is non-empty. `classify` called it an okstra defect while another signal pointed elsewhere.
43
+ - **이미 보고됨** — `conflictingSignals` is empty and `existingIssue` is set. An open issue already covers the last occurrence, so there is nothing new to say.
44
+
45
+ | # | fingerprint | errorType / phase | 발생 | 보류 사유 | 지지 신호 | 반대 신호 / 기존 이슈 |
46
+ |---|---|---|---:|---|---|---|
47
+
48
+ **본문** — then, for every `등록 예정` row, print that candidate's `body` **whole and verbatim** in a fenced block, one block per row, labelled with the same `#`. Never elide a section. The table summarizes; the body is what actually gets posted, and "승인 화면에서 본 것과 올라가는 것이 같다" is this facet's whole claim — a title plus a signal list does not carry it.
49
+
50
+ Bodies are bounded by construction: every line is derived (counts, signal values, invariant names) except the one representative message, which shares its source with the `title` in the same row. The raw error records are deliberately NOT in the body — their free text carries the reporting project's ticket ids and source filenames, and no shape rule can separate those from okstra's own vocabulary. They stay in the candidate's `records` in `plan.json` and in the `okstra error-zip` archive. If you need to vet them, read `plan.json` locally; do not paste them into the body.
51
+
52
+ Then ask via `AskUserQuestion` (options: 전부 등록 / 일부만 선택 / 등록 안 함). Deselected candidates are removed from the plan file before submit.
53
+
54
+ A held candidate is promoted only when the user explicitly says so, and the action you set depends on `existingIssue`:
55
+
56
+ - `existingIssue` is null → set `action` to `create`.
57
+ - `existingIssue` is set → set `action` to `comment`. **Never `create` over an existing issue** — that files a duplicate carrying the same `okstra-fingerprint` marker, and from then on `find_existing_issue` matches whichever of the two the list returns first, so every later run reports against an arbitrary one.
58
+
59
+ Say which you set and why.
60
+
61
+ Surface `unreachableRuns > 0` — no silent omission.
62
+
63
+ ### error-issue.4 — Submit
64
+
65
+ Only after an explicit approval:
66
+
67
+ ```bash
68
+ okstra error-issue submit --plan ~/.okstra/error-issue-plan.json
69
+ ```
70
+
71
+ Report `created` / `commented` / `skipped` / `rejected` / `failed`. If `rejected > 0`, print each rejection's `reasons` verbatim — a rejection means the outbound text or the evidence failed the final gate, and it is never something to work around.
72
+
73
+ If `failed > 0`, print each failure's `error` and **do not re-run `submit` on the same plan file**. A failure is a candidate that crashed or whose `gh` call errored; the candidates before it were already filed, and their `action` in the plan file still says `create`. Re-run `error-issue plan` instead — the fingerprint lookup turns the already-filed ones into `comment`/`skip`, which is the only thing that keeps a retry from filing duplicates.
74
+
75
+ ### error-issue.5 — Next step
76
+
77
+ End with: "이 이슈를 실제로 고치려면 `/okstra-brief-gen` 에 이슈 URL 을 주고, okstra 레포에서 `okstra-run --task-type error-analysis` 로 진행하세요."
@@ -0,0 +1,34 @@
1
+ # okstra-inspect facet — run-audit
2
+
3
+ Loaded lazily by the dispatch table in `SKILL.md` (core). Shared rules — Step 0 preflight, the standard task-key resolution rule (0/1/N), the no-task fallback, and Output Rules — live in the core file and still apply here.
4
+
5
+ ## run-audit
6
+
7
+ Trigger phrases: "okstra run-audit", "run audit", "진행 점검", "태스크들 제대로 가고 있나", "silent failures".
8
+
9
+ Check every run's artifacts against progress invariants. This catches what the error log cannot: a run that never logged a failure but still ended wrong. Read-only.
10
+
11
+ ### run-audit.1 — Run the audit
12
+
13
+ ```bash
14
+ okstra run-audit
15
+ ```
16
+
17
+ ### run-audit.2 — Report (Korean)
18
+
19
+ Group the `violations` array by `invariant` and report one section each:
20
+
21
+ | invariant | 위반 태스크 수 | 프로젝트 수 |
22
+ |---|---:|---|
23
+
24
+ Under each, list the affected `taskKey` values with their `detail` and `source`. Always cite `source` — the reader must be able to open the file the verdict came from.
25
+
26
+ ### run-audit.3 — Split the two outcomes
27
+
28
+ State plainly which violations are this task's own state versus a repeated okstra defect:
29
+
30
+ - an invariant broken in **one** task → that task's own state; tell the user what to do about that task
31
+ - an invariant broken across **two or more** tasks → an okstra defect; point at `/okstra-inspect error-issue` to file it
32
+ - `approval-not-forgotten` is the exception and never routes to `error-issue` no matter how far it spreads. It means a human has not approved yet, not that okstra did something wrong — `error_issue.NEVER_AN_ISSUE` drops it, so pointing the user at `error-issue` for it promises a filing that will never happen. Report it as a to-do list of tasks awaiting the user's approval instead.
33
+
34
+ Never file an issue from this facet. Filing is `error-issue`'s job and it has its own approval gate.