okstra 0.164.0 → 0.165.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/README.md +1 -1
  2. package/docs/architecture.md +12 -8
  3. package/docs/cli.md +7 -3
  4. package/docs/for-ai/README.md +2 -2
  5. package/docs/for-ai/skills/okstra-inspect.md +2 -2
  6. package/docs/for-ai/skills/okstra-user-response.md +2 -2
  7. package/docs/project-structure-overview.md +15 -9
  8. package/package.json +1 -1
  9. package/runtime/BUILD.json +2 -2
  10. package/runtime/agents/workers/antigravity-worker.md +9 -7
  11. package/runtime/agents/workers/codex-worker.md +9 -7
  12. package/runtime/agents/workers/grok-worker.md +6 -4
  13. package/runtime/agents/workers/kimi-worker.md +6 -4
  14. package/runtime/bin/okstra-antigravity-exec.sh +1 -340
  15. package/runtime/bin/okstra-claude-exec.sh +1 -178
  16. package/runtime/bin/okstra-codex-exec.sh +1 -467
  17. package/runtime/bin/okstra-provider-exec.py +165 -190
  18. package/runtime/bin/okstra-trace-cleanup.sh +14 -7
  19. package/runtime/bin/okstra-wrapper-status.py +26 -19
  20. package/runtime/prompts/lead/adapters/cmux.md +1 -1
  21. package/runtime/prompts/lead/convergence.md +36 -8
  22. package/runtime/prompts/lead/okstra-lead-contract.md +23 -1
  23. package/runtime/prompts/lead/plan-body-verification.md +9 -1
  24. package/runtime/prompts/lead/report-writer.md +1 -0
  25. package/runtime/prompts/lead/team-contract.md +3 -3
  26. package/runtime/prompts/profiles/_common-contract.md +9 -1
  27. package/runtime/prompts/profiles/_coverage-critic.md +1 -1
  28. package/runtime/prompts/profiles/_implementation-diff-review.md +3 -1
  29. package/runtime/prompts/profiles/_implementation-self-check.md +1 -1
  30. package/runtime/prompts/profiles/_implementation-verifier.md +3 -1
  31. package/runtime/prompts/profiles/implementation-planning.md +5 -3
  32. package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +1 -1
  33. package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +1 -1
  34. package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +148 -0
  35. package/runtime/python/okstra_ctl/adapters/providers/claude/adapter.py +55 -0
  36. package/runtime/python/okstra_ctl/adapters/providers/codex/adapter.py +41 -0
  37. package/runtime/python/okstra_ctl/adapters/providers/grok/adapter.py +44 -0
  38. package/runtime/python/okstra_ctl/adapters/providers/kimi/adapter.py +42 -0
  39. package/runtime/python/okstra_ctl/dispatch_core.py +5 -1
  40. package/runtime/python/okstra_ctl/dispatch_state.py +10 -0
  41. package/runtime/python/okstra_ctl/domain/provider.py +5 -1
  42. package/runtime/python/okstra_ctl/domain/worker_exec.py +102 -0
  43. package/runtime/python/okstra_ctl/domain/worker_role.py +34 -0
  44. package/runtime/python/okstra_ctl/domain/worker_stream.py +261 -0
  45. package/runtime/python/okstra_ctl/incremental_scope.py +16 -4
  46. package/runtime/python/okstra_ctl/report_html/common.py +71 -25
  47. package/runtime/python/okstra_ctl/report_html/models.py +5 -0
  48. package/runtime/python/okstra_ctl/report_html/render.py +1 -1
  49. package/runtime/python/okstra_ctl/report_html/run_usage.py +19 -0
  50. package/runtime/python/okstra_ctl/report_html/view_models/implementation_planning.py +14 -0
  51. package/runtime/python/okstra_ctl/report_views.py +44 -16
  52. package/runtime/python/okstra_ctl/stage_citations.py +52 -15
  53. package/runtime/python/okstra_ctl/user_response.py +45 -29
  54. package/runtime/python/okstra_ctl/wizard.py +13 -9
  55. package/runtime/python/okstra_ctl/worker_prompt_policy.py +10 -3
  56. package/runtime/python/okstra_ctl/worker_request.py +140 -0
  57. package/runtime/python/okstra_ctl/worker_runner.py +622 -0
  58. package/runtime/python/okstra_token_usage/collect.py +8 -1
  59. package/runtime/python/okstra_token_usage/report.py +42 -0
  60. package/runtime/python/okstra_token_usage/task_totals.py +88 -0
  61. package/runtime/schemas/final-report-v1.0.schema.json +70 -0
  62. package/runtime/schemas/final-report-v2.0.schema.json +90 -0
  63. package/runtime/skills/okstra-inspect/SKILL.md +1 -2
  64. package/runtime/skills/okstra-inspect/facets/logs.md +5 -5
  65. package/runtime/skills/okstra-inspect/facets/run-audit.md +3 -3
  66. package/runtime/skills/okstra-run/SKILL.md +1 -1
  67. package/runtime/skills/okstra-user-response/SKILL.md +15 -5
  68. package/runtime/templates/report-writer-prompt-preamble.md +1 -0
  69. package/runtime/templates/reports/html/assets/base.css +8 -4
  70. package/runtime/templates/reports/html/base.template.html +12 -6
  71. package/runtime/templates/reports/html/i18n/en.json +29 -6
  72. package/runtime/templates/reports/html/i18n/ko.json +29 -6
  73. package/runtime/templates/reports/html/macros/forms.html +9 -3
  74. package/runtime/templates/reports/html/tasks/implementation-planning.template.html +14 -19
  75. package/runtime/templates/reports/report.js +59 -26
  76. package/runtime/templates/reports/user-response.template.md +12 -8
  77. package/runtime/validators/validate-run.py +88 -7
  78. package/runtime/validators/validate_session_conformance.py +62 -1
  79. package/src/cli-registry.mjs +0 -7
  80. package/runtime/bin/okstra-wrapper-agy-stream.py +0 -61
  81. package/runtime/python/okstra_ctl/error_issue.py +0 -640
  82. package/runtime/python/okstra_ctl/issue_signals.py +0 -186
  83. package/runtime/skills/okstra-inspect/facets/error-issue.md +0 -77
  84. package/src/commands/inspect/error-issue.mjs +0 -27
@@ -5,21 +5,36 @@ import re
5
5
 
6
6
 
7
7
  # Every block whose rows prose cites but no section of its own renders, with
8
- # the keys each one names its content and its provenance by. A key hunt across
9
- # all of them would misread `crossVerification.consensus`, whose `evidence` key
10
- # holds provenance while `evidence.primary` uses the same word for content —
11
- # so the block a row came from is named here rather than guessed.
8
+ # the keys each one names its content, its provenance, and its confidence by. A
9
+ # key hunt across all of them would misread `crossVerification.consensus`, whose
10
+ # `evidence` key holds provenance while `evidence.primary` uses the same word
11
+ # for content — so the block a row came from is named here rather than guessed.
12
+ #
13
+ # Source and confidence are separate columns because most blocks carry only one
14
+ # of the two: `evidence.primary` cites a file and never rates itself, while
15
+ # `evidence.secondary` rates itself and cites nothing. Folding them into one
16
+ # field is what made the ledger label every row "source · confidence" and then
17
+ # print a single value under it.
18
+ #
19
+ # `endStateCoverage` is deliberately absent. Its rows hold an id pair
20
+ # (`EB-001` covered by `R-001`) rather than a statement, so the ledger rendered
21
+ # them as an id with no body, and the requirement-coverage table already names
22
+ # every one of them in its Source column.
23
+ #
24
+ # The kind each row carries is a vocabulary key, not the words the reader sees:
25
+ # the labels live in the i18n `ledgerKind` table so they arrive in the reader's
26
+ # language. `evidence.primary` was previously stamped "unclassified", which
27
+ # named the renderer's own indecision rather than anything about the row.
12
28
  _LEDGER_BLOCKS = (
13
- (("evidence", "primary"), "evidence", "source", "unclassified"),
14
- (("evidence", "secondary"), "hypothesis", "confidence", "hypothesis"),
15
- (("analysisCommon", "confirmedFacts"), "statement", "", "confirmed fact"),
16
- (("analysisCommon", "inferences"), "statement", "confidence", "inference"),
17
- (("analysisCommon", "unknowns"), "question", "reason", "unknown"),
18
- (("crossVerification", "consensus"), "statement", "evidence", "cross-check consensus"),
19
- (("crossVerification", "differences"), "disagreement", "workersPosition", "cross-check dissent"),
20
- (("missingInformation",), "item", "risk", "missing information"),
21
- (("endStateCoverage",), "coveredBy", "evidence", "exit contract"),
22
- (("followUpTasks",), "title", "reason", "follow-up"),
29
+ (("evidence", "primary"), "evidence", "source", "", "code-evidence"),
30
+ (("evidence", "secondary"), "hypothesis", "", "confidence", "hypothesis"),
31
+ (("analysisCommon", "confirmedFacts"), "statement", "", "", "confirmed-fact"),
32
+ (("analysisCommon", "inferences"), "statement", "", "confidence", "inference"),
33
+ (("analysisCommon", "unknowns"), "question", "reason", "", "unknown"),
34
+ (("crossVerification", "consensus"), "statement", "evidence", "", "cross-check-consensus"),
35
+ (("crossVerification", "differences"), "disagreement", "workersPosition", "", "cross-check-dissent"),
36
+ (("missingInformation",), "item", "risk", "", "missing-information"),
37
+ (("followUpTasks",), "title", "reason", "", "follow-up"),
23
38
  )
24
39
 
25
40
 
@@ -32,23 +47,39 @@ def _dig(data: dict, path: tuple[str, ...]) -> list:
32
47
  return node if isinstance(node, list) else []
33
48
 
34
49
 
35
- def _ledger_row(row: dict, text_key: str, source_key: str, kind: str) -> dict[str, object]:
50
+ def _joined(value: object) -> str:
51
+ """One reader-facing string for a cell a block may fill with either a
52
+ sentence or a list of citations. `str()` on a list prints Python's own
53
+ repr — quotes, brackets and all — straight into the report.
54
+ """
55
+ if isinstance(value, (list, tuple)):
56
+ return " · ".join(str(item) for item in value if item)
57
+ return str(value or "")
58
+
59
+
60
+ def _ledger_row(
61
+ row: dict, text_key: str, source_key: str, confidence_key: str, kind: str
62
+ ) -> dict[str, object]:
36
63
  return {
37
64
  "id": row.get("id", ""),
38
65
  "kind": kind,
39
- "text": str(row.get(text_key) or ""),
66
+ "text": _joined(row.get(text_key)),
40
67
  "codeEvidence": row.get("currentCodeEvidence") or [],
41
- "source": str(row.get(source_key) or "") if source_key else "",
68
+ "source": _joined(row.get(source_key)) if source_key else "",
69
+ "confidence": _joined(row.get(confidence_key)) if confidence_key else "",
42
70
  }
43
71
 
44
72
 
45
- def _own_section_ids(data: dict) -> set[str]:
73
+ def _own_section_ids(data: dict, omitted_fields: tuple[str, ...] = ()) -> set[str]:
46
74
  """Ids a section of the report already renders and anchors.
47
75
 
48
76
  The ledger is the fallback home for a cited row, so it must not claim an id
49
77
  that has one — two elements with the same anchor send half the links to the
50
78
  wrong place. `crossVerification.consensus` numbers its rows `C-001` in some
51
79
  runs, exactly where a clarification lives.
80
+
81
+ `omitted_fields` names task-block fields the HTML template does not render,
82
+ so their ids do not count as anchored.
52
83
  """
53
84
  from ..report_contract import TASK_TYPE_DATA_PROPERTY
54
85
 
@@ -56,20 +87,34 @@ def _own_section_ids(data: dict) -> set[str]:
56
87
  _collect_ids(data.get("clarificationItems", []), found)
57
88
  property_name = TASK_TYPE_DATA_PROPERTY.get(data.get("header", {}).get("taskType", ""))
58
89
  if property_name:
59
- _collect_ids(data.get(property_name, {}), found)
90
+ block = data.get(property_name, {})
91
+ if omitted_fields and isinstance(block, dict):
92
+ block = {
93
+ key: value for key, value in block.items() if key not in omitted_fields
94
+ }
95
+ _collect_ids(block, found)
60
96
  return found
61
97
 
62
98
 
63
99
  def evidence_index(data: dict) -> dict[str, object]:
100
+ """The rows the ledger carries, keyed by id.
101
+
102
+ A row whose statement came out empty is dropped rather than listed: the
103
+ ledger exists so a cited id resolves to something the reader can read, and
104
+ an id above a blank line resolves to nothing.
105
+ """
64
106
  owned = _own_section_ids(data)
65
107
  rows: dict[str, object] = {}
66
- for path, text_key, source_key, kind in _LEDGER_BLOCKS:
108
+ for path, text_key, source_key, confidence_key, kind in _LEDGER_BLOCKS:
67
109
  for row in _dig(data, path):
68
110
  if not isinstance(row, dict):
69
111
  continue
70
112
  row_id = row.get("id")
71
- if row_id and row_id not in owned and row_id not in rows:
72
- rows[row_id] = _ledger_row(row, text_key, source_key, kind)
113
+ if not row_id or row_id in owned or row_id in rows:
114
+ continue
115
+ entry = _ledger_row(row, text_key, source_key, confidence_key, kind)
116
+ if entry["text"]:
117
+ rows[row_id] = entry
73
118
  return rows
74
119
 
75
120
 
@@ -88,7 +133,7 @@ def _collect_ids(value: object, found: set[str]) -> None:
88
133
  _ROW_ID = re.compile(r"[A-Z]{1,3}-\d+")
89
134
 
90
135
 
91
- def anchor_index(data: dict) -> dict[str, str]:
136
+ def anchor_index(data: dict, omitted_fields: tuple[str, ...] = ()) -> dict[str, str]:
92
137
  """Map every row a reader can reach to the anchor name that lands on it.
93
138
 
94
139
  Prose cites ids across section boundaries — a hotspot names a
@@ -98,9 +143,10 @@ def anchor_index(data: dict) -> dict[str, str]:
98
143
 
99
144
  It stops there. `summary` is the AI-facing digest and
100
145
  `analysisCommon.scope` describes the analysis target rather than listing
101
- rows; neither renders, so a link to one would land nowhere.
146
+ rows; neither renders, so a link to one would land nowhere. Same for the
147
+ blocks a template declares in `omitted_fields`.
102
148
  """
103
- found = _own_section_ids(data) | set(evidence_index(data))
149
+ found = _own_section_ids(data, omitted_fields) | set(evidence_index(data))
104
150
  return {row_id: f"id-{row_id}" for row_id in sorted(found) if _ROW_ID.fullmatch(row_id)}
105
151
 
106
152
 
@@ -66,6 +66,11 @@ class HumanReportView:
66
66
  template_name: str
67
67
  context: dict[str, object]
68
68
  figures: tuple[FigureModel, ...]
69
+ # Task-block fields this template leaves out of the HTML. Their rows still
70
+ # exist in the data and the markdown, so prose keeps citing their ids — but
71
+ # an anchor to a row this document never renders scrolls nowhere, which
72
+ # reads as a broken report rather than as a deliberate omission.
73
+ omitted_fields: tuple[str, ...] = ()
69
74
 
70
75
 
71
76
  @dataclass(frozen=True)
@@ -130,7 +130,7 @@ def render_v2_html_view(
130
130
  env.policies["json.dumps_kwargs"] = {"sort_keys": True, "ensure_ascii": False}
131
131
  # Binding the index here is what lets a template cite an id without
132
132
  # threading the index through every macro and call site.
133
- anchors = anchor_index(data)
133
+ anchors = anchor_index(data, view.omitted_fields)
134
134
  chrome = load_dictionary(lang, HTML_DICTIONARY_REL)
135
135
  translate = make_jinja_global(chrome)
136
136
  env.globals["t"] = translate
@@ -23,6 +23,7 @@ def _totals_row(row: object) -> dict[str, str]:
23
23
  values = row if isinstance(row, dict) else {}
24
24
  return {
25
25
  "rawTokens": format_int(values.get("totalTokens")),
26
+ "cacheReadTokens": format_int(values.get("cacheReadTokens")),
26
27
  "billableTokens": format_int(values.get("billableTokens")),
27
28
  "cost": format_usd(values.get("costUsd")),
28
29
  }
@@ -43,6 +44,7 @@ def _agent_row(row: dict) -> dict[str, str]:
43
44
  "model": str(row.get("model") or ""),
44
45
  "status": str(row.get("status") or ""),
45
46
  "rawTokens": format_int(row.get("totalTokens")),
47
+ "cacheReadTokens": format_int(row.get("cacheReadTokens")),
46
48
  "billableTokens": format_int(row.get("billableTokens")),
47
49
  "cost": format_usd(row.get("costUsd")),
48
50
  "duration": format_duration_ms(row.get("durationMs")),
@@ -70,15 +72,31 @@ def _unaccounted(rows: list[dict], grand: dict) -> dict[str, str] | None:
70
72
  gap = grand_tokens - _sum(rows, "totalTokens")
71
73
  if gap <= 0:
72
74
  return None
75
+ cache_gap = (_number(grand.get("cacheReadTokens")) or 0) - _sum(rows, "cacheReadTokens")
73
76
  billable_gap = (_number(grand.get("billableTokens")) or 0) - _sum(rows, "billableTokens")
74
77
  cost_gap = (_number(grand.get("costUsd")) or 0) - _sum(rows, "costUsd")
75
78
  return {
76
79
  "rawTokens": format_int(gap),
80
+ "cacheReadTokens": format_int(max(0, cache_gap)),
77
81
  "billableTokens": format_int(max(0, billable_gap)),
78
82
  "cost": format_usd(max(0.0, cost_gap)),
79
83
  }
80
84
 
81
85
 
86
+ def _task_cumulative(row: object) -> dict[str, str] | None:
87
+ """What the task has spent over every run, when more than this one exists.
88
+
89
+ A single-run task would repeat the grand total word for word, and a reader
90
+ seeing the same figure twice reads it as a second charge.
91
+ """
92
+ if not isinstance(row, dict):
93
+ return None
94
+ run_count = _number(row.get("runCount")) or 0
95
+ if run_count < 2:
96
+ return None
97
+ return {**_totals_row(row), "runCount": format_int(run_count)}
98
+
99
+
82
100
  _MEASURED_KEYS = ("totalTokens", "billableTokens", "costUsd", "durationMs")
83
101
 
84
102
 
@@ -106,5 +124,6 @@ def run_usage(data: dict) -> dict[str, object] | None:
106
124
  "rows": [_agent_row(row) for row in rows],
107
125
  "unaccounted": _unaccounted(rows, usage.get("grand") or {}),
108
126
  **totals,
127
+ "taskCumulative": _task_cumulative(usage.get("taskCumulative")),
109
128
  "cliCost": format_usd(cli_cost) if cli_cost else "",
110
129
  }
@@ -74,6 +74,19 @@ def _stage_figure(planning: dict):
74
74
  return stage_map_figure(nodes=nodes, edges=edges, title="Implementation stage dependencies")
75
75
 
76
76
 
77
+ # Blocks the approver's view leaves to the markdown report: the checklist the
78
+ # implementer works from, the dependency and rollback tables an incident reader
79
+ # needs, and the verification rounds the audit trail keeps. Declared here so the
80
+ # ids they carry stop being anchor targets in this document.
81
+ _OMITTED_FIELDS = (
82
+ "validationChecklist",
83
+ "crossProjectDependencies",
84
+ "dependencyMigrationRisk",
85
+ "rollbackStrategy",
86
+ "planBodyVerification",
87
+ )
88
+
89
+
77
90
  def build_implementation_planning_view(data: dict) -> HumanReportView:
78
91
  planning = data["implementationPlanning"]
79
92
  figure = _stage_figure(planning)
@@ -97,4 +110,5 @@ def build_implementation_planning_view(data: dict) -> HumanReportView:
97
110
  "html/tasks/implementation-planning.template.html",
98
111
  context,
99
112
  (figure,),
113
+ _OMITTED_FIELDS,
100
114
  )
@@ -1050,11 +1050,16 @@ class UserResponseEntry:
1050
1050
 
1051
1051
 
1052
1052
  @dataclass(frozen=True)
1053
- class UserResponseApproval:
1054
- """HTML Plan Approval 위젯의 Export 결과. ``implementation_option`` 이 빈
1055
- 문자열이면 라인을 생략한다 (소비 측은 Recommended Option 폴백)."""
1056
- approved: bool
1053
+ class UserPlanDecision:
1054
+ """HTML 계획 결정 위젯의 Export 결과.
1055
+
1056
+ ``status`` 가 승인이 아니면 ``reason`` 이 필수다 — 사유 없는 반려는 다음
1057
+ run 이 무엇을 고쳐야 하는지 알 수 없어 되돌아올 수밖에 없다.
1058
+ ``implementation_option`` 이 빈 문자열이면 라인을 생략한다 (소비 측은
1059
+ Recommended Option 폴백)."""
1060
+ status: str
1057
1061
  implementation_option: str = ""
1062
+ reason: str = ""
1058
1063
 
1059
1064
 
1060
1065
  @dataclass(frozen=True)
@@ -1072,6 +1077,13 @@ _ANALYSIS_REVIEW_STATUSES = frozenset({
1072
1077
  "rejected",
1073
1078
  })
1074
1079
 
1080
+ PLAN_DECISION_APPROVED = "approved"
1081
+ _PLAN_DECISION_STATUSES = frozenset({
1082
+ PLAN_DECISION_APPROVED,
1083
+ "revision-requested",
1084
+ "rejected",
1085
+ })
1086
+
1075
1087
 
1076
1088
  def _quoted_sidecar_field(label: str, value: str) -> str:
1077
1089
  cleaned = value.strip()
@@ -1100,12 +1112,25 @@ def _serialize_analysis_review(review: UserResponseAnalysisReview) -> str:
1100
1112
  )
1101
1113
 
1102
1114
 
1115
+ def _serialize_plan_decision(decision: UserPlanDecision) -> str:
1116
+ if decision.status not in _PLAN_DECISION_STATUSES:
1117
+ raise ValueError(f"invalid PLAN DECISION status: {decision.status}")
1118
+ if decision.status != PLAN_DECISION_APPROVED and not decision.reason.strip():
1119
+ raise ValueError(f"PLAN DECISION {decision.status} requires a Reason")
1120
+ chunk = f"\n## PLAN DECISION\n- Status: {decision.status}\n"
1121
+ if decision.implementation_option:
1122
+ chunk += f"- Implementation-Option: {decision.implementation_option.strip()}\n"
1123
+ if decision.reason.strip():
1124
+ chunk += _quoted_sidecar_field("Reason", decision.reason)
1125
+ return chunk
1126
+
1127
+
1103
1128
  def serialize_user_response(
1104
1129
  *,
1105
1130
  run_meta: RunMeta,
1106
1131
  entries: list[UserResponseEntry],
1107
1132
  created_at: str,
1108
- approval: UserResponseApproval | None = None,
1133
+ plan_decision: UserPlanDecision | None = None,
1109
1134
  analysis_review: UserResponseAnalysisReview | None = None,
1110
1135
  ) -> str:
1111
1136
  """Return the canonical markdown text the HTML 'Export user
@@ -1125,7 +1150,7 @@ def serialize_user_response(
1125
1150
  "\n"
1126
1151
  "# User Response\n"
1127
1152
  )
1128
- has_approval = approval is not None and approval.approved
1153
+ has_plan_decision = plan_decision is not None
1129
1154
  has_analysis_review = analysis_review is not None
1130
1155
  body_chunks: list[str] = []
1131
1156
  for e in entries:
@@ -1137,13 +1162,10 @@ def serialize_user_response(
1137
1162
  if e.rationale:
1138
1163
  chunk += f"- Rationale: {e.rationale.strip()}\n"
1139
1164
  body_chunks.append(chunk)
1140
- if not entries and not has_approval and not has_analysis_review:
1165
+ if not entries and not has_plan_decision and not has_analysis_review:
1141
1166
  body_chunks.append("\n_(No user responses recorded.)_\n")
1142
- if has_approval:
1143
- chunk = "\n## APPROVAL\n- Approved: true\n"
1144
- if approval.implementation_option:
1145
- chunk += f"- Implementation-Option: {approval.implementation_option.strip()}\n"
1146
- body_chunks.append(chunk)
1167
+ if plan_decision is not None:
1168
+ body_chunks.append(_serialize_plan_decision(plan_decision))
1147
1169
  if analysis_review is not None:
1148
1170
  body_chunks.append(_serialize_analysis_review(analysis_review))
1149
1171
  return head + "".join(body_chunks)
@@ -1369,11 +1391,17 @@ def _plan_approval_section(ctx: PlanApprovalContext, run_meta: RunMeta) -> str:
1369
1391
  )
1370
1392
  return (
1371
1393
  '<section id="plan-approval">\n'
1372
- " <h2>Plan Approval</h2>\n"
1394
+ " <h2>Plan Decision</h2>\n"
1373
1395
  f' <label>구현 옵션: <select id="approval-option"{disabled}>{"".join(opts)}</select></label>\n'
1374
- f' <label><input type="checkbox" id="approval-checkbox"{disabled}> '
1375
- "이 plan 을 승인합니다 — Export 시 sidecar 에 APPROVAL 블록으로 기록되고, "
1376
- "implementation 시작 마법사가 확인 후 적용합니다.</label>"
1396
+ ' <fieldset id="plan-decision"><legend>판정</legend>'
1397
+ f'<label><input type="radio" name="plan-decision-status" value="approved"{disabled}> '
1398
+ "이 plan 을 승인합니다</label>"
1399
+ '<label><input type="radio" name="plan-decision-status" '
1400
+ 'value="revision-requested"> 고쳐서 다시 가져오게 합니다</label>'
1401
+ '<label><input type="radio" name="plan-decision-status" value="rejected"> '
1402
+ "이 plan 을 반려합니다</label></fieldset>\n"
1403
+ ' <label>사유 — 반려하거나 다시 고치게 할 때는 반드시 적어야 합니다'
1404
+ '<textarea id="plan-decision-reason" rows="4"></textarea></label>'
1377
1405
  f"{reason_html}\n"
1378
1406
  "</section>\n"
1379
1407
  )
@@ -9,6 +9,7 @@ which decides whether a re-run narrows or stays full.
9
9
  from __future__ import annotations
10
10
 
11
11
  import re
12
+ from collections.abc import Iterator
12
13
 
13
14
  _RANGE = r"(?:[-–]|\bto\b|\bthrough\b)"
14
15
  _ITEM = rf"\d+(?:\s*{_RANGE}\s*\d+)?"
@@ -23,25 +24,61 @@ _ITEM_RE = re.compile(rf"(\d+)(?:\s*{_RANGE}\s*(\d+))?", re.IGNORECASE)
23
24
  RANGE_MAX_SPAN = 64
24
25
 
25
26
 
27
+ def _stage_citation_items(text: str) -> Iterator[tuple[int, int | None]]:
28
+ """Each `stage`-anchored citation in *text* as `(start, end)`.
29
+
30
+ *end* is None for a bare number and the far endpoint for a range. Numbers
31
+ stay anchored to a word-initial `stage`/`stages` token on the same line;
32
+ harvesting bare numbers — or letting the anchor reach across a line break,
33
+ or match the tail of `Substage`/`Backstage` — would let any prose, including
34
+ a row disclaiming every stage, justify any stage.
35
+ """
36
+ for span in _LIST_RE.finditer(text):
37
+ for item in _ITEM_RE.finditer(span.group(1)):
38
+ end = item.group(2)
39
+ yield int(item.group(1)), int(end) if end is not None else None
40
+
41
+
26
42
  def cited_stage_numbers(text: str) -> set[int]:
27
43
  """Stage numbers *text* cites, in every prose form a planner writes.
28
44
 
29
- Numbers stay anchored to a word-initial `stage`/`stages` token on the same
30
- line; harvesting bare numbers — or letting the anchor reach across a line
31
- break, or match the tail of `Substage`/`Backstage` — would let any prose,
32
- including a row disclaiming every stage, justify any stage.
45
+ A range reaches every number between its endpoints here, which is what a
46
+ reader asking "could this touch stage 5" needs.
33
47
  """
34
48
  cited: set[int] = set()
49
+ for start, end in _stage_citation_items(text):
50
+ cited.add(start)
51
+ if end is None:
52
+ continue
53
+ cited.add(end)
54
+ # A reversed or absurdly wide range is a typo, not a citation of
55
+ # everything between its endpoints.
56
+ if 0 <= end - start <= RANGE_MAX_SPAN:
57
+ cited.update(range(start, end))
58
+ return cited
59
+
60
+
61
+ def enumerated_stage_numbers(text: str) -> set[int]:
62
+ """Stage numbers *text* names one by one — a range's interior excluded.
63
+
64
+ The two readers of this grammar ask opposite questions, and a range answers
65
+ only one of them. The incremental-scope back-trace asks "could this answer
66
+ reach stage 5", so it must read `Stages 1-8` as reaching it — that is
67
+ `cited_stage_numbers`, and widening there is the safe direction. Coverage
68
+ provenance asks "did the planner confirm stage 5 satisfies this
69
+ requirement", and a range answers that for free: one `Stages 1-64` cell
70
+ stamps every stage in the map without the planner looking at any of them.
71
+
72
+ So the interior is dropped here and only hand-written numbers count. A
73
+ range's endpoints ARE hand-written and stay, which keeps `Stages 7-8`
74
+ honest while forcing the wide case to be spelled out. The cost of a
75
+ legitimately broad requirement is typing each number, and that typing is
76
+ the confirmation this check is asking for.
77
+ """
78
+ enumerated: set[int] = set()
35
79
  for span in _LIST_RE.finditer(text):
36
80
  for item in _ITEM_RE.finditer(span.group(1)):
37
- start = int(item.group(1))
38
- cited.add(start)
39
- if item.group(2) is None:
40
- continue
41
- end = int(item.group(2))
42
- cited.add(end)
43
- # A reversed or absurdly wide range is a typo, not a citation of
44
- # everything between its endpoints.
45
- if 0 <= end - start <= RANGE_MAX_SPAN:
46
- cited.update(range(start, end))
47
- return cited
81
+ enumerated.add(int(item.group(1)))
82
+ if item.group(2) is not None:
83
+ enumerated.add(int(item.group(2)))
84
+ return enumerated
@@ -3,7 +3,7 @@
3
3
  The sidecar format is documented in ``templates/reports/user-response.template.md``
4
4
  and produced byte-identically by ``report_views.serialize_user_response`` (Python)
5
5
  and ``templates/reports/report.js`` (browser). This module owns the read side of
6
- the ``## APPROVAL`` block used by the implementation wizard and the optional
6
+ the ``## PLAN DECISION`` block used by the implementation wizard and the optional
7
7
  ``## ANALYSIS REVIEW`` block used by analysis reruns.
8
8
  """
9
9
  from __future__ import annotations
@@ -19,7 +19,8 @@ from pathlib import Path
19
19
  from typing import Optional
20
20
 
21
21
  from okstra_ctl.report_views import (
22
- serialize_user_response, UserResponseEntry, UserResponseApproval, infer_run_meta,
22
+ PLAN_DECISION_APPROVED,
23
+ serialize_user_response, UserResponseEntry, UserPlanDecision, infer_run_meta,
23
24
  parse_expected_form_options,
24
25
  )
25
26
  from okstra_ctl.report_view_artifacts import user_responses_dir_for_report
@@ -32,7 +33,7 @@ from okstra_ctl.clarification_items import (
32
33
  _section_1_slice,
33
34
  )
34
35
 
35
- _APPROVAL_HEADING_RE = re.compile(r"^## APPROVAL\s*$", re.MULTILINE)
36
+ _PLAN_DECISION_HEADING_RE = re.compile(r"^## PLAN DECISION\s*$", re.MULTILINE)
36
37
  _NEXT_RESPONSE_HEADING_RE = re.compile(r"^## ", re.MULTILINE)
37
38
  _ANALYSIS_REVIEW_HEADING_RE = re.compile(r"^## ANALYSIS REVIEW\s*$", re.MULTILINE)
38
39
  _ANALYSIS_SIDECAR_HEADING_RE = re.compile(
@@ -50,13 +51,18 @@ class UserResponseError(ValueError):
50
51
 
51
52
 
52
53
  @dataclass(frozen=True)
53
- class UserResponseApprovalRecord:
54
- """sidecar 의 ``## APPROVAL`` 블록 + 매칭에 필요한 frontmatter 필드."""
55
- approved: bool
54
+ class PlanDecisionRecord:
55
+ """sidecar 의 ``## PLAN DECISION`` 블록 + 매칭에 필요한 frontmatter 필드."""
56
+ status: str
56
57
  implementation_option: str
58
+ reason: str
57
59
  source_report: str
58
60
  seq: str
59
61
 
62
+ @property
63
+ def approved(self) -> bool:
64
+ return self.status == PLAN_DECISION_APPROVED
65
+
60
66
 
61
67
  @dataclass(frozen=True)
62
68
  class AnalysisReviewRecord:
@@ -353,31 +359,37 @@ def load_authoritative_analysis_review(
353
359
  return review
354
360
 
355
361
 
356
- def parse_user_response_approval(
357
- sidecar_text: str,
358
- ) -> Optional[UserResponseApprovalRecord]:
359
- """``## APPROVAL`` 블록이 ``- Approved: true`` 를 가질 때만 record 를
360
- 반환한다. 옵션 라인은 블록 내부에서만 읽는다 (다른 응답 본문의 우연한
361
- 동일 문구를 옵션으로 오인하지 않도록).
362
+ _PLAN_DECISION_STATUS_RE = re.compile(
363
+ r"^- Status:\s*(approved|revision-requested|rejected)\s*$", re.MULTILINE
364
+ )
365
+
366
+
367
+ def parse_plan_decision(sidecar_text: str) -> Optional[PlanDecisionRecord]:
368
+ """``## PLAN DECISION`` 블록을 읽어 record 로 돌려준다. 승인·재작업·반려를
369
+ 구분하지 않고 그대로 싣는다 — 어느 판정을 받아들일지는 소비 측 정책이다.
370
+ 옵션·사유 라인은 블록 내부에서만 읽는다 (다른 응답 본문의 우연한 동일
371
+ 문구를 옵션으로 오인하지 않도록).
362
372
 
363
- strictness 는 의도적이다: producer 출력과 byte-identical 한 소문자
364
- ``true`` 만 인정하며, 손편집 변형(``TRUE``/``True``)은 fail-closed 로
373
+ strictness 는 의도적이다: producer 출력과 byte-identical 한 소문자 status
374
+ 만 인정하며, 손편집 변형(``Approved``/``REJECTED``)은 fail-closed 로
365
375
  불인정한다."""
366
- m = _APPROVAL_HEADING_RE.search(sidecar_text)
376
+ m = _PLAN_DECISION_HEADING_RE.search(sidecar_text)
367
377
  if not m:
368
378
  return None
369
379
  block = sidecar_text[m.end():]
370
380
  nxt = _NEXT_RESPONSE_HEADING_RE.search(block)
371
381
  if nxt:
372
382
  block = block[: nxt.start()]
373
- if re.search(r"^- Approved:\s*true\s*$", block, re.MULTILINE) is None:
383
+ status = _PLAN_DECISION_STATUS_RE.search(block)
384
+ if status is None:
374
385
  return None
375
386
  om = re.search(r"^- Implementation-Option:\s*(\S.*?)\s*$", block, re.MULTILINE)
376
387
  sm = re.search(r"^seq:\s*(\S+)\s*$", sidecar_text, re.MULTILINE)
377
388
  rm = re.search(r"^source-report:\s*(\S.*?)\s*$", sidecar_text, re.MULTILINE)
378
- return UserResponseApprovalRecord(
379
- approved=True,
389
+ return PlanDecisionRecord(
390
+ status=status.group(1),
380
391
  implementation_option=om.group(1) if om else "",
392
+ reason=_quoted_review_value(block, "Reason"),
381
393
  source_report=rm.group(1) if rm else "",
382
394
  seq=sm.group(1) if sm else "",
383
395
  )
@@ -404,8 +416,8 @@ def _value(block: str) -> str:
404
416
 
405
417
  def parse_user_response_entries(sidecar_text: str) -> list[UserResponseEntry]:
406
418
  """Reverse of ``serialize_user_response`` for the per-response ``## C-*``
407
- blocks. The ``## APPROVAL`` block is skipped (read separately by
408
- ``parse_user_response_approval``)."""
419
+ blocks. The ``## PLAN DECISION`` block is skipped (read separately by
420
+ ``parse_plan_decision``)."""
409
421
  entries: list[UserResponseEntry] = []
410
422
  matches = list(_RESPONSE_HEADING_RE.finditer(sidecar_text))
411
423
  for i, m in enumerate(matches):
@@ -581,7 +593,7 @@ def show_open_rows(report_path: Path) -> dict:
581
593
 
582
594
 
583
595
  def write_sidecar(report_path: Path, answers: list[dict],
584
- approval: Optional[dict], created_at: str,
596
+ plan_decision: Optional[dict], created_at: str,
585
597
  task_key: str = "") -> Path:
586
598
  run_meta = infer_run_meta(report_path, task_key=task_key or None)
587
599
  out_dir = user_responses_dir_for_report(report_path)
@@ -597,13 +609,15 @@ def write_sidecar(report_path: Path, answers: list[dict],
597
609
  response_id=a["id"], kind=a.get("kind", ""), value=a["value"],
598
610
  rationale=a.get("rationale"), disposition=a.get("disposition", "answer"))
599
611
 
600
- appr = None
601
- if approval and approval.get("approved"):
602
- appr = UserResponseApproval(
603
- approved=True, implementation_option=approval.get("implementationOption", ""))
612
+ decision = None
613
+ if plan_decision and plan_decision.get("status"):
614
+ decision = UserPlanDecision(
615
+ status=plan_decision["status"],
616
+ implementation_option=plan_decision.get("implementationOption", ""),
617
+ reason=plan_decision.get("reason", ""))
604
618
  sidecar.write_text(
605
619
  serialize_user_response(run_meta=run_meta, entries=list(merged.values()),
606
- created_at=created_at, approval=appr),
620
+ created_at=created_at, plan_decision=decision),
607
621
  encoding="utf-8")
608
622
  return sidecar
609
623
 
@@ -623,7 +637,9 @@ def main(argv: Optional[list[str]] = None) -> int:
623
637
  pw = sub.add_parser("write")
624
638
  pw.add_argument("--report", required=True)
625
639
  pw.add_argument("--answers", required=True, help="JSON array of answer entries")
626
- pw.add_argument("--approval", default="", help="JSON approval object")
640
+ pw.add_argument(
641
+ "--plan-decision", default="",
642
+ help='JSON plan decision, e.g. {"status":"rejected","reason":"..."}')
627
643
  pw.add_argument("--task-key", default="", help="task-key from list/show context")
628
644
 
629
645
  ns = parser.parse_args(argv)
@@ -636,9 +652,9 @@ def main(argv: Optional[list[str]] = None) -> int:
636
652
  return 0
637
653
  if ns.cmd == "write":
638
654
  answers = json.loads(ns.answers)
639
- approval = json.loads(ns.approval) if ns.approval else None
655
+ decision = json.loads(ns.plan_decision) if ns.plan_decision else None
640
656
  created_at = dt.datetime.now(dt.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
641
- p = write_sidecar(Path(ns.report), answers, approval, created_at,
657
+ p = write_sidecar(Path(ns.report), answers, decision, created_at,
642
658
  task_key=ns.task_key)
643
659
  json.dump({"sidecar": str(p)}, sys.stdout, ensure_ascii=False)
644
660
  return 0