okstra 0.164.0 → 0.165.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/docs/architecture.md +12 -8
- package/docs/cli.md +7 -3
- package/docs/for-ai/README.md +2 -2
- package/docs/for-ai/skills/okstra-inspect.md +2 -2
- package/docs/for-ai/skills/okstra-user-response.md +2 -2
- package/docs/project-structure-overview.md +15 -9
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/workers/antigravity-worker.md +9 -7
- package/runtime/agents/workers/codex-worker.md +9 -7
- package/runtime/agents/workers/grok-worker.md +6 -4
- package/runtime/agents/workers/kimi-worker.md +6 -4
- package/runtime/bin/okstra-antigravity-exec.sh +1 -340
- package/runtime/bin/okstra-claude-exec.sh +1 -178
- package/runtime/bin/okstra-codex-exec.sh +1 -467
- package/runtime/bin/okstra-provider-exec.py +165 -190
- package/runtime/bin/okstra-trace-cleanup.sh +14 -7
- package/runtime/bin/okstra-wrapper-status.py +26 -19
- package/runtime/prompts/lead/adapters/cmux.md +1 -1
- package/runtime/prompts/lead/convergence.md +36 -8
- package/runtime/prompts/lead/okstra-lead-contract.md +23 -1
- package/runtime/prompts/lead/plan-body-verification.md +9 -1
- package/runtime/prompts/lead/report-writer.md +1 -0
- package/runtime/prompts/lead/team-contract.md +3 -3
- package/runtime/prompts/profiles/_common-contract.md +9 -1
- package/runtime/prompts/profiles/_coverage-critic.md +1 -1
- package/runtime/prompts/profiles/_implementation-diff-review.md +3 -1
- package/runtime/prompts/profiles/_implementation-self-check.md +1 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +3 -1
- package/runtime/prompts/profiles/implementation-planning.md +5 -3
- package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +148 -0
- package/runtime/python/okstra_ctl/adapters/providers/claude/adapter.py +55 -0
- package/runtime/python/okstra_ctl/adapters/providers/codex/adapter.py +41 -0
- package/runtime/python/okstra_ctl/adapters/providers/grok/adapter.py +44 -0
- package/runtime/python/okstra_ctl/adapters/providers/kimi/adapter.py +42 -0
- package/runtime/python/okstra_ctl/dispatch_core.py +5 -1
- package/runtime/python/okstra_ctl/dispatch_state.py +10 -0
- package/runtime/python/okstra_ctl/domain/provider.py +5 -1
- package/runtime/python/okstra_ctl/domain/worker_exec.py +102 -0
- package/runtime/python/okstra_ctl/domain/worker_role.py +34 -0
- package/runtime/python/okstra_ctl/domain/worker_stream.py +261 -0
- package/runtime/python/okstra_ctl/incremental_scope.py +16 -4
- package/runtime/python/okstra_ctl/report_html/common.py +71 -25
- package/runtime/python/okstra_ctl/report_html/models.py +5 -0
- package/runtime/python/okstra_ctl/report_html/render.py +1 -1
- package/runtime/python/okstra_ctl/report_html/run_usage.py +19 -0
- package/runtime/python/okstra_ctl/report_html/view_models/implementation_planning.py +14 -0
- package/runtime/python/okstra_ctl/report_views.py +44 -16
- package/runtime/python/okstra_ctl/stage_citations.py +52 -15
- package/runtime/python/okstra_ctl/user_response.py +45 -29
- package/runtime/python/okstra_ctl/wizard.py +13 -9
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +10 -3
- package/runtime/python/okstra_ctl/worker_request.py +140 -0
- package/runtime/python/okstra_ctl/worker_runner.py +622 -0
- package/runtime/python/okstra_token_usage/collect.py +8 -1
- package/runtime/python/okstra_token_usage/report.py +42 -0
- package/runtime/python/okstra_token_usage/task_totals.py +88 -0
- package/runtime/schemas/final-report-v1.0.schema.json +70 -0
- package/runtime/schemas/final-report-v2.0.schema.json +90 -0
- package/runtime/skills/okstra-inspect/SKILL.md +1 -2
- package/runtime/skills/okstra-inspect/facets/logs.md +5 -5
- package/runtime/skills/okstra-inspect/facets/run-audit.md +3 -3
- package/runtime/skills/okstra-run/SKILL.md +1 -1
- package/runtime/skills/okstra-user-response/SKILL.md +15 -5
- package/runtime/templates/report-writer-prompt-preamble.md +1 -0
- package/runtime/templates/reports/html/assets/base.css +8 -4
- package/runtime/templates/reports/html/base.template.html +12 -6
- package/runtime/templates/reports/html/i18n/en.json +29 -6
- package/runtime/templates/reports/html/i18n/ko.json +29 -6
- package/runtime/templates/reports/html/macros/forms.html +9 -3
- package/runtime/templates/reports/html/tasks/implementation-planning.template.html +14 -19
- package/runtime/templates/reports/report.js +59 -26
- package/runtime/templates/reports/user-response.template.md +12 -8
- package/runtime/validators/validate-run.py +88 -7
- package/runtime/validators/validate_session_conformance.py +62 -1
- package/src/cli-registry.mjs +0 -7
- package/runtime/bin/okstra-wrapper-agy-stream.py +0 -61
- package/runtime/python/okstra_ctl/error_issue.py +0 -640
- package/runtime/python/okstra_ctl/issue_signals.py +0 -186
- package/runtime/skills/okstra-inspect/facets/error-issue.md +0 -77
- package/src/commands/inspect/error-issue.mjs +0 -27
|
@@ -5,21 +5,36 @@ import re
|
|
|
5
5
|
|
|
6
6
|
|
|
7
7
|
# Every block whose rows prose cites but no section of its own renders, with
|
|
8
|
-
# the keys each one names its content and its
|
|
9
|
-
# all of them would misread `crossVerification.consensus`, whose
|
|
10
|
-
# holds provenance while `evidence.primary` uses the same word
|
|
11
|
-
# so the block a row came from is named here rather than guessed.
|
|
8
|
+
# the keys each one names its content, its provenance, and its confidence by. A
|
|
9
|
+
# key hunt across all of them would misread `crossVerification.consensus`, whose
|
|
10
|
+
# `evidence` key holds provenance while `evidence.primary` uses the same word
|
|
11
|
+
# for content — so the block a row came from is named here rather than guessed.
|
|
12
|
+
#
|
|
13
|
+
# Source and confidence are separate columns because most blocks carry only one
|
|
14
|
+
# of the two: `evidence.primary` cites a file and never rates itself, while
|
|
15
|
+
# `evidence.secondary` rates itself and cites nothing. Folding them into one
|
|
16
|
+
# field is what made the ledger label every row "source · confidence" and then
|
|
17
|
+
# print a single value under it.
|
|
18
|
+
#
|
|
19
|
+
# `endStateCoverage` is deliberately absent. Its rows hold an id pair
|
|
20
|
+
# (`EB-001` covered by `R-001`) rather than a statement, so the ledger rendered
|
|
21
|
+
# them as an id with no body, and the requirement-coverage table already names
|
|
22
|
+
# every one of them in its Source column.
|
|
23
|
+
#
|
|
24
|
+
# The kind each row carries is a vocabulary key, not the words the reader sees:
|
|
25
|
+
# the labels live in the i18n `ledgerKind` table so they arrive in the reader's
|
|
26
|
+
# language. `evidence.primary` was previously stamped "unclassified", which
|
|
27
|
+
# named the renderer's own indecision rather than anything about the row.
|
|
12
28
|
_LEDGER_BLOCKS = (
|
|
13
|
-
(("evidence", "primary"), "evidence", "source", "
|
|
14
|
-
(("evidence", "secondary"), "hypothesis", "confidence", "hypothesis"),
|
|
15
|
-
(("analysisCommon", "confirmedFacts"), "statement", "", "confirmed
|
|
16
|
-
(("analysisCommon", "inferences"), "statement", "confidence", "inference"),
|
|
17
|
-
(("analysisCommon", "unknowns"), "question", "reason", "unknown"),
|
|
18
|
-
(("crossVerification", "consensus"), "statement", "evidence", "cross-check
|
|
19
|
-
(("crossVerification", "differences"), "disagreement", "workersPosition", "cross-check
|
|
20
|
-
(("missingInformation",), "item", "risk", "missing
|
|
21
|
-
(("
|
|
22
|
-
(("followUpTasks",), "title", "reason", "follow-up"),
|
|
29
|
+
(("evidence", "primary"), "evidence", "source", "", "code-evidence"),
|
|
30
|
+
(("evidence", "secondary"), "hypothesis", "", "confidence", "hypothesis"),
|
|
31
|
+
(("analysisCommon", "confirmedFacts"), "statement", "", "", "confirmed-fact"),
|
|
32
|
+
(("analysisCommon", "inferences"), "statement", "", "confidence", "inference"),
|
|
33
|
+
(("analysisCommon", "unknowns"), "question", "reason", "", "unknown"),
|
|
34
|
+
(("crossVerification", "consensus"), "statement", "evidence", "", "cross-check-consensus"),
|
|
35
|
+
(("crossVerification", "differences"), "disagreement", "workersPosition", "", "cross-check-dissent"),
|
|
36
|
+
(("missingInformation",), "item", "risk", "", "missing-information"),
|
|
37
|
+
(("followUpTasks",), "title", "reason", "", "follow-up"),
|
|
23
38
|
)
|
|
24
39
|
|
|
25
40
|
|
|
@@ -32,23 +47,39 @@ def _dig(data: dict, path: tuple[str, ...]) -> list:
|
|
|
32
47
|
return node if isinstance(node, list) else []
|
|
33
48
|
|
|
34
49
|
|
|
35
|
-
def
|
|
50
|
+
def _joined(value: object) -> str:
|
|
51
|
+
"""One reader-facing string for a cell a block may fill with either a
|
|
52
|
+
sentence or a list of citations. `str()` on a list prints Python's own
|
|
53
|
+
repr — quotes, brackets and all — straight into the report.
|
|
54
|
+
"""
|
|
55
|
+
if isinstance(value, (list, tuple)):
|
|
56
|
+
return " · ".join(str(item) for item in value if item)
|
|
57
|
+
return str(value or "")
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _ledger_row(
|
|
61
|
+
row: dict, text_key: str, source_key: str, confidence_key: str, kind: str
|
|
62
|
+
) -> dict[str, object]:
|
|
36
63
|
return {
|
|
37
64
|
"id": row.get("id", ""),
|
|
38
65
|
"kind": kind,
|
|
39
|
-
"text":
|
|
66
|
+
"text": _joined(row.get(text_key)),
|
|
40
67
|
"codeEvidence": row.get("currentCodeEvidence") or [],
|
|
41
|
-
"source":
|
|
68
|
+
"source": _joined(row.get(source_key)) if source_key else "",
|
|
69
|
+
"confidence": _joined(row.get(confidence_key)) if confidence_key else "",
|
|
42
70
|
}
|
|
43
71
|
|
|
44
72
|
|
|
45
|
-
def _own_section_ids(data: dict) -> set[str]:
|
|
73
|
+
def _own_section_ids(data: dict, omitted_fields: tuple[str, ...] = ()) -> set[str]:
|
|
46
74
|
"""Ids a section of the report already renders and anchors.
|
|
47
75
|
|
|
48
76
|
The ledger is the fallback home for a cited row, so it must not claim an id
|
|
49
77
|
that has one — two elements with the same anchor send half the links to the
|
|
50
78
|
wrong place. `crossVerification.consensus` numbers its rows `C-001` in some
|
|
51
79
|
runs, exactly where a clarification lives.
|
|
80
|
+
|
|
81
|
+
`omitted_fields` names task-block fields the HTML template does not render,
|
|
82
|
+
so their ids do not count as anchored.
|
|
52
83
|
"""
|
|
53
84
|
from ..report_contract import TASK_TYPE_DATA_PROPERTY
|
|
54
85
|
|
|
@@ -56,20 +87,34 @@ def _own_section_ids(data: dict) -> set[str]:
|
|
|
56
87
|
_collect_ids(data.get("clarificationItems", []), found)
|
|
57
88
|
property_name = TASK_TYPE_DATA_PROPERTY.get(data.get("header", {}).get("taskType", ""))
|
|
58
89
|
if property_name:
|
|
59
|
-
|
|
90
|
+
block = data.get(property_name, {})
|
|
91
|
+
if omitted_fields and isinstance(block, dict):
|
|
92
|
+
block = {
|
|
93
|
+
key: value for key, value in block.items() if key not in omitted_fields
|
|
94
|
+
}
|
|
95
|
+
_collect_ids(block, found)
|
|
60
96
|
return found
|
|
61
97
|
|
|
62
98
|
|
|
63
99
|
def evidence_index(data: dict) -> dict[str, object]:
|
|
100
|
+
"""The rows the ledger carries, keyed by id.
|
|
101
|
+
|
|
102
|
+
A row whose statement came out empty is dropped rather than listed: the
|
|
103
|
+
ledger exists so a cited id resolves to something the reader can read, and
|
|
104
|
+
an id above a blank line resolves to nothing.
|
|
105
|
+
"""
|
|
64
106
|
owned = _own_section_ids(data)
|
|
65
107
|
rows: dict[str, object] = {}
|
|
66
|
-
for path, text_key, source_key, kind in _LEDGER_BLOCKS:
|
|
108
|
+
for path, text_key, source_key, confidence_key, kind in _LEDGER_BLOCKS:
|
|
67
109
|
for row in _dig(data, path):
|
|
68
110
|
if not isinstance(row, dict):
|
|
69
111
|
continue
|
|
70
112
|
row_id = row.get("id")
|
|
71
|
-
if row_id
|
|
72
|
-
|
|
113
|
+
if not row_id or row_id in owned or row_id in rows:
|
|
114
|
+
continue
|
|
115
|
+
entry = _ledger_row(row, text_key, source_key, confidence_key, kind)
|
|
116
|
+
if entry["text"]:
|
|
117
|
+
rows[row_id] = entry
|
|
73
118
|
return rows
|
|
74
119
|
|
|
75
120
|
|
|
@@ -88,7 +133,7 @@ def _collect_ids(value: object, found: set[str]) -> None:
|
|
|
88
133
|
_ROW_ID = re.compile(r"[A-Z]{1,3}-\d+")
|
|
89
134
|
|
|
90
135
|
|
|
91
|
-
def anchor_index(data: dict) -> dict[str, str]:
|
|
136
|
+
def anchor_index(data: dict, omitted_fields: tuple[str, ...] = ()) -> dict[str, str]:
|
|
92
137
|
"""Map every row a reader can reach to the anchor name that lands on it.
|
|
93
138
|
|
|
94
139
|
Prose cites ids across section boundaries — a hotspot names a
|
|
@@ -98,9 +143,10 @@ def anchor_index(data: dict) -> dict[str, str]:
|
|
|
98
143
|
|
|
99
144
|
It stops there. `summary` is the AI-facing digest and
|
|
100
145
|
`analysisCommon.scope` describes the analysis target rather than listing
|
|
101
|
-
rows; neither renders, so a link to one would land nowhere.
|
|
146
|
+
rows; neither renders, so a link to one would land nowhere. Same for the
|
|
147
|
+
blocks a template declares in `omitted_fields`.
|
|
102
148
|
"""
|
|
103
|
-
found = _own_section_ids(data) | set(evidence_index(data))
|
|
149
|
+
found = _own_section_ids(data, omitted_fields) | set(evidence_index(data))
|
|
104
150
|
return {row_id: f"id-{row_id}" for row_id in sorted(found) if _ROW_ID.fullmatch(row_id)}
|
|
105
151
|
|
|
106
152
|
|
|
@@ -66,6 +66,11 @@ class HumanReportView:
|
|
|
66
66
|
template_name: str
|
|
67
67
|
context: dict[str, object]
|
|
68
68
|
figures: tuple[FigureModel, ...]
|
|
69
|
+
# Task-block fields this template leaves out of the HTML. Their rows still
|
|
70
|
+
# exist in the data and the markdown, so prose keeps citing their ids — but
|
|
71
|
+
# an anchor to a row this document never renders scrolls nowhere, which
|
|
72
|
+
# reads as a broken report rather than as a deliberate omission.
|
|
73
|
+
omitted_fields: tuple[str, ...] = ()
|
|
69
74
|
|
|
70
75
|
|
|
71
76
|
@dataclass(frozen=True)
|
|
@@ -130,7 +130,7 @@ def render_v2_html_view(
|
|
|
130
130
|
env.policies["json.dumps_kwargs"] = {"sort_keys": True, "ensure_ascii": False}
|
|
131
131
|
# Binding the index here is what lets a template cite an id without
|
|
132
132
|
# threading the index through every macro and call site.
|
|
133
|
-
anchors = anchor_index(data)
|
|
133
|
+
anchors = anchor_index(data, view.omitted_fields)
|
|
134
134
|
chrome = load_dictionary(lang, HTML_DICTIONARY_REL)
|
|
135
135
|
translate = make_jinja_global(chrome)
|
|
136
136
|
env.globals["t"] = translate
|
|
@@ -23,6 +23,7 @@ def _totals_row(row: object) -> dict[str, str]:
|
|
|
23
23
|
values = row if isinstance(row, dict) else {}
|
|
24
24
|
return {
|
|
25
25
|
"rawTokens": format_int(values.get("totalTokens")),
|
|
26
|
+
"cacheReadTokens": format_int(values.get("cacheReadTokens")),
|
|
26
27
|
"billableTokens": format_int(values.get("billableTokens")),
|
|
27
28
|
"cost": format_usd(values.get("costUsd")),
|
|
28
29
|
}
|
|
@@ -43,6 +44,7 @@ def _agent_row(row: dict) -> dict[str, str]:
|
|
|
43
44
|
"model": str(row.get("model") or ""),
|
|
44
45
|
"status": str(row.get("status") or ""),
|
|
45
46
|
"rawTokens": format_int(row.get("totalTokens")),
|
|
47
|
+
"cacheReadTokens": format_int(row.get("cacheReadTokens")),
|
|
46
48
|
"billableTokens": format_int(row.get("billableTokens")),
|
|
47
49
|
"cost": format_usd(row.get("costUsd")),
|
|
48
50
|
"duration": format_duration_ms(row.get("durationMs")),
|
|
@@ -70,15 +72,31 @@ def _unaccounted(rows: list[dict], grand: dict) -> dict[str, str] | None:
|
|
|
70
72
|
gap = grand_tokens - _sum(rows, "totalTokens")
|
|
71
73
|
if gap <= 0:
|
|
72
74
|
return None
|
|
75
|
+
cache_gap = (_number(grand.get("cacheReadTokens")) or 0) - _sum(rows, "cacheReadTokens")
|
|
73
76
|
billable_gap = (_number(grand.get("billableTokens")) or 0) - _sum(rows, "billableTokens")
|
|
74
77
|
cost_gap = (_number(grand.get("costUsd")) or 0) - _sum(rows, "costUsd")
|
|
75
78
|
return {
|
|
76
79
|
"rawTokens": format_int(gap),
|
|
80
|
+
"cacheReadTokens": format_int(max(0, cache_gap)),
|
|
77
81
|
"billableTokens": format_int(max(0, billable_gap)),
|
|
78
82
|
"cost": format_usd(max(0.0, cost_gap)),
|
|
79
83
|
}
|
|
80
84
|
|
|
81
85
|
|
|
86
|
+
def _task_cumulative(row: object) -> dict[str, str] | None:
|
|
87
|
+
"""What the task has spent over every run, when more than this one exists.
|
|
88
|
+
|
|
89
|
+
A single-run task would repeat the grand total word for word, and a reader
|
|
90
|
+
seeing the same figure twice reads it as a second charge.
|
|
91
|
+
"""
|
|
92
|
+
if not isinstance(row, dict):
|
|
93
|
+
return None
|
|
94
|
+
run_count = _number(row.get("runCount")) or 0
|
|
95
|
+
if run_count < 2:
|
|
96
|
+
return None
|
|
97
|
+
return {**_totals_row(row), "runCount": format_int(run_count)}
|
|
98
|
+
|
|
99
|
+
|
|
82
100
|
_MEASURED_KEYS = ("totalTokens", "billableTokens", "costUsd", "durationMs")
|
|
83
101
|
|
|
84
102
|
|
|
@@ -106,5 +124,6 @@ def run_usage(data: dict) -> dict[str, object] | None:
|
|
|
106
124
|
"rows": [_agent_row(row) for row in rows],
|
|
107
125
|
"unaccounted": _unaccounted(rows, usage.get("grand") or {}),
|
|
108
126
|
**totals,
|
|
127
|
+
"taskCumulative": _task_cumulative(usage.get("taskCumulative")),
|
|
109
128
|
"cliCost": format_usd(cli_cost) if cli_cost else "",
|
|
110
129
|
}
|
|
@@ -74,6 +74,19 @@ def _stage_figure(planning: dict):
|
|
|
74
74
|
return stage_map_figure(nodes=nodes, edges=edges, title="Implementation stage dependencies")
|
|
75
75
|
|
|
76
76
|
|
|
77
|
+
# Blocks the approver's view leaves to the markdown report: the checklist the
|
|
78
|
+
# implementer works from, the dependency and rollback tables an incident reader
|
|
79
|
+
# needs, and the verification rounds the audit trail keeps. Declared here so the
|
|
80
|
+
# ids they carry stop being anchor targets in this document.
|
|
81
|
+
_OMITTED_FIELDS = (
|
|
82
|
+
"validationChecklist",
|
|
83
|
+
"crossProjectDependencies",
|
|
84
|
+
"dependencyMigrationRisk",
|
|
85
|
+
"rollbackStrategy",
|
|
86
|
+
"planBodyVerification",
|
|
87
|
+
)
|
|
88
|
+
|
|
89
|
+
|
|
77
90
|
def build_implementation_planning_view(data: dict) -> HumanReportView:
|
|
78
91
|
planning = data["implementationPlanning"]
|
|
79
92
|
figure = _stage_figure(planning)
|
|
@@ -97,4 +110,5 @@ def build_implementation_planning_view(data: dict) -> HumanReportView:
|
|
|
97
110
|
"html/tasks/implementation-planning.template.html",
|
|
98
111
|
context,
|
|
99
112
|
(figure,),
|
|
113
|
+
_OMITTED_FIELDS,
|
|
100
114
|
)
|
|
@@ -1050,11 +1050,16 @@ class UserResponseEntry:
|
|
|
1050
1050
|
|
|
1051
1051
|
|
|
1052
1052
|
@dataclass(frozen=True)
|
|
1053
|
-
class
|
|
1054
|
-
"""HTML
|
|
1055
|
-
|
|
1056
|
-
|
|
1053
|
+
class UserPlanDecision:
|
|
1054
|
+
"""HTML 계획 결정 위젯의 Export 결과.
|
|
1055
|
+
|
|
1056
|
+
``status`` 가 승인이 아니면 ``reason`` 이 필수다 — 사유 없는 반려는 다음
|
|
1057
|
+
run 이 무엇을 고쳐야 하는지 알 수 없어 되돌아올 수밖에 없다.
|
|
1058
|
+
``implementation_option`` 이 빈 문자열이면 라인을 생략한다 (소비 측은
|
|
1059
|
+
Recommended Option 폴백)."""
|
|
1060
|
+
status: str
|
|
1057
1061
|
implementation_option: str = ""
|
|
1062
|
+
reason: str = ""
|
|
1058
1063
|
|
|
1059
1064
|
|
|
1060
1065
|
@dataclass(frozen=True)
|
|
@@ -1072,6 +1077,13 @@ _ANALYSIS_REVIEW_STATUSES = frozenset({
|
|
|
1072
1077
|
"rejected",
|
|
1073
1078
|
})
|
|
1074
1079
|
|
|
1080
|
+
PLAN_DECISION_APPROVED = "approved"
|
|
1081
|
+
_PLAN_DECISION_STATUSES = frozenset({
|
|
1082
|
+
PLAN_DECISION_APPROVED,
|
|
1083
|
+
"revision-requested",
|
|
1084
|
+
"rejected",
|
|
1085
|
+
})
|
|
1086
|
+
|
|
1075
1087
|
|
|
1076
1088
|
def _quoted_sidecar_field(label: str, value: str) -> str:
|
|
1077
1089
|
cleaned = value.strip()
|
|
@@ -1100,12 +1112,25 @@ def _serialize_analysis_review(review: UserResponseAnalysisReview) -> str:
|
|
|
1100
1112
|
)
|
|
1101
1113
|
|
|
1102
1114
|
|
|
1115
|
+
def _serialize_plan_decision(decision: UserPlanDecision) -> str:
|
|
1116
|
+
if decision.status not in _PLAN_DECISION_STATUSES:
|
|
1117
|
+
raise ValueError(f"invalid PLAN DECISION status: {decision.status}")
|
|
1118
|
+
if decision.status != PLAN_DECISION_APPROVED and not decision.reason.strip():
|
|
1119
|
+
raise ValueError(f"PLAN DECISION {decision.status} requires a Reason")
|
|
1120
|
+
chunk = f"\n## PLAN DECISION\n- Status: {decision.status}\n"
|
|
1121
|
+
if decision.implementation_option:
|
|
1122
|
+
chunk += f"- Implementation-Option: {decision.implementation_option.strip()}\n"
|
|
1123
|
+
if decision.reason.strip():
|
|
1124
|
+
chunk += _quoted_sidecar_field("Reason", decision.reason)
|
|
1125
|
+
return chunk
|
|
1126
|
+
|
|
1127
|
+
|
|
1103
1128
|
def serialize_user_response(
|
|
1104
1129
|
*,
|
|
1105
1130
|
run_meta: RunMeta,
|
|
1106
1131
|
entries: list[UserResponseEntry],
|
|
1107
1132
|
created_at: str,
|
|
1108
|
-
|
|
1133
|
+
plan_decision: UserPlanDecision | None = None,
|
|
1109
1134
|
analysis_review: UserResponseAnalysisReview | None = None,
|
|
1110
1135
|
) -> str:
|
|
1111
1136
|
"""Return the canonical markdown text the HTML 'Export user
|
|
@@ -1125,7 +1150,7 @@ def serialize_user_response(
|
|
|
1125
1150
|
"\n"
|
|
1126
1151
|
"# User Response\n"
|
|
1127
1152
|
)
|
|
1128
|
-
|
|
1153
|
+
has_plan_decision = plan_decision is not None
|
|
1129
1154
|
has_analysis_review = analysis_review is not None
|
|
1130
1155
|
body_chunks: list[str] = []
|
|
1131
1156
|
for e in entries:
|
|
@@ -1137,13 +1162,10 @@ def serialize_user_response(
|
|
|
1137
1162
|
if e.rationale:
|
|
1138
1163
|
chunk += f"- Rationale: {e.rationale.strip()}\n"
|
|
1139
1164
|
body_chunks.append(chunk)
|
|
1140
|
-
if not entries and not
|
|
1165
|
+
if not entries and not has_plan_decision and not has_analysis_review:
|
|
1141
1166
|
body_chunks.append("\n_(No user responses recorded.)_\n")
|
|
1142
|
-
if
|
|
1143
|
-
|
|
1144
|
-
if approval.implementation_option:
|
|
1145
|
-
chunk += f"- Implementation-Option: {approval.implementation_option.strip()}\n"
|
|
1146
|
-
body_chunks.append(chunk)
|
|
1167
|
+
if plan_decision is not None:
|
|
1168
|
+
body_chunks.append(_serialize_plan_decision(plan_decision))
|
|
1147
1169
|
if analysis_review is not None:
|
|
1148
1170
|
body_chunks.append(_serialize_analysis_review(analysis_review))
|
|
1149
1171
|
return head + "".join(body_chunks)
|
|
@@ -1369,11 +1391,17 @@ def _plan_approval_section(ctx: PlanApprovalContext, run_meta: RunMeta) -> str:
|
|
|
1369
1391
|
)
|
|
1370
1392
|
return (
|
|
1371
1393
|
'<section id="plan-approval">\n'
|
|
1372
|
-
" <h2>Plan
|
|
1394
|
+
" <h2>Plan Decision</h2>\n"
|
|
1373
1395
|
f' <label>구현 옵션: <select id="approval-option"{disabled}>{"".join(opts)}</select></label>\n'
|
|
1374
|
-
|
|
1375
|
-
"
|
|
1376
|
-
"
|
|
1396
|
+
' <fieldset id="plan-decision"><legend>판정</legend>'
|
|
1397
|
+
f'<label><input type="radio" name="plan-decision-status" value="approved"{disabled}> '
|
|
1398
|
+
"이 plan 을 승인합니다</label>"
|
|
1399
|
+
'<label><input type="radio" name="plan-decision-status" '
|
|
1400
|
+
'value="revision-requested"> 고쳐서 다시 가져오게 합니다</label>'
|
|
1401
|
+
'<label><input type="radio" name="plan-decision-status" value="rejected"> '
|
|
1402
|
+
"이 plan 을 반려합니다</label></fieldset>\n"
|
|
1403
|
+
' <label>사유 — 반려하거나 다시 고치게 할 때는 반드시 적어야 합니다'
|
|
1404
|
+
'<textarea id="plan-decision-reason" rows="4"></textarea></label>'
|
|
1377
1405
|
f"{reason_html}\n"
|
|
1378
1406
|
"</section>\n"
|
|
1379
1407
|
)
|
|
@@ -9,6 +9,7 @@ which decides whether a re-run narrows or stays full.
|
|
|
9
9
|
from __future__ import annotations
|
|
10
10
|
|
|
11
11
|
import re
|
|
12
|
+
from collections.abc import Iterator
|
|
12
13
|
|
|
13
14
|
_RANGE = r"(?:[-–]|\bto\b|\bthrough\b)"
|
|
14
15
|
_ITEM = rf"\d+(?:\s*{_RANGE}\s*\d+)?"
|
|
@@ -23,25 +24,61 @@ _ITEM_RE = re.compile(rf"(\d+)(?:\s*{_RANGE}\s*(\d+))?", re.IGNORECASE)
|
|
|
23
24
|
RANGE_MAX_SPAN = 64
|
|
24
25
|
|
|
25
26
|
|
|
27
|
+
def _stage_citation_items(text: str) -> Iterator[tuple[int, int | None]]:
|
|
28
|
+
"""Each `stage`-anchored citation in *text* as `(start, end)`.
|
|
29
|
+
|
|
30
|
+
*end* is None for a bare number and the far endpoint for a range. Numbers
|
|
31
|
+
stay anchored to a word-initial `stage`/`stages` token on the same line;
|
|
32
|
+
harvesting bare numbers — or letting the anchor reach across a line break,
|
|
33
|
+
or match the tail of `Substage`/`Backstage` — would let any prose, including
|
|
34
|
+
a row disclaiming every stage, justify any stage.
|
|
35
|
+
"""
|
|
36
|
+
for span in _LIST_RE.finditer(text):
|
|
37
|
+
for item in _ITEM_RE.finditer(span.group(1)):
|
|
38
|
+
end = item.group(2)
|
|
39
|
+
yield int(item.group(1)), int(end) if end is not None else None
|
|
40
|
+
|
|
41
|
+
|
|
26
42
|
def cited_stage_numbers(text: str) -> set[int]:
|
|
27
43
|
"""Stage numbers *text* cites, in every prose form a planner writes.
|
|
28
44
|
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
break, or match the tail of `Substage`/`Backstage` — would let any prose,
|
|
32
|
-
including a row disclaiming every stage, justify any stage.
|
|
45
|
+
A range reaches every number between its endpoints here, which is what a
|
|
46
|
+
reader asking "could this touch stage 5" needs.
|
|
33
47
|
"""
|
|
34
48
|
cited: set[int] = set()
|
|
49
|
+
for start, end in _stage_citation_items(text):
|
|
50
|
+
cited.add(start)
|
|
51
|
+
if end is None:
|
|
52
|
+
continue
|
|
53
|
+
cited.add(end)
|
|
54
|
+
# A reversed or absurdly wide range is a typo, not a citation of
|
|
55
|
+
# everything between its endpoints.
|
|
56
|
+
if 0 <= end - start <= RANGE_MAX_SPAN:
|
|
57
|
+
cited.update(range(start, end))
|
|
58
|
+
return cited
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def enumerated_stage_numbers(text: str) -> set[int]:
|
|
62
|
+
"""Stage numbers *text* names one by one — a range's interior excluded.
|
|
63
|
+
|
|
64
|
+
The two readers of this grammar ask opposite questions, and a range answers
|
|
65
|
+
only one of them. The incremental-scope back-trace asks "could this answer
|
|
66
|
+
reach stage 5", so it must read `Stages 1-8` as reaching it — that is
|
|
67
|
+
`cited_stage_numbers`, and widening there is the safe direction. Coverage
|
|
68
|
+
provenance asks "did the planner confirm stage 5 satisfies this
|
|
69
|
+
requirement", and a range answers that for free: one `Stages 1-64` cell
|
|
70
|
+
stamps every stage in the map without the planner looking at any of them.
|
|
71
|
+
|
|
72
|
+
So the interior is dropped here and only hand-written numbers count. A
|
|
73
|
+
range's endpoints ARE hand-written and stay, which keeps `Stages 7-8`
|
|
74
|
+
honest while forcing the wide case to be spelled out. The cost of a
|
|
75
|
+
legitimately broad requirement is typing each number, and that typing is
|
|
76
|
+
the confirmation this check is asking for.
|
|
77
|
+
"""
|
|
78
|
+
enumerated: set[int] = set()
|
|
35
79
|
for span in _LIST_RE.finditer(text):
|
|
36
80
|
for item in _ITEM_RE.finditer(span.group(1)):
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
end = int(item.group(2))
|
|
42
|
-
cited.add(end)
|
|
43
|
-
# A reversed or absurdly wide range is a typo, not a citation of
|
|
44
|
-
# everything between its endpoints.
|
|
45
|
-
if 0 <= end - start <= RANGE_MAX_SPAN:
|
|
46
|
-
cited.update(range(start, end))
|
|
47
|
-
return cited
|
|
81
|
+
enumerated.add(int(item.group(1)))
|
|
82
|
+
if item.group(2) is not None:
|
|
83
|
+
enumerated.add(int(item.group(2)))
|
|
84
|
+
return enumerated
|
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
The sidecar format is documented in ``templates/reports/user-response.template.md``
|
|
4
4
|
and produced byte-identically by ``report_views.serialize_user_response`` (Python)
|
|
5
5
|
and ``templates/reports/report.js`` (browser). This module owns the read side of
|
|
6
|
-
the ``##
|
|
6
|
+
the ``## PLAN DECISION`` block used by the implementation wizard and the optional
|
|
7
7
|
``## ANALYSIS REVIEW`` block used by analysis reruns.
|
|
8
8
|
"""
|
|
9
9
|
from __future__ import annotations
|
|
@@ -19,7 +19,8 @@ from pathlib import Path
|
|
|
19
19
|
from typing import Optional
|
|
20
20
|
|
|
21
21
|
from okstra_ctl.report_views import (
|
|
22
|
-
|
|
22
|
+
PLAN_DECISION_APPROVED,
|
|
23
|
+
serialize_user_response, UserResponseEntry, UserPlanDecision, infer_run_meta,
|
|
23
24
|
parse_expected_form_options,
|
|
24
25
|
)
|
|
25
26
|
from okstra_ctl.report_view_artifacts import user_responses_dir_for_report
|
|
@@ -32,7 +33,7 @@ from okstra_ctl.clarification_items import (
|
|
|
32
33
|
_section_1_slice,
|
|
33
34
|
)
|
|
34
35
|
|
|
35
|
-
|
|
36
|
+
_PLAN_DECISION_HEADING_RE = re.compile(r"^## PLAN DECISION\s*$", re.MULTILINE)
|
|
36
37
|
_NEXT_RESPONSE_HEADING_RE = re.compile(r"^## ", re.MULTILINE)
|
|
37
38
|
_ANALYSIS_REVIEW_HEADING_RE = re.compile(r"^## ANALYSIS REVIEW\s*$", re.MULTILINE)
|
|
38
39
|
_ANALYSIS_SIDECAR_HEADING_RE = re.compile(
|
|
@@ -50,13 +51,18 @@ class UserResponseError(ValueError):
|
|
|
50
51
|
|
|
51
52
|
|
|
52
53
|
@dataclass(frozen=True)
|
|
53
|
-
class
|
|
54
|
-
"""sidecar 의 ``##
|
|
55
|
-
|
|
54
|
+
class PlanDecisionRecord:
|
|
55
|
+
"""sidecar 의 ``## PLAN DECISION`` 블록 + 매칭에 필요한 frontmatter 필드."""
|
|
56
|
+
status: str
|
|
56
57
|
implementation_option: str
|
|
58
|
+
reason: str
|
|
57
59
|
source_report: str
|
|
58
60
|
seq: str
|
|
59
61
|
|
|
62
|
+
@property
|
|
63
|
+
def approved(self) -> bool:
|
|
64
|
+
return self.status == PLAN_DECISION_APPROVED
|
|
65
|
+
|
|
60
66
|
|
|
61
67
|
@dataclass(frozen=True)
|
|
62
68
|
class AnalysisReviewRecord:
|
|
@@ -353,31 +359,37 @@ def load_authoritative_analysis_review(
|
|
|
353
359
|
return review
|
|
354
360
|
|
|
355
361
|
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
)
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
+
_PLAN_DECISION_STATUS_RE = re.compile(
|
|
363
|
+
r"^- Status:\s*(approved|revision-requested|rejected)\s*$", re.MULTILINE
|
|
364
|
+
)
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
def parse_plan_decision(sidecar_text: str) -> Optional[PlanDecisionRecord]:
|
|
368
|
+
"""``## PLAN DECISION`` 블록을 읽어 record 로 돌려준다. 승인·재작업·반려를
|
|
369
|
+
구분하지 않고 그대로 싣는다 — 어느 판정을 받아들일지는 소비 측 정책이다.
|
|
370
|
+
옵션·사유 라인은 블록 내부에서만 읽는다 (다른 응답 본문의 우연한 동일
|
|
371
|
+
문구를 옵션으로 오인하지 않도록).
|
|
362
372
|
|
|
363
|
-
strictness 는 의도적이다: producer 출력과 byte-identical 한 소문자
|
|
364
|
-
|
|
373
|
+
strictness 는 의도적이다: producer 출력과 byte-identical 한 소문자 status
|
|
374
|
+
만 인정하며, 손편집 변형(``Approved``/``REJECTED``)은 fail-closed 로
|
|
365
375
|
불인정한다."""
|
|
366
|
-
m =
|
|
376
|
+
m = _PLAN_DECISION_HEADING_RE.search(sidecar_text)
|
|
367
377
|
if not m:
|
|
368
378
|
return None
|
|
369
379
|
block = sidecar_text[m.end():]
|
|
370
380
|
nxt = _NEXT_RESPONSE_HEADING_RE.search(block)
|
|
371
381
|
if nxt:
|
|
372
382
|
block = block[: nxt.start()]
|
|
373
|
-
|
|
383
|
+
status = _PLAN_DECISION_STATUS_RE.search(block)
|
|
384
|
+
if status is None:
|
|
374
385
|
return None
|
|
375
386
|
om = re.search(r"^- Implementation-Option:\s*(\S.*?)\s*$", block, re.MULTILINE)
|
|
376
387
|
sm = re.search(r"^seq:\s*(\S+)\s*$", sidecar_text, re.MULTILINE)
|
|
377
388
|
rm = re.search(r"^source-report:\s*(\S.*?)\s*$", sidecar_text, re.MULTILINE)
|
|
378
|
-
return
|
|
379
|
-
|
|
389
|
+
return PlanDecisionRecord(
|
|
390
|
+
status=status.group(1),
|
|
380
391
|
implementation_option=om.group(1) if om else "",
|
|
392
|
+
reason=_quoted_review_value(block, "Reason"),
|
|
381
393
|
source_report=rm.group(1) if rm else "",
|
|
382
394
|
seq=sm.group(1) if sm else "",
|
|
383
395
|
)
|
|
@@ -404,8 +416,8 @@ def _value(block: str) -> str:
|
|
|
404
416
|
|
|
405
417
|
def parse_user_response_entries(sidecar_text: str) -> list[UserResponseEntry]:
|
|
406
418
|
"""Reverse of ``serialize_user_response`` for the per-response ``## C-*``
|
|
407
|
-
blocks. The ``##
|
|
408
|
-
``
|
|
419
|
+
blocks. The ``## PLAN DECISION`` block is skipped (read separately by
|
|
420
|
+
``parse_plan_decision``)."""
|
|
409
421
|
entries: list[UserResponseEntry] = []
|
|
410
422
|
matches = list(_RESPONSE_HEADING_RE.finditer(sidecar_text))
|
|
411
423
|
for i, m in enumerate(matches):
|
|
@@ -581,7 +593,7 @@ def show_open_rows(report_path: Path) -> dict:
|
|
|
581
593
|
|
|
582
594
|
|
|
583
595
|
def write_sidecar(report_path: Path, answers: list[dict],
|
|
584
|
-
|
|
596
|
+
plan_decision: Optional[dict], created_at: str,
|
|
585
597
|
task_key: str = "") -> Path:
|
|
586
598
|
run_meta = infer_run_meta(report_path, task_key=task_key or None)
|
|
587
599
|
out_dir = user_responses_dir_for_report(report_path)
|
|
@@ -597,13 +609,15 @@ def write_sidecar(report_path: Path, answers: list[dict],
|
|
|
597
609
|
response_id=a["id"], kind=a.get("kind", ""), value=a["value"],
|
|
598
610
|
rationale=a.get("rationale"), disposition=a.get("disposition", "answer"))
|
|
599
611
|
|
|
600
|
-
|
|
601
|
-
if
|
|
602
|
-
|
|
603
|
-
|
|
612
|
+
decision = None
|
|
613
|
+
if plan_decision and plan_decision.get("status"):
|
|
614
|
+
decision = UserPlanDecision(
|
|
615
|
+
status=plan_decision["status"],
|
|
616
|
+
implementation_option=plan_decision.get("implementationOption", ""),
|
|
617
|
+
reason=plan_decision.get("reason", ""))
|
|
604
618
|
sidecar.write_text(
|
|
605
619
|
serialize_user_response(run_meta=run_meta, entries=list(merged.values()),
|
|
606
|
-
created_at=created_at,
|
|
620
|
+
created_at=created_at, plan_decision=decision),
|
|
607
621
|
encoding="utf-8")
|
|
608
622
|
return sidecar
|
|
609
623
|
|
|
@@ -623,7 +637,9 @@ def main(argv: Optional[list[str]] = None) -> int:
|
|
|
623
637
|
pw = sub.add_parser("write")
|
|
624
638
|
pw.add_argument("--report", required=True)
|
|
625
639
|
pw.add_argument("--answers", required=True, help="JSON array of answer entries")
|
|
626
|
-
pw.add_argument(
|
|
640
|
+
pw.add_argument(
|
|
641
|
+
"--plan-decision", default="",
|
|
642
|
+
help='JSON plan decision, e.g. {"status":"rejected","reason":"..."}')
|
|
627
643
|
pw.add_argument("--task-key", default="", help="task-key from list/show context")
|
|
628
644
|
|
|
629
645
|
ns = parser.parse_args(argv)
|
|
@@ -636,9 +652,9 @@ def main(argv: Optional[list[str]] = None) -> int:
|
|
|
636
652
|
return 0
|
|
637
653
|
if ns.cmd == "write":
|
|
638
654
|
answers = json.loads(ns.answers)
|
|
639
|
-
|
|
655
|
+
decision = json.loads(ns.plan_decision) if ns.plan_decision else None
|
|
640
656
|
created_at = dt.datetime.now(dt.timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
641
|
-
p = write_sidecar(Path(ns.report), answers,
|
|
657
|
+
p = write_sidecar(Path(ns.report), answers, decision, created_at,
|
|
642
658
|
task_key=ns.task_key)
|
|
643
659
|
json.dump({"sidecar": str(p)}, sys.stdout, ensure_ascii=False)
|
|
644
660
|
return 0
|