okstra 0.195.1 → 0.195.3
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/architecture.md +1 -0
- package/docs/cli.md +1 -0
- package/docs/project-structure-overview.md +1 -0
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/lead/convergence.md +2 -0
- package/runtime/prompts/lead/report-writer.md +1 -1
- package/runtime/prompts/wizard/prompts.ko.json +2 -27
- package/runtime/python/okstra_ctl/agent/prompt_cli/dynamic_verifier.py +11 -3
- package/runtime/python/okstra_ctl/agent/prompt_cli/inputs.py +8 -1
- package/runtime/python/okstra_ctl/agent/prompt_cli/materialize.py +47 -3
- package/runtime/python/okstra_ctl/blocking_checks.py +12 -0
- package/runtime/python/okstra_ctl/convergence.py +77 -0
- package/runtime/python/okstra_ctl/convergence_critic_prompt.py +2 -2
- package/runtime/python/okstra_ctl/convergence_critic_verify_prompt.py +221 -0
- package/runtime/python/okstra_ctl/convergence_engine.py +2 -2
- package/runtime/python/okstra_ctl/convergence_reverify_prompt.py +4 -4
- package/runtime/python/okstra_ctl/convergence_store.py +8 -1
- package/runtime/python/okstra_ctl/dispatch_state.py +4 -2
- package/runtime/python/okstra_ctl/final_report_schema.py +59 -0
- package/runtime/python/okstra_ctl/report_corrections.py +1 -1
- package/runtime/python/okstra_ctl/report_html/render.py +10 -4
- package/runtime/python/okstra_ctl/report_narrative.py +4 -1
- package/runtime/python/okstra_ctl/report_projections.py +19 -1
- package/runtime/python/okstra_ctl/report_synthesis_packet.py +20 -8
- package/runtime/python/okstra_ctl/wizard/engine.py +13 -4
- package/runtime/python/okstra_ctl/wizard/picker_navigation.py +14 -36
- package/runtime/python/okstra_ctl/wizard/roles.py +26 -82
- package/runtime/python/okstra_ctl/wizard/state.py +1 -5
- package/runtime/python/okstra_ctl/wizard/steps_roles.py +0 -1
- package/runtime/python/okstra_ctl/worker_prompt_contract.py +22 -1
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +14 -1
- package/runtime/python/okstra_token_usage/collect.py +27 -6
- package/runtime/python/okstra_token_usage/pricing.py +39 -6
- package/runtime/skills/okstra-run/SKILL.md +1 -1
- package/runtime/templates/reports/html/base.template.html +1 -1
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
"""coverage-critic gap 검증 지시문을 gap 배치와 로스터에서 결정적으로 렌더한다.
|
|
2
|
+
|
|
3
|
+
`prompts/lead/convergence.md` §"Gap verification" 은 critic 의 gap 마다 Phase 4
|
|
4
|
+
분석자 한 명이 1라운드 adversarial 반박을 하라고 요구한다. 그런데 그 라운드는
|
|
5
|
+
번호 라운드 원장 밖이라 `plan-round` 가 계획을 내지 않고, `reverify-prompt` 는
|
|
6
|
+
계획 행만 렌더하며, 검증 audience 의 프롬프트는 렌더러 서명 없이는 materialize
|
|
7
|
+
가 거절한다. 실측(2026-09-09 dev-10642 requirements-discovery 001): 리드가 세
|
|
8
|
+
경로를 다 시도해 전부 거부됐고 gap 3건이 `gapsUnverified` 로 남았다.
|
|
9
|
+
|
|
10
|
+
이 모듈이 그 지시문을 만든다. 입력은 리드가 `apply-critic-gaps` 에 줄 것과 같은
|
|
11
|
+
coverage 배치(투표 전, `gaps[]` 만 채운 상태)와 Round 0 그룹(분석 로스터)이다.
|
|
12
|
+
배정은 엔진과 같은 규칙(`critic_gap_assignees`, 로스터 순서 round-robin)이라
|
|
13
|
+
`apply-critic-gaps` 의 커버리지 검사와 어긋나지 않는다. gap 마다 critic 결과
|
|
14
|
+
파일의 `### [<gapId>]` 절과 critic 감사 사이드카를 실어, 검증자가 리드의 전사가
|
|
15
|
+
아니라 critic 이 실제로 인용한 것을 판단하게 한다. 응답 형식은 번호 라운드와
|
|
16
|
+
같은 adversarial 정본이다 — `apply-critic-gaps` 는 gap 표를 언제나
|
|
17
|
+
adversarial 로 읽는다(`_parse_critic_gap` 의 `adversarial=True`).
|
|
18
|
+
|
|
19
|
+
산출물은 프롬프트 materializer 의 `--instruction` 이 받는 본문이다. `## Instructions`
|
|
20
|
+
로 시작하므로 `complete_reverify_instruction` 이 모델·task type·금지 목록을 그
|
|
21
|
+
앞에 붙이고 출력 계약을 뒤에 덧붙인다.
|
|
22
|
+
"""
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from dataclasses import dataclass
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import Any, Mapping, Sequence
|
|
28
|
+
|
|
29
|
+
from .convergence_critic_prompt import analysis_roster
|
|
30
|
+
from .convergence_engine import critic_gap_assignees
|
|
31
|
+
from .convergence_provenance import worker_result_suffix
|
|
32
|
+
from .convergence_reverify_prompt import (
|
|
33
|
+
ADVERSARIAL_MANDATE,
|
|
34
|
+
ADVERSARIAL_RESPONSE,
|
|
35
|
+
)
|
|
36
|
+
from .worker_artifact_paths import WorkerArtifactPathError, audit_sidecar_rel
|
|
37
|
+
from .worker_prompt_policy import CRITIC_VERIFY_DISPATCH_KIND
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class CriticVerifyPromptError(ValueError):
|
|
41
|
+
"""지시문을 결정적으로 만들 수 없다."""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
# 렌더된 지시문의 서명. `validate_reverify_prompt` 가 dispatch kind
|
|
45
|
+
# `critic-verify` 에 이 줄을 요구한다 — 번호 라운드의 `reverify-prompt` 서명과
|
|
46
|
+
# 같은 역할이고, 손으로 쓴 지시문은 materialize 에서 거절된다.
|
|
47
|
+
RENDERED_BY_LINE = "**Rendered by:** okstra convergence critic-verify-prompt"
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
@dataclass(frozen=True)
|
|
51
|
+
class CriticGap:
|
|
52
|
+
"""검증 큐의 gap 하나와, critic 결과 파일 안의 실물 인용 위치."""
|
|
53
|
+
|
|
54
|
+
gap_id: str
|
|
55
|
+
summary: str
|
|
56
|
+
category: str
|
|
57
|
+
ticket_ids: tuple[str, ...]
|
|
58
|
+
origin_evidence: str
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def _nonempty_string(value: Any) -> str:
|
|
62
|
+
return value if isinstance(value, str) and value.strip() else ""
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _project_relative(project_root: Path, path: Path) -> str:
|
|
66
|
+
try:
|
|
67
|
+
return path.resolve().relative_to(project_root.resolve()).as_posix()
|
|
68
|
+
except ValueError as exc:
|
|
69
|
+
raise CriticVerifyPromptError(
|
|
70
|
+
f"critic result resolves outside the project root: {path}"
|
|
71
|
+
) from exc
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def critic_result_paths(
|
|
75
|
+
groups: Mapping[str, Any],
|
|
76
|
+
*,
|
|
77
|
+
critic_provider: str,
|
|
78
|
+
project_root: Path,
|
|
79
|
+
run_dir: Path,
|
|
80
|
+
) -> tuple[str, str]:
|
|
81
|
+
"""critic 결과 파일과 그 감사 사이드카의 프로젝트 상대 경로.
|
|
82
|
+
|
|
83
|
+
critic 결과는 `<provider>-worker-critic-<task-type>-<workerResults seq>.md`
|
|
84
|
+
로 쓰인다(`-worker-` 토큰이 사이드카 이름을 결정한다 — convergence.md
|
|
85
|
+
§"Coverage critic pass"). 접미사는 분석자 결과와 같은 규칙으로 그룹의
|
|
86
|
+
`runManifestPath` 에서 읽는다. 파일이 없으면 critic 결과가 아직 수집되지
|
|
87
|
+
않은 것이라 거절한다 — gap 검증은 critic 결과 뒤에만 온다.
|
|
88
|
+
"""
|
|
89
|
+
if not _nonempty_string(critic_provider):
|
|
90
|
+
raise CriticVerifyPromptError("run manifest has no critic assignment provider")
|
|
91
|
+
suffix = worker_result_suffix(Path(run_dir), groups)
|
|
92
|
+
if suffix is None:
|
|
93
|
+
raise CriticVerifyPromptError(
|
|
94
|
+
"cannot resolve the worker-result suffix from the grouping's runManifestPath"
|
|
95
|
+
)
|
|
96
|
+
result = Path(run_dir) / "worker-results" / f"{critic_provider}-worker-critic-{suffix}.md"
|
|
97
|
+
if not result.is_file():
|
|
98
|
+
raise CriticVerifyPromptError(
|
|
99
|
+
f"critic result is not collected yet: {result}; gap verification "
|
|
100
|
+
"follows the critic result"
|
|
101
|
+
)
|
|
102
|
+
result_rel = _project_relative(project_root, result)
|
|
103
|
+
try:
|
|
104
|
+
return result_rel, audit_sidecar_rel(result_rel)
|
|
105
|
+
except WorkerArtifactPathError as exc:
|
|
106
|
+
raise CriticVerifyPromptError(str(exc)) from exc
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def critic_verify_gaps(
|
|
110
|
+
batch: Mapping[str, Any],
|
|
111
|
+
groups: Mapping[str, Any],
|
|
112
|
+
worker_id: str,
|
|
113
|
+
) -> list[CriticGap]:
|
|
114
|
+
"""이 워커에게 배정된 gap 을 배치 순서대로.
|
|
115
|
+
|
|
116
|
+
배정은 `critic_gap_assignees` 와 같다: 선언된 중복을 뺀 gap *i* 가
|
|
117
|
+
`roster[i % len(roster)]` 에게 간다. 로스터는 Round 0 그룹의 분석 audience
|
|
118
|
+
워커 순서다 — `apply-critic-gaps` 가 `analyserRoster` 로 적는 것과 같은
|
|
119
|
+
순서다.
|
|
120
|
+
"""
|
|
121
|
+
gaps_value = batch.get("gaps")
|
|
122
|
+
if not isinstance(gaps_value, list) or not gaps_value:
|
|
123
|
+
raise CriticVerifyPromptError("coverage batch has no gaps array")
|
|
124
|
+
roster = analysis_roster(groups)
|
|
125
|
+
# `<worker>-worker` 슬러그도 같은 워커다 — reverify 의 `plan_row_for_worker`
|
|
126
|
+
# 와 validate-run 의 `_plan_dispatch_finding_ids` 가 같은 규칙으로 맞춘다.
|
|
127
|
+
if worker_id not in roster and worker_id.removesuffix("-worker") in roster:
|
|
128
|
+
worker_id = worker_id.removesuffix("-worker")
|
|
129
|
+
if worker_id not in roster:
|
|
130
|
+
raise CriticVerifyPromptError(
|
|
131
|
+
f"`{worker_id}` is not an analysis worker of this run; roster: "
|
|
132
|
+
f"{', '.join(roster) or 'none'}"
|
|
133
|
+
)
|
|
134
|
+
verifiable = [
|
|
135
|
+
gap for gap in gaps_value
|
|
136
|
+
if isinstance(gap, Mapping) and not gap.get("duplicateOf")
|
|
137
|
+
]
|
|
138
|
+
assignees = critic_gap_assignees(roster, verifiable)
|
|
139
|
+
assigned: list[CriticGap] = []
|
|
140
|
+
seen: set[str] = set()
|
|
141
|
+
for gap, assignee in zip(verifiable, assignees):
|
|
142
|
+
gap_id = _nonempty_string(gap.get("gapId"))
|
|
143
|
+
if not gap_id:
|
|
144
|
+
raise CriticVerifyPromptError("coverage batch gap has no gapId")
|
|
145
|
+
if gap_id in seen:
|
|
146
|
+
raise CriticVerifyPromptError(f"duplicate critic gapId: {gap_id}")
|
|
147
|
+
seen.add(gap_id)
|
|
148
|
+
if assignee != worker_id:
|
|
149
|
+
continue
|
|
150
|
+
ticket_ids = gap.get("ticketIds")
|
|
151
|
+
assigned.append(CriticGap(
|
|
152
|
+
gap_id=gap_id,
|
|
153
|
+
summary=_nonempty_string(gap.get("summary")),
|
|
154
|
+
category=_nonempty_string(gap.get("category")),
|
|
155
|
+
ticket_ids=tuple(
|
|
156
|
+
str(ticket) for ticket in ticket_ids
|
|
157
|
+
) if isinstance(ticket_ids, list) else (),
|
|
158
|
+
origin_evidence=_nonempty_string(gap.get("originEvidence")),
|
|
159
|
+
))
|
|
160
|
+
if not assigned:
|
|
161
|
+
raise CriticVerifyPromptError(
|
|
162
|
+
f"round-robin assigns no gap to `{worker_id}`; assignees in batch "
|
|
163
|
+
f"order: {', '.join(assignees) or 'none'}"
|
|
164
|
+
)
|
|
165
|
+
return assigned
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
_EVIDENCE_ACCESS = """The `**Cited evidence**` line is the lead's summary of what the critic cited.
|
|
169
|
+
The complete citation is the critic's own item: before judging, open the
|
|
170
|
+
`**Origin item**` file at the named `### [<gap-id>]` section and read every path,
|
|
171
|
+
line, command, and quote it cites. The `**Origin audit sidecar**` records the
|
|
172
|
+
read-only commands the critic ran and their output; it counts as cited evidence
|
|
173
|
+
and you may open it. A gap claims that something was NOT covered — to refute it,
|
|
174
|
+
show where the coverage exists (an analyser result item, a file, a test); to
|
|
175
|
+
let it survive, confirm the coverage is absent where the critic says it is."""
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def critic_verify_prompt_body(
|
|
179
|
+
*,
|
|
180
|
+
task_key: str,
|
|
181
|
+
critic_worker: str,
|
|
182
|
+
critic_result_path: str,
|
|
183
|
+
critic_audit_path: str,
|
|
184
|
+
gaps: Sequence[CriticGap],
|
|
185
|
+
) -> str:
|
|
186
|
+
"""critic gap 검증 지시문 본문. 같은 입력이면 같은 바이트를 낸다."""
|
|
187
|
+
if not _nonempty_string(task_key):
|
|
188
|
+
raise CriticVerifyPromptError("run manifest carries no taskKey")
|
|
189
|
+
if not gaps:
|
|
190
|
+
raise CriticVerifyPromptError("no gaps to verify")
|
|
191
|
+
rows = [
|
|
192
|
+
"## Instructions\n\n",
|
|
193
|
+
f"{RENDERED_BY_LINE}\n\n",
|
|
194
|
+
f"Perform ADVERSARIAL coverage-gap verification for {task_key} "
|
|
195
|
+
f"(dispatch kind `{CRITIC_VERIFY_DISPATCH_KIND}`, one round).\n\n",
|
|
196
|
+
ADVERSARIAL_MANDATE, "\n\n",
|
|
197
|
+
_EVIDENCE_ACCESS, "\n\n",
|
|
198
|
+
"## Findings to verify\n",
|
|
199
|
+
]
|
|
200
|
+
for gap in gaps:
|
|
201
|
+
rows.append(f"\n### {gap.gap_id}: {gap.summary or '(no summary)'}\n")
|
|
202
|
+
rows.append(f"**Origin**: {critic_worker}\n")
|
|
203
|
+
rows.append(f"**Category**: {gap.category or '(none recorded)'}\n")
|
|
204
|
+
if gap.ticket_ids:
|
|
205
|
+
rows.append(f"**Tickets**: {', '.join(gap.ticket_ids)}\n")
|
|
206
|
+
rows.append(f"**Cited evidence**: {gap.origin_evidence or '(none recorded)'}\n")
|
|
207
|
+
rows.append(
|
|
208
|
+
f"**Origin item**: `{critic_result_path}` — section `### [{gap.gap_id}]`\n"
|
|
209
|
+
)
|
|
210
|
+
rows.append(f"**Origin audit sidecar**: `{critic_audit_path}`\n")
|
|
211
|
+
rows.append("\n## Response format\n\n")
|
|
212
|
+
rows.append(
|
|
213
|
+
"One block per gap, headed by the gap id at exactly three hashes. "
|
|
214
|
+
"Field labels are bold with the colon outside (`**Verdict**: …`); the "
|
|
215
|
+
"collector also reads `**Verdict:** …` and `- Verdict: …` as the same field.\n\n"
|
|
216
|
+
)
|
|
217
|
+
rows.append(ADVERSARIAL_RESPONSE.replace("<finding-id>", gaps[0].gap_id))
|
|
218
|
+
rows.append("\n")
|
|
219
|
+
if len(gaps) > 1:
|
|
220
|
+
rows.append(f"\n### {gaps[1].gap_id}\n**Verdict**: ...\n")
|
|
221
|
+
return "".join(rows)
|
|
@@ -952,7 +952,7 @@ def _parse_critic_dispatches(
|
|
|
952
952
|
return dispatches, completed
|
|
953
953
|
|
|
954
954
|
|
|
955
|
-
def
|
|
955
|
+
def critic_gap_assignees(
|
|
956
956
|
roster_order: list[str],
|
|
957
957
|
gaps: list[Mapping[str, Any]],
|
|
958
958
|
) -> list[str]:
|
|
@@ -990,7 +990,7 @@ def _critic_gap_coverage_errors(
|
|
|
990
990
|
계약 위반이 아니라 더 본 것이고, 그 사이 방치된 gap 은 표가 없어
|
|
991
991
|
`unverifiedGaps` 로 스스로 드러난다.
|
|
992
992
|
"""
|
|
993
|
-
assignees = set(
|
|
993
|
+
assignees = set(critic_gap_assignees(roster_order, gaps))
|
|
994
994
|
if dispatched == set(roster_order) or dispatched == assignees:
|
|
995
995
|
return []
|
|
996
996
|
missing = sorted(assignees - dispatched)
|
|
@@ -53,7 +53,7 @@ class ReverifyFinding:
|
|
|
53
53
|
origin_audit_path: str
|
|
54
54
|
|
|
55
55
|
|
|
56
|
-
|
|
56
|
+
ADVERSARIAL_MANDATE = """Your job is to BREAK each finding below, not to confirm it. For EACH finding,
|
|
57
57
|
open the cited evidence directly and actively search for evidence that the claim
|
|
58
58
|
is wrong, overstated, or unproven. Then respond with exactly one verdict:
|
|
59
59
|
|
|
@@ -95,7 +95,7 @@ commands that worker ran and their output; it counts as cited evidence and you m
|
|
|
95
95
|
open it. Judge the claim against what the origin worker actually cited, never against
|
|
96
96
|
the summary line alone."""
|
|
97
97
|
|
|
98
|
-
|
|
98
|
+
ADVERSARIAL_RESPONSE = """### <finding-id>
|
|
99
99
|
**Verdict**: REFUTED | SURVIVES | SURVIVES-WITH-CAVEAT | UNVERIFIABLE
|
|
100
100
|
**Basis** (only if REFUTED): counter-evidence | burden-not-met
|
|
101
101
|
**Explanation**: <2-3 sentences; for counter-evidence include the file:line you found>"""
|
|
@@ -206,8 +206,8 @@ def reverify_prompt_body(
|
|
|
206
206
|
if not findings:
|
|
207
207
|
raise ReverifyPromptError("no findings to verify")
|
|
208
208
|
mode = "ADVERSARIAL re-verification" if adversarial else "re-verification"
|
|
209
|
-
mandate =
|
|
210
|
-
response =
|
|
209
|
+
mandate = ADVERSARIAL_MANDATE if adversarial else _COLLABORATIVE_MANDATE
|
|
210
|
+
response = ADVERSARIAL_RESPONSE if adversarial else _COLLABORATIVE_RESPONSE
|
|
211
211
|
rows = [
|
|
212
212
|
"## Instructions\n\n",
|
|
213
213
|
f"{RENDERED_BY_LINE}\n\n",
|
|
@@ -106,9 +106,16 @@ def reserve_dynamic_verifier(
|
|
|
106
106
|
*,
|
|
107
107
|
input_digest: str,
|
|
108
108
|
invocation_ref: str | None = None,
|
|
109
|
+
dispatch_kind: str | None = None,
|
|
109
110
|
) -> tuple[RoleExecution, Invocation]:
|
|
110
111
|
"""Reserve one provider-neutral verifier identity for a logical round.
|
|
111
112
|
|
|
113
|
+
``dispatch_kind`` 를 주지 않으면 번호 라운드(`reverify-r<N>`)다. critic gap
|
|
114
|
+
검증(`critic-verify`)은 같은 verifier 신원을 예약하되 kind 를 그대로 적는다 —
|
|
115
|
+
validate-run 이 team-state 의 dispatch kind 와 예약된 invocation 의
|
|
116
|
+
`dispatchKind` 를 대조하므로 예약이 `reverify-r1` 로 남으면 그 디스패치가
|
|
117
|
+
거부된다.
|
|
118
|
+
|
|
112
119
|
예약은 정체성만 잡는다. 쓰기 계약은 디스패치가 attempt 를 열 때 한 번만
|
|
113
120
|
계산해 그 attempt 행에 적는다 — 예약이 같은 값을 두 번째로 계산하던 동안,
|
|
114
121
|
두 계산의 입력(worktree)이 갈리면 같은 invocationRef 가 `invocationRef
|
|
@@ -156,7 +163,7 @@ def reserve_dynamic_verifier(
|
|
|
156
163
|
source_invocation_ref=None,
|
|
157
164
|
recovery_ref=None,
|
|
158
165
|
duty_id=duty_id,
|
|
159
|
-
dispatch_kind=f"reverify-r{round_number}",
|
|
166
|
+
dispatch_kind=dispatch_kind or f"reverify-r{round_number}",
|
|
160
167
|
round=round_number,
|
|
161
168
|
input_digest=input_digest,
|
|
162
169
|
)
|
|
@@ -61,6 +61,7 @@ from .worker_prompt_contract import (
|
|
|
61
61
|
validate_prompt_model_header,
|
|
62
62
|
validate_reverify_prompt,
|
|
63
63
|
)
|
|
64
|
+
from .worker_prompt_policy import is_verification_dispatch_kind
|
|
64
65
|
from .worker_runner import LIVE, QUIET
|
|
65
66
|
from .worker_request import verifier_extra_dirs
|
|
66
67
|
from .worker_artifact_paths import audit_sidecar_rel
|
|
@@ -1880,13 +1881,13 @@ def validate_dispatch_prompts(
|
|
|
1880
1881
|
if isinstance(manifest.get("agentContract"), Mapping):
|
|
1881
1882
|
_validate_agent_invocations(manifest, jobs)
|
|
1882
1883
|
initial_jobs = [
|
|
1883
|
-
job for job in jobs if not job.dispatch_kind
|
|
1884
|
+
job for job in jobs if not is_verification_dispatch_kind(job.dispatch_kind)
|
|
1884
1885
|
]
|
|
1885
1886
|
if initial_jobs:
|
|
1886
1887
|
validate_initial_prompts(manifest, initial_jobs)
|
|
1887
1888
|
|
|
1888
1889
|
reverify_jobs = [
|
|
1889
|
-
job for job in jobs if job.dispatch_kind
|
|
1890
|
+
job for job in jobs if is_verification_dispatch_kind(job.dispatch_kind)
|
|
1890
1891
|
]
|
|
1891
1892
|
if not reverify_jobs:
|
|
1892
1893
|
return
|
|
@@ -1919,6 +1920,7 @@ def validate_dispatch_prompts(
|
|
|
1919
1920
|
task_type=task_type,
|
|
1920
1921
|
forbidden_actions=forbidden_actions,
|
|
1921
1922
|
expected_model=job.model_execution_value or None,
|
|
1923
|
+
dispatch_kind=job.dispatch_kind,
|
|
1922
1924
|
)
|
|
1923
1925
|
)
|
|
1924
1926
|
if errors:
|
|
@@ -334,6 +334,65 @@ class _Validator:
|
|
|
334
334
|
self.errors.append(f"{_format_path(path)}: {message}")
|
|
335
335
|
|
|
336
336
|
|
|
337
|
+
def follow_up_task_rules(schema: Mapping[str, Any], task_type: str) -> tuple[str, ...]:
|
|
338
|
+
"""이 task type 의 `followUpTasks` 에 스키마가 못 박은 행 규칙, 저작 문장으로.
|
|
339
|
+
|
|
340
|
+
`allOf` 의 if/then 가지 하나가 비종결 task type 에 phase-continuation 행
|
|
341
|
+
하나를 요구하고(`minItems`, `contains.origin.const`), `FollowUpRow` 의
|
|
342
|
+
가지가 그 행의 `autoSpawn` 을 `no` 로 고정한다. 작성자는 둘 다 읽지 못해
|
|
343
|
+
빈 배열을 냈고 조립이 `array length 0 < minItems 1` 로 거절했다(2026-09-09
|
|
344
|
+
실측, dev-10642 requirements-discovery: 이 계열로만 라운드 3회). 가지가
|
|
345
|
+
이 task type 에 없으면 빈 튜플이다 — 제약이 없다는 뜻이지 실패가 아니다.
|
|
346
|
+
"""
|
|
347
|
+
def _task_matches(condition: Any) -> bool:
|
|
348
|
+
if not isinstance(condition, dict):
|
|
349
|
+
return False
|
|
350
|
+
header = (condition.get("properties") or {}).get("header") or {}
|
|
351
|
+
selector = (header.get("properties") or {}).get("taskType") or {}
|
|
352
|
+
allowed = selector.get("enum")
|
|
353
|
+
if allowed is None and "const" in selector:
|
|
354
|
+
allowed = [selector["const"]]
|
|
355
|
+
return isinstance(allowed, list) and task_type in allowed
|
|
356
|
+
|
|
357
|
+
rules: list[str] = []
|
|
358
|
+
for branch in schema.get("allOf") or []:
|
|
359
|
+
if not isinstance(branch, dict) or not _task_matches(branch.get("if")):
|
|
360
|
+
continue
|
|
361
|
+
follow_up = ((branch.get("then") or {}).get("properties") or {}).get("followUpTasks")
|
|
362
|
+
if not isinstance(follow_up, dict):
|
|
363
|
+
continue
|
|
364
|
+
min_items = follow_up.get("minItems")
|
|
365
|
+
origin = (
|
|
366
|
+
((follow_up.get("contains") or {}).get("properties") or {}).get("origin") or {}
|
|
367
|
+
).get("const")
|
|
368
|
+
if isinstance(min_items, int) and min_items > 0:
|
|
369
|
+
rules.append(
|
|
370
|
+
f"`Follow Up Tasks`: at least {min_items} `- Item N` row(s) for task "
|
|
371
|
+
f"type `{task_type}`; an empty list is refused."
|
|
372
|
+
)
|
|
373
|
+
if isinstance(origin, str) and origin:
|
|
374
|
+
rules.append(
|
|
375
|
+
f"`Follow Up Tasks`: one row must carry `Origin` `{origin}` — the "
|
|
376
|
+
"next phase of this task."
|
|
377
|
+
)
|
|
378
|
+
row_schema = (schema.get("$defs") or {}).get("FollowUpRow") or {}
|
|
379
|
+
for row_branch in row_schema.get("allOf") or []:
|
|
380
|
+
if not isinstance(row_branch, dict):
|
|
381
|
+
continue
|
|
382
|
+
condition = ((row_branch.get("if") or {}).get("properties") or {}).get("origin") or {}
|
|
383
|
+
if condition.get("const") != origin:
|
|
384
|
+
continue
|
|
385
|
+
pinned = ((row_branch.get("then") or {}).get("properties") or {})
|
|
386
|
+
for key, value in pinned.items():
|
|
387
|
+
if isinstance(value, dict) and "const" in value:
|
|
388
|
+
rules.append(
|
|
389
|
+
f"`Follow Up Tasks`: the `{origin}` row's `{key}` "
|
|
390
|
+
f"(schema key; write its Title Case label) is exactly "
|
|
391
|
+
f"`{value['const']}`."
|
|
392
|
+
)
|
|
393
|
+
return tuple(rules)
|
|
394
|
+
|
|
395
|
+
|
|
337
396
|
def verdict_token_rule(schema: Mapping[str, Any], task_type: str) -> tuple[str, ...]:
|
|
338
397
|
"""이 task type 의 `finalVerdict.verdictToken` 에 스키마가 허용하는 값.
|
|
339
398
|
|
|
@@ -35,7 +35,7 @@ from .report_narrative import (
|
|
|
35
35
|
CORRECTION_KINDS = ("replace", "remove", "rewrite")
|
|
36
36
|
MECHANICAL_KINDS = frozenset({"replace", "remove"})
|
|
37
37
|
# okstra 가 렌더하는 절. 지시문 본문에 있으면 두 절이 갈라지므로 거절한다.
|
|
38
|
-
OKSTRA_OWNED_SECTIONS = ("## Corrections", "## Output")
|
|
38
|
+
OKSTRA_OWNED_SECTIONS = ("## Corrections", "## Previous Attempt", "## Output")
|
|
39
39
|
|
|
40
40
|
_SCHEMA_RELATIVE = ("schemas", "report-writer-corrections-v1.0.schema.json")
|
|
41
41
|
_SEGMENT_RE = re.compile(r"^([A-Za-z][A-Za-z0-9]*)((?:\[\d+\])*)$")
|
|
@@ -191,6 +191,7 @@ def render_v2_html_view(
|
|
|
191
191
|
response_js = (root / "report.js").read_text(encoding="utf-8")
|
|
192
192
|
base_js = (root / "html/assets/base.js").read_text(encoding="utf-8")
|
|
193
193
|
source_data = final_report_data_path(Path(run_meta.source_report)).as_posix()
|
|
194
|
+
user_responses_dir = user_responses_dir_for_report(data_path.resolve())
|
|
194
195
|
context = {
|
|
195
196
|
**view.context,
|
|
196
197
|
"runMeta": run_meta,
|
|
@@ -215,6 +216,11 @@ def render_v2_html_view(
|
|
|
215
216
|
# Every task type ends with the same run-cost section, so it is bound
|
|
216
217
|
# here rather than in ten view models that would each rebuild it.
|
|
217
218
|
"runUsage": run_usage(data),
|
|
219
|
+
# 바닥글이 "여기에 두면" 이라고 가리키는 디렉터리. 상대 경로
|
|
220
|
+
# `runs/<type>/user-responses/` 는 task 디렉터리 기준이라는 말이 없어
|
|
221
|
+
# 프로젝트 루트에서 찾으면 없고, implementation 은 stage 아래라 그 경로
|
|
222
|
+
# 자체가 틀렸다(실측 2026-09-09) — 보고서 파일에서 계산한 절대 경로를 찍는다.
|
|
223
|
+
"userResponseDir": user_responses_dir.as_posix(),
|
|
218
224
|
"executionRoles": data.get("executionRoles") or [],
|
|
219
225
|
"css": (root / "html/assets/base.css").read_text(encoding="utf-8"),
|
|
220
226
|
"js": response_js + "\n" + base_js,
|
|
@@ -223,8 +229,8 @@ def render_v2_html_view(
|
|
|
223
229
|
document = env.get_template(route.template_name).render(**context)
|
|
224
230
|
document = inject_report_index(document, label=translate("base.contents"))
|
|
225
231
|
output_path.write_text(document, encoding="utf-8")
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
232
|
+
# The footer tells the reader to drop the exported file here, so the
|
|
233
|
+
# directory has to exist before they go looking for it — on every report,
|
|
234
|
+
# not only those with clarification rows: the footer is rendered on all.
|
|
235
|
+
user_responses_dir.mkdir(parents=True, exist_ok=True)
|
|
230
236
|
return output_path
|
|
@@ -40,7 +40,10 @@ NARRATIVE_GRAMMAR_INSTRUCTIONS: tuple[str, ...] = (
|
|
|
40
40
|
"nest a child by indenting two more spaces), `- Item <N>` (one array entry, "
|
|
41
41
|
"numbered 1..N without gaps), and `> value` (one scalar; repeat the line for a "
|
|
42
42
|
"multi-line value; `> _none_` for null, an empty object, or an empty array). "
|
|
43
|
-
"
|
|
43
|
+
"A `> value` line is indented exactly two spaces deeper than the `- **Label**` "
|
|
44
|
+
"or `- Item N` line it belongs to — a label at column 0 takes its value at "
|
|
45
|
+
"column 2, a label at column 2 takes it at column 4; a value at column 0 is "
|
|
46
|
+
"outside every field and the file is refused. Blank lines are ignored.",
|
|
44
47
|
"Every other line is rejected — YAML frontmatter (`---` blocks), Markdown "
|
|
45
48
|
"headings (`#`, `##`, `###`), pipe tables at column 0, code fences, bare "
|
|
46
49
|
"paragraphs, JSON. Put such text inside a `> ` value instead. Report assembly "
|
|
@@ -49,6 +49,24 @@ def _agent_label(row: Mapping[str, Any]) -> str:
|
|
|
49
49
|
return _AGENT_LABELS.get(value.lower(), value)
|
|
50
50
|
|
|
51
51
|
|
|
52
|
+
def _row_model(row: Mapping[str, Any], usage: Mapping[str, Any]) -> str:
|
|
53
|
+
"""행이 이름한 모델, 없으면 사용량 수집기가 트랜스크립트에서 읽은 모델.
|
|
54
|
+
|
|
55
|
+
`/okstra-run` 으로 현재 세션에서 도는 리드는 prepare 시점에 모델을 모른다 —
|
|
56
|
+
매니페스트 `leadModel` 과 team-state `lead.model` 이 문자 그대로 `unknown`
|
|
57
|
+
이다. 수집기는 그 세션의 jsonl 에서 `claude-opus-5` 를 읽어 `leadUsage.model`
|
|
58
|
+
에 적어 두므로, 표는 그 값을 쓴다(실측 2026-09-09: 리드 행 모델 `unknown`).
|
|
59
|
+
"""
|
|
60
|
+
named = row.get("model") or row.get("modelExecutionValue")
|
|
61
|
+
if isinstance(named, str) and named.strip() and named.strip() != "unknown":
|
|
62
|
+
return named.strip()
|
|
63
|
+
if usage.get("source") != "unavailable":
|
|
64
|
+
measured = usage.get("model")
|
|
65
|
+
if isinstance(measured, str) and measured.strip():
|
|
66
|
+
return measured.strip()
|
|
67
|
+
return "unknown"
|
|
68
|
+
|
|
69
|
+
|
|
52
70
|
def _execution_row(
|
|
53
71
|
row: Mapping[str, Any],
|
|
54
72
|
*,
|
|
@@ -65,7 +83,7 @@ def _execution_row(
|
|
|
65
83
|
result = {
|
|
66
84
|
"agent": _agent_label(row),
|
|
67
85
|
"role": role,
|
|
68
|
-
"model":
|
|
86
|
+
"model": _row_model(row, source),
|
|
69
87
|
"status": status,
|
|
70
88
|
"summary": _text(
|
|
71
89
|
row.get("summary") or row.get("reason"),
|
|
@@ -26,7 +26,7 @@ from .report_narrative import writer_owned_data
|
|
|
26
26
|
from .scope_provenance import brief_end_state_id_sequence
|
|
27
27
|
|
|
28
28
|
from .exact_coverage import COVERAGE_VERDICT_PRECEDENCE
|
|
29
|
-
from .final_report_schema import task_block_rules, verdict_token_rule
|
|
29
|
+
from .final_report_schema import follow_up_task_rules, task_block_rules, verdict_token_rule
|
|
30
30
|
from .report_contract import TASK_TYPE_DATA_PROPERTY
|
|
31
31
|
from .report_markdown import humanise
|
|
32
32
|
from .report_narrative import NarrativeContractError, allowed_top_level_fields
|
|
@@ -87,6 +87,10 @@ class ReportSynthesisPacket:
|
|
|
87
87
|
# (2026-09-03 실측: implementation-option-selection 네 회차).
|
|
88
88
|
block_key: str = ""
|
|
89
89
|
block_rules: tuple[str, ...] = ()
|
|
90
|
+
# 비종결 task type 의 `followUpTasks` 행 규칙(최소 행 수, phase-continuation
|
|
91
|
+
# 행, 그 행의 autoSpawn). 스키마의 if/then 가지라 작성자에게 도달하지 않았고
|
|
92
|
+
# 빈 배열이 조립에서야 거절됐다(2026-09-09 실측, dev-10642).
|
|
93
|
+
follow_up_rules: tuple[str, ...] = ()
|
|
90
94
|
# 작성자가 최상위에 쓸 수 있는 필드 전체(서사 스키마의 properties). 필수만
|
|
91
95
|
# 적던 동안 리드가 어느 task type 에도 없는 절을 지시했고, 작성자는 조립이
|
|
92
96
|
# 거절할 때까지 그 지시를 거를 근거가 없었다(2026-09-03 실측).
|
|
@@ -142,6 +146,7 @@ class ReportSynthesisPacket:
|
|
|
142
146
|
f"contain: {labels}. Report assembly refuses any other top-level "
|
|
143
147
|
"field, whichever instruction asked for it."
|
|
144
148
|
)
|
|
149
|
+
lines.extend(self.follow_up_rules)
|
|
145
150
|
if self.block_rules:
|
|
146
151
|
lines.append(
|
|
147
152
|
f"Shape of the `{self.block_key}` block, one line per object, from "
|
|
@@ -736,8 +741,8 @@ def build_report_synthesis_packet(
|
|
|
736
741
|
for worker in roster
|
|
737
742
|
if _string(worker) and _string(worker) != "report-writer"
|
|
738
743
|
)
|
|
739
|
-
required_top_level, verdict_tokens, block_rules =
|
|
740
|
-
sources, task_type
|
|
744
|
+
required_top_level, verdict_tokens, block_rules, follow_up_rules = (
|
|
745
|
+
_schema_authoring_rules(sources, task_type)
|
|
741
746
|
)
|
|
742
747
|
return ReportSynthesisPacket(
|
|
743
748
|
task_key=_string(manifest.get("taskKey")),
|
|
@@ -745,6 +750,7 @@ def build_report_synthesis_packet(
|
|
|
745
750
|
result_path=_relative(project_root, narrative_path),
|
|
746
751
|
required_top_level=required_top_level,
|
|
747
752
|
verdict_tokens=verdict_tokens,
|
|
753
|
+
follow_up_rules=follow_up_rules,
|
|
748
754
|
block_key=TASK_TYPE_DATA_PROPERTY.get(task_type, "") if block_rules else "",
|
|
749
755
|
block_rules=block_rules,
|
|
750
756
|
allowed_top_level=_allowed_top_level_labels(),
|
|
@@ -758,12 +764,13 @@ def build_report_synthesis_packet(
|
|
|
758
764
|
|
|
759
765
|
def _schema_authoring_rules(
|
|
760
766
|
sources: tuple[ReportSynthesisSource, ...], task_type: str,
|
|
761
|
-
) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...]]:
|
|
762
|
-
"""넘겨받은(동결된) 완성 리포트 스키마에서 작성자 몫의 규칙
|
|
767
|
+
) -> tuple[tuple[str, ...], tuple[str, ...], tuple[str, ...], tuple[str, ...]]:
|
|
768
|
+
"""넘겨받은(동결된) 완성 리포트 스키마에서 작성자 몫의 규칙 네 가지를 뽑는다.
|
|
763
769
|
|
|
764
770
|
최상위 `required` 가운데 작성자 소유 이름(서사 스키마의 properties)만
|
|
765
|
-
사람이 읽는 라벨로, 이 task type 의 `Verdict Token` 허용값,
|
|
766
|
-
|
|
771
|
+
사람이 읽는 라벨로, 이 task type 의 `Verdict Token` 허용값, 이 task type 의
|
|
772
|
+
데이터 블록 안쪽 모양(`task_block_rules`), 그리고 `followUpTasks` 행 규칙
|
|
773
|
+
(`follow_up_task_rules`). 스키마 소스가 없거나
|
|
767
774
|
JSON 이 아니면 빈 값이다 — 이 함수는 조립 검증을 대신하지 않고 도달하지
|
|
768
775
|
못하던 규칙을 저작 계약에 옮길 뿐이다.
|
|
769
776
|
"""
|
|
@@ -789,7 +796,12 @@ def _schema_authoring_rules(
|
|
|
789
796
|
)
|
|
790
797
|
block_key = TASK_TYPE_DATA_PROPERTY.get(task_type, "")
|
|
791
798
|
block_rules = task_block_rules(schema, block_key) if block_key else ()
|
|
792
|
-
return
|
|
799
|
+
return (
|
|
800
|
+
required_top_level,
|
|
801
|
+
verdict_token_rule(schema, task_type),
|
|
802
|
+
block_rules,
|
|
803
|
+
follow_up_task_rules(schema, task_type),
|
|
804
|
+
)
|
|
793
805
|
|
|
794
806
|
|
|
795
807
|
def _allowed_top_level_labels() -> tuple[str, ...]:
|
|
@@ -132,13 +132,23 @@ def next_prompt(state: WizardState) -> Prompt:
|
|
|
132
132
|
|
|
133
133
|
|
|
134
134
|
def _native_picker_screen(state: WizardState, prompt: Prompt) -> Prompt:
|
|
135
|
+
"""호스트 네이티브 선택기 한도에 맞춘 화면.
|
|
136
|
+
|
|
137
|
+
단일 선택이 한도를 넘으면 쪽으로 나눈다(`present_picker`). 체크박스(`multi`)는
|
|
138
|
+
나누지 않는다 — 네이티브 체크박스에 못 실으면 `CapabilityInteractionPort.plan`
|
|
139
|
+
이 `numbered-multi` 로 내려 전체 목록을 한 번에 보인다. 종전엔 체크박스도
|
|
140
|
+
한 줄씩 토글하는 쪽으로 내렸는데, claude-code 한도 4 에서 후보 12개는 쪽당
|
|
141
|
+
2개가 됐고, 쪽 사본이 추천 표시를 단 채 단일 선택이 돼 `Prompt` 의 추천
|
|
142
|
+
불변식(단일 선택은 추천 정확히 하나)에 걸려 화면이 열리지 않았다(실측
|
|
143
|
+
2026-09-09, verifier 전체 후보 화면).
|
|
144
|
+
"""
|
|
135
145
|
if "native_single_select" not in state.available_functions:
|
|
136
146
|
return prompt
|
|
137
147
|
if prompt.kind == "pick_group":
|
|
138
148
|
if _interaction_plan(state, prompt).kind == "native-group":
|
|
139
149
|
return prompt
|
|
140
150
|
prompt = prompt.questions[0]
|
|
141
|
-
if
|
|
151
|
+
if prompt.multi:
|
|
142
152
|
return prompt
|
|
143
153
|
limit = default_host_registry().resolve(state.host_runtime).interaction().native_option_limit
|
|
144
154
|
return present_picker(state, prompt, limit=limit)
|
|
@@ -377,9 +387,8 @@ def submit(state: WizardState, value: str) -> dict[str, Any]:
|
|
|
377
387
|
if prompt.kind == "pick_group":
|
|
378
388
|
return _submit_group(state, prompt, value)
|
|
379
389
|
if _is_role_selection_step(prompt.step):
|
|
380
|
-
# 화면은 호스트 한도에 맞춰 쪽으로 잘린 사본일 수
|
|
381
|
-
# 답은 잘리지 않은 원본의 선택지로
|
|
382
|
-
# 쪽에 걸쳐 고른 값이다.
|
|
390
|
+
# 고정 단일 역할의 화면은 호스트 한도에 맞춰 쪽으로 잘린 사본일 수
|
|
391
|
+
# 있다(`present_picker`). 답은 잘리지 않은 원본의 선택지로 검증한다.
|
|
383
392
|
original = next_role_prompt(state)
|
|
384
393
|
if original is not None and original.step == prompt.step:
|
|
385
394
|
prompt = original
|