okstra 0.189.2 → 0.189.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/architecture.md +1 -1
- package/docs/cli.md +3 -3
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/lead/report-writer.md +8 -1
- package/runtime/prompts/profiles/_common-contract.md +1 -1
- package/runtime/prompts/profiles/_implementation-deliverable.md +1 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +2 -2
- package/runtime/prompts/profiles/implementation-planning.md +1 -1
- package/runtime/python/okstra_ctl/agent/prompt_cli/cli.py +5 -1
- package/runtime/python/okstra_ctl/plan_items_cli.py +12 -0
- package/runtime/python/okstra_ctl/render.py +24 -7
- package/runtime/python/okstra_ctl/report_assembly.py +10 -4
- package/runtime/python/okstra_ctl/report_html/common.py +241 -113
- package/runtime/python/okstra_ctl/report_html/context_links.py +121 -0
- package/runtime/python/okstra_ctl/report_html/models.py +11 -5
- package/runtime/python/okstra_ctl/report_html/render.py +54 -13
- package/runtime/python/okstra_ctl/report_html/view_models/change_impact_analysis.py +8 -1
- package/runtime/python/okstra_ctl/report_html/view_models/error_analysis.py +8 -1
- package/runtime/python/okstra_ctl/report_html/view_models/feature_analysis.py +11 -1
- package/runtime/python/okstra_ctl/report_html/view_models/final_verification.py +13 -1
- package/runtime/python/okstra_ctl/report_html/view_models/implementation.py +9 -1
- package/runtime/python/okstra_ctl/report_html/view_models/implementation_option_selection.py +29 -2
- package/runtime/python/okstra_ctl/report_html/view_models/implementation_planning.py +14 -9
- package/runtime/python/okstra_ctl/report_html/view_models/improvement_discovery.py +8 -1
- package/runtime/python/okstra_ctl/report_html/view_models/project_analysis.py +15 -1
- package/runtime/python/okstra_ctl/report_html/view_models/release_handoff.py +7 -1
- package/runtime/python/okstra_ctl/report_html/view_models/requirements_discovery.py +9 -1
- package/runtime/python/okstra_ctl/report_translation.py +58 -1
- package/runtime/python/okstra_ctl/run.py +6 -0
- package/runtime/python/okstra_ctl/worker_audit_check.py +23 -8
- package/runtime/schemas/final-report-v2.0.schema.json +4 -0
- package/runtime/schemas/final-report-v3.0.schema.json +4 -0
- package/runtime/templates/reports/html/base.template.html +13 -3
- package/runtime/templates/reports/html/i18n/en.json +43 -5
- package/runtime/templates/reports/html/i18n/ko.json +43 -5
- package/runtime/templates/reports/html/macros/forms.html +4 -4
- package/runtime/templates/reports/html/tasks/final-verification.template.html +11 -1
- package/runtime/templates/reports/html/tasks/implementation-option-selection.template.html +16 -15
- package/runtime/templates/reports/html/tasks/implementation-planning.template.html +5 -5
- package/runtime/templates/reports/html/tasks/requirements-discovery.template.html +1 -1
- package/runtime/validators/validate-report-views.py +30 -1
- package/runtime/validators/validate-run.py +220 -3
|
@@ -3084,6 +3084,9 @@ def validate_final_report_data(
|
|
|
3084
3084
|
failures.extend(f"{task_type}: {error}" for error in analysis_result.errors)
|
|
3085
3085
|
|
|
3086
3086
|
_validate_no_opaque_id_references(data, failures)
|
|
3087
|
+
_validate_no_null_literals_in_prose(data, failures)
|
|
3088
|
+
for warning in _unbridged_worker_finding_refs(data):
|
|
3089
|
+
print(f"validate-run: warning: {warning}", file=sys.stderr)
|
|
3087
3090
|
# Phase-agnostic: the coverage critic runs in every finding-producing phase.
|
|
3088
3091
|
_validate_unverified_critic_gaps_recorded(data, failures)
|
|
3089
3092
|
# Called here rather than from a task-type branch: four profiles raise
|
|
@@ -3093,6 +3096,9 @@ def validate_final_report_data(
|
|
|
3093
3096
|
|
|
3094
3097
|
task_type = (data.get("header") or {}).get("taskType")
|
|
3095
3098
|
_validate_verifier_fail_blocks_verdict(data, failures)
|
|
3099
|
+
_validate_verifier_discrepancy_names_checklist_phase(
|
|
3100
|
+
data, report_path, project_root, failures
|
|
3101
|
+
)
|
|
3096
3102
|
if task_type == "implementation-option-selection":
|
|
3097
3103
|
selection = data.get("implementationOptionSelection") or {}
|
|
3098
3104
|
validation_root = project_root or report_path.parent
|
|
@@ -3159,7 +3165,9 @@ def validate_final_report_data(
|
|
|
3159
3165
|
carried = _carried_decision_map(
|
|
3160
3166
|
manifest, project_root=project_root, report_path=report_path
|
|
3161
3167
|
)
|
|
3162
|
-
_validate_supersession_ledger(
|
|
3168
|
+
_validate_supersession_ledger(
|
|
3169
|
+
data, failures, carried=carried, new_plan=selected_direction_contract
|
|
3170
|
+
)
|
|
3163
3171
|
_validate_approval_clarification_backtrace(data, failures)
|
|
3164
3172
|
_validate_rerun_guidance(data, failures)
|
|
3165
3173
|
_validate_approval_guidance(data, failures)
|
|
@@ -3239,6 +3247,97 @@ def _validate_no_opaque_id_references(data: dict, failures: list[str]) -> None:
|
|
|
3239
3247
|
)
|
|
3240
3248
|
|
|
3241
3249
|
|
|
3250
|
+
# A stringified null is not prose. Lowercase `none` is a token the clarification
|
|
3251
|
+
# option contract uses on purpose ("reverses nothing"), so only the capitalised
|
|
3252
|
+
# and language-native spellings count — 2026-09-05 audit: `"None"` in 16
|
|
3253
|
+
# trade-off cells and 2 option fields across the shipped reports.
|
|
3254
|
+
_NULL_LITERALS = frozenset({"None", "null", "undefined", "NaN"})
|
|
3255
|
+
_FINDING_REF_RE = re.compile(r"^F-\d{3,}$")
|
|
3256
|
+
# Citation lists the human page renders as links or plain text.
|
|
3257
|
+
_CITATION_LIST_KEYS = ("evidenceRefs", "supportingEvidence", "falsifyingEvidenceChecked")
|
|
3258
|
+
|
|
3259
|
+
|
|
3260
|
+
def _prose_pointers(node: object, key: str = "", pointer: str = "") -> list[tuple[str, str]]:
|
|
3261
|
+
"""(JSON pointer, value) for every string under a reader-facing prose key."""
|
|
3262
|
+
from okstra_ctl.report_translation import PROSE_KEYS
|
|
3263
|
+
|
|
3264
|
+
out: list[tuple[str, str]] = []
|
|
3265
|
+
if isinstance(node, dict):
|
|
3266
|
+
for child_key, value in node.items():
|
|
3267
|
+
out.extend(_prose_pointers(value, str(child_key), f"{pointer}/{child_key}"))
|
|
3268
|
+
elif isinstance(node, list):
|
|
3269
|
+
for index, value in enumerate(node):
|
|
3270
|
+
out.extend(_prose_pointers(value, key, f"{pointer}/{index}"))
|
|
3271
|
+
elif isinstance(node, str) and key in PROSE_KEYS:
|
|
3272
|
+
out.append((pointer, node))
|
|
3273
|
+
return out
|
|
3274
|
+
|
|
3275
|
+
|
|
3276
|
+
def _validate_no_null_literals_in_prose(data: dict, failures: list[str]) -> None:
|
|
3277
|
+
"""A prose cell holding `None`/`null`/`undefined` is a serialisation
|
|
3278
|
+
artefact the page prints verbatim — nlpvibe planning-009…014 showed
|
|
3279
|
+
`None` under Test cost and Rollout cost. An empty value is spelled by
|
|
3280
|
+
omitting the field or leaving it empty, never by naming the language's
|
|
3281
|
+
null."""
|
|
3282
|
+
offenders = [
|
|
3283
|
+
f"{pointer} = {value.strip()!r}"
|
|
3284
|
+
for pointer, value in _prose_pointers(data)
|
|
3285
|
+
if value.strip() in _NULL_LITERALS
|
|
3286
|
+
]
|
|
3287
|
+
if offenders:
|
|
3288
|
+
failures.append(
|
|
3289
|
+
"final-report data.json: prose field(s) hold a null literal — "
|
|
3290
|
+
+ ", ".join(offenders[:8])
|
|
3291
|
+
+ (f" (+{len(offenders) - 8} more)" if len(offenders) > 8 else "")
|
|
3292
|
+
+ ". Leave the field empty or omit it; the page prints the literal as text."
|
|
3293
|
+
)
|
|
3294
|
+
|
|
3295
|
+
|
|
3296
|
+
def _unbridged_worker_finding_refs(data: dict) -> list[str]:
|
|
3297
|
+
"""Advisory: citation lists naming a worker's own finding number (`F-NNN`)
|
|
3298
|
+
that no `evidence.primary[].sourceItems` entry carries.
|
|
3299
|
+
|
|
3300
|
+
The page links a bare `F-NNN` only when exactly one promoted evidence row
|
|
3301
|
+
cites it as `<worker>:F-NNN` (`report_html/common.py`
|
|
3302
|
+
`worker_finding_links`); every other one is dead text. The 2026-09-05
|
|
3303
|
+
audit found 22 to 156 such citations per shipped report, so this stays a
|
|
3304
|
+
warning until the writer contracts have caught up — cite the promoted
|
|
3305
|
+
`E-` row, or add the finding to that row's `sourceItems`.
|
|
3306
|
+
"""
|
|
3307
|
+
bridged: set[str] = set()
|
|
3308
|
+
for row in ((data.get("evidence") or {}).get("primary") or []):
|
|
3309
|
+
if not isinstance(row, dict):
|
|
3310
|
+
continue
|
|
3311
|
+
for item in row.get("sourceItems") or []:
|
|
3312
|
+
for token in re.findall(r"\b(F-\d{3,})\b", str(item)):
|
|
3313
|
+
bridged.add(token)
|
|
3314
|
+
unbridged: dict[str, list[str]] = {}
|
|
3315
|
+
|
|
3316
|
+
def walk(node: object, key: str = "", pointer: str = "") -> None:
|
|
3317
|
+
if isinstance(node, dict):
|
|
3318
|
+
for child_key, value in node.items():
|
|
3319
|
+
walk(value, str(child_key), f"{pointer}/{child_key}")
|
|
3320
|
+
elif isinstance(node, list):
|
|
3321
|
+
if key in _CITATION_LIST_KEYS:
|
|
3322
|
+
for value in node:
|
|
3323
|
+
if isinstance(value, str) and _FINDING_REF_RE.match(value.strip()) and value.strip() not in bridged:
|
|
3324
|
+
unbridged.setdefault(value.strip(), []).append(pointer)
|
|
3325
|
+
else:
|
|
3326
|
+
for index, value in enumerate(node):
|
|
3327
|
+
walk(value, key, f"{pointer}/{index}")
|
|
3328
|
+
|
|
3329
|
+
walk(data)
|
|
3330
|
+
if not unbridged:
|
|
3331
|
+
return []
|
|
3332
|
+
listed = ", ".join(f"{ref} ({len(where)}×)" for ref, where in sorted(unbridged.items())[:10])
|
|
3333
|
+
return [
|
|
3334
|
+
f"citation lists name worker finding number(s) no evidence.primary "
|
|
3335
|
+
f"sourceItems entry carries — {listed}"
|
|
3336
|
+
+ (f" (+{len(unbridged) - 10} more)" if len(unbridged) > 10 else "")
|
|
3337
|
+
+ "; cite the promoted E- row or add `<worker>:F-NNN` to its sourceItems"
|
|
3338
|
+
]
|
|
3339
|
+
|
|
3340
|
+
|
|
3242
3341
|
def _validate_implementation_planning_cross_project(data: dict, failures: list[str]) -> None:
|
|
3243
3342
|
"""타 프로젝트 의존을 DM 행(`kind == 'cross-project'`)으로 선언했다면
|
|
3244
3343
|
`crossProjectDependencies` 에 `direction == 'upstream-precondition'` 행이
|
|
@@ -4791,11 +4890,22 @@ def _answered_clarification_ids(data: dict) -> list[str]:
|
|
|
4791
4890
|
|
|
4792
4891
|
|
|
4793
4892
|
def _validate_supersession_ledger(
|
|
4794
|
-
data: dict,
|
|
4893
|
+
data: dict,
|
|
4894
|
+
failures: list[str],
|
|
4895
|
+
*,
|
|
4896
|
+
carried: dict | None = None,
|
|
4897
|
+
new_plan: bool = False,
|
|
4795
4898
|
) -> None:
|
|
4796
4899
|
"""Incorporating an answer means retiring what it invalidates, not only
|
|
4797
4900
|
adding what it decides.
|
|
4798
4901
|
|
|
4902
|
+
`new_plan` marks a plan built from a selected direction. Prepare seeds
|
|
4903
|
+
that run's ledger with every answer the option-selection record carried
|
|
4904
|
+
(2026-09-05), and a first plan has no earlier statement those answers
|
|
4905
|
+
could retire — an entry per carried row would be `no-dependent-statement`
|
|
4906
|
+
by construction. Those ids are exempt; answers the plan itself raised and
|
|
4907
|
+
settled still need their entry.
|
|
4908
|
+
|
|
4799
4909
|
A re-run reconciles each `C-*` row's `Status` and writes the new decision
|
|
4800
4910
|
into the plan, but nothing required it to remove the sentences the answer
|
|
4801
4911
|
made false. The result is one plan carrying two opposite instructions for
|
|
@@ -4809,7 +4919,10 @@ def _validate_supersession_ledger(
|
|
|
4809
4919
|
if not isinstance(ip, dict):
|
|
4810
4920
|
return
|
|
4811
4921
|
answered = set(_answered_clarification_ids(data))
|
|
4812
|
-
|
|
4922
|
+
if new_plan:
|
|
4923
|
+
answered.difference_update((carried or {}).keys())
|
|
4924
|
+
else:
|
|
4925
|
+
answered.update((carried or {}).keys())
|
|
4813
4926
|
if not answered:
|
|
4814
4927
|
return
|
|
4815
4928
|
ledger = [e for e in (ip.get("supersessionLedger") or []) if isinstance(e, dict)]
|
|
@@ -5149,6 +5262,110 @@ def _validate_verifier_discrepancy_is_not_passed(
|
|
|
5149
5262
|
)
|
|
5150
5263
|
|
|
5151
5264
|
|
|
5265
|
+
_CHECKLIST_ID_RE = re.compile(r"\bVC-\d{3,}\b")
|
|
5266
|
+
_CHECKLIST_PHASE_RE = re.compile(r"\bphase\W{0,3}(pre|mid|post)\b", re.IGNORECASE)
|
|
5267
|
+
|
|
5268
|
+
|
|
5269
|
+
def _approved_plan_record(
|
|
5270
|
+
data: dict, report_path: Path, project_root: Path | None
|
|
5271
|
+
) -> dict | None:
|
|
5272
|
+
"""이 구현 리포트가 가리키는 승인 계획 레코드(data.json). 못 찾으면 None.
|
|
5273
|
+
|
|
5274
|
+
`approvedPlanReference.planFile` 은 실물에서 프로젝트 상대(`.okstra/tasks/...`)로,
|
|
5275
|
+
fixture 에서 태스크 상대(`runs/implementation-planning/...`)로 나오고 확장자는
|
|
5276
|
+
`.md` 와 `.data.json` 둘 다 쓰인다. 어느 형태든 레코드로 되짚고, 없으면 None 을
|
|
5277
|
+
돌려 호출자가 판정을 건너뛰게 한다 — 계획 부재는 다른 검사의 몫이고, 없는
|
|
5278
|
+
파일을 여기서 위반으로 세지 않는다.
|
|
5279
|
+
"""
|
|
5280
|
+
implementation = data.get("implementation")
|
|
5281
|
+
if not isinstance(implementation, dict):
|
|
5282
|
+
return None
|
|
5283
|
+
reference = implementation.get("approvedPlanReference")
|
|
5284
|
+
plan_file = reference.get("planFile") if isinstance(reference, dict) else None
|
|
5285
|
+
if not isinstance(plan_file, str) or not plan_file.strip():
|
|
5286
|
+
return None
|
|
5287
|
+
from okstra_ctl.final_report_paths import final_report_data_path
|
|
5288
|
+
|
|
5289
|
+
candidate = Path(plan_file.strip())
|
|
5290
|
+
if candidate.name.endswith(".md"):
|
|
5291
|
+
candidate = final_report_data_path(candidate)
|
|
5292
|
+
task_root = next(
|
|
5293
|
+
(parent.parent for parent in report_path.parents if parent.name == "runs"),
|
|
5294
|
+
None,
|
|
5295
|
+
)
|
|
5296
|
+
roots: list[Path | None] = (
|
|
5297
|
+
[None] if candidate.is_absolute()
|
|
5298
|
+
else [root for root in (project_root, task_root) if root is not None]
|
|
5299
|
+
)
|
|
5300
|
+
for root in roots:
|
|
5301
|
+
path = candidate if root is None else root / candidate
|
|
5302
|
+
if not path.is_file():
|
|
5303
|
+
continue
|
|
5304
|
+
try:
|
|
5305
|
+
loaded = json.loads(path.read_text(encoding="utf-8"))
|
|
5306
|
+
except (OSError, ValueError):
|
|
5307
|
+
return None
|
|
5308
|
+
return loaded if isinstance(loaded, dict) else None
|
|
5309
|
+
return None
|
|
5310
|
+
|
|
5311
|
+
|
|
5312
|
+
def _validate_verifier_discrepancy_names_checklist_phase(
|
|
5313
|
+
data: dict,
|
|
5314
|
+
report_path: Path,
|
|
5315
|
+
project_root: Path | None,
|
|
5316
|
+
failures: list[str],
|
|
5317
|
+
) -> None:
|
|
5318
|
+
"""계획 `validationChecklist` 행을 근거로 적은 divergence 는 그 행의 `phase` 를 인용한다.
|
|
5319
|
+
|
|
5320
|
+
2026-09-05 실측(fontsninja-v3-site dev-10626 stage-1): codex 검증자가 `VC-003` 의
|
|
5321
|
+
`git diff --name-only` 가 커밋 뒤 빈 출력이라며 FAIL 을 냈다. 그 행은 계획 레코드에
|
|
5322
|
+
`phase: mid` — 편집과 커밋 사이의 체크포인트 — 로 선언돼 있어, 커밋 뒤의 빈 출력은
|
|
5323
|
+
계획의 단계 순서 그 자체였다. 수렴에서 제기자 본인이 반대 읽기에 AGREE 했지만 FAIL
|
|
5324
|
+
행은 남아 stage 가 `failed` 로 갔고, 리드도 그 주장을 열어 보지 않고 라우팅에 옮겼다.
|
|
5325
|
+
`pre`/`mid`/`post` 는 행이 언제 성립하는지를 정하므로, 행을 인용하는 문장이 그 값을
|
|
5326
|
+
함께 적어야 한다 — 읽지 않은 행을 근거로 쓰는 문장은 그러면 쓸 수 없다.
|
|
5327
|
+
|
|
5328
|
+
행에 `phase` 가 없거나 계획 레코드를 못 찾으면 판정하지 않는다.
|
|
5329
|
+
"""
|
|
5330
|
+
plan = _approved_plan_record(data, report_path, project_root)
|
|
5331
|
+
if plan is None:
|
|
5332
|
+
return
|
|
5333
|
+
planning = plan.get("implementationPlanning")
|
|
5334
|
+
rows = planning.get("validationChecklist") if isinstance(planning, dict) else None
|
|
5335
|
+
phases = {
|
|
5336
|
+
str(row["id"]): str(row["phase"]).strip().lower()
|
|
5337
|
+
for row in (rows if isinstance(rows, list) else [])
|
|
5338
|
+
if isinstance(row, dict)
|
|
5339
|
+
and isinstance(row.get("id"), str)
|
|
5340
|
+
and isinstance(row.get("phase"), str)
|
|
5341
|
+
}
|
|
5342
|
+
if not phases:
|
|
5343
|
+
return
|
|
5344
|
+
for who, row in _verifier_rows(data):
|
|
5345
|
+
discrepancy = row.get("discrepancy")
|
|
5346
|
+
if not isinstance(discrepancy, str) or not discrepancy.strip():
|
|
5347
|
+
continue
|
|
5348
|
+
cited = sorted(set(_CHECKLIST_ID_RE.findall(discrepancy)) & set(phases))
|
|
5349
|
+
if not cited:
|
|
5350
|
+
continue
|
|
5351
|
+
named = {
|
|
5352
|
+
match.group(1).lower()
|
|
5353
|
+
for match in _CHECKLIST_PHASE_RE.finditer(discrepancy)
|
|
5354
|
+
}
|
|
5355
|
+
missing = [
|
|
5356
|
+
f"{row_id} (phase: {phases[row_id]})"
|
|
5357
|
+
for row_id in cited
|
|
5358
|
+
if phases[row_id] not in named
|
|
5359
|
+
]
|
|
5360
|
+
if missing:
|
|
5361
|
+
failures.append(
|
|
5362
|
+
f"verifier-discrepancy: {who} 가 계획 체크리스트 행을 근거로 divergence 를 "
|
|
5363
|
+
f"적었지만 그 행의 phase 를 인용하지 않았다 — {', '.join(missing)}. "
|
|
5364
|
+
"`pre`/`mid`/`post` 는 행이 언제 성립하는지를 정하므로 인용 문장에 "
|
|
5365
|
+
"`VC-NNN (phase: <값>)` 으로 적는다 (`_implementation-verifier.md` § Tier 1)."
|
|
5366
|
+
)
|
|
5367
|
+
|
|
5368
|
+
|
|
5152
5369
|
def _verifier_rows(data: dict):
|
|
5153
5370
|
"""(표시 이름, verifierResults 행) 쌍."""
|
|
5154
5371
|
implementation = data.get("implementation")
|