okstra 0.143.0 → 0.145.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. package/README.md +4 -1
  2. package/docs/architecture.md +18 -2
  3. package/docs/cli.md +39 -2
  4. package/docs/project-structure-overview.md +19 -6
  5. package/package.json +1 -1
  6. package/runtime/BUILD.json +2 -2
  7. package/runtime/prompts/coding-preflight/overview.md +1 -1
  8. package/runtime/prompts/lead/convergence.md +11 -3
  9. package/runtime/prompts/lead/okstra-lead-contract.md +7 -1
  10. package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -1
  11. package/runtime/prompts/profiles/_common-contract.md +1 -1
  12. package/runtime/prompts/profiles/_implementation-verifier.md +48 -2
  13. package/runtime/prompts/profiles/change-impact-analysis.md +24 -0
  14. package/runtime/prompts/profiles/feature-analysis.md +24 -0
  15. package/runtime/prompts/profiles/forbidden-actions.json +18 -0
  16. package/runtime/prompts/profiles/project-analysis.md +24 -0
  17. package/runtime/prompts/wizard/prompts.ko.json +44 -1
  18. package/runtime/python/okstra_ctl/analysis_inputs.py +369 -0
  19. package/runtime/python/okstra_ctl/clarification_items.py +74 -1
  20. package/runtime/python/okstra_ctl/mutation_probe.py +1263 -0
  21. package/runtime/python/okstra_ctl/render.py +77 -4
  22. package/runtime/python/okstra_ctl/render_final_report.py +13 -4
  23. package/runtime/python/okstra_ctl/report_views.py +134 -3
  24. package/runtime/python/okstra_ctl/run.py +118 -0
  25. package/runtime/python/okstra_ctl/run_context.py +34 -2
  26. package/runtime/python/okstra_ctl/schema_excerpt.py +12 -4
  27. package/runtime/python/okstra_ctl/self_mock_signals.py +183 -0
  28. package/runtime/python/okstra_ctl/user_response.py +309 -3
  29. package/runtime/python/okstra_ctl/wizard.py +545 -32
  30. package/runtime/python/okstra_ctl/worker_prompt_policy.py +3 -0
  31. package/runtime/python/okstra_ctl/workflow.py +22 -0
  32. package/runtime/schemas/final-report-v1.0.schema.json +849 -3
  33. package/runtime/skills/okstra-run/SKILL.md +13 -1
  34. package/runtime/templates/reports/change-impact-analysis-input.template.md +58 -0
  35. package/runtime/templates/reports/feature-analysis-input.template.md +59 -0
  36. package/runtime/templates/reports/final-report.template.md +220 -0
  37. package/runtime/templates/reports/i18n/en.json +8 -0
  38. package/runtime/templates/reports/i18n/ko.json +8 -0
  39. package/runtime/templates/reports/project-analysis-input.template.md +58 -0
  40. package/runtime/templates/reports/report.js +84 -5
  41. package/runtime/templates/reports/user-response.template.md +19 -1
  42. package/runtime/validators/detect_self_mock.py +220 -0
  43. package/runtime/validators/validate-report-views.py +61 -7
  44. package/runtime/validators/validate-run.py +518 -0
  45. package/runtime/validators/validate_analysis_report.py +864 -0
  46. package/src/commands/execute/render-bundle.mjs +3 -0
@@ -0,0 +1,220 @@
1
+ #!/usr/bin/env python3
2
+ """Self-mock detector. Runs both gates and writes the run's ``qa/`` sidecar.
3
+
4
+ Gate A is static: it scans changed TEST files for SUT-stub signals. Gate B is
5
+ mutation-based (``okstra_ctl.mutation_probe``) and runs over the ``--changed-file``
6
+ set — the stage's WHOLE diff, because each mutation adapter picks its own
7
+ production sources out of it. ``overall`` is FAIL when either gate fails;
8
+ ``unsupported(...)`` from gate B is not a failure, it means that gate could not
9
+ run here at all.
10
+
11
+ Signals come from the SSOT in ``okstra_ctl.self_mock_signals`` — never redefine a
12
+ pattern here. Each file is matched as ONE whole-file string rather than line by
13
+ line, because some signals span lines (java ``injectmocks-spy`` is two annotations
14
+ on separate lines); the reported line number is derived from the match offset.
15
+
16
+ Writes a ``qa/`` sidecar JSON and prints ``QA-RESULT: PASS|FAIL`` as its last
17
+ stdout line. The exit code follows ``overall``, not gate A alone: 0 = PASS,
18
+ 1 = FAIL from EITHER gate. A mutation failure with a clean static scan still
19
+ exits 1, because the verifier records this exit code as the command's outcome.
20
+
21
+ Alongside the hits the sidecar records every ``--test-file`` it received, split
22
+ into ``scannedFiles`` and ``skippedFiles``: hits alone cannot distinguish
23
+ "scanned the changed test files and found nothing" from "scanned an empty or
24
+ wrong input" — both are an empty hit list. Reporting the skipped ones too is
25
+ what keeps the gate from blocking a run whose changed test file the detector
26
+ legitimately cannot read.
27
+
28
+ ``--waivers`` is the false-positive escape hatch. A regex gate produces some
29
+ false accusations, and with no channel for them one wedges the stage forever —
30
+ the sidecar is detector-written and hand-editing it is a contract violation.
31
+ A waived hit moves out of ``staticDetect.hits`` into ``staticDetect.waived`` and
32
+ stops counting toward the verdict. The detector matches only; it never judges
33
+ whether the waiver is legitimate. ``validate-run.py::_validate_selfmock`` does
34
+ that, by requiring a ``reason`` and a user ``acknowledgedBy`` on every waived
35
+ entry — so an entry lacking either is carried through to that gate rather than
36
+ dropped here, and an agent cannot clear its own finding by writing a waiver.
37
+ The argument itself is recorded as ``staticDetect.waiverSource`` for the same
38
+ reason: the gate pins it to the task's own ``qa/self-mock-waivers.json``, so
39
+ redirecting ``--waivers`` at a self-authored file blocks instead of passing.
40
+ """
41
+ from __future__ import annotations
42
+
43
+ import argparse
44
+ import json
45
+ import sys
46
+ from datetime import datetime, timezone
47
+ from pathlib import Path
48
+
49
+ # scripts/ (repo) and python/ (installed under ~/.okstra/lib) are not packages;
50
+ # insert whichever exists so okstra_ctl is importable directly.
51
+ _VALIDATORS_DIR = Path(__file__).resolve().parent
52
+ for _ssot_dir in (_VALIDATORS_DIR.parent / "scripts", _VALIDATORS_DIR.parent / "python"):
53
+ if _ssot_dir.is_dir() and str(_ssot_dir) not in sys.path:
54
+ sys.path.insert(0, str(_ssot_dir))
55
+
56
+ from okstra_ctl.mutation_probe import probe_changed_files
57
+ from okstra_ctl.self_mock_signals import (
58
+ EXT_TO_LANG,
59
+ SIGNALS,
60
+ partition_waived_entries,
61
+ selfmock_path_key,
62
+ )
63
+
64
+
65
+ def overall_verdict(static_status: str, mutation_status: str) -> str:
66
+ """Fold both gates into the single verdict `validate-run.py` reads.
67
+
68
+ Either gate failing fails the stage. `unsupported(...)` is not a failure —
69
+ it means that gate could not run at all (no adapter for the language, no
70
+ tool installed), and blocking on it would wedge every repo that has no
71
+ mutation tooling. The reason stays in the sidecar for audit either way.
72
+ """
73
+ return "FAIL" if "FAIL" in (static_status, mutation_status) else "PASS"
74
+
75
+
76
+ def is_scannable(path: Path) -> bool:
77
+ """True when the scan can actually read this path.
78
+
79
+ A file whose extension has no signal set (Go, Ruby, a JSON fixture) and one
80
+ a stage deleted are skipped rather than treated as failures. Both are still
81
+ reported — as ``skippedFiles`` — because the gate needs to know the detector
82
+ RECEIVED them: its trigger for a changed test file is extension- and
83
+ existence-agnostic, so demanding those appear among the scanned ones would
84
+ block every such run.
85
+ """
86
+ return EXT_TO_LANG.get(path.suffix) is not None and path.is_file()
87
+
88
+
89
+ def scannable_files(paths: list[Path]) -> list[Path]:
90
+ """Return the subset a scan can actually read, in the caller's own order."""
91
+ return [p for p in paths if is_scannable(p)]
92
+
93
+
94
+ def scan_files(paths: list[Path]) -> list[dict]:
95
+ """Return one hit dict ``{file, line, signal}`` per signal match."""
96
+ hits: list[dict] = []
97
+ for p in scannable_files(paths):
98
+ lang = EXT_TO_LANG[p.suffix]
99
+ text = p.read_text(encoding="utf-8", errors="replace")
100
+ file_hits: list[dict] = []
101
+ for sig in SIGNALS.get(lang, []):
102
+ for m in sig.pattern.finditer(text):
103
+ line = text[: m.start()].count("\n") + 1
104
+ file_hits.append({"file": str(p), "line": line, "signal": sig.name})
105
+ hits.extend(sorted(file_hits, key=lambda h: (h["line"], h["signal"])))
106
+ return hits
107
+
108
+
109
+ def load_waivers(path: Path | None) -> list[dict]:
110
+ """Read the user-acknowledged false-positive entries; `[]` when there are none.
111
+
112
+ An absent file is the normal case — the flag is authored unconditionally in
113
+ `prompts/profiles/_implementation-verifier.md` while most stages never need a
114
+ waiver. An unreadable or malformed one raises instead: silently falling back
115
+ to "no waivers" would leave the operator re-reading an unchanged FAIL with no
116
+ hint that their acknowledgement never parsed.
117
+ """
118
+ if path is None or not path.is_file():
119
+ return []
120
+ try:
121
+ data = json.loads(path.read_text(encoding="utf-8"))
122
+ except (OSError, json.JSONDecodeError) as exc:
123
+ raise ValueError(f"waiver file unreadable at {path}: {exc}") from exc
124
+ if not isinstance(data, list) or any(not isinstance(e, dict) for e in data):
125
+ raise ValueError(
126
+ f"waiver file at {path} must be a JSON array of "
127
+ "{file, line, signal, reason, acknowledgedBy} objects"
128
+ )
129
+ return data
130
+
131
+
132
+ def partition_waived(
133
+ hits: list[dict], waivers: list[dict]
134
+ ) -> tuple[list[dict], list[dict]]:
135
+ """Gate A's half of the shared waiver matching: keyed on `signal`."""
136
+ return partition_waived_entries(hits, waivers, "signal")
137
+
138
+
139
+ def main(argv=None) -> int:
140
+ ap = argparse.ArgumentParser(
141
+ description="Detect self-mocked tests (SUT stubbed by its own test)."
142
+ )
143
+ ap.add_argument("--test-file", action="append", default=[], dest="test_files")
144
+ # Gate B needs the WHOLE changed set, not the test files gate A scans: each
145
+ # mutation adapter selects its own production sources out of it. Passing only
146
+ # the test files leaves every adapter with nothing to mutate, which reports a
147
+ # vacuous PASS while gate B is silently dead.
148
+ ap.add_argument("--changed-file", action="append", default=[], dest="changed_files")
149
+ ap.add_argument("--sidecar", required=True)
150
+ ap.add_argument("--stage-name", default=None)
151
+ ap.add_argument("--waivers", default=None)
152
+ ap.add_argument("--diff", default=None)
153
+ ap.add_argument("--worktree", default=None)
154
+ args = ap.parse_args(argv)
155
+
156
+ try:
157
+ waivers = load_waivers(Path(args.waivers) if args.waivers else None)
158
+ except ValueError as exc:
159
+ ap.error(str(exc))
160
+
161
+ # Paths are recorded as the caller spelled them (never resolved): the gate in
162
+ # validate-run.py compares these against the report's repo-relative §5.7.3
163
+ # rows, and both come from `git diff --name-only` in the worktree cwd.
164
+ received = [Path(f) for f in args.test_files]
165
+ # Gate B's own record of its input, the counterpart to gate A's
166
+ # scannedFiles/skippedFiles: without it a run that forgot `--changed-file`
167
+ # produces a sidecar indistinguishable from one where gate B had nothing to
168
+ # flag. Spelled exactly as passed, like the gate A lists.
169
+ changed = [Path(f) for f in args.changed_files]
170
+ scanned = [p for p in received if is_scannable(p)]
171
+ skipped = [p for p in received if not is_scannable(p)]
172
+ hits, waived = partition_waived(scan_files(scanned), waivers)
173
+ static_status = "FAIL" if hits else "PASS"
174
+ # The SAME waiver list feeds both gates: one user-managed file, and a single
175
+ # `waiverSource` for the gate to pin. Gate A reads its `signal` entries, gate
176
+ # B its `mutant` ones.
177
+ mutation = probe_changed_files(
178
+ changed,
179
+ Path(args.diff) if args.diff else None,
180
+ Path(args.worktree) if args.worktree else None,
181
+ waivers,
182
+ )
183
+ # Recorded for the same reason `staticDetect.waiverSource` is: the gate pins
184
+ # it to the task's own file, so redirecting `--waivers` at something the run
185
+ # wrote itself blocks instead of passing.
186
+ mutation = {**mutation, "waiverSource": args.waivers}
187
+ status = overall_verdict(static_status, str(mutation["status"]))
188
+ sidecar = {
189
+ "stageName": args.stage_name,
190
+ "overall": status,
191
+ "ranAt": datetime.now(timezone.utc).isoformat(),
192
+ "scannedFiles": [str(p) for p in scanned],
193
+ "skippedFiles": [str(p) for p in skipped],
194
+ "changedFiles": [str(p) for p in changed],
195
+ # `waiverSource` is the `--waivers` argument exactly as passed. The gate
196
+ # pins it to the task's own `qa/self-mock-waivers.json`, which is what
197
+ # stops a run from reading its acknowledgement out of a file it wrote
198
+ # itself somewhere the task bundle never records.
199
+ "staticDetect": {
200
+ "status": static_status,
201
+ "hits": hits,
202
+ "waived": waived,
203
+ "waiverSource": args.waivers,
204
+ },
205
+ "mutation": mutation,
206
+ }
207
+ sidecar_path = Path(args.sidecar)
208
+ sidecar_path.parent.mkdir(parents=True, exist_ok=True)
209
+ sidecar_path.write_text(json.dumps(sidecar, indent=2), encoding="utf-8")
210
+
211
+ for h in hits:
212
+ print(f"SELF-MOCK {h['file']}:{h['line']} {h['signal']}")
213
+ for s in mutation["survived"]:
214
+ print(f"MUTANT-SURVIVED {s['file']}:{s['line']} {s['mutant']} ({s['status']})")
215
+ print(f"QA-RESULT: {status}")
216
+ return 1 if status == "FAIL" else 0
217
+
218
+
219
+ if __name__ == "__main__":
220
+ raise SystemExit(main())
@@ -3,7 +3,8 @@
3
3
  (``scripts/okstra-render-report-views.py``).
4
4
 
5
5
  Checks, for a given final-report MD path:
6
- 1. ``*.html`` sibling exists.
6
+ 1. ``*.html`` sibling exists when the report has clarification rows,
7
+ an analysis-review contract, or a plan-approval contract.
7
8
  2. HTML's ``source-sha256`` in run-meta matches the current MD body —
8
9
  stale html detection.
9
10
  3. HTML's §5.6 / §5.7 / §5.8 deliverable regions contain no
@@ -13,6 +14,8 @@ Checks, for a given final-report MD path:
13
14
  ``<img src=>`` — self-contained guarantee.
14
15
  5. Every Response ID in HTML matches the §1 Clarification Items table
15
16
  of the source MD (1:1).
17
+ 6. Analysis reports contain all three Analysis Review decisions and an
18
+ affected-ID selector exactly matching the structured IDs in data.json.
16
19
 
17
20
  Exit codes: 0 on success, 1 on any failure. Failures are printed one
18
21
  per line to stderr.
@@ -34,6 +37,7 @@ from okstra_ctl.clarification_items import ( # noqa: E402
34
37
  section_1_present_but_unparsed,
35
38
  )
36
39
  from okstra_ctl.report_views import ( # noqa: E402
40
+ analysis_review_context,
37
41
  extract_html_digest,
38
42
  plan_approval_context,
39
43
  source_digest,
@@ -47,6 +51,18 @@ _EXTERNAL_URL_RE = re.compile(
47
51
  )
48
52
 
49
53
  _RESPONSE_ID_ATTR_RE = re.compile(r'data-response-id="(C-\d+)"')
54
+ _ANALYSIS_REVIEW_SECTION_RE = re.compile(
55
+ r'<section id="analysis-review">(?P<body>.*?)</section>', re.DOTALL
56
+ )
57
+ _ANALYSIS_REVIEW_STATUS_RE = re.compile(
58
+ r'name="analysis-review-status"\s+value="([^"]+)"'
59
+ )
60
+ _ANALYSIS_REVIEW_SELECTOR_RE = re.compile(
61
+ r'<select id="analysis-review-affected-ids"[^>]*>(?P<body>.*?)</select>',
62
+ re.DOTALL,
63
+ )
64
+ _OPTION_VALUE_RE = re.compile(r'<option value="([^"]+)">')
65
+ _ANALYSIS_REVIEW_STATUSES = ["accepted", "rejected", "revision-requested"]
50
66
 
51
67
 
52
68
  def _main_body(html_text: str) -> str:
@@ -78,6 +94,34 @@ def _no_form_sections(html_body: str) -> list[str]:
78
94
  return chunks
79
95
 
80
96
 
97
+ def _validate_analysis_review_html(
98
+ html_text: str, expected_ids: tuple[str, ...]
99
+ ) -> list[str]:
100
+ section = _ANALYSIS_REVIEW_SECTION_RE.search(html_text)
101
+ if section is None:
102
+ return ["analysis report html is missing the Analysis Review control"]
103
+ body = section.group("body")
104
+ failures: list[str] = []
105
+ statuses = sorted(_ANALYSIS_REVIEW_STATUS_RE.findall(body))
106
+ if statuses != _ANALYSIS_REVIEW_STATUSES:
107
+ failures.append(
108
+ "Analysis Review controls must be exactly accepted, "
109
+ "revision-requested, and rejected"
110
+ )
111
+ selector = _ANALYSIS_REVIEW_SELECTOR_RE.search(body)
112
+ actual_ids = (
113
+ sorted(_OPTION_VALUE_RE.findall(selector.group("body")))
114
+ if selector is not None
115
+ else []
116
+ )
117
+ if actual_ids != sorted(expected_ids):
118
+ failures.append(
119
+ f"Analysis Review selector mismatch: data.json has "
120
+ f"{sorted(expected_ids)}, HTML has {actual_ids}"
121
+ )
122
+ return failures
123
+
124
+
81
125
  def validate(report_path: Path) -> list[str]:
82
126
  failures: list[str] = []
83
127
  if not report_path.is_file():
@@ -94,14 +138,16 @@ def validate(report_path: Path) -> list[str]:
94
138
  "form parity. Re-render the report so §1 matches the schema."
95
139
  ]
96
140
  md_ids = _md_response_ids(md)
141
+ review_ctx = analysis_review_context(report_path)
97
142
 
98
- # (1) sibling artifact exists — conditional on §1 clarification rows.
99
- # The html view's sole value over the MD is its embedded form widgets
100
- # for §1 C-* rows, so a clarification-free report intentionally has no
101
- # html sibling (see report_views.report_has_clarification_items). When
102
- # there are no C-* rows and no html, that is the expected skip state.
143
+ # (1) sibling artifact exists when clarification rows or a structured
144
+ # analysis-review / plan-approval contract require interactive controls.
103
145
  if not html_path.is_file():
104
- if not md_ids and plan_approval_context(report_path, md) is None:
146
+ if (
147
+ not md_ids
148
+ and plan_approval_context(report_path, md) is None
149
+ and review_ctx is None
150
+ ):
105
151
  return []
106
152
  return [f"missing html artifact: {html_path}"]
107
153
 
@@ -146,6 +192,14 @@ def validate(report_path: Path) -> list[str]:
146
192
  f"Response ID mismatch: MD §1 has {md_ids}, HTML has {html_ids}"
147
193
  )
148
194
 
195
+ review_section_present = _ANALYSIS_REVIEW_SECTION_RE.search(html_text) is not None
196
+ if review_ctx is not None:
197
+ failures.extend(
198
+ _validate_analysis_review_html(html_text, review_ctx.selector_ids)
199
+ )
200
+ elif review_section_present:
201
+ failures.append("non-analysis report html must not contain Analysis Review controls")
202
+
149
203
  return failures
150
204
 
151
205