okstra 0.159.0 → 0.161.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/docs/architecture/storage-model.md +2 -0
- package/docs/architecture.md +2 -1
- package/docs/cli.md +8 -3
- package/docs/for-ai/README.md +2 -2
- package/docs/for-ai/skills/okstra-inspect.md +3 -0
- package/docs/for-ai/skills/okstra-run.md +2 -1
- package/docs/for-ai/skills/okstra-user-response.md +5 -5
- package/docs/project-structure-overview.md +5 -1
- package/docs/task-process/implementation.md +28 -0
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/bin/okstra-claude-exec.sh +4 -1
- package/runtime/prompts/host-orchestration/README.md +18 -0
- package/runtime/prompts/host-orchestration/implementation.md +57 -0
- package/runtime/prompts/launch.template.md +10 -1
- package/runtime/prompts/lead/adapters/claude-code.md +1 -1
- package/runtime/prompts/lead/adapters/cmux.md +67 -0
- package/runtime/prompts/lead/context-loader.md +5 -2
- package/runtime/prompts/lead/convergence.md +3 -1
- package/runtime/prompts/lead/plan-body-verification.md +21 -2
- package/runtime/prompts/lead/team-contract.md +2 -1
- package/runtime/prompts/profiles/_clarification-recommendation.md +11 -1
- package/runtime/prompts/profiles/_common-contract.md +3 -1
- package/runtime/prompts/profiles/implementation-planning.md +2 -0
- package/runtime/prompts/profiles/requirements-discovery.md +1 -1
- package/runtime/prompts/wizard/prompts.ko.json +3 -0
- package/runtime/python/okstra_ctl/clarification_items.py +9 -0
- package/runtime/python/okstra_ctl/cmux.py +531 -0
- package/runtime/python/okstra_ctl/codex_dispatch.py +6 -6
- package/runtime/python/okstra_ctl/convergence.py +168 -11
- package/runtime/python/okstra_ctl/dispatch_core.py +76 -7
- package/runtime/python/okstra_ctl/dispatch_state.py +16 -0
- package/runtime/python/okstra_ctl/error_issue.py +640 -0
- package/runtime/python/okstra_ctl/error_report.py +56 -0
- package/runtime/python/okstra_ctl/error_zip.py +23 -10
- package/runtime/python/okstra_ctl/incremental_scope.py +159 -19
- package/runtime/python/okstra_ctl/initial_prompt_materialization.py +18 -5
- package/runtime/python/okstra_ctl/issue_signals.py +186 -0
- package/runtime/python/okstra_ctl/lead_runtime.py +30 -2
- package/runtime/python/okstra_ctl/paths.py +38 -0
- package/runtime/python/okstra_ctl/plan_items_cli.py +167 -3
- package/runtime/python/okstra_ctl/profile_show.py +134 -0
- package/runtime/python/okstra_ctl/recap.py +63 -0
- package/runtime/python/okstra_ctl/render.py +7 -2
- package/runtime/python/okstra_ctl/render_final_report.py +7 -22
- package/runtime/python/okstra_ctl/report_translation.py +4 -0
- package/runtime/python/okstra_ctl/report_views.py +7 -3
- package/runtime/python/okstra_ctl/run.py +54 -3
- package/runtime/python/okstra_ctl/run_audit.py +477 -0
- package/runtime/python/okstra_ctl/team.py +50 -11
- package/runtime/python/okstra_ctl/user_response.py +25 -10
- package/runtime/python/okstra_ctl/verdict_blocks.py +183 -0
- package/runtime/python/okstra_ctl/wizard.py +64 -10
- package/runtime/python/okstra_ctl/worker_audit_check.py +44 -0
- package/runtime/python/okstra_ctl/worker_audit_ledger.py +207 -0
- package/runtime/python/okstra_ctl/worker_heartbeat.py +9 -3
- package/runtime/python/okstra_ctl/worker_liveness.py +81 -9
- package/runtime/schemas/final-report-v1.0.schema.json +14 -0
- package/runtime/schemas/final-report-v2.0.schema.json +51 -1
- package/runtime/skills/okstra-inspect/SKILL.md +3 -1
- package/runtime/skills/okstra-inspect/facets/error-issue.md +77 -0
- package/runtime/skills/okstra-inspect/facets/run-audit.md +34 -0
- package/runtime/skills/okstra-run/SKILL.md +28 -10
- package/runtime/skills/okstra-user-response/SKILL.md +18 -18
- package/runtime/templates/reports/final-report.template.md +4 -0
- package/runtime/templates/reports/html/i18n/en.json +5 -1
- package/runtime/templates/reports/html/i18n/ko.json +5 -1
- package/runtime/templates/reports/html/macros/forms.html +15 -0
- package/runtime/templates/reports/html/tasks/implementation-planning.template.html +1 -0
- package/runtime/templates/reports/i18n/en.json +2 -0
- package/runtime/validators/validate-run.py +267 -208
- package/runtime/validators/validate-workflow.sh +6 -0
- package/runtime/validators/validate_session_conformance.py +135 -31
- package/src/cli-registry.mjs +34 -0
- package/src/commands/execute/incremental-scope.mjs +10 -0
- package/src/commands/execute/worker-audit-check.mjs +35 -0
- package/src/commands/inspect/error-issue.mjs +27 -0
- package/src/commands/inspect/profile-show.mjs +29 -0
- package/src/commands/inspect/run-audit.mjs +26 -0
|
@@ -1,4 +1,9 @@
|
|
|
1
|
-
"""Neutral okstra team CLI for
|
|
1
|
+
"""Neutral okstra team CLI for pane-backed worker dispatch.
|
|
2
|
+
|
|
3
|
+
Outside cmux this is the external lead's door onto tmux panes. Under cmux it
|
|
4
|
+
is every lead's door onto cmux surfaces, because okstra owns the panes there
|
|
5
|
+
rather than the host. Which backend a run uses is read from its run manifest.
|
|
6
|
+
"""
|
|
2
7
|
from __future__ import annotations
|
|
3
8
|
|
|
4
9
|
import argparse
|
|
@@ -7,8 +12,10 @@ import sys
|
|
|
7
12
|
from pathlib import Path
|
|
8
13
|
from typing import Any, Mapping, Sequence
|
|
9
14
|
|
|
15
|
+
from . import cmux
|
|
10
16
|
from . import tmux
|
|
11
17
|
from .dispatch_core import (
|
|
18
|
+
BACKEND_CMUX_PANE,
|
|
12
19
|
BACKEND_TMUX_PANE,
|
|
13
20
|
DispatchError,
|
|
14
21
|
DispatchPlan,
|
|
@@ -51,7 +58,7 @@ def _parser() -> argparse.ArgumentParser:
|
|
|
51
58
|
|
|
52
59
|
|
|
53
60
|
def _add_dispatch_parser(sub) -> None:
|
|
54
|
-
parser = sub.add_parser("dispatch", help="dispatch
|
|
61
|
+
parser = sub.add_parser("dispatch", help="dispatch pane-backed workers")
|
|
55
62
|
_add_run_args(parser)
|
|
56
63
|
parser.add_argument("--workers", default="")
|
|
57
64
|
parser.add_argument("--jobs-file", default="")
|
|
@@ -61,7 +68,7 @@ def _add_dispatch_parser(sub) -> None:
|
|
|
61
68
|
|
|
62
69
|
|
|
63
70
|
def _add_await_parser(sub) -> None:
|
|
64
|
-
parser = sub.add_parser("await", help="wait for
|
|
71
|
+
parser = sub.add_parser("await", help="wait for pane-backed workers")
|
|
65
72
|
_add_run_args(parser)
|
|
66
73
|
parser.add_argument("--poll-interval-seconds", type=int, default=5)
|
|
67
74
|
parser.add_argument("--timeout-seconds", type=int, default=None)
|
|
@@ -70,7 +77,7 @@ def _add_await_parser(sub) -> None:
|
|
|
70
77
|
|
|
71
78
|
|
|
72
79
|
def _add_teardown_parser(sub) -> None:
|
|
73
|
-
parser = sub.add_parser("teardown", help="
|
|
80
|
+
parser = sub.add_parser("teardown", help="reclaim this run's worker panes")
|
|
74
81
|
_add_run_args(parser)
|
|
75
82
|
parser.add_argument("--dry-run", action="store_true")
|
|
76
83
|
parser.add_argument("--json", action="store_true")
|
|
@@ -94,8 +101,8 @@ def _dispatch(args) -> int:
|
|
|
94
101
|
okstra_bin=Path(args.okstra_bin),
|
|
95
102
|
requested_workers=requested,
|
|
96
103
|
idle_timeout_seconds=args.idle_timeout_seconds,
|
|
97
|
-
required_lead_runtime="external",
|
|
98
|
-
default_backend=
|
|
104
|
+
required_lead_runtime=None if _is_cmux_run(manifest) else "external",
|
|
105
|
+
default_backend=_manifest_backend(manifest),
|
|
99
106
|
supported_worker_wrappers=_SUPPORTED_WRAPPERS,
|
|
100
107
|
unsupported_worker_label="external lead",
|
|
101
108
|
dispatch_kind=args.dispatch_kind,
|
|
@@ -133,13 +140,13 @@ def _teardown(args) -> int:
|
|
|
133
140
|
team_state_path = _resolve_project_path(project_root, _require_string(manifest, "teamStatePath"))
|
|
134
141
|
team_state = _load_json(team_state_path, "team-state")
|
|
135
142
|
run_dir = _resolve_project_path(project_root, _require_string(manifest, "runDirectoryPath"))
|
|
136
|
-
|
|
137
|
-
panes = _teardown_panes(team_state, run_dir, lead_pane)
|
|
143
|
+
panes = _reclaimable_panes(manifest, team_state, run_dir)
|
|
138
144
|
if args.dry_run:
|
|
139
145
|
_emit_teardown(args.json, panes)
|
|
140
146
|
return 0
|
|
147
|
+
reclaim = cmux.close_surface if _is_cmux_run(manifest) else tmux.kill_pane
|
|
141
148
|
for pane in panes:
|
|
142
|
-
|
|
149
|
+
reclaim(pane["paneId"])
|
|
143
150
|
_mark_teardown_errors(team_state_path)
|
|
144
151
|
_emit_teardown(args.json, panes)
|
|
145
152
|
return 0
|
|
@@ -158,7 +165,7 @@ def _plan_for_existing(
|
|
|
158
165
|
lead_events_path=_resolve_project_path(root, _require_string(manifest, "leadEventsPath")),
|
|
159
166
|
manifest=manifest,
|
|
160
167
|
jobs=(),
|
|
161
|
-
default_backend=
|
|
168
|
+
default_backend=_manifest_backend(manifest),
|
|
162
169
|
)
|
|
163
170
|
|
|
164
171
|
|
|
@@ -174,12 +181,25 @@ def _await_payload(plan: DispatchPlan, completed: bool) -> dict[str, Any]:
|
|
|
174
181
|
}
|
|
175
182
|
|
|
176
183
|
|
|
177
|
-
def
|
|
184
|
+
def _reclaimable_panes(
|
|
185
|
+
manifest: Mapping[str, Any], team_state: Mapping[str, Any], run_dir: Path
|
|
186
|
+
) -> list[dict[str, str]]:
|
|
187
|
+
"""Everything this run owns and may close.
|
|
188
|
+
|
|
189
|
+
Under cmux the recorded ids are the whole list. There is no per-pane tag API
|
|
190
|
+
to sweep with, and scanning by title would be worse than nothing: cmux labels
|
|
191
|
+
its own agent surfaces with the same glyph okstra's tmux cleanup treats as a
|
|
192
|
+
teammate marker, so a sweep could close the lead. Only surfaces okstra
|
|
193
|
+
created are recorded, so only those can be closed.
|
|
194
|
+
"""
|
|
178
195
|
seen: set[str] = set()
|
|
179
196
|
panes: list[dict[str, str]] = []
|
|
180
197
|
for record in team_state.get("workerDispatches", []):
|
|
181
198
|
if isinstance(record, dict):
|
|
182
199
|
_append_pane(panes, seen, str(record.get("paneId", "")), "worker")
|
|
200
|
+
if _is_cmux_run(manifest):
|
|
201
|
+
return panes
|
|
202
|
+
lead_pane = tmux.resolve_caller_pane()
|
|
183
203
|
for pane in tmux.list_run_panes(run_dir, lead_pane=lead_pane):
|
|
184
204
|
_append_pane(panes, seen, pane.pane_id, pane.kind)
|
|
185
205
|
return panes
|
|
@@ -208,10 +228,29 @@ def _emit_teardown(as_json: bool, panes: list[dict[str, str]]) -> None:
|
|
|
208
228
|
print(f"{pane['paneId']}\t{pane['kind']}")
|
|
209
229
|
|
|
210
230
|
|
|
231
|
+
def _manifest_backend(manifest: Mapping[str, Any]) -> str:
|
|
232
|
+
"""The backend prepare recorded for this run.
|
|
233
|
+
|
|
234
|
+
Deliberately not a flag on this command: the manifest already answers it,
|
|
235
|
+
and a flag would be a second answer free to disagree. A manifest written
|
|
236
|
+
before the field existed reads as tmux, which is what those runs used.
|
|
237
|
+
"""
|
|
238
|
+
return str(manifest.get("terminalBackend") or "") or BACKEND_TMUX_PANE
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _is_cmux_run(manifest: Mapping[str, Any]) -> bool:
|
|
242
|
+
return _manifest_backend(manifest) == BACKEND_CMUX_PANE
|
|
243
|
+
|
|
244
|
+
|
|
211
245
|
def _validate_external_manifest(manifest: Mapping[str, Any]) -> None:
|
|
212
246
|
runtime = manifest.get("leadRuntime")
|
|
213
247
|
if runtime == "external":
|
|
214
248
|
return
|
|
249
|
+
if _is_cmux_run(manifest):
|
|
250
|
+
# Under cmux okstra owns the panes for every lead, so this command is no
|
|
251
|
+
# longer the external lead's private door and the advice below no longer
|
|
252
|
+
# applies — there is nowhere else for a codex or Claude Code lead to go.
|
|
253
|
+
return
|
|
215
254
|
if runtime == "codex":
|
|
216
255
|
raise DispatchError("use okstra codex-dispatch for leadRuntime=codex")
|
|
217
256
|
if runtime == "claude-code":
|
|
@@ -20,6 +20,7 @@ from typing import Optional
|
|
|
20
20
|
|
|
21
21
|
from okstra_ctl.report_views import (
|
|
22
22
|
serialize_user_response, UserResponseEntry, UserResponseApproval, infer_run_meta,
|
|
23
|
+
parse_expected_form_options,
|
|
23
24
|
)
|
|
24
25
|
from okstra_ctl.report_view_artifacts import user_responses_dir_for_report
|
|
25
26
|
from okstra_ctl.listing import list_runs, absolute_final_report_path
|
|
@@ -473,18 +474,31 @@ def list_awaiting_tasks(home: Path, project_id: str, limit: int) -> list[dict]:
|
|
|
473
474
|
# match. `§x.y` and `path.ext:line` are the other two ref shapes.
|
|
474
475
|
_SECTION_REF_RE = re.compile(r"§[\d.]+|[A-Z]{1,4}-\d+|[\w./-]+\.\w+:\d+")
|
|
475
476
|
_ID_TOKEN_RE = re.compile(r"^[A-Z]{1,4}-\d+$")
|
|
476
|
-
_ALTERNATIVES_CUE = "Alternatives:"
|
|
477
477
|
_DEFINITION_SNIPPET_CAP = 200
|
|
478
478
|
_SNIPPET_NOISE_RE = re.compile(r'<a id="[^"]*"></a>|`|\*\*')
|
|
479
|
+
_OPTION_LETTER_LABEL_RE = re.compile(r"^\([a-z]\)\s*")
|
|
479
480
|
|
|
480
481
|
|
|
481
|
-
def
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
482
|
+
def _options_from_expected_form(expected_form: str) -> list[dict]:
|
|
483
|
+
"""Rebuild a schema-v1 row's options from its ``Expected form`` cell.
|
|
484
|
+
|
|
485
|
+
v1 keeps the choices as one string and has nowhere to record their impact,
|
|
486
|
+
so those fields come back empty and the picker reports them as unstated
|
|
487
|
+
rather than inventing them. Splitting goes through the canonical
|
|
488
|
+
``parse_expected_form_options`` — the parser the HTML view already uses and
|
|
489
|
+
the only one under test.
|
|
490
|
+
"""
|
|
491
|
+
return [
|
|
492
|
+
{
|
|
493
|
+
"role": "recommended" if value == "recommended" else "alternative",
|
|
494
|
+
"answer": _OPTION_LETTER_LABEL_RE.sub("", label).strip(),
|
|
495
|
+
"rationale": "",
|
|
496
|
+
"scopeImpact": [],
|
|
497
|
+
"addedWork": "",
|
|
498
|
+
"directionChange": "",
|
|
499
|
+
}
|
|
500
|
+
for value, label in parse_expected_form_options(expected_form)
|
|
501
|
+
]
|
|
488
502
|
|
|
489
503
|
|
|
490
504
|
def _clean_snippet(line: str) -> str:
|
|
@@ -558,8 +572,9 @@ def show_open_rows(report_path: Path) -> dict:
|
|
|
558
572
|
refs = sorted(set(_SECTION_REF_RE.findall(statement + " " + expected)))
|
|
559
573
|
rows.append({"id": it.row_id, "kind": it.kind, "blocks": it.blocks,
|
|
560
574
|
"status": it.status, "statement": statement,
|
|
561
|
-
"
|
|
562
|
-
|
|
575
|
+
"expectedForm": expected,
|
|
576
|
+
# v2 authors the choices; v1 only ever had the string.
|
|
577
|
+
"options": r["options"] or _options_from_expected_form(expected),
|
|
563
578
|
"contextRefs": refs,
|
|
564
579
|
"resolvedRefs": resolve_refs(text, refs)})
|
|
565
580
|
return {"reportPath": str(report_path), "rows": rows}
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
"""Parser for the worker verdict block that plan-body and convergence share.
|
|
2
|
+
|
|
3
|
+
The response shape is fixed by contract (`prompts/lead/plan-body-verification.md`
|
|
4
|
+
§"Response format"), so the parser belongs here rather than in each lead. A
|
|
5
|
+
per-round ad-hoc regex makes the round's fidelity depend on whoever wrote it,
|
|
6
|
+
and its failure mode is silence: dev-10400 lost 19 of 37 assigned items to a
|
|
7
|
+
no-match that nothing reported. Every shape this module cannot read is an error.
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
|
|
14
|
+
VERDICT_TOKENS = frozenset({
|
|
15
|
+
"AGREE", "DISAGREE", "SUPPLEMENT", "UNVERIFIABLE", "VERIFICATION-ERROR",
|
|
16
|
+
})
|
|
17
|
+
FIXABILITY_VALUES = frozenset({"planner-fixable", "needs-user-input"})
|
|
18
|
+
|
|
19
|
+
# The convergence reverify prompts use their own vocabularies. Collaborative and
|
|
20
|
+
# full-reanalysis rounds speak the schema's own words; the adversarial round asks
|
|
21
|
+
# the verifier to break the finding, so it answers in break/survive terms that
|
|
22
|
+
# have to be translated back (`prompts/lead/convergence.md`
|
|
23
|
+
# §"Adversarial Re-verification Prompt").
|
|
24
|
+
COLLABORATIVE_VERDICTS = {
|
|
25
|
+
"AGREE": "agree",
|
|
26
|
+
"DISAGREE": "disagree",
|
|
27
|
+
"SUPPLEMENT": "supplement",
|
|
28
|
+
"UNVERIFIABLE": "unverifiable",
|
|
29
|
+
"VERIFICATION-ERROR": "verification-error",
|
|
30
|
+
}
|
|
31
|
+
ADVERSARIAL_VERDICTS = {
|
|
32
|
+
"SURVIVES": "agree",
|
|
33
|
+
"SURVIVES-WITH-CAVEAT": "supplement",
|
|
34
|
+
"REFUTED": "disagree",
|
|
35
|
+
# A verifier that looked and could not check is not a verifier that failed.
|
|
36
|
+
# `verification-error` drops the vote from the participating count, which
|
|
37
|
+
# shrinks the roster without saying so.
|
|
38
|
+
"UNVERIFIABLE": "unverifiable",
|
|
39
|
+
"VERIFICATION-ERROR": "verification-error",
|
|
40
|
+
}
|
|
41
|
+
DISAGREE_BASES = frozenset({"counter-evidence", "burden-not-met"})
|
|
42
|
+
|
|
43
|
+
_ITEM_RE = re.compile(r"^###[ \t]+(?P<id>[^\s:]+)[ \t]*:?.*$", re.MULTILINE)
|
|
44
|
+
# The contract writes some labels with a parenthetical qualifier —
|
|
45
|
+
# `**Fixability** (only when DISAGREE):` — so the colon may trail a `(...)`.
|
|
46
|
+
_FIELD_RE = re.compile(
|
|
47
|
+
r"^\*\*(?P<key>Verdict|Fixability|Note|Explanation|Prior dissent|Basis"
|
|
48
|
+
r"|Your evidence)\*\*"
|
|
49
|
+
r"(?:[ \t]*\([^)]*\))?[ \t]*:[ \t]*(?P<value>.*)$",
|
|
50
|
+
re.MULTILINE,
|
|
51
|
+
)
|
|
52
|
+
_VERDICT_RE = re.compile(r"^(?P<token>[A-Z-]+)(?:\((?P<kind>[a-f])\))?$")
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
class VerdictBlockError(ValueError):
|
|
56
|
+
"""Raised when a worker response does not match the contract shape."""
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass(frozen=True)
|
|
60
|
+
class FindingVote:
|
|
61
|
+
"""One worker's vote on one convergence finding, in schema vocabulary."""
|
|
62
|
+
|
|
63
|
+
finding_id: str
|
|
64
|
+
verdict: str
|
|
65
|
+
disagree_basis: str | None
|
|
66
|
+
explanation: str
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _scan_blocks(text: str) -> dict[str, dict[str, str]]:
|
|
70
|
+
"""Every `### <id>` block in *text*, as id → field map, in document order."""
|
|
71
|
+
matches = list(_ITEM_RE.finditer(text))
|
|
72
|
+
blocks: dict[str, dict[str, str]] = {}
|
|
73
|
+
for index, match in enumerate(matches):
|
|
74
|
+
end = matches[index + 1].start() if index + 1 < len(matches) else len(text)
|
|
75
|
+
item_id = match.group("id")
|
|
76
|
+
if item_id in blocks:
|
|
77
|
+
raise VerdictBlockError(f"item `{item_id}` appears twice in one response")
|
|
78
|
+
body = text[match.end():end]
|
|
79
|
+
blocks[item_id] = {
|
|
80
|
+
m.group("key"): m.group("value").strip() for m in _FIELD_RE.finditer(body)
|
|
81
|
+
}
|
|
82
|
+
return blocks
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def parse_finding_votes(text: str, *, adversarial: bool) -> dict[str, FindingVote]:
|
|
86
|
+
"""Convergence reverify votes in *text*, keyed by finding id.
|
|
87
|
+
|
|
88
|
+
*adversarial* selects the vocabulary; guessing it from the content would make
|
|
89
|
+
an unfamiliar token silently read as a different verdict.
|
|
90
|
+
"""
|
|
91
|
+
vocabulary = ADVERSARIAL_VERDICTS if adversarial else COLLABORATIVE_VERDICTS
|
|
92
|
+
return {
|
|
93
|
+
finding_id: _finding_vote(finding_id, fields, vocabulary)
|
|
94
|
+
for finding_id, fields in _scan_blocks(text).items()
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def _finding_vote(
|
|
99
|
+
finding_id: str, fields: dict[str, str], vocabulary: dict[str, str]
|
|
100
|
+
) -> FindingVote:
|
|
101
|
+
raw = fields.get("Verdict", "")
|
|
102
|
+
token = raw.strip().strip("`").strip().upper()
|
|
103
|
+
if token not in vocabulary:
|
|
104
|
+
raise VerdictBlockError(
|
|
105
|
+
f"finding `{finding_id}` has an unknown verdict: {raw or '(missing)'} "
|
|
106
|
+
f"— expected one of {sorted(vocabulary)}"
|
|
107
|
+
)
|
|
108
|
+
verdict = vocabulary[token]
|
|
109
|
+
explanation = fields.get("Explanation", "")
|
|
110
|
+
if not explanation:
|
|
111
|
+
raise VerdictBlockError(
|
|
112
|
+
f"finding `{finding_id}` has no `**Explanation**:` line — every vote "
|
|
113
|
+
f"the classifier reads carries one"
|
|
114
|
+
)
|
|
115
|
+
basis = fields.get("Basis", "").strip() or None
|
|
116
|
+
if verdict != "disagree":
|
|
117
|
+
basis = None
|
|
118
|
+
elif basis is not None and basis not in DISAGREE_BASES:
|
|
119
|
+
raise VerdictBlockError(
|
|
120
|
+
f"finding `{finding_id}` has basis `{basis}` — expected one of "
|
|
121
|
+
f"{sorted(DISAGREE_BASES)}"
|
|
122
|
+
)
|
|
123
|
+
return FindingVote(finding_id, verdict, basis, explanation)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
@dataclass(frozen=True)
|
|
127
|
+
class VerdictBlock:
|
|
128
|
+
"""One worker's verdict on one item."""
|
|
129
|
+
|
|
130
|
+
item_id: str
|
|
131
|
+
verdict: str
|
|
132
|
+
breakage_kind: str
|
|
133
|
+
fixability: str
|
|
134
|
+
note: str
|
|
135
|
+
explanation: str
|
|
136
|
+
prior_dissent: str
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def parse_verdict_blocks(text: str) -> dict[str, VerdictBlock]:
|
|
140
|
+
"""Every `### <item-id>` plan-body verdict block in *text*, keyed by item id."""
|
|
141
|
+
return {
|
|
142
|
+
item_id: _block(item_id, fields)
|
|
143
|
+
for item_id, fields in _scan_blocks(text).items()
|
|
144
|
+
}
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _verdict_token(item_id: str, raw: str) -> tuple[str, str]:
|
|
148
|
+
parsed = _VERDICT_RE.match(raw.strip().strip("`").strip())
|
|
149
|
+
if parsed is None or parsed.group("token") not in VERDICT_TOKENS:
|
|
150
|
+
raise VerdictBlockError(
|
|
151
|
+
f"item `{item_id}` has an unknown verdict: {raw or '(missing)'}"
|
|
152
|
+
)
|
|
153
|
+
return parsed.group("token"), parsed.group("kind") or ""
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def _block(item_id: str, fields: dict[str, str]) -> VerdictBlock:
|
|
157
|
+
raw = fields.get("Verdict", "")
|
|
158
|
+
if not raw:
|
|
159
|
+
raise VerdictBlockError(f"item `{item_id}` has no `**Verdict**:` line")
|
|
160
|
+
token, kind = _verdict_token(item_id, raw)
|
|
161
|
+
fixability = fields.get("Fixability", "")
|
|
162
|
+
if token == "DISAGREE":
|
|
163
|
+
if not kind:
|
|
164
|
+
raise VerdictBlockError(
|
|
165
|
+
f"item `{item_id}` is DISAGREE with no breakage kind — the gate "
|
|
166
|
+
f"reads the kind to decide whether one vote blocks, so a bare "
|
|
167
|
+
f"`DISAGREE` cannot be scored"
|
|
168
|
+
)
|
|
169
|
+
if fixability not in FIXABILITY_VALUES:
|
|
170
|
+
raise VerdictBlockError(
|
|
171
|
+
f"item `{item_id}` is DISAGREE with fixability "
|
|
172
|
+
f"`{fixability or '(missing)'}` — expected one of "
|
|
173
|
+
f"{sorted(FIXABILITY_VALUES)}"
|
|
174
|
+
)
|
|
175
|
+
return VerdictBlock(
|
|
176
|
+
item_id=item_id,
|
|
177
|
+
verdict=token,
|
|
178
|
+
breakage_kind=kind,
|
|
179
|
+
fixability=fixability,
|
|
180
|
+
note=fields.get("Note", ""),
|
|
181
|
+
explanation=fields.get("Explanation", ""),
|
|
182
|
+
prior_dissent=fields.get("Prior dissent", ""),
|
|
183
|
+
)
|
|
@@ -53,8 +53,10 @@ from okstra_ctl.lead_runtime import lead_runtime_info
|
|
|
53
53
|
from okstra_ctl.runner_resolution import native_provider_for_host
|
|
54
54
|
from okstra_ctl.clarification_items import (
|
|
55
55
|
scan_approval_gate,
|
|
56
|
+
sidecar_answers,
|
|
56
57
|
user_response_sidecars,
|
|
57
58
|
)
|
|
59
|
+
from okstra_ctl.incremental_scope import preview_link_availability_for_report
|
|
58
60
|
from okstra_ctl.design_prep import (
|
|
59
61
|
DesignPrepError,
|
|
60
62
|
load_design_prep_items,
|
|
@@ -94,6 +96,7 @@ from okstra_ctl.workers import (
|
|
|
94
96
|
from okstra_ctl.workflow import PHASE_SEQUENCE
|
|
95
97
|
from okstra_ctl.wizard_stage_intent import (
|
|
96
98
|
WHOLE_TASK_STAGE,
|
|
99
|
+
WizardStageIntent,
|
|
97
100
|
WizardStageIntentError,
|
|
98
101
|
resolve_wizard_stage_intent,
|
|
99
102
|
wizard_stage_confirmation_label,
|
|
@@ -4570,6 +4573,23 @@ def submit(state: WizardState, value: str) -> dict[str, Any]:
|
|
|
4570
4573
|
return {"echo": echo or "", "next": prompt_payload(state, nxt)}
|
|
4571
4574
|
|
|
4572
4575
|
|
|
4576
|
+
def _stage_intent(state: WizardState) -> WizardStageIntent:
|
|
4577
|
+
"""This state's stage selection, resolved once for every consumer.
|
|
4578
|
+
|
|
4579
|
+
`render_args` needs the single stage this run prepares; `wizard_outcome`
|
|
4580
|
+
needs the chain the skill drives. Deriving them separately would let the two
|
|
4581
|
+
disagree about which stage the run is for.
|
|
4582
|
+
"""
|
|
4583
|
+
try:
|
|
4584
|
+
return resolve_wizard_stage_intent(
|
|
4585
|
+
task_type=state.task_type,
|
|
4586
|
+
selected_stage=state.selected_stage,
|
|
4587
|
+
selected_stages=state.selected_stages,
|
|
4588
|
+
)
|
|
4589
|
+
except WizardStageIntentError as exc:
|
|
4590
|
+
raise WizardError(str(exc)) from exc
|
|
4591
|
+
|
|
4592
|
+
|
|
4573
4593
|
def render_args(state: WizardState) -> dict[str, str]:
|
|
4574
4594
|
"""Convert finalized state into ``okstra render-bundle`` argument map."""
|
|
4575
4595
|
if state.aborted:
|
|
@@ -4584,14 +4604,7 @@ def render_args(state: WizardState) -> dict[str, str]:
|
|
|
4584
4604
|
if state.reuse_worktree or state.task_type == "final-verification"
|
|
4585
4605
|
else state.base_ref
|
|
4586
4606
|
)
|
|
4587
|
-
|
|
4588
|
-
stage_intent = resolve_wizard_stage_intent(
|
|
4589
|
-
task_type=state.task_type,
|
|
4590
|
-
selected_stage=state.selected_stage,
|
|
4591
|
-
selected_stages=state.selected_stages,
|
|
4592
|
-
)
|
|
4593
|
-
except WizardStageIntentError as exc:
|
|
4594
|
-
raise WizardError(str(exc)) from exc
|
|
4607
|
+
stage_intent = _stage_intent(state)
|
|
4595
4608
|
pr_template = (
|
|
4596
4609
|
state.pr_template_path
|
|
4597
4610
|
if state.task_type == "release-handoff"
|
|
@@ -4624,7 +4637,6 @@ def render_args(state: WizardState) -> dict[str, str]:
|
|
|
4624
4637
|
"approved-plan": state.approved_plan_path,
|
|
4625
4638
|
"stage": stage_intent.stage,
|
|
4626
4639
|
"stages": state.handoff_stages,
|
|
4627
|
-
"chain-stages": stage_intent.chain_stages,
|
|
4628
4640
|
"base-ref": base_ref,
|
|
4629
4641
|
"workers": workers,
|
|
4630
4642
|
"directive": state.directive,
|
|
@@ -4643,6 +4655,37 @@ def render_args(state: WizardState) -> dict[str, str]:
|
|
|
4643
4655
|
}
|
|
4644
4656
|
|
|
4645
4657
|
|
|
4658
|
+
def _reverify_scope_line(state: WizardState) -> Optional[str]:
|
|
4659
|
+
"""이번 clarification 재실행이 좁혀질지 — 확인 단계에서 미리 보여주는 줄.
|
|
4660
|
+
|
|
4661
|
+
이 판정의 절반(답변된 id 가 stage 로 되짚어지는지)은 base SHA 없이 결정되고
|
|
4662
|
+
직전 리포트만 있으면 이미 정해져 있다. 그런데 지금까지는 `okstra recap
|
|
4663
|
+
assemble` 을 따로 돌려야만 보였고, run 이 시작된 뒤에 full 로 밝혀지면 두
|
|
4664
|
+
시간을 물린 뒤였다. 확인 단계는 그 전에 되돌릴 수 있는 마지막 지점이다.
|
|
4665
|
+
"""
|
|
4666
|
+
if state.task_type != "implementation-planning":
|
|
4667
|
+
return None
|
|
4668
|
+
if not state.clarification_response_path or not state.project_root:
|
|
4669
|
+
return None
|
|
4670
|
+
report = _resolve_path(
|
|
4671
|
+
state.clarification_response_path, Path(state.project_root)
|
|
4672
|
+
)
|
|
4673
|
+
if not report.is_file():
|
|
4674
|
+
return None
|
|
4675
|
+
preview = preview_link_availability_for_report(
|
|
4676
|
+
report, set(sidecar_answers(report))
|
|
4677
|
+
)
|
|
4678
|
+
if not preview["wouldForceFull"]:
|
|
4679
|
+
return _msg(state.workspace_root, "confirmation",
|
|
4680
|
+
"reverify_scope_incremental")
|
|
4681
|
+
if preview["unlinkedIds"]:
|
|
4682
|
+
return _msg(state.workspace_root, "confirmation",
|
|
4683
|
+
"reverify_scope_unlinked",
|
|
4684
|
+
ids=", ".join(preview["unlinkedIds"]))
|
|
4685
|
+
return _msg(state.workspace_root, "confirmation", "reverify_scope_full",
|
|
4686
|
+
reason=preview["reason"])
|
|
4687
|
+
|
|
4688
|
+
|
|
4646
4689
|
def confirmation_block(state: WizardState) -> str:
|
|
4647
4690
|
"""Human-readable echo of the resolved selections (for the Confirm step)."""
|
|
4648
4691
|
header = _msg(state.workspace_root, "confirmation", "header")
|
|
@@ -4727,6 +4770,9 @@ def confirmation_block(state: WizardState) -> str:
|
|
|
4727
4770
|
lines.append(f" stage : {stage}")
|
|
4728
4771
|
if state.clarification_response_path:
|
|
4729
4772
|
lines.append(f" clarification : {state.clarification_response_path}")
|
|
4773
|
+
reverify_line = _reverify_scope_line(state)
|
|
4774
|
+
if reverify_line is not None:
|
|
4775
|
+
lines.append(reverify_line)
|
|
4730
4776
|
if state.task_type == "release-handoff" and state.handoff_mode:
|
|
4731
4777
|
scope = (
|
|
4732
4778
|
_msg(state.workspace_root, "confirmation",
|
|
@@ -4760,13 +4806,21 @@ def _wizard_persist_actions(state: WizardState) -> list[dict[str, str]]:
|
|
|
4760
4806
|
|
|
4761
4807
|
|
|
4762
4808
|
def wizard_outcome(state: WizardState) -> dict[str, Any]:
|
|
4763
|
-
"""Public outcome for callers that need launch data and follow-up writes.
|
|
4809
|
+
"""Public outcome for callers that need launch data and follow-up writes.
|
|
4810
|
+
|
|
4811
|
+
`renderArgs` carries only what `okstra render-bundle` accepts, so a caller
|
|
4812
|
+
can pass every entry through unfiltered — which is exactly what the
|
|
4813
|
+
okstra-run skill is told to do. Signals the skill consumes itself, like the
|
|
4814
|
+
unattended stage chain, live under `orchestration`; mixing them into
|
|
4815
|
+
`renderArgs` made the renderer reject the wizard's own output.
|
|
4816
|
+
"""
|
|
4764
4817
|
if state.aborted:
|
|
4765
4818
|
raise WizardError("wizard was aborted by the user — outcome is unavailable")
|
|
4766
4819
|
if state.confirmed is not True:
|
|
4767
4820
|
raise WizardError("wizard is not complete — outcome is unavailable")
|
|
4768
4821
|
return {
|
|
4769
4822
|
"renderArgs": render_args(state),
|
|
4823
|
+
"orchestration": {"chainStages": _stage_intent(state).chain_stages},
|
|
4770
4824
|
"persistActions": _wizard_persist_actions(state),
|
|
4771
4825
|
"confirmationText": confirmation_block(state),
|
|
4772
4826
|
}
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""CLI adapter for the worker audit-sidecar contract (`okstra worker-audit-check`).
|
|
2
|
+
|
|
3
|
+
Phase 7 runs the same rules through `validate-run.py`, but by then the worker
|
|
4
|
+
session is gone and the only remedies left are a retroactive edit — which breaks
|
|
5
|
+
the audit chain — or a failed run. Called the moment a worker returns, the same
|
|
6
|
+
rules cost one message to a worker that is still listening.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import argparse
|
|
11
|
+
import json
|
|
12
|
+
import sys
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from okstra_ctl.worker_audit_ledger import check_worker_results_audit
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _parser() -> argparse.ArgumentParser:
|
|
19
|
+
parser = argparse.ArgumentParser(
|
|
20
|
+
prog="okstra worker-audit-check",
|
|
21
|
+
description="Check one run's worker audit sidecars (read-only).",
|
|
22
|
+
)
|
|
23
|
+
parser.add_argument("--run-dir", type=Path, required=True,
|
|
24
|
+
help="runs/<task-type>/ for this run")
|
|
25
|
+
parser.add_argument("--task-type", required=True)
|
|
26
|
+
parser.add_argument("--seq", required=True,
|
|
27
|
+
help="this run's 3-digit seq")
|
|
28
|
+
parser.add_argument("--worker", default=None,
|
|
29
|
+
help="check only this worker id (default: every worker)")
|
|
30
|
+
return parser
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def main(argv: list[str] | None = None) -> int:
|
|
34
|
+
args = _parser().parse_args(argv)
|
|
35
|
+
failures = check_worker_results_audit(
|
|
36
|
+
args.run_dir, args.task_type, args.seq, worker=args.worker
|
|
37
|
+
)
|
|
38
|
+
print(json.dumps({"ok": not failures, "failures": failures},
|
|
39
|
+
ensure_ascii=False, indent=2))
|
|
40
|
+
return 2 if failures else 0
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
if __name__ == "__main__":
|
|
44
|
+
raise SystemExit(main(sys.argv[1:]))
|