okstra 0.169.0 → 0.170.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/architecture.md +17 -1
- package/docs/cli.md +11 -1
- package/docs/for-ai/skills/okstra-setup.md +8 -0
- package/docs/project-structure-overview.md +3 -1
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/bin/okstra-error-log.py +38 -282
- package/runtime/prompts/duties/acceptance-critic.md +25 -5
- package/runtime/prompts/duties/acceptance-verifier.md +25 -5
- package/runtime/prompts/duties/analysis-worker.md +25 -5
- package/runtime/prompts/duties/code-reviewer.md +25 -5
- package/runtime/prompts/duties/common.md +15 -11
- package/runtime/prompts/duties/diagnosis-worker.md +44 -0
- package/runtime/prompts/duties/discovery-worker.md +44 -0
- package/runtime/prompts/duties/implementation-executor.md +25 -5
- package/runtime/prompts/duties/implementation-verifier.md +25 -5
- package/runtime/prompts/duties/lead.md +25 -5
- package/runtime/prompts/duties/planning-worker.md +44 -0
- package/runtime/prompts/duties/report-writer.md +25 -5
- package/runtime/prompts/duties/reverification-worker.md +25 -5
- package/runtime/prompts/duties/schedule-verifier.md +25 -5
- package/runtime/prompts/duties/scope-critic.md +25 -5
- package/runtime/prompts/duties/translator.md +25 -5
- package/runtime/prompts/lead/convergence.md +53 -7
- package/runtime/prompts/lead/okstra-lead-contract.md +1 -1
- package/runtime/prompts/lead/plan-body-verification.md +5 -1
- package/runtime/prompts/lead/report-writer.md +1 -1
- package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
- package/runtime/prompts/profiles/final-verification.md +1 -1
- package/runtime/prompts/profiles/implementation-planning.md +2 -2
- package/runtime/python/okstra_ctl/agent_invocation.py +146 -6
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +38 -0
- package/runtime/python/okstra_ctl/cmux.py +36 -19
- package/runtime/python/okstra_ctl/dispatch_core.py +317 -27
- package/runtime/python/okstra_ctl/dispatch_state.py +143 -9
- package/runtime/python/okstra_ctl/doctor.py +31 -0
- package/runtime/python/okstra_ctl/error_log_write.py +308 -0
- package/runtime/python/okstra_ctl/plan_derivations.py +94 -0
- package/runtime/python/okstra_ctl/plan_items_cli.py +114 -3
- package/runtime/python/okstra_ctl/run.py +7 -1
- package/runtime/python/okstra_ctl/schema_excerpt.py +34 -0
- package/runtime/python/okstra_ctl/verdict_blocks.py +17 -0
- package/runtime/python/okstra_ctl/worker_audit_check.py +26 -4
- package/runtime/python/okstra_ctl/worker_audit_ledger.py +59 -9
- package/runtime/python/okstra_ctl/worker_prompt_contract.py +24 -1
- package/runtime/python/okstra_ctl/worker_prompt_headers.py +2 -2
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +12 -1
- package/runtime/python/okstra_project/resolver.py +34 -0
- package/runtime/skills/okstra-setup/references/project-config.md +38 -0
- package/runtime/validators/lib/fixtures.sh +9 -1
- package/runtime/validators/validate-run.py +37 -2
|
@@ -14,8 +14,15 @@ import sys
|
|
|
14
14
|
from typing import Any
|
|
15
15
|
|
|
16
16
|
from .convergence_store import write_json_atomic
|
|
17
|
+
from .plan_derivations import extract_tokens, find_derivations
|
|
17
18
|
from .plan_items import PlanItemContractError, extract_plan_items
|
|
18
|
-
from .
|
|
19
|
+
from .user_response import parse_user_response_entries
|
|
20
|
+
from .verdict_blocks import (
|
|
21
|
+
PLAN_ITEM_VERDICTS,
|
|
22
|
+
VerdictBlock,
|
|
23
|
+
VerdictBlockError,
|
|
24
|
+
parse_verdict_blocks,
|
|
25
|
+
)
|
|
19
26
|
|
|
20
27
|
|
|
21
28
|
def _load_json_object(path: Path) -> dict[str, Any]:
|
|
@@ -64,6 +71,20 @@ def _parser() -> argparse.ArgumentParser:
|
|
|
64
71
|
help="one worker's plan-verify result file (repeatable)")
|
|
65
72
|
collect.add_argument("--items", type=Path, required=True)
|
|
66
73
|
collect.add_argument("--output", type=Path, required=True)
|
|
74
|
+
derivations = commands.add_parser(
|
|
75
|
+
"derivations",
|
|
76
|
+
help="list plan statements an answered clarification may have falsified",
|
|
77
|
+
)
|
|
78
|
+
derivations.add_argument("--data", type=Path, required=True)
|
|
79
|
+
derivations.add_argument("--response", type=Path, required=True,
|
|
80
|
+
help="the user-responses sidecar for this run")
|
|
81
|
+
derivations.add_argument("--clarification", default=None,
|
|
82
|
+
help="only this C-id (default: every answered one)")
|
|
83
|
+
seed = commands.add_parser(
|
|
84
|
+
"seed",
|
|
85
|
+
help="create the planBodyVerification.planItems[] rows a round lands in",
|
|
86
|
+
)
|
|
87
|
+
seed.add_argument("--data", type=Path, required=True)
|
|
67
88
|
apply_verdicts = commands.add_parser(
|
|
68
89
|
"apply-verdicts",
|
|
69
90
|
help="overwrite planBodyVerification.planItems[].verdicts in data.json",
|
|
@@ -112,8 +133,18 @@ def _assigned_item_ids(items_path: Path) -> list[str]:
|
|
|
112
133
|
|
|
113
134
|
def _verdict_row(worker: str, block: VerdictBlock) -> dict[str, Any]:
|
|
114
135
|
"""One `planItems[].verdicts[]` row. Optional fields stay absent when empty
|
|
115
|
-
so the recorded table shows what the worker actually said.
|
|
116
|
-
|
|
136
|
+
so the recorded table shows what the worker actually said.
|
|
137
|
+
|
|
138
|
+
The verdict crosses a vocabulary boundary here: a worker answers
|
|
139
|
+
`UNVERIFIABLE`, and the schema persists that as `verification-error`
|
|
140
|
+
(`PLAN_ITEM_VERDICTS`). Writing the worker's token straight through produced
|
|
141
|
+
a data.json its own schema rejects, and the mapping the contract prescribes
|
|
142
|
+
had to be applied by hand every round.
|
|
143
|
+
"""
|
|
144
|
+
row: dict[str, Any] = {
|
|
145
|
+
"worker": worker,
|
|
146
|
+
"verdict": PLAN_ITEM_VERDICTS[block.verdict],
|
|
147
|
+
}
|
|
117
148
|
for key, value in (
|
|
118
149
|
("breakageKind", block.breakage_kind),
|
|
119
150
|
("fixability", block.fixability),
|
|
@@ -194,6 +225,84 @@ def _plan_body_items(data: dict[str, Any], data_path: Path) -> list[dict[str, An
|
|
|
194
225
|
return items
|
|
195
226
|
|
|
196
227
|
|
|
228
|
+
def _derivations(args: argparse.Namespace) -> dict[str, Any]:
|
|
229
|
+
"""Candidate statements each answered clarification may have falsified.
|
|
230
|
+
|
|
231
|
+
Advisory by construction: it reports where a decision's subject is mentioned
|
|
232
|
+
and never which mentions are now wrong. Both contracts require the author to
|
|
233
|
+
enumerate before editing; this supplies the enumeration, which is the half
|
|
234
|
+
that was being skipped, and leaves the judgement where it belongs.
|
|
235
|
+
"""
|
|
236
|
+
try:
|
|
237
|
+
sidecar = args.response.read_text(encoding="utf-8")
|
|
238
|
+
except (OSError, UnicodeError) as exc:
|
|
239
|
+
raise PlanItemContractError(
|
|
240
|
+
f"cannot read user-response sidecar {args.response}: {exc}"
|
|
241
|
+
) from exc
|
|
242
|
+
planning = _planning(_load_json_object(args.data))
|
|
243
|
+
entries = [
|
|
244
|
+
entry for entry in parse_user_response_entries(sidecar)
|
|
245
|
+
if args.clarification is None or entry.response_id == args.clarification
|
|
246
|
+
]
|
|
247
|
+
if args.clarification is not None and not entries:
|
|
248
|
+
raise PlanItemContractError(
|
|
249
|
+
f"{args.response} has no response block for {args.clarification}"
|
|
250
|
+
)
|
|
251
|
+
clarifications = []
|
|
252
|
+
for entry in entries:
|
|
253
|
+
tokens = extract_tokens(f"{entry.value}\n{entry.rationale or ''}")
|
|
254
|
+
clarifications.append({
|
|
255
|
+
"id": entry.response_id,
|
|
256
|
+
"disposition": entry.disposition,
|
|
257
|
+
"tokens": tokens,
|
|
258
|
+
"candidates": find_derivations(planning, tokens),
|
|
259
|
+
})
|
|
260
|
+
return {
|
|
261
|
+
"ok": True,
|
|
262
|
+
"operation": "derivations",
|
|
263
|
+
"advisory": True,
|
|
264
|
+
"clarifications": clarifications,
|
|
265
|
+
}
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def _seed(args: argparse.Namespace) -> dict[str, Any]:
|
|
269
|
+
"""Create the `planBodyVerification.planItems[]` rows a round lands in.
|
|
270
|
+
|
|
271
|
+
`apply-verdicts` refuses a verdict whose item has no row — correctly, since
|
|
272
|
+
the gate is re-derived from that table and a verdict with nowhere to land
|
|
273
|
+
would score as never cast. But nothing created the rows: the report writer
|
|
274
|
+
leaves `planItems: []` (§5.5.9 is a lead substep that runs after it), and
|
|
275
|
+
there was no step, in code or in the contract, that filled them. Every round
|
|
276
|
+
had to be hand-seeded before the CLI would accept its own output.
|
|
277
|
+
|
|
278
|
+
Idempotent by id. An existing row keeps everything it carries — verdicts
|
|
279
|
+
already applied, `carriedForwardFromSeq`, `selfFixNote` — because a re-seed
|
|
280
|
+
between rounds must not erase the round before it.
|
|
281
|
+
"""
|
|
282
|
+
extracted = _envelope(_load_json_object(args.data))["items"]
|
|
283
|
+
data = _load_json_object(args.data)
|
|
284
|
+
recorded = _plan_body_items(data, args.data)
|
|
285
|
+
known = {
|
|
286
|
+
item.get("id")
|
|
287
|
+
for item in recorded
|
|
288
|
+
if isinstance(item, Mapping)
|
|
289
|
+
}
|
|
290
|
+
added = [
|
|
291
|
+
{**item, "verdicts": []}
|
|
292
|
+
for item in extracted
|
|
293
|
+
if item["id"] not in known
|
|
294
|
+
]
|
|
295
|
+
recorded.extend(added)
|
|
296
|
+
write_json_atomic(args.data, data)
|
|
297
|
+
return {
|
|
298
|
+
"ok": True,
|
|
299
|
+
"operation": "seed",
|
|
300
|
+
"path": str(args.data),
|
|
301
|
+
"seeded": len(added),
|
|
302
|
+
"existing": len(known),
|
|
303
|
+
}
|
|
304
|
+
|
|
305
|
+
|
|
197
306
|
def _apply_verdicts(args: argparse.Namespace) -> dict[str, Any]:
|
|
198
307
|
data = _load_json_object(args.data)
|
|
199
308
|
incoming = _load_json_object(args.verdicts).get("planItems")
|
|
@@ -226,6 +335,8 @@ _HANDLERS = {
|
|
|
226
335
|
"extract": _extract,
|
|
227
336
|
"validate": _validate,
|
|
228
337
|
"collect-verdicts": _collect_verdicts,
|
|
338
|
+
"derivations": _derivations,
|
|
339
|
+
"seed": _seed,
|
|
229
340
|
"apply-verdicts": _apply_verdicts,
|
|
230
341
|
}
|
|
231
342
|
|
|
@@ -68,6 +68,7 @@ from .domain.host import (
|
|
|
68
68
|
ProviderUnavailable,
|
|
69
69
|
)
|
|
70
70
|
from .model_discovery import normalize_execution_for_dispatch
|
|
71
|
+
from .worker_prompt_policy import ANALYSIS_DUTY_BY_TASK_TYPE
|
|
71
72
|
from .models import (
|
|
72
73
|
ModelAssignment,
|
|
73
74
|
UnknownProviderError,
|
|
@@ -2523,7 +2524,12 @@ def _allowed_agent_audiences(
|
|
|
2523
2524
|
elif inp.task_type == "final-verification":
|
|
2524
2525
|
audiences.add("acceptance-verifier")
|
|
2525
2526
|
else:
|
|
2526
|
-
|
|
2527
|
+
# Same map the prompt policy resolves the duty from. Allowing a
|
|
2528
|
+
# different audience here than the one the policy will ask for makes
|
|
2529
|
+
# every worker prompt in the phase fail materialization.
|
|
2530
|
+
audiences.add(
|
|
2531
|
+
ANALYSIS_DUTY_BY_TASK_TYPE.get(inp.task_type, "analysis-worker")
|
|
2532
|
+
)
|
|
2527
2533
|
if models.critic_choice not in {"", "off"}:
|
|
2528
2534
|
audiences.update({"scope-critic", "acceptance-critic"})
|
|
2529
2535
|
return sorted(audiences)
|
|
@@ -23,6 +23,7 @@ from __future__ import annotations
|
|
|
23
23
|
|
|
24
24
|
import json
|
|
25
25
|
import re
|
|
26
|
+
from pathlib import Path
|
|
26
27
|
|
|
27
28
|
from .report_contract import TASK_TYPE_DATA_PROPERTY
|
|
28
29
|
|
|
@@ -78,6 +79,39 @@ def excerpt_cut_from_version(excerpt: dict) -> str:
|
|
|
78
79
|
return value if isinstance(value, str) else ""
|
|
79
80
|
|
|
80
81
|
|
|
82
|
+
def excerpt_version_skew(excerpt_path: Path, installed: str) -> str:
|
|
83
|
+
"""The version the bundle excerpt was cut from, when it is not *installed*.
|
|
84
|
+
|
|
85
|
+
Empty means no actionable skew: the file is absent or unreadable, carries no
|
|
86
|
+
stamp, or matches. A non-empty return is the older version, which the caller
|
|
87
|
+
turns into its own message.
|
|
88
|
+
|
|
89
|
+
The comparison used to live only in the renderer's error decorator, so it ran
|
|
90
|
+
in Phase 6 — after a worker had already authored a whole report against a
|
|
91
|
+
stale excerpt. The same two values are available much earlier, and the fix
|
|
92
|
+
(re-prepare the bundle) is the same either way.
|
|
93
|
+
"""
|
|
94
|
+
if not installed:
|
|
95
|
+
return ""
|
|
96
|
+
try:
|
|
97
|
+
excerpt = json.loads(excerpt_path.read_text(encoding="utf-8"))
|
|
98
|
+
except (OSError, json.JSONDecodeError):
|
|
99
|
+
return ""
|
|
100
|
+
if not isinstance(excerpt, dict):
|
|
101
|
+
return ""
|
|
102
|
+
cut_from = excerpt_cut_from_version(excerpt)
|
|
103
|
+
return cut_from if cut_from and cut_from != installed else ""
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def bundle_excerpt_path(start: Path) -> Path | None:
|
|
107
|
+
"""The task bundle's schema excerpt, found by walking up from *start*."""
|
|
108
|
+
for ancestor in Path(start).resolve().parents:
|
|
109
|
+
candidate = ancestor / "instruction-set" / "final-report-schema.json"
|
|
110
|
+
if candidate.is_file():
|
|
111
|
+
return candidate
|
|
112
|
+
return None
|
|
113
|
+
|
|
114
|
+
|
|
81
115
|
def build_schema_excerpt(schema: dict, task_type: str, cut_from_version: str = "") -> dict:
|
|
82
116
|
"""Return a task-type-scoped copy of *schema*.
|
|
83
117
|
|
|
@@ -38,6 +38,23 @@ ADVERSARIAL_VERDICTS = {
|
|
|
38
38
|
"UNVERIFIABLE": "unverifiable",
|
|
39
39
|
"VERIFICATION-ERROR": "verification-error",
|
|
40
40
|
}
|
|
41
|
+
# Plan-body verdicts persist in their own vocabulary, which is NOT the
|
|
42
|
+
# convergence one above: the schema keeps the first three tokens uppercase
|
|
43
|
+
# (`$defs/PlanBodyVerification/properties/planItems/items/properties/verdicts/
|
|
44
|
+
# items/properties/verdict`) and has no `UNVERIFIABLE` member at all.
|
|
45
|
+
# `plan-body-verification.md` §"Planning-time environment gap" states the
|
|
46
|
+
# mapping — "Recording `UNVERIFIABLE` is the honest outcome — it is persisted as
|
|
47
|
+
# `verification-error`" — and it lives here rather than inside the transcription
|
|
48
|
+
# CLI so the worker-facing token and the persisted token are decided in one
|
|
49
|
+
# place. Writing the raw token through put a value in data.json that its own
|
|
50
|
+
# schema rejects.
|
|
51
|
+
PLAN_ITEM_VERDICTS = {
|
|
52
|
+
"AGREE": "AGREE",
|
|
53
|
+
"DISAGREE": "DISAGREE",
|
|
54
|
+
"SUPPLEMENT": "SUPPLEMENT",
|
|
55
|
+
"UNVERIFIABLE": "verification-error",
|
|
56
|
+
"VERIFICATION-ERROR": "verification-error",
|
|
57
|
+
}
|
|
41
58
|
DISAGREE_BASES = frozenset({"counter-evidence", "burden-not-met"})
|
|
42
59
|
|
|
43
60
|
_ITEM_RE = re.compile(r"^###[ \t]+(?P<id>[^\s:]+)[ \t]*:?.*$", re.MULTILINE)
|
|
@@ -12,7 +12,10 @@ import json
|
|
|
12
12
|
import sys
|
|
13
13
|
from pathlib import Path
|
|
14
14
|
|
|
15
|
-
from okstra_ctl.worker_audit_ledger import
|
|
15
|
+
from okstra_ctl.worker_audit_ledger import (
|
|
16
|
+
check_worker_results_audit,
|
|
17
|
+
worker_result_files,
|
|
18
|
+
)
|
|
16
19
|
|
|
17
20
|
|
|
18
21
|
def _parser() -> argparse.ArgumentParser:
|
|
@@ -26,7 +29,8 @@ def _parser() -> argparse.ArgumentParser:
|
|
|
26
29
|
parser.add_argument("--seq", required=True,
|
|
27
30
|
help="this run's 3-digit seq")
|
|
28
31
|
parser.add_argument("--worker", default=None,
|
|
29
|
-
help="check only this worker id
|
|
32
|
+
help="check only this worker id, with or without the "
|
|
33
|
+
"`-worker` suffix (default: every worker)")
|
|
30
34
|
return parser
|
|
31
35
|
|
|
32
36
|
|
|
@@ -35,8 +39,26 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
35
39
|
failures = check_worker_results_audit(
|
|
36
40
|
args.run_dir, args.task_type, args.seq, worker=args.worker
|
|
37
41
|
)
|
|
38
|
-
|
|
39
|
-
|
|
42
|
+
# A selector that narrows to nothing also produces no failures, so `ok`
|
|
43
|
+
# alone cannot tell a real pass from a check that judged zero files —
|
|
44
|
+
# a mistyped `--worker` used to read as a clean bill of health. The count
|
|
45
|
+
# is what makes the two distinguishable at a glance.
|
|
46
|
+
inspected = [
|
|
47
|
+
path.name
|
|
48
|
+
for path, _role, _seq in worker_result_files(
|
|
49
|
+
args.run_dir, args.task_type, args.seq, args.worker
|
|
50
|
+
)
|
|
51
|
+
]
|
|
52
|
+
print(json.dumps(
|
|
53
|
+
{
|
|
54
|
+
"ok": not failures,
|
|
55
|
+
"inspected": len(inspected),
|
|
56
|
+
"inspectedFiles": inspected,
|
|
57
|
+
"failures": failures,
|
|
58
|
+
},
|
|
59
|
+
ensure_ascii=False,
|
|
60
|
+
indent=2,
|
|
61
|
+
))
|
|
40
62
|
return 2 if failures else 0
|
|
41
63
|
|
|
42
64
|
|
|
@@ -11,6 +11,7 @@ so the rules live here rather than inside either one — the same split
|
|
|
11
11
|
from __future__ import annotations
|
|
12
12
|
|
|
13
13
|
import re
|
|
14
|
+
from dataclasses import dataclass
|
|
14
15
|
from pathlib import Path
|
|
15
16
|
|
|
16
17
|
from okstra_ctl.worker_prompt_headers import EVIDENCE_LEDGER_HEADER
|
|
@@ -148,26 +149,75 @@ def _ledger_path_hint(cited_path: str, read_paths: set[str]) -> str:
|
|
|
148
149
|
)
|
|
149
150
|
|
|
150
151
|
|
|
151
|
-
|
|
152
|
+
@dataclass(frozen=True)
|
|
153
|
+
class WorkerResultName:
|
|
154
|
+
"""The three fields a canonical worker-result basename carries."""
|
|
155
|
+
|
|
156
|
+
worker_role: str
|
|
157
|
+
task_type: str
|
|
158
|
+
seq: str
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def parse_worker_result_name(basename: str) -> WorkerResultName | None:
|
|
162
|
+
"""Split `<role>-worker-<task-type>-<seq>.md`, or None when non-canonical.
|
|
163
|
+
|
|
164
|
+
Callers that already hold one worker's result path — the dispatcher settling
|
|
165
|
+
that worker — read the audit check's arguments from here instead of
|
|
166
|
+
re-deriving them from the manifest. The `worker_role` this returns carries
|
|
167
|
+
the `-worker` suffix, which is what `check_worker_results_audit(worker=...)`
|
|
168
|
+
matches on; a bare provider id like `claude` matches nothing.
|
|
169
|
+
"""
|
|
170
|
+
if "-audit-" in basename:
|
|
171
|
+
return None
|
|
172
|
+
match = _WORKER_RESULT_BASENAME_RE.match(basename)
|
|
173
|
+
if match is None:
|
|
174
|
+
return None
|
|
175
|
+
return WorkerResultName(
|
|
176
|
+
worker_role=match.group("worker"),
|
|
177
|
+
task_type=match.group("task_type"),
|
|
178
|
+
seq=match.group("seq"),
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def normalize_worker_filter(worker: str | None) -> str | None:
|
|
183
|
+
"""Accept a worker id the way every other okstra surface spells it.
|
|
184
|
+
|
|
185
|
+
Result files are named `<role>-worker-...`, so the filter matches
|
|
186
|
+
`claude-worker`. But `claude` is what a worker is called everywhere a lead
|
|
187
|
+
reads or types one — `--workers claude,codex`, `workerId`, the roster in the
|
|
188
|
+
profile — so `--worker claude` was the natural thing to pass, and it matched
|
|
189
|
+
nothing. A zero-match filter produces no failures, which the CLI reports as
|
|
190
|
+
`{"ok": true}` with exit 0: indistinguishable from a real pass, at exactly
|
|
191
|
+
the point `team-contract` tells the lead to run this check before deciding
|
|
192
|
+
whether to re-dispatch. Accepting both spellings removes the trap rather
|
|
193
|
+
than documenting it.
|
|
194
|
+
"""
|
|
195
|
+
if worker is None or worker.endswith("-worker"):
|
|
196
|
+
return worker
|
|
197
|
+
return f"{worker}-worker"
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def worker_result_files(
|
|
201
|
+
run_dir: Path, task_type: str, seq: str | None, worker: str | None
|
|
202
|
+
):
|
|
152
203
|
"""Every worker-results file in *run_dir* this check owns, in name order."""
|
|
204
|
+
worker = normalize_worker_filter(worker)
|
|
153
205
|
for path in sorted((run_dir / "worker-results").glob("*.md")):
|
|
154
|
-
|
|
155
|
-
continue
|
|
156
|
-
match = _WORKER_RESULT_BASENAME_RE.match(path.name)
|
|
206
|
+
match = parse_worker_result_name(path.name)
|
|
157
207
|
if match is None:
|
|
158
208
|
# Files that don't match the canonical pattern (e.g. ad-hoc notes
|
|
159
209
|
# left by the operator) are out of contract scope.
|
|
160
210
|
continue
|
|
161
|
-
if match.
|
|
211
|
+
if match.task_type != task_type:
|
|
162
212
|
# Cross-phase artifacts shouldn't appear here; skip rather than
|
|
163
213
|
# fail to keep the check focused on the current phase.
|
|
164
214
|
continue
|
|
165
|
-
if seq is not None and match.
|
|
215
|
+
if seq is not None and match.seq != seq:
|
|
166
216
|
# A prior run's artifact. Its contract was judged when it ran.
|
|
167
217
|
continue
|
|
168
|
-
if worker is not None and match.
|
|
218
|
+
if worker is not None and match.worker_role != worker:
|
|
169
219
|
continue
|
|
170
|
-
yield path, match.
|
|
220
|
+
yield path, match.worker_role, match.seq
|
|
171
221
|
|
|
172
222
|
|
|
173
223
|
def check_worker_results_audit(
|
|
@@ -196,7 +246,7 @@ def check_worker_results_audit(
|
|
|
196
246
|
# `release-handoff`, which is single-lead). Nothing to enforce.
|
|
197
247
|
return failures
|
|
198
248
|
|
|
199
|
-
for path, worker_role, result_seq in
|
|
249
|
+
for path, worker_role, result_seq in worker_result_files(run_dir, task_type, seq, worker):
|
|
200
250
|
rel = path.name
|
|
201
251
|
try:
|
|
202
252
|
content = path.read_text()
|
|
@@ -230,7 +230,9 @@ def validate_reverify_prompt(
|
|
|
230
230
|
read_scope_position < 0 or read_scope_position > boundary_position
|
|
231
231
|
):
|
|
232
232
|
errors.append("phase boundary block must follow the reverify anchor headers")
|
|
233
|
-
first_heading = re.
|
|
233
|
+
first_heading = re.compile(r"(?m)^##\s+").search(
|
|
234
|
+
normalized, _task_instructions_offset(normalized)
|
|
235
|
+
)
|
|
234
236
|
if (
|
|
235
237
|
boundary_position >= 0
|
|
236
238
|
and first_heading is not None
|
|
@@ -240,6 +242,27 @@ def validate_reverify_prompt(
|
|
|
240
242
|
return errors
|
|
241
243
|
|
|
242
244
|
|
|
245
|
+
def _task_instructions_offset(text: str) -> int:
|
|
246
|
+
"""Where the lead's own instruction body starts.
|
|
247
|
+
|
|
248
|
+
The `agent-prompt` materializer composes every prompt as anchors →
|
|
249
|
+
`## Duty Contract` → `## Task Instructions`, and convergence's
|
|
250
|
+
materialization gate makes that the only body a reverify dispatch may send.
|
|
251
|
+
The duty section's heading is therefore always the document's first `##`,
|
|
252
|
+
which left the check below with no satisfiable input: the composer writes a
|
|
253
|
+
heading above anything the lead can author, so a whole-document "first
|
|
254
|
+
heading" test failed every materialized prompt regardless of where the lead
|
|
255
|
+
put the phase boundary.
|
|
256
|
+
|
|
257
|
+
These rules judge what the lead wrote, so they start where the lead's text
|
|
258
|
+
starts — the same region `_validate_model_header` already reads. A prompt
|
|
259
|
+
without the marker is judged whole.
|
|
260
|
+
"""
|
|
261
|
+
marker = "\n\n## Task Instructions\n\n"
|
|
262
|
+
index = text.find(marker)
|
|
263
|
+
return 0 if index < 0 else index + len(marker)
|
|
264
|
+
|
|
265
|
+
|
|
243
266
|
def _section_values(text: str, header: str) -> list[str]:
|
|
244
267
|
lines = text.splitlines()
|
|
245
268
|
values: list[str] = []
|
|
@@ -91,7 +91,7 @@ def worker_prompt_headers(
|
|
|
91
91
|
project_root,
|
|
92
92
|
audit_source_rel or result_rel,
|
|
93
93
|
)
|
|
94
|
-
errors_log_path =
|
|
94
|
+
errors_log_path = resolve_errors_log_path(project_root, manifest, active_context)
|
|
95
95
|
errors_sidecar_path = _worker_errors_sidecar_path(
|
|
96
96
|
project_root,
|
|
97
97
|
manifest,
|
|
@@ -224,7 +224,7 @@ def _improvement_grilling_log_path(
|
|
|
224
224
|
return path
|
|
225
225
|
|
|
226
226
|
|
|
227
|
-
def
|
|
227
|
+
def resolve_errors_log_path(
|
|
228
228
|
project_root: Path,
|
|
229
229
|
manifest: Mapping[str, Any],
|
|
230
230
|
active_context: Mapping[str, Any],
|
|
@@ -58,6 +58,17 @@ SUPPORTED_TASK_TYPES = frozenset({
|
|
|
58
58
|
"release-handoff",
|
|
59
59
|
*ANALYSIS_TASK_TYPES,
|
|
60
60
|
})
|
|
61
|
+
# One duty per analysis ROLE, not per phase. Phases that ask their worker for the
|
|
62
|
+
# same kind of judgement share a contract — requirements- and improvement-discovery
|
|
63
|
+
# both hand over candidates they do not start — and a phase whose worker decides
|
|
64
|
+
# something else gets its own. A task type absent from this map takes the
|
|
65
|
+
# observational default below: describe the area, do not design for it.
|
|
66
|
+
ANALYSIS_DUTY_BY_TASK_TYPE: dict[str, AgentAudience] = {
|
|
67
|
+
"requirements-discovery": "discovery-worker",
|
|
68
|
+
"improvement-discovery": "discovery-worker",
|
|
69
|
+
"error-analysis": "diagnosis-worker",
|
|
70
|
+
"implementation-planning": "planning-worker",
|
|
71
|
+
}
|
|
61
72
|
WORKER_PREAMBLE_FILENAME_BY_AUDIENCE = {
|
|
62
73
|
"analysis": "worker-prompt-preamble.md",
|
|
63
74
|
"implementation-executor": "implementation-worker-preamble.md",
|
|
@@ -113,7 +124,7 @@ def resolve_prompt_plan(
|
|
|
113
124
|
if dispatch_kind == "critic"
|
|
114
125
|
else "acceptance-verifier"
|
|
115
126
|
if task_type == "final-verification"
|
|
116
|
-
else "analysis-worker"
|
|
127
|
+
else ANALYSIS_DUTY_BY_TASK_TYPE.get(task_type, "analysis-worker")
|
|
117
128
|
)
|
|
118
129
|
if task_type == "implementation" and not executor_worker_id:
|
|
119
130
|
raise ValueError("implementation executor worker ID is required")
|
|
@@ -108,6 +108,40 @@ def resolve_architecture(project_root: Path | str) -> str:
|
|
|
108
108
|
return style if style in _ARCHITECTURE_STYLES else "none"
|
|
109
109
|
|
|
110
110
|
|
|
111
|
+
def resolve_review_rule_packs(project_root: Path | str) -> tuple[str, ...]:
|
|
112
|
+
"""Return the project's declared review rule pack paths, else ``()``.
|
|
113
|
+
|
|
114
|
+
A review rule pack is a project's own review standard — the file a phase
|
|
115
|
+
reads before judging a diff. It used to reach a run only when the task
|
|
116
|
+
brief cited its exact path, so a team standard applied or not depending on
|
|
117
|
+
who wrote the brief. Declaring it here makes it apply to every run in the
|
|
118
|
+
project; the brief citation still works and the two are a union.
|
|
119
|
+
|
|
120
|
+
Mirrors resolve_architecture's failure posture: any read/parse problem or a
|
|
121
|
+
non-list value degrades to "none declared" rather than raising inside a run.
|
|
122
|
+
An entry that is not an absolute path is dropped, because workers run with
|
|
123
|
+
a worktree as cwd — a relative path would resolve against a tree that does
|
|
124
|
+
not contain the pack, and would mean a different file per worker.
|
|
125
|
+
"""
|
|
126
|
+
try:
|
|
127
|
+
payload = json.loads(project_json_path(Path(project_root)).read_text(encoding="utf-8"))
|
|
128
|
+
except (OSError, ValueError):
|
|
129
|
+
return ()
|
|
130
|
+
if not isinstance(payload, dict):
|
|
131
|
+
return ()
|
|
132
|
+
declared = payload.get("reviewRulePacks")
|
|
133
|
+
if not isinstance(declared, list):
|
|
134
|
+
return ()
|
|
135
|
+
packs = []
|
|
136
|
+
for entry in declared:
|
|
137
|
+
if not isinstance(entry, str) or not entry.strip():
|
|
138
|
+
continue
|
|
139
|
+
candidate = Path(entry.strip()).expanduser()
|
|
140
|
+
if candidate.is_absolute():
|
|
141
|
+
packs.append(str(candidate))
|
|
142
|
+
return tuple(dict.fromkeys(packs))
|
|
143
|
+
|
|
144
|
+
|
|
111
145
|
def upsert_project_json(project_root: Path, project_id: str, *,
|
|
112
146
|
now: Optional[str] = None) -> dict:
|
|
113
147
|
"""project.json 을 읽거나 새로 만든다.
|
|
@@ -252,3 +252,41 @@ rules from advisory to a binding planning + verification constraint:
|
|
|
252
252
|
|
|
253
253
|
The full two-layer model lives in `docs/architecture.md` § Project
|
|
254
254
|
self-registration.
|
|
255
|
+
|
|
256
|
+
## G. Project review rule packs (`reviewRulePacks`)
|
|
257
|
+
|
|
258
|
+
`reviewRulePacks` declares the review standards this project's phases read
|
|
259
|
+
before they judge a plan or a diff — a team PR-review skill's `SKILL.md`, for
|
|
260
|
+
example. Without it a pack reaches a run only when the task brief cites its
|
|
261
|
+
exact path, so whether the team standard applied came down to who wrote the
|
|
262
|
+
brief. A declared pack applies to every run in the project; the brief citation
|
|
263
|
+
keeps working, and the two channels are a union.
|
|
264
|
+
|
|
265
|
+
`okstra setup` never writes it. Hand-add it to `project.json` and the upsert
|
|
266
|
+
preserves it:
|
|
267
|
+
|
|
268
|
+
```json
|
|
269
|
+
{
|
|
270
|
+
"reviewRulePacks": [
|
|
271
|
+
"/Users/me/.claude/skills/team-pr-reviewer/SKILL.md"
|
|
272
|
+
]
|
|
273
|
+
}
|
|
274
|
+
```
|
|
275
|
+
|
|
276
|
+
- **Absolute paths only.** A worker runs with a task worktree as its cwd, so a
|
|
277
|
+
relative path names a different file there — relative entries are dropped by
|
|
278
|
+
`scripts/okstra_project/resolver.py::resolve_review_rule_packs`, which also
|
|
279
|
+
falls back to "none declared" on an unreadable or malformed `project.json`.
|
|
280
|
+
A leading `~` is expanded.
|
|
281
|
+
- **Who reads it.** `implementation-planning` (to plan away the findings before
|
|
282
|
+
the code exists), the implementation executor's coding-conventions preflight,
|
|
283
|
+
and the static review passes of the implementation verifier and
|
|
284
|
+
`final-verification`. Each reads only the declared file and the
|
|
285
|
+
`references/*.md` files it directly names — never a parent directory or a
|
|
286
|
+
host skill catalog.
|
|
287
|
+
- **What is machine-checked.** `okstra doctor --phase <phase>` fails its
|
|
288
|
+
`review rule packs` check when a declared path is not a readable file,
|
|
289
|
+
because a stale path otherwise costs the whole pack in silence: the phase
|
|
290
|
+
records that it read no rules and the run still passes. Whether a pack that
|
|
291
|
+
does resolve was actually read and applied stays the phase's own
|
|
292
|
+
`project-review-rules:` record — no machine check reads a worker's reasoning.
|
|
@@ -544,6 +544,14 @@ if WORKSPACE_ROOT:
|
|
|
544
544
|
link_agent_dispatch_result,
|
|
545
545
|
record_verified_agent_dispatch,
|
|
546
546
|
)
|
|
547
|
+
# The analysis duty depends on the task type, exactly as it does in a real
|
|
548
|
+
# run — a fixture that assumed one audience for every phase would be
|
|
549
|
+
# rejected by the manifest's own allowlist.
|
|
550
|
+
from okstra_ctl.worker_prompt_policy import ANALYSIS_DUTY_BY_TASK_TYPE
|
|
551
|
+
|
|
552
|
+
analysis_audience = ANALYSIS_DUTY_BY_TASK_TYPE.get(
|
|
553
|
+
str(run_manifest.get("taskType") or ""), "analysis-worker"
|
|
554
|
+
)
|
|
547
555
|
|
|
548
556
|
lead_record = record_verified_agent_dispatch(
|
|
549
557
|
project_root=project_root,
|
|
@@ -580,7 +588,7 @@ if WORKSPACE_ROOT:
|
|
|
580
588
|
worker_id=worker_id,
|
|
581
589
|
audience=(
|
|
582
590
|
"report-writer" if worker_id == "report-writer"
|
|
583
|
-
else
|
|
591
|
+
else analysis_audience
|
|
584
592
|
),
|
|
585
593
|
assignment_ref=assignment_ref,
|
|
586
594
|
purpose=None,
|
|
@@ -5470,6 +5470,27 @@ def _plan_verify_result_workers(report_path: Path, task_type: str) -> set[str] |
|
|
|
5470
5470
|
return workers
|
|
5471
5471
|
|
|
5472
5472
|
|
|
5473
|
+
def _plan_verify_seq_near_misses(report_path: Path, task_type: str) -> list[str]:
|
|
5474
|
+
"""Plan-verify results in the directory that this run's seq filter excluded.
|
|
5475
|
+
|
|
5476
|
+
The seq comes from the report filename, and a run whose `reports` and
|
|
5477
|
+
`workerResults` sequences differ makes the other one look equally plausible
|
|
5478
|
+
to a lead naming the file by hand. Naming a near miss is the difference
|
|
5479
|
+
between "the file is missing" and "the file is there under a different seq"
|
|
5480
|
+
— the first sends a lead to re-dispatch two workers that already ran.
|
|
5481
|
+
"""
|
|
5482
|
+
seq = _report_run_seq(report_path)
|
|
5483
|
+
if not seq:
|
|
5484
|
+
return []
|
|
5485
|
+
directory = report_path.parent.parent / "worker-results"
|
|
5486
|
+
matched = set(directory.glob(f"*-plan-verify-r*-{task_type}-{seq}.md"))
|
|
5487
|
+
return sorted(
|
|
5488
|
+
path.name
|
|
5489
|
+
for path in directory.glob(f"*-plan-verify-r*-{task_type}-*.md")
|
|
5490
|
+
if path not in matched
|
|
5491
|
+
)
|
|
5492
|
+
|
|
5493
|
+
|
|
5473
5494
|
def _validate_plan_body_verdict_provenance(
|
|
5474
5495
|
data: dict,
|
|
5475
5496
|
report_path: Path,
|
|
@@ -5514,14 +5535,28 @@ def _validate_plan_body_verdict_provenance(
|
|
|
5514
5535
|
"final-report data.json: planBodyVerification records verdicts from "
|
|
5515
5536
|
f"{unbacked} but no matching plan-body reverify result file exists "
|
|
5516
5537
|
f"under `runs/{task_type}/worker-results/` "
|
|
5517
|
-
f"(
|
|
5518
|
-
"
|
|
5538
|
+
f"(globbed `*-plan-verify-r*-{task_type}-"
|
|
5539
|
+
f"{_report_run_seq(report_path) or '*'}.md`, where the seq is the "
|
|
5540
|
+
f"report's own — `final-report-{task_type}-<seq>` — not this run's "
|
|
5541
|
+
f"`workerResults` sequence)"
|
|
5542
|
+
+ _near_miss_clause(report_path, task_type)
|
|
5543
|
+
+ ". A vote the gate is computed from MUST trace back to a dispatch "
|
|
5519
5544
|
"that actually returned — otherwise the round can be skipped and "
|
|
5520
5545
|
'the gate still read `passed` (plan-body-verification.md §"Round '
|
|
5521
5546
|
'protocol" step 3).'
|
|
5522
5547
|
)
|
|
5523
5548
|
|
|
5524
5549
|
|
|
5550
|
+
def _near_miss_clause(report_path: Path, task_type: str) -> str:
|
|
5551
|
+
near = _plan_verify_seq_near_misses(report_path, task_type)
|
|
5552
|
+
if not near:
|
|
5553
|
+
return ""
|
|
5554
|
+
return (
|
|
5555
|
+
f"; the directory does hold {near}, which the seq filter excluded — "
|
|
5556
|
+
f"rename to this run's report seq rather than re-dispatching"
|
|
5557
|
+
)
|
|
5558
|
+
|
|
5559
|
+
|
|
5525
5560
|
_UNIFORM_VERIFIER_MIN_ITEMS = 5
|
|
5526
5561
|
|
|
5527
5562
|
|