okstra 0.169.1 → 0.170.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/architecture.md +17 -1
- package/docs/cli.md +12 -2
- package/docs/for-ai/skills/okstra-setup.md +8 -0
- package/docs/project-structure-overview.md +3 -1
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/duties/acceptance-critic.md +25 -5
- package/runtime/prompts/duties/acceptance-verifier.md +25 -5
- package/runtime/prompts/duties/analysis-worker.md +25 -5
- package/runtime/prompts/duties/code-reviewer.md +25 -5
- package/runtime/prompts/duties/common.md +15 -11
- package/runtime/prompts/duties/diagnosis-worker.md +44 -0
- package/runtime/prompts/duties/discovery-worker.md +44 -0
- package/runtime/prompts/duties/implementation-executor.md +25 -5
- package/runtime/prompts/duties/implementation-verifier.md +25 -5
- package/runtime/prompts/duties/lead.md +25 -5
- package/runtime/prompts/duties/planning-worker.md +44 -0
- package/runtime/prompts/duties/report-writer.md +25 -5
- package/runtime/prompts/duties/reverification-worker.md +25 -5
- package/runtime/prompts/duties/schedule-verifier.md +25 -5
- package/runtime/prompts/duties/scope-critic.md +25 -5
- package/runtime/prompts/duties/translator.md +25 -5
- package/runtime/prompts/lead/plan-body-verification.md +7 -2
- package/runtime/prompts/lead/report-writer.md +1 -1
- package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -1
- package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
- package/runtime/prompts/profiles/final-verification.md +1 -1
- package/runtime/prompts/profiles/implementation-planning.md +2 -2
- package/runtime/python/okstra_ctl/agent_invocation.py +92 -3
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +30 -0
- package/runtime/python/okstra_ctl/cmux.py +36 -19
- package/runtime/python/okstra_ctl/dispatch_core.py +92 -22
- package/runtime/python/okstra_ctl/dispatch_state.py +143 -9
- package/runtime/python/okstra_ctl/doctor.py +31 -0
- package/runtime/python/okstra_ctl/plan_derivations.py +94 -0
- package/runtime/python/okstra_ctl/plan_items_cli.py +131 -4
- package/runtime/python/okstra_ctl/run.py +7 -1
- package/runtime/python/okstra_ctl/schema_excerpt.py +34 -0
- package/runtime/python/okstra_ctl/verdict_blocks.py +17 -0
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +12 -1
- package/runtime/python/okstra_project/resolver.py +34 -0
- package/runtime/schemas/final-report-v2.0.schema.json +5 -0
- package/runtime/skills/okstra-setup/references/project-config.md +38 -0
- package/runtime/validators/lib/fixtures.sh +9 -1
- package/runtime/validators/validate-run.py +108 -2
|
@@ -33,6 +33,11 @@ from .agent_invocation import (
|
|
|
33
33
|
agent_model_assignment_from_payload,
|
|
34
34
|
verify_agent_invocation,
|
|
35
35
|
)
|
|
36
|
+
from .final_report_paths import (
|
|
37
|
+
final_report_data_path,
|
|
38
|
+
final_report_markdown_path,
|
|
39
|
+
)
|
|
40
|
+
from .worker_prompt_body import REPORT_WRITER_WORKER_ID
|
|
36
41
|
from .worker_prompt_contract import (
|
|
37
42
|
PromptRecord,
|
|
38
43
|
validate_initial_prompt_records,
|
|
@@ -504,9 +509,20 @@ def link_agent_dispatch_result(
|
|
|
504
509
|
item for item in links
|
|
505
510
|
if isinstance(item, Mapping) and item.get("resultPath") == result_relative
|
|
506
511
|
]
|
|
507
|
-
|
|
512
|
+
# A superseded link is the record of a result the lead rejected, kept so
|
|
513
|
+
# the chain shows the corrective round happened rather than hiding it.
|
|
514
|
+
# It no longer owns the path, so the re-dispatch may claim it.
|
|
515
|
+
live_conflicts = [
|
|
516
|
+
item for item in same_result
|
|
517
|
+
if not item.get("supersededBy") and dict(item) != link
|
|
518
|
+
]
|
|
519
|
+
if live_conflicts:
|
|
508
520
|
raise DispatchError(
|
|
509
|
-
f"agent result is already linked to another dispatch:
|
|
521
|
+
f"agent result is already linked to another dispatch: "
|
|
522
|
+
f"{result_relative}. If that result was rejected and re-dispatched, "
|
|
523
|
+
f"record the rejection with `okstra agent-prompt reject-result "
|
|
524
|
+
f"--dispatch-id {live_conflicts[0].get('dispatchId')} "
|
|
525
|
+
f"--superseded-by {dispatch_id}` first"
|
|
510
526
|
)
|
|
511
527
|
if any(
|
|
512
528
|
isinstance(item, Mapping)
|
|
@@ -523,6 +539,69 @@ def link_agent_dispatch_result(
|
|
|
523
539
|
return link
|
|
524
540
|
|
|
525
541
|
|
|
542
|
+
def reject_agent_dispatch_result(
|
|
543
|
+
*,
|
|
544
|
+
project_root: Path,
|
|
545
|
+
run_manifest_path: Path,
|
|
546
|
+
dispatch_id: str,
|
|
547
|
+
superseded_by: str,
|
|
548
|
+
reason: str,
|
|
549
|
+
) -> dict[str, str]:
|
|
550
|
+
"""Record that a linked result was rejected and re-dispatched.
|
|
551
|
+
|
|
552
|
+
The contract tells the lead to re-dispatch a worker whose answer violated the
|
|
553
|
+
response format, with a correction paragraph appended. That path did not
|
|
554
|
+
finish: the prompt is immutable and already dispatched, so
|
|
555
|
+
`--replace-undispatched` refuses (correctly), and a fresh invocation id
|
|
556
|
+
produces a worker that writes the same result path — where `link-result`
|
|
557
|
+
refused because the path already belonged to the first dispatch. The worker
|
|
558
|
+
ran, wrote a good result, and the ledger could not accept it.
|
|
559
|
+
|
|
560
|
+
Rejection is a fact worth recording rather than routing around, so it is
|
|
561
|
+
written rather than allowed implicitly: the first link stays in the ledger
|
|
562
|
+
marked `supersededBy` with the lead's reason, and only then may the
|
|
563
|
+
re-dispatch claim the path. Nothing is deleted, so the chain still shows both
|
|
564
|
+
attempts and why the second exists.
|
|
565
|
+
"""
|
|
566
|
+
if not superseded_by.strip():
|
|
567
|
+
raise DispatchError("superseding dispatch ID is required")
|
|
568
|
+
if not reason.strip():
|
|
569
|
+
raise DispatchError(
|
|
570
|
+
"a rejection reason is required — the ledger has to say why a "
|
|
571
|
+
"returned result was not accepted"
|
|
572
|
+
)
|
|
573
|
+
project_root = project_root.resolve()
|
|
574
|
+
manifest_path = resolve_project_path(
|
|
575
|
+
project_root, str(run_manifest_path)
|
|
576
|
+
).resolve(strict=True)
|
|
577
|
+
manifest = load_json_object(manifest_path, "run manifest")
|
|
578
|
+
team_state_path = resolve_required_path(project_root, manifest, "teamStatePath")
|
|
579
|
+
with _team_state_lock(team_state_path):
|
|
580
|
+
team_state = load_json_object(team_state_path, "team-state")
|
|
581
|
+
links = team_state.get("agentResultLinks")
|
|
582
|
+
if not isinstance(links, list):
|
|
583
|
+
raise DispatchError("team-state agentResultLinks must be an array")
|
|
584
|
+
matches = [
|
|
585
|
+
item for item in links
|
|
586
|
+
if isinstance(item, Mapping) and item.get("dispatchId") == dispatch_id
|
|
587
|
+
]
|
|
588
|
+
if len(matches) != 1:
|
|
589
|
+
raise DispatchError(
|
|
590
|
+
f"expected exactly one agent result link for {dispatch_id}, "
|
|
591
|
+
f"found {len(matches)}"
|
|
592
|
+
)
|
|
593
|
+
row = matches[0]
|
|
594
|
+
if row.get("supersededBy"):
|
|
595
|
+
raise DispatchError(
|
|
596
|
+
f"agent result link is already superseded by "
|
|
597
|
+
f"{row['supersededBy']}: {dispatch_id}"
|
|
598
|
+
)
|
|
599
|
+
row["supersededBy"] = superseded_by
|
|
600
|
+
row["rejectionReason"] = reason
|
|
601
|
+
write_json(team_state_path, team_state)
|
|
602
|
+
return dict(row)
|
|
603
|
+
|
|
604
|
+
|
|
526
605
|
def _relative_project_path(project_root: Path, path: Path) -> str:
|
|
527
606
|
try:
|
|
528
607
|
return path.resolve(strict=False).relative_to(project_root).as_posix()
|
|
@@ -644,6 +723,54 @@ def missing_completion_paths(job: WorkerJob) -> tuple[Path, ...]:
|
|
|
644
723
|
return tuple(path for path in job.completion_paths if not path.is_file())
|
|
645
724
|
|
|
646
725
|
|
|
726
|
+
def dispatch_result_path(
|
|
727
|
+
worker_id: str,
|
|
728
|
+
worker_result_path: Path,
|
|
729
|
+
manifest: Mapping[str, Any],
|
|
730
|
+
project_root: Path,
|
|
731
|
+
) -> Path:
|
|
732
|
+
"""What `resultPath` means for this worker.
|
|
733
|
+
|
|
734
|
+
For everyone but the report writer it is the worker-result file. The report
|
|
735
|
+
writer authors three artifacts and its canonical result is the final-report
|
|
736
|
+
data.json, not its own `.md` pointer — a distinction that is not cosmetic:
|
|
737
|
+
`dispatch_core` hands `job.result_path` to report finalization as the data
|
|
738
|
+
path, so a `.md` there is parsed as JSON and a complete report settles
|
|
739
|
+
`error`.
|
|
740
|
+
|
|
741
|
+
This lives beside `worker_jobs_from_file` because both job constructors must
|
|
742
|
+
apply it. It was `dispatch_core`-private while the roster path applied it and
|
|
743
|
+
the `--jobs-file` path took whatever the file said, which is how the two
|
|
744
|
+
produced different jobs for the same worker.
|
|
745
|
+
"""
|
|
746
|
+
if worker_id != REPORT_WRITER_WORKER_ID:
|
|
747
|
+
return worker_result_path
|
|
748
|
+
return final_report_data_path(
|
|
749
|
+
resolve_required_path(project_root, manifest, "expectedReportPath")
|
|
750
|
+
)
|
|
751
|
+
|
|
752
|
+
|
|
753
|
+
def dispatch_completion_paths(
|
|
754
|
+
worker_id: str,
|
|
755
|
+
worker_result_path: Path,
|
|
756
|
+
manifest: Mapping[str, Any],
|
|
757
|
+
project_root: Path,
|
|
758
|
+
) -> tuple[Path, ...]:
|
|
759
|
+
"""Every artifact that must exist before this worker counts as done.
|
|
760
|
+
|
|
761
|
+
The report writer's three are the data.json, its rendered Markdown sibling,
|
|
762
|
+
and the worker-result pointer (`prompts/lead/report-writer.md` §"Completion
|
|
763
|
+
detection"). Same reason as `dispatch_result_path`: both constructors need
|
|
764
|
+
the same answer.
|
|
765
|
+
"""
|
|
766
|
+
if worker_id != REPORT_WRITER_WORKER_ID:
|
|
767
|
+
return (worker_result_path,)
|
|
768
|
+
data_json = final_report_data_path(
|
|
769
|
+
resolve_required_path(project_root, manifest, "expectedReportPath")
|
|
770
|
+
)
|
|
771
|
+
return (data_json, final_report_markdown_path(data_json), worker_result_path)
|
|
772
|
+
|
|
773
|
+
|
|
647
774
|
def validate_initial_prompts(
|
|
648
775
|
manifest: Mapping[str, Any], jobs: Sequence[WorkerJob]
|
|
649
776
|
) -> None:
|
|
@@ -798,6 +925,7 @@ def worker_jobs_from_file(
|
|
|
798
925
|
project_root: Path,
|
|
799
926
|
jobs_file: Path,
|
|
800
927
|
*,
|
|
928
|
+
manifest: Mapping[str, Any],
|
|
801
929
|
backend: str,
|
|
802
930
|
idle_timeout_seconds: int,
|
|
803
931
|
default_dispatch_kind: str,
|
|
@@ -816,6 +944,7 @@ def worker_jobs_from_file(
|
|
|
816
944
|
_worker_job_from_file(
|
|
817
945
|
project_root,
|
|
818
946
|
item,
|
|
947
|
+
manifest=manifest,
|
|
819
948
|
backend=backend,
|
|
820
949
|
idle_timeout_seconds=idle_timeout_seconds,
|
|
821
950
|
dispatch_kind=dispatch_kind,
|
|
@@ -831,6 +960,7 @@ def _worker_job_from_file(
|
|
|
831
960
|
project_root: Path,
|
|
832
961
|
item: Mapping[str, Any],
|
|
833
962
|
*,
|
|
963
|
+
manifest: Mapping[str, Any],
|
|
834
964
|
backend: str,
|
|
835
965
|
idle_timeout_seconds: int,
|
|
836
966
|
dispatch_kind: str,
|
|
@@ -842,16 +972,20 @@ def _worker_job_from_file(
|
|
|
842
972
|
prompt_path = resolve_project_path(
|
|
843
973
|
project_root, require_string(item, "promptPath")
|
|
844
974
|
)
|
|
845
|
-
result_path = resolve_project_path(
|
|
846
|
-
project_root, require_string(item, "resultPath")
|
|
847
|
-
)
|
|
848
975
|
worker_result_path = resolve_project_path(
|
|
849
976
|
project_root, require_string(item, "workerResultPath")
|
|
850
977
|
)
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
854
|
-
|
|
978
|
+
# The file's own `resultPath` / `completionPaths` are advisory: the same
|
|
979
|
+
# derivation the roster path applies decides them, so a jobs-file dispatch
|
|
980
|
+
# and a roster dispatch of one worker cannot disagree about which artifact
|
|
981
|
+
# is the result. A file that names the report writer's `.md` as its result
|
|
982
|
+
# used to reach report finalization as the data path.
|
|
983
|
+
result_path = dispatch_result_path(
|
|
984
|
+
worker_id, worker_result_path, manifest, project_root
|
|
985
|
+
)
|
|
986
|
+
completion_paths = dispatch_completion_paths(
|
|
987
|
+
worker_id, worker_result_path, manifest, project_root
|
|
988
|
+
)
|
|
855
989
|
digests = item.get("digests")
|
|
856
990
|
digest_values = digests if isinstance(digests, Mapping) else {}
|
|
857
991
|
host_model_value = item.get("hostModelValue")
|
|
@@ -9,6 +9,7 @@ from pathlib import Path
|
|
|
9
9
|
from typing import Iterable
|
|
10
10
|
|
|
11
11
|
from okstra_project import ResolverError, project_json_path, resolve_project_root
|
|
12
|
+
from okstra_project.resolver import resolve_review_rule_packs
|
|
12
13
|
|
|
13
14
|
from . import improvement_lenses, worktree_registry
|
|
14
15
|
from .models import provider_wrappers
|
|
@@ -99,6 +100,9 @@ def _project_checks(cwd: Path) -> tuple[Path | None, list[DoctorCheck]]:
|
|
|
99
100
|
architecture = _architecture_declaration_check(project_root)
|
|
100
101
|
if architecture is not None:
|
|
101
102
|
checks.append(architecture)
|
|
103
|
+
review_packs = _review_rule_pack_check(project_root)
|
|
104
|
+
if review_packs is not None:
|
|
105
|
+
checks.append(review_packs)
|
|
102
106
|
return project_root, checks
|
|
103
107
|
|
|
104
108
|
|
|
@@ -158,6 +162,33 @@ def _architecture_declaration_check(project_root: Path) -> DoctorCheck | None:
|
|
|
158
162
|
)
|
|
159
163
|
|
|
160
164
|
|
|
165
|
+
def _review_rule_pack_check(project_root: Path) -> DoctorCheck | None:
|
|
166
|
+
"""Report a declared review rule pack that no worker will be able to open.
|
|
167
|
+
|
|
168
|
+
`reviewRulePacks` is what makes a project's own review standard apply
|
|
169
|
+
without every brief citing it, so a stale path costs the whole pack in
|
|
170
|
+
silence: the phase records `project-review-rules: declared <path>
|
|
171
|
+
unreadable` at best, and a run that reviewed against nothing still passes.
|
|
172
|
+
Absent declaration stays quiet — brief-cited packs remain the other, equally
|
|
173
|
+
valid, channel.
|
|
174
|
+
|
|
175
|
+
What this cannot check is whether a pack that does resolve was actually read
|
|
176
|
+
and applied; that stays the worker's own `project-review-rules:` record.
|
|
177
|
+
"""
|
|
178
|
+
declared = resolve_review_rule_packs(project_root)
|
|
179
|
+
if not declared:
|
|
180
|
+
return None
|
|
181
|
+
missing = [path for path in declared if not Path(path).is_file()]
|
|
182
|
+
if not missing:
|
|
183
|
+
return _ok("review rule packs", f"{len(declared)} declared, all readable")
|
|
184
|
+
return _fail(
|
|
185
|
+
"review rule packs",
|
|
186
|
+
f"declared but not readable: {', '.join(missing)} — phases skip a pack "
|
|
187
|
+
"they cannot open, so the run reviews against fewer rules than declared. "
|
|
188
|
+
f"Fix the path in {project_json_path(project_root)} or drop the entry.",
|
|
189
|
+
)
|
|
190
|
+
|
|
191
|
+
|
|
161
192
|
def _profile_check(workspace: Path, phase: str) -> DoctorCheck:
|
|
162
193
|
profile = _profile_path(workspace, phase)
|
|
163
194
|
if profile.is_file():
|
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""Find the statements an answered clarification may have just falsified.
|
|
2
|
+
|
|
3
|
+
`_common-contract.md` §"Supersession" and `report-writer.md` §"Self-fix rewrite"
|
|
4
|
+
both say the same thing: an answer does not merely add a decision, it invalidates
|
|
5
|
+
whatever the plan wrote under the opposite assumption, so before editing you must
|
|
6
|
+
"grep the constant, the symbol, the path, the requirement ID across the whole
|
|
7
|
+
plan body". Both leave that grep to the author's diligence, and it is the step
|
|
8
|
+
that gets skipped — in one observed run, 17 of 23 blocked plan items were a
|
|
9
|
+
recorded decision whose derivations were never swept.
|
|
10
|
+
|
|
11
|
+
This does the grep. It is deliberately advisory: it returns candidate locations,
|
|
12
|
+
never a verdict about which ones are now false. Deciding that is the author's
|
|
13
|
+
job, and a tool that guessed would be trading one silent failure for another.
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import re
|
|
18
|
+
from typing import Any, Iterator, Mapping, Sequence
|
|
19
|
+
|
|
20
|
+
# A backticked span is how both contracts tell an author to write a symbol, a
|
|
21
|
+
# path, or a constant, so it is the highest-signal thing to extract. Bare prose
|
|
22
|
+
# words are deliberately not extracted: they match everywhere and would bury the
|
|
23
|
+
# hits that matter.
|
|
24
|
+
_BACKTICKED_RE = re.compile(r"`([^`\n]+)`")
|
|
25
|
+
# Ids the plan carries in its own rows (`R-001`, `P-Step-3`, `VC-002`) and the
|
|
26
|
+
# ticket ids the brief uses (`DEV-10174`, `PROD-1623`).
|
|
27
|
+
_ID_RE = re.compile(r"\b(?:[A-Z][A-Za-z]*-[A-Za-z0-9]+(?:-[A-Za-z0-9]+)*)\b")
|
|
28
|
+
# Below this length a token matches too much to be worth reading: `id`, `db`,
|
|
29
|
+
# and `ko` each hit dozens of unrelated rows.
|
|
30
|
+
_MIN_TOKEN_LENGTH = 3
|
|
31
|
+
_EXCERPT_MAX = 160
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def extract_tokens(text: str) -> list[str]:
|
|
35
|
+
"""The symbols, paths, and ids an answer names, longest first.
|
|
36
|
+
|
|
37
|
+
Longest first so a caller reading the report sees the specific token before
|
|
38
|
+
the general one it contains (`src/domains/font` before `src/domains`).
|
|
39
|
+
"""
|
|
40
|
+
found: set[str] = set()
|
|
41
|
+
for raw in _BACKTICKED_RE.findall(text):
|
|
42
|
+
token = raw.strip()
|
|
43
|
+
# A backticked sentence is prose in code font, not a symbol.
|
|
44
|
+
if len(token) >= _MIN_TOKEN_LENGTH and " " not in token:
|
|
45
|
+
found.add(token)
|
|
46
|
+
for token in _ID_RE.findall(text):
|
|
47
|
+
if len(token) >= _MIN_TOKEN_LENGTH:
|
|
48
|
+
found.add(token)
|
|
49
|
+
return sorted(found, key=lambda token: (-len(token), token))
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _string_leaves(node: Any, pointer: str = "") -> Iterator[tuple[str, str]]:
|
|
53
|
+
if isinstance(node, str):
|
|
54
|
+
yield pointer, node
|
|
55
|
+
elif isinstance(node, Mapping):
|
|
56
|
+
for key, value in node.items():
|
|
57
|
+
yield from _string_leaves(value, f"{pointer}/{key}")
|
|
58
|
+
elif isinstance(node, Sequence) and not isinstance(node, (str, bytes)):
|
|
59
|
+
for index, value in enumerate(node):
|
|
60
|
+
yield from _string_leaves(value, f"{pointer}/{index}")
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _excerpt(text: str, token: str) -> str:
|
|
64
|
+
index = text.find(token)
|
|
65
|
+
if index < 0:
|
|
66
|
+
return text[:_EXCERPT_MAX]
|
|
67
|
+
start = max(0, index - _EXCERPT_MAX // 3)
|
|
68
|
+
excerpt = text[start:start + _EXCERPT_MAX]
|
|
69
|
+
return ("…" if start else "") + excerpt + ("…" if len(text) > start + _EXCERPT_MAX else "")
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def find_derivations(
|
|
73
|
+
plan: Mapping[str, Any],
|
|
74
|
+
tokens: Sequence[str],
|
|
75
|
+
*,
|
|
76
|
+
exclude_pointer_prefix: str = "",
|
|
77
|
+
) -> list[dict[str, str]]:
|
|
78
|
+
"""Every string in *plan* that mentions one of *tokens*.
|
|
79
|
+
|
|
80
|
+
One hit per (pointer, token): a row naming two affected symbols is two things
|
|
81
|
+
to re-check, not one.
|
|
82
|
+
"""
|
|
83
|
+
hits: list[dict[str, str]] = []
|
|
84
|
+
for pointer, text in _string_leaves(plan):
|
|
85
|
+
if exclude_pointer_prefix and pointer.startswith(exclude_pointer_prefix):
|
|
86
|
+
continue
|
|
87
|
+
for token in tokens:
|
|
88
|
+
if token in text:
|
|
89
|
+
hits.append({
|
|
90
|
+
"token": token,
|
|
91
|
+
"pointer": pointer,
|
|
92
|
+
"excerpt": _excerpt(text, token),
|
|
93
|
+
})
|
|
94
|
+
return hits
|
|
@@ -14,8 +14,15 @@ import sys
|
|
|
14
14
|
from typing import Any
|
|
15
15
|
|
|
16
16
|
from .convergence_store import write_json_atomic
|
|
17
|
+
from .plan_derivations import extract_tokens, find_derivations
|
|
17
18
|
from .plan_items import PlanItemContractError, extract_plan_items
|
|
18
|
-
from .
|
|
19
|
+
from .user_response import parse_user_response_entries
|
|
20
|
+
from .verdict_blocks import (
|
|
21
|
+
PLAN_ITEM_VERDICTS,
|
|
22
|
+
VerdictBlock,
|
|
23
|
+
VerdictBlockError,
|
|
24
|
+
parse_verdict_blocks,
|
|
25
|
+
)
|
|
19
26
|
|
|
20
27
|
|
|
21
28
|
def _load_json_object(path: Path) -> dict[str, Any]:
|
|
@@ -64,12 +71,31 @@ def _parser() -> argparse.ArgumentParser:
|
|
|
64
71
|
help="one worker's plan-verify result file (repeatable)")
|
|
65
72
|
collect.add_argument("--items", type=Path, required=True)
|
|
66
73
|
collect.add_argument("--output", type=Path, required=True)
|
|
74
|
+
derivations = commands.add_parser(
|
|
75
|
+
"derivations",
|
|
76
|
+
help="list plan statements an answered clarification may have falsified",
|
|
77
|
+
)
|
|
78
|
+
derivations.add_argument("--data", type=Path, required=True)
|
|
79
|
+
derivations.add_argument("--response", type=Path, required=True,
|
|
80
|
+
help="the user-responses sidecar for this run")
|
|
81
|
+
derivations.add_argument("--clarification", default=None,
|
|
82
|
+
help="only this C-id (default: every answered one)")
|
|
83
|
+
seed = commands.add_parser(
|
|
84
|
+
"seed",
|
|
85
|
+
help="create the planBodyVerification.planItems[] rows a round lands in",
|
|
86
|
+
)
|
|
87
|
+
seed.add_argument("--data", type=Path, required=True)
|
|
67
88
|
apply_verdicts = commands.add_parser(
|
|
68
89
|
"apply-verdicts",
|
|
69
90
|
help="overwrite planBodyVerification.planItems[].verdicts in data.json",
|
|
70
91
|
)
|
|
71
92
|
apply_verdicts.add_argument("--data", type=Path, required=True)
|
|
72
93
|
apply_verdicts.add_argument("--verdicts", type=Path, required=True)
|
|
94
|
+
apply_verdicts.add_argument(
|
|
95
|
+
"--round", type=int, required=True, dest="round_number",
|
|
96
|
+
help="the verification round these verdicts were cast in; stamped on "
|
|
97
|
+
"every row so a later self-fix can be told from a current judgement",
|
|
98
|
+
)
|
|
73
99
|
return parser
|
|
74
100
|
|
|
75
101
|
|
|
@@ -112,8 +138,18 @@ def _assigned_item_ids(items_path: Path) -> list[str]:
|
|
|
112
138
|
|
|
113
139
|
def _verdict_row(worker: str, block: VerdictBlock) -> dict[str, Any]:
|
|
114
140
|
"""One `planItems[].verdicts[]` row. Optional fields stay absent when empty
|
|
115
|
-
so the recorded table shows what the worker actually said.
|
|
116
|
-
|
|
141
|
+
so the recorded table shows what the worker actually said.
|
|
142
|
+
|
|
143
|
+
The verdict crosses a vocabulary boundary here: a worker answers
|
|
144
|
+
`UNVERIFIABLE`, and the schema persists that as `verification-error`
|
|
145
|
+
(`PLAN_ITEM_VERDICTS`). Writing the worker's token straight through produced
|
|
146
|
+
a data.json its own schema rejects, and the mapping the contract prescribes
|
|
147
|
+
had to be applied by hand every round.
|
|
148
|
+
"""
|
|
149
|
+
row: dict[str, Any] = {
|
|
150
|
+
"worker": worker,
|
|
151
|
+
"verdict": PLAN_ITEM_VERDICTS[block.verdict],
|
|
152
|
+
}
|
|
117
153
|
for key, value in (
|
|
118
154
|
("breakageKind", block.breakage_kind),
|
|
119
155
|
("fixability", block.fixability),
|
|
@@ -194,6 +230,84 @@ def _plan_body_items(data: dict[str, Any], data_path: Path) -> list[dict[str, An
|
|
|
194
230
|
return items
|
|
195
231
|
|
|
196
232
|
|
|
233
|
+
def _derivations(args: argparse.Namespace) -> dict[str, Any]:
|
|
234
|
+
"""Candidate statements each answered clarification may have falsified.
|
|
235
|
+
|
|
236
|
+
Advisory by construction: it reports where a decision's subject is mentioned
|
|
237
|
+
and never which mentions are now wrong. Both contracts require the author to
|
|
238
|
+
enumerate before editing; this supplies the enumeration, which is the half
|
|
239
|
+
that was being skipped, and leaves the judgement where it belongs.
|
|
240
|
+
"""
|
|
241
|
+
try:
|
|
242
|
+
sidecar = args.response.read_text(encoding="utf-8")
|
|
243
|
+
except (OSError, UnicodeError) as exc:
|
|
244
|
+
raise PlanItemContractError(
|
|
245
|
+
f"cannot read user-response sidecar {args.response}: {exc}"
|
|
246
|
+
) from exc
|
|
247
|
+
planning = _planning(_load_json_object(args.data))
|
|
248
|
+
entries = [
|
|
249
|
+
entry for entry in parse_user_response_entries(sidecar)
|
|
250
|
+
if args.clarification is None or entry.response_id == args.clarification
|
|
251
|
+
]
|
|
252
|
+
if args.clarification is not None and not entries:
|
|
253
|
+
raise PlanItemContractError(
|
|
254
|
+
f"{args.response} has no response block for {args.clarification}"
|
|
255
|
+
)
|
|
256
|
+
clarifications = []
|
|
257
|
+
for entry in entries:
|
|
258
|
+
tokens = extract_tokens(f"{entry.value}\n{entry.rationale or ''}")
|
|
259
|
+
clarifications.append({
|
|
260
|
+
"id": entry.response_id,
|
|
261
|
+
"disposition": entry.disposition,
|
|
262
|
+
"tokens": tokens,
|
|
263
|
+
"candidates": find_derivations(planning, tokens),
|
|
264
|
+
})
|
|
265
|
+
return {
|
|
266
|
+
"ok": True,
|
|
267
|
+
"operation": "derivations",
|
|
268
|
+
"advisory": True,
|
|
269
|
+
"clarifications": clarifications,
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _seed(args: argparse.Namespace) -> dict[str, Any]:
|
|
274
|
+
"""Create the `planBodyVerification.planItems[]` rows a round lands in.
|
|
275
|
+
|
|
276
|
+
`apply-verdicts` refuses a verdict whose item has no row — correctly, since
|
|
277
|
+
the gate is re-derived from that table and a verdict with nowhere to land
|
|
278
|
+
would score as never cast. But nothing created the rows: the report writer
|
|
279
|
+
leaves `planItems: []` (§5.5.9 is a lead substep that runs after it), and
|
|
280
|
+
there was no step, in code or in the contract, that filled them. Every round
|
|
281
|
+
had to be hand-seeded before the CLI would accept its own output.
|
|
282
|
+
|
|
283
|
+
Idempotent by id. An existing row keeps everything it carries — verdicts
|
|
284
|
+
already applied, `carriedForwardFromSeq`, `selfFixNote` — because a re-seed
|
|
285
|
+
between rounds must not erase the round before it.
|
|
286
|
+
"""
|
|
287
|
+
extracted = _envelope(_load_json_object(args.data))["items"]
|
|
288
|
+
data = _load_json_object(args.data)
|
|
289
|
+
recorded = _plan_body_items(data, args.data)
|
|
290
|
+
known = {
|
|
291
|
+
item.get("id")
|
|
292
|
+
for item in recorded
|
|
293
|
+
if isinstance(item, Mapping)
|
|
294
|
+
}
|
|
295
|
+
added = [
|
|
296
|
+
{**item, "verdicts": []}
|
|
297
|
+
for item in extracted
|
|
298
|
+
if item["id"] not in known
|
|
299
|
+
]
|
|
300
|
+
recorded.extend(added)
|
|
301
|
+
write_json_atomic(args.data, data)
|
|
302
|
+
return {
|
|
303
|
+
"ok": True,
|
|
304
|
+
"operation": "seed",
|
|
305
|
+
"path": str(args.data),
|
|
306
|
+
"seeded": len(added),
|
|
307
|
+
"existing": len(known),
|
|
308
|
+
}
|
|
309
|
+
|
|
310
|
+
|
|
197
311
|
def _apply_verdicts(args: argparse.Namespace) -> dict[str, Any]:
|
|
198
312
|
data = _load_json_object(args.data)
|
|
199
313
|
incoming = _load_json_object(args.verdicts).get("planItems")
|
|
@@ -213,11 +327,22 @@ def _apply_verdicts(args: argparse.Namespace) -> dict[str, Any]:
|
|
|
213
327
|
f"gate is re-derived from that table, so a verdict with nowhere to "
|
|
214
328
|
f"land would be scored as if it were never cast"
|
|
215
329
|
)
|
|
330
|
+
if args.round_number < 1:
|
|
331
|
+
raise PlanItemContractError("--round must be 1 or greater")
|
|
216
332
|
for item in recorded:
|
|
217
333
|
if isinstance(item, Mapping) and item.get("id") in rows:
|
|
218
334
|
# Overwrite, never merge: the contract records one round at a time,
|
|
219
335
|
# and a merged table lets a previous round's votes keep voting.
|
|
220
|
-
|
|
336
|
+
#
|
|
337
|
+
# Each row carries the round it was cast in. A self-fix round
|
|
338
|
+
# rewrites the plan *after* a verification round, so an item left out
|
|
339
|
+
# of a later round keeps a verdict on text that has since changed —
|
|
340
|
+
# invisibly, because the gate reads the table without knowing any
|
|
341
|
+
# row's vintage. Stamping it here is what lets the validator tell a
|
|
342
|
+
# current judgement from one two rewrites old.
|
|
343
|
+
item["verdicts"] = [
|
|
344
|
+
{**row, "round": args.round_number} for row in rows[item["id"]]
|
|
345
|
+
]
|
|
221
346
|
write_json_atomic(args.data, data)
|
|
222
347
|
return {"ok": True, "operation": "apply-verdicts", "path": str(args.data)}
|
|
223
348
|
|
|
@@ -226,6 +351,8 @@ _HANDLERS = {
|
|
|
226
351
|
"extract": _extract,
|
|
227
352
|
"validate": _validate,
|
|
228
353
|
"collect-verdicts": _collect_verdicts,
|
|
354
|
+
"derivations": _derivations,
|
|
355
|
+
"seed": _seed,
|
|
229
356
|
"apply-verdicts": _apply_verdicts,
|
|
230
357
|
}
|
|
231
358
|
|
|
@@ -68,6 +68,7 @@ from .domain.host import (
|
|
|
68
68
|
ProviderUnavailable,
|
|
69
69
|
)
|
|
70
70
|
from .model_discovery import normalize_execution_for_dispatch
|
|
71
|
+
from .worker_prompt_policy import ANALYSIS_DUTY_BY_TASK_TYPE
|
|
71
72
|
from .models import (
|
|
72
73
|
ModelAssignment,
|
|
73
74
|
UnknownProviderError,
|
|
@@ -2523,7 +2524,12 @@ def _allowed_agent_audiences(
|
|
|
2523
2524
|
elif inp.task_type == "final-verification":
|
|
2524
2525
|
audiences.add("acceptance-verifier")
|
|
2525
2526
|
else:
|
|
2526
|
-
|
|
2527
|
+
# Same map the prompt policy resolves the duty from. Allowing a
|
|
2528
|
+
# different audience here than the one the policy will ask for makes
|
|
2529
|
+
# every worker prompt in the phase fail materialization.
|
|
2530
|
+
audiences.add(
|
|
2531
|
+
ANALYSIS_DUTY_BY_TASK_TYPE.get(inp.task_type, "analysis-worker")
|
|
2532
|
+
)
|
|
2527
2533
|
if models.critic_choice not in {"", "off"}:
|
|
2528
2534
|
audiences.update({"scope-critic", "acceptance-critic"})
|
|
2529
2535
|
return sorted(audiences)
|
|
@@ -23,6 +23,7 @@ from __future__ import annotations
|
|
|
23
23
|
|
|
24
24
|
import json
|
|
25
25
|
import re
|
|
26
|
+
from pathlib import Path
|
|
26
27
|
|
|
27
28
|
from .report_contract import TASK_TYPE_DATA_PROPERTY
|
|
28
29
|
|
|
@@ -78,6 +79,39 @@ def excerpt_cut_from_version(excerpt: dict) -> str:
|
|
|
78
79
|
return value if isinstance(value, str) else ""
|
|
79
80
|
|
|
80
81
|
|
|
82
|
+
def excerpt_version_skew(excerpt_path: Path, installed: str) -> str:
|
|
83
|
+
"""The version the bundle excerpt was cut from, when it is not *installed*.
|
|
84
|
+
|
|
85
|
+
Empty means no actionable skew: the file is absent or unreadable, carries no
|
|
86
|
+
stamp, or matches. A non-empty return is the older version, which the caller
|
|
87
|
+
turns into its own message.
|
|
88
|
+
|
|
89
|
+
The comparison used to live only in the renderer's error decorator, so it ran
|
|
90
|
+
in Phase 6 — after a worker had already authored a whole report against a
|
|
91
|
+
stale excerpt. The same two values are available much earlier, and the fix
|
|
92
|
+
(re-prepare the bundle) is the same either way.
|
|
93
|
+
"""
|
|
94
|
+
if not installed:
|
|
95
|
+
return ""
|
|
96
|
+
try:
|
|
97
|
+
excerpt = json.loads(excerpt_path.read_text(encoding="utf-8"))
|
|
98
|
+
except (OSError, json.JSONDecodeError):
|
|
99
|
+
return ""
|
|
100
|
+
if not isinstance(excerpt, dict):
|
|
101
|
+
return ""
|
|
102
|
+
cut_from = excerpt_cut_from_version(excerpt)
|
|
103
|
+
return cut_from if cut_from and cut_from != installed else ""
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def bundle_excerpt_path(start: Path) -> Path | None:
|
|
107
|
+
"""The task bundle's schema excerpt, found by walking up from *start*."""
|
|
108
|
+
for ancestor in Path(start).resolve().parents:
|
|
109
|
+
candidate = ancestor / "instruction-set" / "final-report-schema.json"
|
|
110
|
+
if candidate.is_file():
|
|
111
|
+
return candidate
|
|
112
|
+
return None
|
|
113
|
+
|
|
114
|
+
|
|
81
115
|
def build_schema_excerpt(schema: dict, task_type: str, cut_from_version: str = "") -> dict:
|
|
82
116
|
"""Return a task-type-scoped copy of *schema*.
|
|
83
117
|
|
|
@@ -38,6 +38,23 @@ ADVERSARIAL_VERDICTS = {
|
|
|
38
38
|
"UNVERIFIABLE": "unverifiable",
|
|
39
39
|
"VERIFICATION-ERROR": "verification-error",
|
|
40
40
|
}
|
|
41
|
+
# Plan-body verdicts persist in their own vocabulary, which is NOT the
|
|
42
|
+
# convergence one above: the schema keeps the first three tokens uppercase
|
|
43
|
+
# (`$defs/PlanBodyVerification/properties/planItems/items/properties/verdicts/
|
|
44
|
+
# items/properties/verdict`) and has no `UNVERIFIABLE` member at all.
|
|
45
|
+
# `plan-body-verification.md` §"Planning-time environment gap" states the
|
|
46
|
+
# mapping — "Recording `UNVERIFIABLE` is the honest outcome — it is persisted as
|
|
47
|
+
# `verification-error`" — and it lives here rather than inside the transcription
|
|
48
|
+
# CLI so the worker-facing token and the persisted token are decided in one
|
|
49
|
+
# place. Writing the raw token through put a value in data.json that its own
|
|
50
|
+
# schema rejects.
|
|
51
|
+
PLAN_ITEM_VERDICTS = {
|
|
52
|
+
"AGREE": "AGREE",
|
|
53
|
+
"DISAGREE": "DISAGREE",
|
|
54
|
+
"SUPPLEMENT": "SUPPLEMENT",
|
|
55
|
+
"UNVERIFIABLE": "verification-error",
|
|
56
|
+
"VERIFICATION-ERROR": "verification-error",
|
|
57
|
+
}
|
|
41
58
|
DISAGREE_BASES = frozenset({"counter-evidence", "burden-not-met"})
|
|
42
59
|
|
|
43
60
|
_ITEM_RE = re.compile(r"^###[ \t]+(?P<id>[^\s:]+)[ \t]*:?.*$", re.MULTILINE)
|
|
@@ -58,6 +58,17 @@ SUPPORTED_TASK_TYPES = frozenset({
|
|
|
58
58
|
"release-handoff",
|
|
59
59
|
*ANALYSIS_TASK_TYPES,
|
|
60
60
|
})
|
|
61
|
+
# One duty per analysis ROLE, not per phase. Phases that ask their worker for the
|
|
62
|
+
# same kind of judgement share a contract — requirements- and improvement-discovery
|
|
63
|
+
# both hand over candidates they do not start — and a phase whose worker decides
|
|
64
|
+
# something else gets its own. A task type absent from this map takes the
|
|
65
|
+
# observational default below: describe the area, do not design for it.
|
|
66
|
+
ANALYSIS_DUTY_BY_TASK_TYPE: dict[str, AgentAudience] = {
|
|
67
|
+
"requirements-discovery": "discovery-worker",
|
|
68
|
+
"improvement-discovery": "discovery-worker",
|
|
69
|
+
"error-analysis": "diagnosis-worker",
|
|
70
|
+
"implementation-planning": "planning-worker",
|
|
71
|
+
}
|
|
61
72
|
WORKER_PREAMBLE_FILENAME_BY_AUDIENCE = {
|
|
62
73
|
"analysis": "worker-prompt-preamble.md",
|
|
63
74
|
"implementation-executor": "implementation-worker-preamble.md",
|
|
@@ -113,7 +124,7 @@ def resolve_prompt_plan(
|
|
|
113
124
|
if dispatch_kind == "critic"
|
|
114
125
|
else "acceptance-verifier"
|
|
115
126
|
if task_type == "final-verification"
|
|
116
|
-
else "analysis-worker"
|
|
127
|
+
else ANALYSIS_DUTY_BY_TASK_TYPE.get(task_type, "analysis-worker")
|
|
117
128
|
)
|
|
118
129
|
if task_type == "implementation" and not executor_worker_id:
|
|
119
130
|
raise ValueError("implementation executor worker ID is required")
|