okstra 0.169.0 → 0.170.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. package/docs/architecture.md +17 -1
  2. package/docs/cli.md +11 -1
  3. package/docs/for-ai/skills/okstra-setup.md +8 -0
  4. package/docs/project-structure-overview.md +3 -1
  5. package/package.json +1 -1
  6. package/runtime/BUILD.json +2 -2
  7. package/runtime/bin/okstra-error-log.py +38 -282
  8. package/runtime/prompts/duties/acceptance-critic.md +25 -5
  9. package/runtime/prompts/duties/acceptance-verifier.md +25 -5
  10. package/runtime/prompts/duties/analysis-worker.md +25 -5
  11. package/runtime/prompts/duties/code-reviewer.md +25 -5
  12. package/runtime/prompts/duties/common.md +15 -11
  13. package/runtime/prompts/duties/diagnosis-worker.md +44 -0
  14. package/runtime/prompts/duties/discovery-worker.md +44 -0
  15. package/runtime/prompts/duties/implementation-executor.md +25 -5
  16. package/runtime/prompts/duties/implementation-verifier.md +25 -5
  17. package/runtime/prompts/duties/lead.md +25 -5
  18. package/runtime/prompts/duties/planning-worker.md +44 -0
  19. package/runtime/prompts/duties/report-writer.md +25 -5
  20. package/runtime/prompts/duties/reverification-worker.md +25 -5
  21. package/runtime/prompts/duties/schedule-verifier.md +25 -5
  22. package/runtime/prompts/duties/scope-critic.md +25 -5
  23. package/runtime/prompts/duties/translator.md +25 -5
  24. package/runtime/prompts/lead/convergence.md +53 -7
  25. package/runtime/prompts/lead/okstra-lead-contract.md +1 -1
  26. package/runtime/prompts/lead/plan-body-verification.md +5 -1
  27. package/runtime/prompts/lead/report-writer.md +1 -1
  28. package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -1
  29. package/runtime/prompts/profiles/_implementation-verifier.md +1 -1
  30. package/runtime/prompts/profiles/final-verification.md +1 -1
  31. package/runtime/prompts/profiles/implementation-planning.md +2 -2
  32. package/runtime/python/okstra_ctl/agent_invocation.py +146 -6
  33. package/runtime/python/okstra_ctl/agent_prompt_cli.py +38 -0
  34. package/runtime/python/okstra_ctl/cmux.py +36 -19
  35. package/runtime/python/okstra_ctl/dispatch_core.py +317 -27
  36. package/runtime/python/okstra_ctl/dispatch_state.py +143 -9
  37. package/runtime/python/okstra_ctl/doctor.py +31 -0
  38. package/runtime/python/okstra_ctl/error_log_write.py +308 -0
  39. package/runtime/python/okstra_ctl/plan_derivations.py +94 -0
  40. package/runtime/python/okstra_ctl/plan_items_cli.py +114 -3
  41. package/runtime/python/okstra_ctl/run.py +7 -1
  42. package/runtime/python/okstra_ctl/schema_excerpt.py +34 -0
  43. package/runtime/python/okstra_ctl/verdict_blocks.py +17 -0
  44. package/runtime/python/okstra_ctl/worker_audit_check.py +26 -4
  45. package/runtime/python/okstra_ctl/worker_audit_ledger.py +59 -9
  46. package/runtime/python/okstra_ctl/worker_prompt_contract.py +24 -1
  47. package/runtime/python/okstra_ctl/worker_prompt_headers.py +2 -2
  48. package/runtime/python/okstra_ctl/worker_prompt_policy.py +12 -1
  49. package/runtime/python/okstra_project/resolver.py +34 -0
  50. package/runtime/skills/okstra-setup/references/project-config.md +38 -0
  51. package/runtime/validators/lib/fixtures.sh +9 -1
  52. package/runtime/validators/validate-run.py +37 -2
@@ -33,6 +33,11 @@ from .agent_invocation import (
33
33
  agent_model_assignment_from_payload,
34
34
  verify_agent_invocation,
35
35
  )
36
+ from .final_report_paths import (
37
+ final_report_data_path,
38
+ final_report_markdown_path,
39
+ )
40
+ from .worker_prompt_body import REPORT_WRITER_WORKER_ID
36
41
  from .worker_prompt_contract import (
37
42
  PromptRecord,
38
43
  validate_initial_prompt_records,
@@ -504,9 +509,20 @@ def link_agent_dispatch_result(
504
509
  item for item in links
505
510
  if isinstance(item, Mapping) and item.get("resultPath") == result_relative
506
511
  ]
507
- if same_result and any(dict(item) != link for item in same_result):
512
+ # A superseded link is the record of a result the lead rejected, kept so
513
+ # the chain shows the corrective round happened rather than hiding it.
514
+ # It no longer owns the path, so the re-dispatch may claim it.
515
+ live_conflicts = [
516
+ item for item in same_result
517
+ if not item.get("supersededBy") and dict(item) != link
518
+ ]
519
+ if live_conflicts:
508
520
  raise DispatchError(
509
- f"agent result is already linked to another dispatch: {result_relative}"
521
+ f"agent result is already linked to another dispatch: "
522
+ f"{result_relative}. If that result was rejected and re-dispatched, "
523
+ f"record the rejection with `okstra agent-prompt reject-result "
524
+ f"--dispatch-id {live_conflicts[0].get('dispatchId')} "
525
+ f"--superseded-by {dispatch_id}` first"
510
526
  )
511
527
  if any(
512
528
  isinstance(item, Mapping)
@@ -523,6 +539,69 @@ def link_agent_dispatch_result(
523
539
  return link
524
540
 
525
541
 
542
+ def reject_agent_dispatch_result(
543
+ *,
544
+ project_root: Path,
545
+ run_manifest_path: Path,
546
+ dispatch_id: str,
547
+ superseded_by: str,
548
+ reason: str,
549
+ ) -> dict[str, str]:
550
+ """Record that a linked result was rejected and re-dispatched.
551
+
552
+ The contract tells the lead to re-dispatch a worker whose answer violated the
553
+ response format, with a correction paragraph appended. That path did not
554
+ finish: the prompt is immutable and already dispatched, so
555
+ `--replace-undispatched` refuses (correctly), and a fresh invocation id
556
+ produces a worker that writes the same result path — where `link-result`
557
+ refused because the path already belonged to the first dispatch. The worker
558
+ ran, wrote a good result, and the ledger could not accept it.
559
+
560
+ Rejection is a fact worth recording rather than routing around, so it is
561
+ written rather than allowed implicitly: the first link stays in the ledger
562
+ marked `supersededBy` with the lead's reason, and only then may the
563
+ re-dispatch claim the path. Nothing is deleted, so the chain still shows both
564
+ attempts and why the second exists.
565
+ """
566
+ if not superseded_by.strip():
567
+ raise DispatchError("superseding dispatch ID is required")
568
+ if not reason.strip():
569
+ raise DispatchError(
570
+ "a rejection reason is required — the ledger has to say why a "
571
+ "returned result was not accepted"
572
+ )
573
+ project_root = project_root.resolve()
574
+ manifest_path = resolve_project_path(
575
+ project_root, str(run_manifest_path)
576
+ ).resolve(strict=True)
577
+ manifest = load_json_object(manifest_path, "run manifest")
578
+ team_state_path = resolve_required_path(project_root, manifest, "teamStatePath")
579
+ with _team_state_lock(team_state_path):
580
+ team_state = load_json_object(team_state_path, "team-state")
581
+ links = team_state.get("agentResultLinks")
582
+ if not isinstance(links, list):
583
+ raise DispatchError("team-state agentResultLinks must be an array")
584
+ matches = [
585
+ item for item in links
586
+ if isinstance(item, Mapping) and item.get("dispatchId") == dispatch_id
587
+ ]
588
+ if len(matches) != 1:
589
+ raise DispatchError(
590
+ f"expected exactly one agent result link for {dispatch_id}, "
591
+ f"found {len(matches)}"
592
+ )
593
+ row = matches[0]
594
+ if row.get("supersededBy"):
595
+ raise DispatchError(
596
+ f"agent result link is already superseded by "
597
+ f"{row['supersededBy']}: {dispatch_id}"
598
+ )
599
+ row["supersededBy"] = superseded_by
600
+ row["rejectionReason"] = reason
601
+ write_json(team_state_path, team_state)
602
+ return dict(row)
603
+
604
+
526
605
  def _relative_project_path(project_root: Path, path: Path) -> str:
527
606
  try:
528
607
  return path.resolve(strict=False).relative_to(project_root).as_posix()
@@ -644,6 +723,54 @@ def missing_completion_paths(job: WorkerJob) -> tuple[Path, ...]:
644
723
  return tuple(path for path in job.completion_paths if not path.is_file())
645
724
 
646
725
 
726
+ def dispatch_result_path(
727
+ worker_id: str,
728
+ worker_result_path: Path,
729
+ manifest: Mapping[str, Any],
730
+ project_root: Path,
731
+ ) -> Path:
732
+ """What `resultPath` means for this worker.
733
+
734
+ For everyone but the report writer it is the worker-result file. The report
735
+ writer authors three artifacts and its canonical result is the final-report
736
+ data.json, not its own `.md` pointer — a distinction that is not cosmetic:
737
+ `dispatch_core` hands `job.result_path` to report finalization as the data
738
+ path, so a `.md` there is parsed as JSON and a complete report settles
739
+ `error`.
740
+
741
+ This lives beside `worker_jobs_from_file` because both job constructors must
742
+ apply it. It was `dispatch_core`-private while the roster path applied it and
743
+ the `--jobs-file` path took whatever the file said, which is how the two
744
+ produced different jobs for the same worker.
745
+ """
746
+ if worker_id != REPORT_WRITER_WORKER_ID:
747
+ return worker_result_path
748
+ return final_report_data_path(
749
+ resolve_required_path(project_root, manifest, "expectedReportPath")
750
+ )
751
+
752
+
753
+ def dispatch_completion_paths(
754
+ worker_id: str,
755
+ worker_result_path: Path,
756
+ manifest: Mapping[str, Any],
757
+ project_root: Path,
758
+ ) -> tuple[Path, ...]:
759
+ """Every artifact that must exist before this worker counts as done.
760
+
761
+ The report writer's three are the data.json, its rendered Markdown sibling,
762
+ and the worker-result pointer (`prompts/lead/report-writer.md` §"Completion
763
+ detection"). Same reason as `dispatch_result_path`: both constructors need
764
+ the same answer.
765
+ """
766
+ if worker_id != REPORT_WRITER_WORKER_ID:
767
+ return (worker_result_path,)
768
+ data_json = final_report_data_path(
769
+ resolve_required_path(project_root, manifest, "expectedReportPath")
770
+ )
771
+ return (data_json, final_report_markdown_path(data_json), worker_result_path)
772
+
773
+
647
774
  def validate_initial_prompts(
648
775
  manifest: Mapping[str, Any], jobs: Sequence[WorkerJob]
649
776
  ) -> None:
@@ -798,6 +925,7 @@ def worker_jobs_from_file(
798
925
  project_root: Path,
799
926
  jobs_file: Path,
800
927
  *,
928
+ manifest: Mapping[str, Any],
801
929
  backend: str,
802
930
  idle_timeout_seconds: int,
803
931
  default_dispatch_kind: str,
@@ -816,6 +944,7 @@ def worker_jobs_from_file(
816
944
  _worker_job_from_file(
817
945
  project_root,
818
946
  item,
947
+ manifest=manifest,
819
948
  backend=backend,
820
949
  idle_timeout_seconds=idle_timeout_seconds,
821
950
  dispatch_kind=dispatch_kind,
@@ -831,6 +960,7 @@ def _worker_job_from_file(
831
960
  project_root: Path,
832
961
  item: Mapping[str, Any],
833
962
  *,
963
+ manifest: Mapping[str, Any],
834
964
  backend: str,
835
965
  idle_timeout_seconds: int,
836
966
  dispatch_kind: str,
@@ -842,16 +972,20 @@ def _worker_job_from_file(
842
972
  prompt_path = resolve_project_path(
843
973
  project_root, require_string(item, "promptPath")
844
974
  )
845
- result_path = resolve_project_path(
846
- project_root, require_string(item, "resultPath")
847
- )
848
975
  worker_result_path = resolve_project_path(
849
976
  project_root, require_string(item, "workerResultPath")
850
977
  )
851
- completion_paths = tuple(
852
- resolve_project_path(project_root, path)
853
- for path in string_list(item.get("completionPaths"))
854
- ) or (result_path,)
978
+ # The file's own `resultPath` / `completionPaths` are advisory: the same
979
+ # derivation the roster path applies decides them, so a jobs-file dispatch
980
+ # and a roster dispatch of one worker cannot disagree about which artifact
981
+ # is the result. A file that names the report writer's `.md` as its result
982
+ # used to reach report finalization as the data path.
983
+ result_path = dispatch_result_path(
984
+ worker_id, worker_result_path, manifest, project_root
985
+ )
986
+ completion_paths = dispatch_completion_paths(
987
+ worker_id, worker_result_path, manifest, project_root
988
+ )
855
989
  digests = item.get("digests")
856
990
  digest_values = digests if isinstance(digests, Mapping) else {}
857
991
  host_model_value = item.get("hostModelValue")
@@ -9,6 +9,7 @@ from pathlib import Path
9
9
  from typing import Iterable
10
10
 
11
11
  from okstra_project import ResolverError, project_json_path, resolve_project_root
12
+ from okstra_project.resolver import resolve_review_rule_packs
12
13
 
13
14
  from . import improvement_lenses, worktree_registry
14
15
  from .models import provider_wrappers
@@ -99,6 +100,9 @@ def _project_checks(cwd: Path) -> tuple[Path | None, list[DoctorCheck]]:
99
100
  architecture = _architecture_declaration_check(project_root)
100
101
  if architecture is not None:
101
102
  checks.append(architecture)
103
+ review_packs = _review_rule_pack_check(project_root)
104
+ if review_packs is not None:
105
+ checks.append(review_packs)
102
106
  return project_root, checks
103
107
 
104
108
 
@@ -158,6 +162,33 @@ def _architecture_declaration_check(project_root: Path) -> DoctorCheck | None:
158
162
  )
159
163
 
160
164
 
165
+ def _review_rule_pack_check(project_root: Path) -> DoctorCheck | None:
166
+ """Report a declared review rule pack that no worker will be able to open.
167
+
168
+ `reviewRulePacks` is what makes a project's own review standard apply
169
+ without every brief citing it, so a stale path costs the whole pack in
170
+ silence: the phase records `project-review-rules: declared <path>
171
+ unreadable` at best, and a run that reviewed against nothing still passes.
172
+ Absent declaration stays quiet — brief-cited packs remain the other, equally
173
+ valid, channel.
174
+
175
+ What this cannot check is whether a pack that does resolve was actually read
176
+ and applied; that stays the worker's own `project-review-rules:` record.
177
+ """
178
+ declared = resolve_review_rule_packs(project_root)
179
+ if not declared:
180
+ return None
181
+ missing = [path for path in declared if not Path(path).is_file()]
182
+ if not missing:
183
+ return _ok("review rule packs", f"{len(declared)} declared, all readable")
184
+ return _fail(
185
+ "review rule packs",
186
+ f"declared but not readable: {', '.join(missing)} — phases skip a pack "
187
+ "they cannot open, so the run reviews against fewer rules than declared. "
188
+ f"Fix the path in {project_json_path(project_root)} or drop the entry.",
189
+ )
190
+
191
+
161
192
  def _profile_check(workspace: Path, phase: str) -> DoctorCheck:
162
193
  profile = _profile_path(workspace, phase)
163
194
  if profile.is_file():
@@ -0,0 +1,308 @@
1
+ """Writer core for runs/<task-type>/logs/errors-<task-type>-<seq>.jsonl.
2
+
3
+ Every error record in a run log is appended through this module. It lives here
4
+ rather than inside `scripts/okstra-error-log.py` because the CLI is no longer the
5
+ only caller: the deterministic dispatcher records a worker wrapper's non-zero
6
+ exit as a `cli-failure` in-process (`dispatch_core`), and the contract sentence
7
+ that says it does is only true while both paths share one writer. The script
8
+ keeps the argparse surface and re-exports these names.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import datetime as dt
13
+ import json
14
+ import os
15
+ from pathlib import Path
16
+
17
+ from .models import provider_ids
18
+
19
+ STDERR_EXCERPT_MAX_BYTES = 2048
20
+ TRUNCATION_SUFFIX = "...[truncated]"
21
+ PIPE_BUF_BYTES = 4096
22
+
23
+ ALLOWED_ERROR_TYPES = {"tool-failure", "cli-failure", "contract-violation"}
24
+ # Derived, not listed. The hand-written set had drifted two providers behind the
25
+ # registry: `grok` and `kimi` ship worker definitions and can be dispatched, but
26
+ # their agent names were absent, so every error they reported was rejected at
27
+ # the argument parser — silently, for anyone who did not read the exit code.
28
+ # The registry is where a provider is added, so it is where this follows from.
29
+ ALLOWED_AGENTS = (
30
+ {f"{provider}-worker" for provider in provider_ids("analyser")}
31
+ # The report writer is not an analyser and is named without the suffix,
32
+ # matching `REPORT_WRITER_WORKER_ID`.
33
+ | {"report-writer"}
34
+ # Lead identities come from the selected host adapter rather than the
35
+ # provider registry; `adapters/hosts/*/relay.md` names the value to pass.
36
+ | {"claude-lead"}
37
+ )
38
+ ALLOWED_AGENT_ROLES = {"lead", "worker", "report-writer"}
39
+ SUPPORTED_SIDECAR_SCHEMA_VERSIONS = {1}
40
+
41
+ ALLOWED_CAUSES = {
42
+ "sandbox-denied", "service-unavailable", "auth-failed", "unknown",
43
+ }
44
+ # A `sandbox-denied` claim is only admissible with both probes attached.
45
+ CAUSE_EVIDENCE_FIELDS = ("targetProbe", "controlProbe")
46
+ # Both probes share a record's PIPE_BUF_BYTES budget with stderrExcerpt, so
47
+ # they cannot reuse the 2048 cap that assumes stderrExcerpt owns it alone.
48
+ CAUSE_PROBE_MAX_BYTES = 256
49
+ # Backstop vocabulary: scanned in `message` only — never in stderrExcerpt,
50
+ # where a kernel's real "Operation not permitted" is legitimate content.
51
+ # Deliberately excludes bare "blocked"/"blocks": everyday English that would
52
+ # reject honest records like "test blocked on upstream dependency".
53
+ _BLOCKING_CLAIM_TERMS = (
54
+ "sandbox", "not permitted", "permission denied", "eperm",
55
+ )
56
+ # The backstop targets *unclassified* blocking claims. A worker that declared
57
+ # a specific cause has already done the honest work — `auth-failed` legitimately
58
+ # reads "permission denied" (MySQL 1045).
59
+ _UNCLASSIFIED_CAUSES = (None, "unknown")
60
+
61
+
62
+ def _now_utc():
63
+ return dt.datetime.now(dt.timezone.utc)
64
+
65
+
66
+ def _iso(t):
67
+ return t.isoformat()
68
+
69
+
70
+ def _truncate_utf8(s, limit):
71
+ """Truncate to `limit` bytes without splitting a multibyte character."""
72
+ if s is None:
73
+ return None
74
+ encoded = s.encode("utf-8")
75
+ if len(encoded) <= limit:
76
+ return s
77
+ cut = encoded[:limit]
78
+ while cut:
79
+ try:
80
+ return cut.decode("utf-8") + TRUNCATION_SUFFIX
81
+ except UnicodeDecodeError:
82
+ cut = cut[:-1]
83
+ return TRUNCATION_SUFFIX
84
+
85
+
86
+ def truncate_stderr(s):
87
+ """Truncate stderr text to STDERR_EXCERPT_MAX_BYTES, multibyte-safe."""
88
+ return _truncate_utf8(s, STDERR_EXCERPT_MAX_BYTES)
89
+
90
+
91
+ def normalize_cause_context(context, *, message):
92
+ """Validate a record's cause claim and return the normalized context.
93
+
94
+ Raises ValueError on three conditions:
95
+ - `cause` is set to a value outside ALLOWED_CAUSES;
96
+ - `cause` is 'sandbox-denied' but the two probes are missing or blank —
97
+ those probes are what distinguish a real denial from an unreachable
98
+ or auth-gated target, the misdiagnosis this gate exists to stop;
99
+ - `message` asserts a block in prose while the record left its cause
100
+ unclassified, which would smuggle the same claim past the gate.
101
+ """
102
+ cause = context.get("cause") if isinstance(context, dict) else None
103
+
104
+ if cause is not None and cause not in ALLOWED_CAUSES:
105
+ raise ValueError(
106
+ f"invalid cause: {cause!r} (allowed: {sorted(ALLOWED_CAUSES)})"
107
+ )
108
+
109
+ if cause == "sandbox-denied":
110
+ evidence = context.get("causeEvidence")
111
+ if not isinstance(evidence, dict):
112
+ raise ValueError(
113
+ "cause 'sandbox-denied' requires context.causeEvidence with "
114
+ f"{list(CAUSE_EVIDENCE_FIELDS)}"
115
+ )
116
+ normalized_evidence = {}
117
+ for field in CAUSE_EVIDENCE_FIELDS:
118
+ value = evidence.get(field)
119
+ if not isinstance(value, str) or not value.strip():
120
+ raise ValueError(
121
+ f"cause 'sandbox-denied' requires a non-empty "
122
+ f"context.causeEvidence.{field}: record the command and "
123
+ f"its raw output that proves the claim"
124
+ )
125
+ normalized_evidence[field] = _truncate_utf8(
126
+ value, CAUSE_PROBE_MAX_BYTES
127
+ )
128
+ return {**context, "causeEvidence": normalized_evidence}
129
+
130
+ if message and cause in _UNCLASSIFIED_CAUSES:
131
+ lowered = message.lower()
132
+ hit = next((t for t in _BLOCKING_CLAIM_TERMS if t in lowered), None)
133
+ if hit:
134
+ raise ValueError(
135
+ f"message asserts a blocking claim ({hit!r}) without "
136
+ "context.cause='sandbox-denied' + context.causeEvidence. "
137
+ "Either attach the two probes, or state the cause you "
138
+ "actually verified."
139
+ )
140
+
141
+ return context
142
+
143
+
144
+ def append_jsonl_line(path, record):
145
+ """Append a single JSON record as one line to ``path``.
146
+
147
+ Atomicity guarantee (POSIX only):
148
+ With ``O_APPEND`` and a single ``write()`` syscall, the kernel
149
+ appends the entire payload as one indivisible operation as long as
150
+ the payload size is at most ``PIPE_BUF`` (4096 bytes on Linux and
151
+ macOS). Larger payloads may be split across syscalls and interleave
152
+ with concurrent writers, so this helper rejects them with
153
+ ``ValueError`` rather than silently losing atomicity.
154
+
155
+ The atomicity contract holds only on POSIX filesystems with O_APPEND
156
+ semantics. Concurrent writers using ``O_TRUNC``, ``unlink``, or
157
+ non-append modes against the same path break the contract and are
158
+ out of scope for this helper.
159
+
160
+ Caller responsibilities:
161
+ - Keep records small (this module's stderr excerpt cap of
162
+ ``STDERR_EXCERPT_MAX_BYTES`` exists to keep records well under
163
+ ``PIPE_BUF_BYTES``).
164
+ - Handle ``TypeError`` from ``json.dumps`` for non-serializable values.
165
+
166
+ Creates parent directories as needed.
167
+ """
168
+ p = Path(path)
169
+ p.parent.mkdir(parents=True, exist_ok=True)
170
+ # ensure_ascii=False keeps UTF-8 compact (no \uXXXX escapes).
171
+ # json.dumps escapes literal newlines inside string values, so the
172
+ # only unescaped newline is the record separator we append below.
173
+ line = json.dumps(record, ensure_ascii=False, separators=(",", ":")) + "\n"
174
+ data = line.encode("utf-8")
175
+ if len(data) > PIPE_BUF_BYTES:
176
+ raise ValueError(
177
+ f"record too large for atomic append: {len(data)} bytes > "
178
+ f"PIPE_BUF ({PIPE_BUF_BYTES})"
179
+ )
180
+ # mode 0o644: owner read/write, group/world read-only.
181
+ fd = os.open(str(p), os.O_WRONLY | os.O_CREAT | os.O_APPEND, 0o644)
182
+ try:
183
+ os.write(fd, data)
184
+ finally:
185
+ os.close(fd)
186
+
187
+
188
+ def append_observed(
189
+ *,
190
+ out_path,
191
+ task_key,
192
+ phase,
193
+ agent,
194
+ agent_role,
195
+ model,
196
+ error_type,
197
+ command,
198
+ command_kind,
199
+ exit_code,
200
+ duration_ms,
201
+ message,
202
+ stderr_excerpt,
203
+ context,
204
+ now=None,
205
+ ):
206
+ """Append a lead-observed error event to errors.jsonl."""
207
+ if error_type not in ALLOWED_ERROR_TYPES:
208
+ raise ValueError(f"invalid errorType: {error_type!r}")
209
+ if agent not in ALLOWED_AGENTS:
210
+ raise ValueError(f"invalid agent: {agent!r}")
211
+ if agent_role not in ALLOWED_AGENT_ROLES:
212
+ raise ValueError(f"invalid agentRole: {agent_role!r}")
213
+ # Runs before append_jsonl_line so a rejected claim leaves no trace in the
214
+ # log: a written-then-flagged record is still a record someone can cite.
215
+ context = normalize_cause_context(context, message=message)
216
+ ts = _iso(now or _now_utc())
217
+ rec = {
218
+ "ts": ts,
219
+ "recordedAt": ts,
220
+ "taskKey": task_key,
221
+ "phase": str(phase),
222
+ "agent": agent,
223
+ "agentRole": agent_role,
224
+ "model": model,
225
+ "source": "lead-observed",
226
+ "errorType": error_type,
227
+ "command": command,
228
+ "commandKind": command_kind,
229
+ "exitCode": exit_code,
230
+ "durationMs": duration_ms,
231
+ "message": message,
232
+ "stderrExcerpt": truncate_stderr(stderr_excerpt),
233
+ "context": context,
234
+ }
235
+ append_jsonl_line(out_path, rec)
236
+ return rec
237
+
238
+
239
+ def dump_from_worker_sidecar(
240
+ *,
241
+ sidecar_path,
242
+ out_path,
243
+ task_key,
244
+ agent,
245
+ agent_role,
246
+ model,
247
+ now=None,
248
+ ):
249
+ """Read worker sidecar errors[] and append each to errors.jsonl with
250
+ Lead-side metadata filled in. Returns number of records appended.
251
+
252
+ Raises ValueError if:
253
+ - ``agent`` or ``agent_role`` is not in the allow-lists
254
+ - sidecar ``schemaVersion`` is not in ``SUPPORTED_SIDECAR_SCHEMA_VERSIONS``
255
+ - any entry's ``errorType`` is not in ``ALLOWED_ERROR_TYPES``
256
+ - any entry asserts a blocking cause without its required evidence
257
+
258
+ Returns 0 (no-op) if the sidecar file does not exist or its
259
+ ``errors`` list is empty.
260
+
261
+ Partial-failure semantics: entries are validated and appended in
262
+ order. If entry N fails validation, entries 0..N-1 have already
263
+ been written to ``out_path`` and are NOT rolled back. Callers that
264
+ require atomicity must validate the sidecar payload before invoking
265
+ this function.
266
+ """
267
+ if agent not in ALLOWED_AGENTS:
268
+ raise ValueError(f"invalid agent: {agent!r}")
269
+ if agent_role not in ALLOWED_AGENT_ROLES:
270
+ raise ValueError(f"invalid agentRole: {agent_role!r}")
271
+ p = Path(sidecar_path)
272
+ if not p.exists():
273
+ return 0
274
+ payload = json.loads(p.read_text())
275
+ schema = payload.get("schemaVersion")
276
+ if schema not in SUPPORTED_SIDECAR_SCHEMA_VERSIONS:
277
+ raise ValueError(f"unsupported sidecar schemaVersion: {schema!r}")
278
+ entries = payload.get("errors") or []
279
+ recorded_at = _iso(now or _now_utc())
280
+ count = 0
281
+ for e in entries:
282
+ et = e.get("errorType")
283
+ if et not in ALLOWED_ERROR_TYPES:
284
+ raise ValueError(f"invalid errorType in sidecar: {et!r}")
285
+ entry_context = normalize_cause_context(
286
+ e.get("context"), message=e.get("message")
287
+ )
288
+ rec = {
289
+ "ts": e.get("ts"),
290
+ "recordedAt": recorded_at,
291
+ "taskKey": task_key,
292
+ "phase": str(e.get("phase")) if e.get("phase") is not None else None,
293
+ "agent": agent,
294
+ "agentRole": agent_role,
295
+ "model": model,
296
+ "source": "worker-reported",
297
+ "errorType": et,
298
+ "command": e.get("command"),
299
+ "commandKind": e.get("commandKind"),
300
+ "exitCode": e.get("exitCode"),
301
+ "durationMs": e.get("durationMs"),
302
+ "message": e.get("message"),
303
+ "stderrExcerpt": truncate_stderr(e.get("stderrExcerpt")),
304
+ "context": entry_context,
305
+ }
306
+ append_jsonl_line(out_path, rec)
307
+ count += 1
308
+ return count
@@ -0,0 +1,94 @@
1
+ """Find the statements an answered clarification may have just falsified.
2
+
3
+ `_common-contract.md` §"Supersession" and `report-writer.md` §"Self-fix rewrite"
4
+ both say the same thing: an answer does not merely add a decision, it invalidates
5
+ whatever the plan wrote under the opposite assumption, so before editing you must
6
+ "grep the constant, the symbol, the path, the requirement ID across the whole
7
+ plan body". Both leave that grep to the author's diligence, and it is the step
8
+ that gets skipped — in one observed run, 17 of 23 blocked plan items were a
9
+ recorded decision whose derivations were never swept.
10
+
11
+ This does the grep. It is deliberately advisory: it returns candidate locations,
12
+ never a verdict about which ones are now false. Deciding that is the author's
13
+ job, and a tool that guessed would be trading one silent failure for another.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import re
18
+ from typing import Any, Iterator, Mapping, Sequence
19
+
20
+ # A backticked span is how both contracts tell an author to write a symbol, a
21
+ # path, or a constant, so it is the highest-signal thing to extract. Bare prose
22
+ # words are deliberately not extracted: they match everywhere and would bury the
23
+ # hits that matter.
24
+ _BACKTICKED_RE = re.compile(r"`([^`\n]+)`")
25
+ # Ids the plan carries in its own rows (`R-001`, `P-Step-3`, `VC-002`) and the
26
+ # ticket ids the brief uses (`DEV-10174`, `PROD-1623`).
27
+ _ID_RE = re.compile(r"\b(?:[A-Z][A-Za-z]*-[A-Za-z0-9]+(?:-[A-Za-z0-9]+)*)\b")
28
+ # Below this length a token matches too much to be worth reading: `id`, `db`,
29
+ # and `ko` each hit dozens of unrelated rows.
30
+ _MIN_TOKEN_LENGTH = 3
31
+ _EXCERPT_MAX = 160
32
+
33
+
34
+ def extract_tokens(text: str) -> list[str]:
35
+ """The symbols, paths, and ids an answer names, longest first.
36
+
37
+ Longest first so a caller reading the report sees the specific token before
38
+ the general one it contains (`src/domains/font` before `src/domains`).
39
+ """
40
+ found: set[str] = set()
41
+ for raw in _BACKTICKED_RE.findall(text):
42
+ token = raw.strip()
43
+ # A backticked sentence is prose in code font, not a symbol.
44
+ if len(token) >= _MIN_TOKEN_LENGTH and " " not in token:
45
+ found.add(token)
46
+ for token in _ID_RE.findall(text):
47
+ if len(token) >= _MIN_TOKEN_LENGTH:
48
+ found.add(token)
49
+ return sorted(found, key=lambda token: (-len(token), token))
50
+
51
+
52
+ def _string_leaves(node: Any, pointer: str = "") -> Iterator[tuple[str, str]]:
53
+ if isinstance(node, str):
54
+ yield pointer, node
55
+ elif isinstance(node, Mapping):
56
+ for key, value in node.items():
57
+ yield from _string_leaves(value, f"{pointer}/{key}")
58
+ elif isinstance(node, Sequence) and not isinstance(node, (str, bytes)):
59
+ for index, value in enumerate(node):
60
+ yield from _string_leaves(value, f"{pointer}/{index}")
61
+
62
+
63
+ def _excerpt(text: str, token: str) -> str:
64
+ index = text.find(token)
65
+ if index < 0:
66
+ return text[:_EXCERPT_MAX]
67
+ start = max(0, index - _EXCERPT_MAX // 3)
68
+ excerpt = text[start:start + _EXCERPT_MAX]
69
+ return ("…" if start else "") + excerpt + ("…" if len(text) > start + _EXCERPT_MAX else "")
70
+
71
+
72
+ def find_derivations(
73
+ plan: Mapping[str, Any],
74
+ tokens: Sequence[str],
75
+ *,
76
+ exclude_pointer_prefix: str = "",
77
+ ) -> list[dict[str, str]]:
78
+ """Every string in *plan* that mentions one of *tokens*.
79
+
80
+ One hit per (pointer, token): a row naming two affected symbols is two things
81
+ to re-check, not one.
82
+ """
83
+ hits: list[dict[str, str]] = []
84
+ for pointer, text in _string_leaves(plan):
85
+ if exclude_pointer_prefix and pointer.startswith(exclude_pointer_prefix):
86
+ continue
87
+ for token in tokens:
88
+ if token in text:
89
+ hits.append({
90
+ "token": token,
91
+ "pointer": pointer,
92
+ "excerpt": _excerpt(text, token),
93
+ })
94
+ return hits