okstra 0.169.0 → 0.169.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -11,6 +11,7 @@ so the rules live here rather than inside either one — the same split
11
11
  from __future__ import annotations
12
12
 
13
13
  import re
14
+ from dataclasses import dataclass
14
15
  from pathlib import Path
15
16
 
16
17
  from okstra_ctl.worker_prompt_headers import EVIDENCE_LEDGER_HEADER
@@ -148,26 +149,75 @@ def _ledger_path_hint(cited_path: str, read_paths: set[str]) -> str:
148
149
  )
149
150
 
150
151
 
151
- def _result_files(run_dir: Path, task_type: str, seq: str | None, worker: str | None):
152
+ @dataclass(frozen=True)
153
+ class WorkerResultName:
154
+ """The three fields a canonical worker-result basename carries."""
155
+
156
+ worker_role: str
157
+ task_type: str
158
+ seq: str
159
+
160
+
161
+ def parse_worker_result_name(basename: str) -> WorkerResultName | None:
162
+ """Split `<role>-worker-<task-type>-<seq>.md`, or None when non-canonical.
163
+
164
+ Callers that already hold one worker's result path — the dispatcher settling
165
+ that worker — read the audit check's arguments from here instead of
166
+ re-deriving them from the manifest. The `worker_role` this returns carries
167
+ the `-worker` suffix, which is what `check_worker_results_audit(worker=...)`
168
+ matches on; a bare provider id like `claude` matches nothing.
169
+ """
170
+ if "-audit-" in basename:
171
+ return None
172
+ match = _WORKER_RESULT_BASENAME_RE.match(basename)
173
+ if match is None:
174
+ return None
175
+ return WorkerResultName(
176
+ worker_role=match.group("worker"),
177
+ task_type=match.group("task_type"),
178
+ seq=match.group("seq"),
179
+ )
180
+
181
+
182
+ def normalize_worker_filter(worker: str | None) -> str | None:
183
+ """Accept a worker id the way every other okstra surface spells it.
184
+
185
+ Result files are named `<role>-worker-...`, so the filter matches
186
+ `claude-worker`. But `claude` is what a worker is called everywhere a lead
187
+ reads or types one — `--workers claude,codex`, `workerId`, the roster in the
188
+ profile — so `--worker claude` was the natural thing to pass, and it matched
189
+ nothing. A zero-match filter produces no failures, which the CLI reports as
190
+ `{"ok": true}` with exit 0: indistinguishable from a real pass, at exactly
191
+ the point `team-contract` tells the lead to run this check before deciding
192
+ whether to re-dispatch. Accepting both spellings removes the trap rather
193
+ than documenting it.
194
+ """
195
+ if worker is None or worker.endswith("-worker"):
196
+ return worker
197
+ return f"{worker}-worker"
198
+
199
+
200
+ def worker_result_files(
201
+ run_dir: Path, task_type: str, seq: str | None, worker: str | None
202
+ ):
152
203
  """Every worker-results file in *run_dir* this check owns, in name order."""
204
+ worker = normalize_worker_filter(worker)
153
205
  for path in sorted((run_dir / "worker-results").glob("*.md")):
154
- if "-audit-" in path.name:
155
- continue
156
- match = _WORKER_RESULT_BASENAME_RE.match(path.name)
206
+ match = parse_worker_result_name(path.name)
157
207
  if match is None:
158
208
  # Files that don't match the canonical pattern (e.g. ad-hoc notes
159
209
  # left by the operator) are out of contract scope.
160
210
  continue
161
- if match.group("task_type") != task_type:
211
+ if match.task_type != task_type:
162
212
  # Cross-phase artifacts shouldn't appear here; skip rather than
163
213
  # fail to keep the check focused on the current phase.
164
214
  continue
165
- if seq is not None and match.group("seq") != seq:
215
+ if seq is not None and match.seq != seq:
166
216
  # A prior run's artifact. Its contract was judged when it ran.
167
217
  continue
168
- if worker is not None and match.group("worker") != worker:
218
+ if worker is not None and match.worker_role != worker:
169
219
  continue
170
- yield path, match.group("worker"), match.group("seq")
220
+ yield path, match.worker_role, match.seq
171
221
 
172
222
 
173
223
  def check_worker_results_audit(
@@ -196,7 +246,7 @@ def check_worker_results_audit(
196
246
  # `release-handoff`, which is single-lead). Nothing to enforce.
197
247
  return failures
198
248
 
199
- for path, worker_role, result_seq in _result_files(run_dir, task_type, seq, worker):
249
+ for path, worker_role, result_seq in worker_result_files(run_dir, task_type, seq, worker):
200
250
  rel = path.name
201
251
  try:
202
252
  content = path.read_text()
@@ -230,7 +230,9 @@ def validate_reverify_prompt(
230
230
  read_scope_position < 0 or read_scope_position > boundary_position
231
231
  ):
232
232
  errors.append("phase boundary block must follow the reverify anchor headers")
233
- first_heading = re.search(r"(?m)^##\s+", normalized)
233
+ first_heading = re.compile(r"(?m)^##\s+").search(
234
+ normalized, _task_instructions_offset(normalized)
235
+ )
234
236
  if (
235
237
  boundary_position >= 0
236
238
  and first_heading is not None
@@ -240,6 +242,27 @@ def validate_reverify_prompt(
240
242
  return errors
241
243
 
242
244
 
245
+ def _task_instructions_offset(text: str) -> int:
246
+ """Where the lead's own instruction body starts.
247
+
248
+ The `agent-prompt` materializer composes every prompt as anchors →
249
+ `## Duty Contract` → `## Task Instructions`, and convergence's
250
+ materialization gate makes that the only body a reverify dispatch may send.
251
+ The duty section's heading is therefore always the document's first `##`,
252
+ which left the check below with no satisfiable input: the composer writes a
253
+ heading above anything the lead can author, so a whole-document "first
254
+ heading" test failed every materialized prompt regardless of where the lead
255
+ put the phase boundary.
256
+
257
+ These rules judge what the lead wrote, so they start where the lead's text
258
+ starts — the same region `_validate_model_header` already reads. A prompt
259
+ without the marker is judged whole.
260
+ """
261
+ marker = "\n\n## Task Instructions\n\n"
262
+ index = text.find(marker)
263
+ return 0 if index < 0 else index + len(marker)
264
+
265
+
243
266
  def _section_values(text: str, header: str) -> list[str]:
244
267
  lines = text.splitlines()
245
268
  values: list[str] = []
@@ -91,7 +91,7 @@ def worker_prompt_headers(
91
91
  project_root,
92
92
  audit_source_rel or result_rel,
93
93
  )
94
- errors_log_path = _errors_log_path(project_root, manifest, active_context)
94
+ errors_log_path = resolve_errors_log_path(project_root, manifest, active_context)
95
95
  errors_sidecar_path = _worker_errors_sidecar_path(
96
96
  project_root,
97
97
  manifest,
@@ -224,7 +224,7 @@ def _improvement_grilling_log_path(
224
224
  return path
225
225
 
226
226
 
227
- def _errors_log_path(
227
+ def resolve_errors_log_path(
228
228
  project_root: Path,
229
229
  manifest: Mapping[str, Any],
230
230
  active_context: Mapping[str, Any],