okstra 0.169.0 → 0.169.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/cli.md +1 -1
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/bin/okstra-error-log.py +38 -282
- package/runtime/prompts/lead/convergence.md +53 -7
- package/runtime/prompts/lead/okstra-lead-contract.md +1 -1
- package/runtime/python/okstra_ctl/agent_invocation.py +86 -6
- package/runtime/python/okstra_ctl/agent_prompt_cli.py +8 -0
- package/runtime/python/okstra_ctl/dispatch_core.py +226 -6
- package/runtime/python/okstra_ctl/error_log_write.py +308 -0
- package/runtime/python/okstra_ctl/worker_audit_check.py +26 -4
- package/runtime/python/okstra_ctl/worker_audit_ledger.py +59 -9
- package/runtime/python/okstra_ctl/worker_prompt_contract.py +24 -1
- package/runtime/python/okstra_ctl/worker_prompt_headers.py +2 -2
|
@@ -11,6 +11,7 @@ so the rules live here rather than inside either one — the same split
|
|
|
11
11
|
from __future__ import annotations
|
|
12
12
|
|
|
13
13
|
import re
|
|
14
|
+
from dataclasses import dataclass
|
|
14
15
|
from pathlib import Path
|
|
15
16
|
|
|
16
17
|
from okstra_ctl.worker_prompt_headers import EVIDENCE_LEDGER_HEADER
|
|
@@ -148,26 +149,75 @@ def _ledger_path_hint(cited_path: str, read_paths: set[str]) -> str:
|
|
|
148
149
|
)
|
|
149
150
|
|
|
150
151
|
|
|
151
|
-
|
|
152
|
+
@dataclass(frozen=True)
|
|
153
|
+
class WorkerResultName:
|
|
154
|
+
"""The three fields a canonical worker-result basename carries."""
|
|
155
|
+
|
|
156
|
+
worker_role: str
|
|
157
|
+
task_type: str
|
|
158
|
+
seq: str
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def parse_worker_result_name(basename: str) -> WorkerResultName | None:
|
|
162
|
+
"""Split `<role>-worker-<task-type>-<seq>.md`, or None when non-canonical.
|
|
163
|
+
|
|
164
|
+
Callers that already hold one worker's result path — the dispatcher settling
|
|
165
|
+
that worker — read the audit check's arguments from here instead of
|
|
166
|
+
re-deriving them from the manifest. The `worker_role` this returns carries
|
|
167
|
+
the `-worker` suffix, which is what `check_worker_results_audit(worker=...)`
|
|
168
|
+
matches on; a bare provider id like `claude` matches nothing.
|
|
169
|
+
"""
|
|
170
|
+
if "-audit-" in basename:
|
|
171
|
+
return None
|
|
172
|
+
match = _WORKER_RESULT_BASENAME_RE.match(basename)
|
|
173
|
+
if match is None:
|
|
174
|
+
return None
|
|
175
|
+
return WorkerResultName(
|
|
176
|
+
worker_role=match.group("worker"),
|
|
177
|
+
task_type=match.group("task_type"),
|
|
178
|
+
seq=match.group("seq"),
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
def normalize_worker_filter(worker: str | None) -> str | None:
|
|
183
|
+
"""Accept a worker id the way every other okstra surface spells it.
|
|
184
|
+
|
|
185
|
+
Result files are named `<role>-worker-...`, so the filter matches
|
|
186
|
+
`claude-worker`. But `claude` is what a worker is called everywhere a lead
|
|
187
|
+
reads or types one — `--workers claude,codex`, `workerId`, the roster in the
|
|
188
|
+
profile — so `--worker claude` was the natural thing to pass, and it matched
|
|
189
|
+
nothing. A zero-match filter produces no failures, which the CLI reports as
|
|
190
|
+
`{"ok": true}` with exit 0: indistinguishable from a real pass, at exactly
|
|
191
|
+
the point `team-contract` tells the lead to run this check before deciding
|
|
192
|
+
whether to re-dispatch. Accepting both spellings removes the trap rather
|
|
193
|
+
than documenting it.
|
|
194
|
+
"""
|
|
195
|
+
if worker is None or worker.endswith("-worker"):
|
|
196
|
+
return worker
|
|
197
|
+
return f"{worker}-worker"
|
|
198
|
+
|
|
199
|
+
|
|
200
|
+
def worker_result_files(
|
|
201
|
+
run_dir: Path, task_type: str, seq: str | None, worker: str | None
|
|
202
|
+
):
|
|
152
203
|
"""Every worker-results file in *run_dir* this check owns, in name order."""
|
|
204
|
+
worker = normalize_worker_filter(worker)
|
|
153
205
|
for path in sorted((run_dir / "worker-results").glob("*.md")):
|
|
154
|
-
|
|
155
|
-
continue
|
|
156
|
-
match = _WORKER_RESULT_BASENAME_RE.match(path.name)
|
|
206
|
+
match = parse_worker_result_name(path.name)
|
|
157
207
|
if match is None:
|
|
158
208
|
# Files that don't match the canonical pattern (e.g. ad-hoc notes
|
|
159
209
|
# left by the operator) are out of contract scope.
|
|
160
210
|
continue
|
|
161
|
-
if match.
|
|
211
|
+
if match.task_type != task_type:
|
|
162
212
|
# Cross-phase artifacts shouldn't appear here; skip rather than
|
|
163
213
|
# fail to keep the check focused on the current phase.
|
|
164
214
|
continue
|
|
165
|
-
if seq is not None and match.
|
|
215
|
+
if seq is not None and match.seq != seq:
|
|
166
216
|
# A prior run's artifact. Its contract was judged when it ran.
|
|
167
217
|
continue
|
|
168
|
-
if worker is not None and match.
|
|
218
|
+
if worker is not None and match.worker_role != worker:
|
|
169
219
|
continue
|
|
170
|
-
yield path, match.
|
|
220
|
+
yield path, match.worker_role, match.seq
|
|
171
221
|
|
|
172
222
|
|
|
173
223
|
def check_worker_results_audit(
|
|
@@ -196,7 +246,7 @@ def check_worker_results_audit(
|
|
|
196
246
|
# `release-handoff`, which is single-lead). Nothing to enforce.
|
|
197
247
|
return failures
|
|
198
248
|
|
|
199
|
-
for path, worker_role, result_seq in
|
|
249
|
+
for path, worker_role, result_seq in worker_result_files(run_dir, task_type, seq, worker):
|
|
200
250
|
rel = path.name
|
|
201
251
|
try:
|
|
202
252
|
content = path.read_text()
|
|
@@ -230,7 +230,9 @@ def validate_reverify_prompt(
|
|
|
230
230
|
read_scope_position < 0 or read_scope_position > boundary_position
|
|
231
231
|
):
|
|
232
232
|
errors.append("phase boundary block must follow the reverify anchor headers")
|
|
233
|
-
first_heading = re.
|
|
233
|
+
first_heading = re.compile(r"(?m)^##\s+").search(
|
|
234
|
+
normalized, _task_instructions_offset(normalized)
|
|
235
|
+
)
|
|
234
236
|
if (
|
|
235
237
|
boundary_position >= 0
|
|
236
238
|
and first_heading is not None
|
|
@@ -240,6 +242,27 @@ def validate_reverify_prompt(
|
|
|
240
242
|
return errors
|
|
241
243
|
|
|
242
244
|
|
|
245
|
+
def _task_instructions_offset(text: str) -> int:
|
|
246
|
+
"""Where the lead's own instruction body starts.
|
|
247
|
+
|
|
248
|
+
The `agent-prompt` materializer composes every prompt as anchors →
|
|
249
|
+
`## Duty Contract` → `## Task Instructions`, and convergence's
|
|
250
|
+
materialization gate makes that the only body a reverify dispatch may send.
|
|
251
|
+
The duty section's heading is therefore always the document's first `##`,
|
|
252
|
+
which left the check below with no satisfiable input: the composer writes a
|
|
253
|
+
heading above anything the lead can author, so a whole-document "first
|
|
254
|
+
heading" test failed every materialized prompt regardless of where the lead
|
|
255
|
+
put the phase boundary.
|
|
256
|
+
|
|
257
|
+
These rules judge what the lead wrote, so they start where the lead's text
|
|
258
|
+
starts — the same region `_validate_model_header` already reads. A prompt
|
|
259
|
+
without the marker is judged whole.
|
|
260
|
+
"""
|
|
261
|
+
marker = "\n\n## Task Instructions\n\n"
|
|
262
|
+
index = text.find(marker)
|
|
263
|
+
return 0 if index < 0 else index + len(marker)
|
|
264
|
+
|
|
265
|
+
|
|
243
266
|
def _section_values(text: str, header: str) -> list[str]:
|
|
244
267
|
lines = text.splitlines()
|
|
245
268
|
values: list[str] = []
|
|
@@ -91,7 +91,7 @@ def worker_prompt_headers(
|
|
|
91
91
|
project_root,
|
|
92
92
|
audit_source_rel or result_rel,
|
|
93
93
|
)
|
|
94
|
-
errors_log_path =
|
|
94
|
+
errors_log_path = resolve_errors_log_path(project_root, manifest, active_context)
|
|
95
95
|
errors_sidecar_path = _worker_errors_sidecar_path(
|
|
96
96
|
project_root,
|
|
97
97
|
manifest,
|
|
@@ -224,7 +224,7 @@ def _improvement_grilling_log_path(
|
|
|
224
224
|
return path
|
|
225
225
|
|
|
226
226
|
|
|
227
|
-
def
|
|
227
|
+
def resolve_errors_log_path(
|
|
228
228
|
project_root: Path,
|
|
229
229
|
manifest: Mapping[str, Any],
|
|
230
230
|
active_context: Mapping[str, Any],
|