okstra 0.168.0 → 0.169.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (101) hide show
  1. package/README.md +5 -4
  2. package/docs/architecture/storage-model.md +57 -1
  3. package/docs/architecture.md +70 -2
  4. package/docs/cli.md +8 -4
  5. package/docs/for-ai/skills/okstra-code-review.md +3 -2
  6. package/docs/for-ai/skills/okstra-schedule-gen.md +3 -1
  7. package/docs/project-structure-overview.md +14 -11
  8. package/package.json +1 -1
  9. package/runtime/BUILD.json +2 -2
  10. package/runtime/agents/workers/claude-worker.md +6 -5
  11. package/runtime/agents/workers/report-writer-worker.md +9 -4
  12. package/runtime/agents/workers/translator-worker.md +6 -4
  13. package/runtime/bin/okstra-error-log.py +38 -282
  14. package/runtime/prompts/duties/acceptance-critic.md +24 -0
  15. package/runtime/prompts/duties/acceptance-verifier.md +24 -0
  16. package/runtime/prompts/duties/analysis-worker.md +24 -0
  17. package/runtime/prompts/duties/code-reviewer.md +24 -0
  18. package/runtime/prompts/duties/common.md +35 -0
  19. package/runtime/prompts/duties/implementation-executor.md +24 -0
  20. package/runtime/prompts/duties/implementation-verifier.md +24 -0
  21. package/runtime/prompts/duties/lead.md +24 -0
  22. package/runtime/prompts/duties/report-writer.md +24 -0
  23. package/runtime/prompts/duties/reverification-worker.md +24 -0
  24. package/runtime/prompts/duties/schedule-verifier.md +24 -0
  25. package/runtime/prompts/duties/scope-critic.md +24 -0
  26. package/runtime/prompts/duties/translator.md +24 -0
  27. package/runtime/prompts/lead/convergence.md +104 -14
  28. package/runtime/prompts/lead/okstra-lead-contract.md +11 -21
  29. package/runtime/prompts/lead/plan-body-verification.md +16 -1
  30. package/runtime/prompts/lead/report-writer.md +20 -5
  31. package/runtime/prompts/lead/team-contract.md +13 -13
  32. package/runtime/prompts/profiles/_coding-conventions-preflight.md +1 -1
  33. package/runtime/prompts/profiles/_implementation-diff-review.md +1 -1
  34. package/runtime/prompts/profiles/_implementation-executor.md +1 -1
  35. package/runtime/prompts/profiles/implementation.md +4 -2
  36. package/runtime/python/okstra_ctl/adapters/hosts/antigravity/adapter.py +6 -0
  37. package/runtime/python/okstra_ctl/adapters/hosts/antigravity/relay.md +3 -2
  38. package/runtime/python/okstra_ctl/adapters/hosts/capability_adapter.py +8 -0
  39. package/runtime/python/okstra_ctl/adapters/hosts/claude-code/adapter.py +33 -0
  40. package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +13 -12
  41. package/runtime/python/okstra_ctl/adapters/hosts/codex/adapter.py +6 -0
  42. package/runtime/python/okstra_ctl/adapters/hosts/codex/relay.md +3 -2
  43. package/runtime/python/okstra_ctl/adapters/hosts/external/adapter.py +2 -0
  44. package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +3 -3
  45. package/runtime/python/okstra_ctl/adapters/hosts/grok/adapter.py +6 -0
  46. package/runtime/python/okstra_ctl/adapters/hosts/grok/relay.md +2 -1
  47. package/runtime/python/okstra_ctl/adapters/hosts/kimi/adapter.py +6 -0
  48. package/runtime/python/okstra_ctl/adapters/hosts/kimi/relay.md +2 -1
  49. package/runtime/python/okstra_ctl/agent_invocation.py +1582 -0
  50. package/runtime/python/okstra_ctl/agent_prompt_cli.py +796 -0
  51. package/runtime/python/okstra_ctl/codex_dispatch.py +2 -107
  52. package/runtime/python/okstra_ctl/context_cost.py +46 -5
  53. package/runtime/python/okstra_ctl/dispatch_core.py +538 -43
  54. package/runtime/python/okstra_ctl/dispatch_state.py +461 -36
  55. package/runtime/python/okstra_ctl/doctor.py +90 -16
  56. package/runtime/python/okstra_ctl/entrypoints/hosts.py +87 -9
  57. package/runtime/python/okstra_ctl/error_log_write.py +308 -0
  58. package/runtime/python/okstra_ctl/initial_prompt_materialization.py +214 -23
  59. package/runtime/python/okstra_ctl/path_hints.py +26 -0
  60. package/runtime/python/okstra_ctl/paths.py +20 -0
  61. package/runtime/python/okstra_ctl/ports/__init__.py +8 -0
  62. package/runtime/python/okstra_ctl/ports/host.py +3 -0
  63. package/runtime/python/okstra_ctl/ports/host_model.py +60 -0
  64. package/runtime/python/okstra_ctl/registry/host_registry.py +5 -0
  65. package/runtime/python/okstra_ctl/render.py +217 -12
  66. package/runtime/python/okstra_ctl/report_finalize.py +44 -0
  67. package/runtime/python/okstra_ctl/run.py +368 -51
  68. package/runtime/python/okstra_ctl/session.py +16 -12
  69. package/runtime/python/okstra_ctl/team.py +11 -11
  70. package/runtime/python/okstra_ctl/worker_audit_check.py +26 -4
  71. package/runtime/python/okstra_ctl/worker_audit_ledger.py +59 -9
  72. package/runtime/python/okstra_ctl/worker_dispatch.py +104 -0
  73. package/runtime/python/okstra_ctl/worker_prompt_body.py +5 -38
  74. package/runtime/python/okstra_ctl/worker_prompt_contract.py +56 -3
  75. package/runtime/python/okstra_ctl/worker_prompt_headers.py +2 -2
  76. package/runtime/python/okstra_ctl/worker_prompt_policy.py +38 -1
  77. package/runtime/skills/okstra-code-review/SKILL.md +22 -3
  78. package/runtime/skills/okstra-run/SKILL.md +16 -1
  79. package/runtime/skills/okstra-schedule-gen/SKILL.md +15 -1
  80. package/runtime/templates/implementation-worker-preamble.md +0 -10
  81. package/runtime/templates/report-writer-prompt-preamble.md +0 -9
  82. package/runtime/templates/reports/settings.template.json +0 -11
  83. package/runtime/templates/worker-prompt-preamble.md +0 -10
  84. package/runtime/validators/lib/fixtures.sh +93 -0
  85. package/runtime/validators/lib/validate-assets.sh +0 -8
  86. package/runtime/validators/validate-run.py +182 -0
  87. package/src/cli-registry.mjs +14 -0
  88. package/src/commands/execute/agent-prompt.mjs +25 -0
  89. package/src/commands/execute/codex-dispatch.mjs +6 -63
  90. package/src/commands/execute/worker-dispatch.mjs +76 -0
  91. package/src/commands/lifecycle/doctor.mjs +18 -3
  92. package/src/commands/lifecycle/install.mjs +33 -15
  93. package/src/commands/lifecycle/uninstall.mjs +4 -3
  94. package/src/lib/install-assets.mjs +9 -0
  95. package/runtime/agents/workers/antigravity-worker.md +0 -259
  96. package/runtime/agents/workers/codex-worker.md +0 -259
  97. package/runtime/agents/workers/grok-worker.md +0 -259
  98. package/runtime/agents/workers/kimi-worker.md +0 -259
  99. package/runtime/prompts/coding-preflight/scripts/preedit-check.sh +0 -79
  100. package/runtime/templates/operating-standard.md +0 -22
  101. package/src/lib/worker-agent-render.mjs +0 -50
@@ -16,19 +16,22 @@ from .dispatch_state import (
16
16
  BACKEND_CMUX_PANE,
17
17
  BACKEND_MIXED,
18
18
  BACKEND_TMUX_PANE,
19
+ append_worker_dispatch as _append_worker_dispatch,
19
20
  DispatchError,
20
21
  WorkerJob,
21
22
  dispatch_mode as _dispatch_mode,
22
- dispatch_record_matches,
23
23
  LIVENESS_AUDIT_HEARTBEAT,
24
24
  LIVENESS_WRAPPER_STATUS,
25
25
  load_json_object as _load_json_object,
26
+ link_agent_dispatch_result as _link_agent_dispatch_result,
26
27
  missing_completion_paths as _missing_completion_paths,
28
+ mutate_team_state as _mutate_team_state,
27
29
  require_string as _require_string,
28
30
  resolve_project_path as _resolve_project_path,
29
31
  resolve_required_path as _resolve_required_path,
30
32
  record_dispatch_facts as _record_dispatch_facts,
31
33
  transition_worker_status as _transition_worker_status,
34
+ update_worker_dispatch_status as _update_worker_dispatch_status,
32
35
  string_list as _string_list,
33
36
  string_value as _string_value,
34
37
  TEARDOWN_BEFORE_TERMINAL_REASON,
@@ -37,12 +40,12 @@ from .dispatch_state import (
37
40
  worker_jobs_from_file as _worker_jobs_from_file,
38
41
  worker_state as _worker_state,
39
42
  worktree_path as _worktree_path,
40
- write_json as _write_json,
41
43
  )
42
44
  from .final_report_paths import (
43
45
  final_report_data_path as _final_report_data_path,
44
46
  final_report_markdown_path as _final_report_markdown_path,
45
47
  )
48
+ from .error_log_write import append_observed
46
49
  from .lead_events import LeadEvent, append_lead_event
47
50
  from .initial_prompt_materialization import (
48
51
  InitialPromptMaterializationError,
@@ -58,13 +61,27 @@ from .report_finalize import (
58
61
  FinalizeError,
59
62
  run_finalize,
60
63
  )
64
+ from .worker_audit_ledger import (
65
+ check_worker_results_audit,
66
+ parse_worker_result_name,
67
+ )
61
68
  from .worker_prompt_body import REPORT_WRITER_WORKER_ID
69
+ from .worker_prompt_headers import (
70
+ WorkerPromptHeaderError,
71
+ resolve_errors_log_path,
72
+ )
62
73
  from .worker_artifact_paths import audit_sidecar_rel
63
74
  from .wrapper_status import read_wrapper_status, status_path_for_prompt
64
75
 
65
76
 
66
77
  MAX_WORKER_ATTEMPTS = 2
67
78
  TERMINAL_DISPATCH_STATUSES = {"completed", "timeout", "error", "not-run"}
79
+ # What the error log records for a wrapper the dispatcher timed out, matching
80
+ # the value `team-contract` prescribes for a polling-cap termination.
81
+ _WRAPPER_TIMEOUT_EXIT_CODE = 124
82
+ # The excerpt shares one atomic PIPE_BUF append with the rest of the record, so
83
+ # it is capped far below the writer's own 2048-byte stderr limit.
84
+ _WRAPPER_LOG_TAIL_BYTES = 800
68
85
 
69
86
 
70
87
  @dataclass(frozen=True)
@@ -74,6 +91,7 @@ class WorkerHandle:
74
91
  completed_process: subprocess.CompletedProcess[str] | None
75
92
  status_sidecar_path: Path | None
76
93
  degraded_from: str
94
+ running_process: subprocess.Popen[str] | None = None
77
95
 
78
96
 
79
97
  @dataclass(frozen=True)
@@ -174,7 +192,6 @@ def build_dispatch_plan(
174
192
  )
175
193
  if jobs_file:
176
194
  jobs = _jobs_from_file(project_root, workspace_root, jobs_file, options)
177
- _validate_dispatch_prompts(manifest, active_context, jobs)
178
195
  else:
179
196
  jobs = _jobs_from_roster(
180
197
  project_root,
@@ -186,6 +203,7 @@ def build_dispatch_plan(
186
203
  requested_workers,
187
204
  options,
188
205
  )
206
+ _validate_dispatch_prompts(manifest, active_context, jobs)
189
207
  return DispatchPlan(
190
208
  project_root=project_root,
191
209
  workspace_root=workspace_root.resolve(),
@@ -199,6 +217,7 @@ def build_dispatch_plan(
199
217
 
200
218
 
201
219
  def dispatch_plan(plan: DispatchPlan, *, wait: bool = True) -> int:
220
+ _validate_report_writer_isolation(plan.jobs)
202
221
  if wait:
203
222
  pane_backends = sorted(
204
223
  {
@@ -225,6 +244,161 @@ def dispatch_plan(plan: DispatchPlan, *, wait: bool = True) -> int:
225
244
  return 0
226
245
 
227
246
 
247
+ def dispatch_cli_wrapper_plan(plan: DispatchPlan) -> int:
248
+ """Start one dependency-free CLI batch before collecting any worker."""
249
+ if any(job.backend != BACKEND_CLI_WRAPPER for job in plan.jobs):
250
+ raise DispatchError("concurrent CLI dispatch requires cli-wrapper jobs only")
251
+ _validate_report_writer_isolation(plan.jobs)
252
+ _record_dispatch_facts(plan.team_state_path, _dispatch_mode(plan.jobs))
253
+ final_codes = _dispatch_cli_wrapper_batch(plan, plan.jobs)
254
+ return next((final_codes[job.worker_id] for job in plan.jobs
255
+ if final_codes.get(job.worker_id, 0) != 0), 0)
256
+
257
+
258
+ def _validate_report_writer_isolation(jobs: Sequence[WorkerJob]) -> None:
259
+ if (
260
+ any(job.worker_id == REPORT_WRITER_WORKER_ID for job in jobs)
261
+ and len(jobs) > 1
262
+ ):
263
+ raise DispatchError(
264
+ "report-writer must run in a separate Phase 6 dispatch after "
265
+ "analysis and convergence"
266
+ )
267
+
268
+
269
+ def _dispatch_cli_wrapper_batch(
270
+ plan: DispatchPlan,
271
+ jobs: Sequence[WorkerJob],
272
+ ) -> dict[str, int]:
273
+ """Run one concurrent batch through all retries before the next dependency."""
274
+ pending = [(job, 1) for job in jobs]
275
+ final_codes: dict[str, int] = {}
276
+ while pending:
277
+ active: list[tuple[WorkerHandle, int]] = []
278
+ starting_job: WorkerJob | None = None
279
+ starting_attempt = 1
280
+ try:
281
+ for job, attempt in pending:
282
+ starting_job = job
283
+ starting_attempt = attempt
284
+ active.append((_spawn_cli_job_nonblocking(plan, job, attempt), attempt))
285
+ except (OSError, DispatchError) as exc:
286
+ if starting_job is not None:
287
+ _abort_cli_batch_start(
288
+ plan,
289
+ active,
290
+ starting_job,
291
+ starting_attempt,
292
+ pending,
293
+ exc,
294
+ )
295
+ raise
296
+ outcomes: list[tuple[WorkerHandle, int, WorkerOutcome]] = []
297
+ for handle, attempt in active:
298
+ process = handle.running_process
299
+ if process is None:
300
+ raise DispatchError("concurrent CLI worker process is missing")
301
+ returncode = process.wait()
302
+ completed = subprocess.CompletedProcess(handle.job.command, returncode)
303
+ settled = replace(handle, completed_process=completed, running_process=None)
304
+ outcomes.append((settled, attempt, _outcome_from_completed(settled)))
305
+
306
+ next_round: list[tuple[WorkerJob, int]] = []
307
+ for handle, attempt, outcome in outcomes:
308
+ job = handle.job
309
+ _finish_attempt(plan, job, attempt, outcome)
310
+ if _worker_terminal_status(plan.team_state_path, job.worker_id) == "completed":
311
+ final_codes[job.worker_id] = 0
312
+ continue
313
+ if _should_retry(outcome, attempt):
314
+ _append_event(
315
+ plan,
316
+ "worker-retry-scheduled",
317
+ _retry_details(job, attempt, outcome),
318
+ )
319
+ next_round.append((job, attempt + 1))
320
+ continue
321
+ final_codes[job.worker_id] = outcome.returncode or 1
322
+ pending = next_round
323
+ return final_codes
324
+
325
+
326
+ def _abort_cli_batch_start(
327
+ plan: DispatchPlan,
328
+ active: Sequence[tuple[WorkerHandle, int]],
329
+ failed_job: WorkerJob,
330
+ failed_attempt: int,
331
+ pending: Sequence[tuple[WorkerJob, int]],
332
+ error: Exception,
333
+ ) -> None:
334
+ reason = f"concurrent CLI batch launch failed: {error}"
335
+ for handle, _attempt in active:
336
+ process = handle.running_process
337
+ if process is not None and process.poll() is None:
338
+ process.terminate()
339
+ for handle, attempt in active:
340
+ process = handle.running_process
341
+ if process is not None:
342
+ process.wait()
343
+ _transition_worker_status(
344
+ plan.team_state_path, handle.job.worker_id, "error", reason
345
+ )
346
+ _update_dispatch_status(
347
+ plan.team_state_path, handle.job, attempt, "error", reason
348
+ )
349
+ _transition_worker_status(
350
+ plan.team_state_path, failed_job.worker_id, "error", reason
351
+ )
352
+ if not _update_dispatch_status(
353
+ plan.team_state_path,
354
+ failed_job,
355
+ failed_attempt,
356
+ "error",
357
+ reason,
358
+ ):
359
+ failed_handle = WorkerHandle(
360
+ failed_job,
361
+ "",
362
+ None,
363
+ status_path_for_prompt(failed_job.prompt_path),
364
+ "",
365
+ )
366
+ _record_dispatch(
367
+ plan.team_state_path,
368
+ failed_handle,
369
+ failed_attempt,
370
+ "error",
371
+ reason,
372
+ )
373
+ closed = {
374
+ (handle.job.worker_id, attempt) for handle, attempt in active
375
+ }
376
+ closed.add((failed_job.worker_id, failed_attempt))
377
+ for job, attempt in pending:
378
+ if (job.worker_id, attempt) in closed:
379
+ continue
380
+ _transition_worker_status(
381
+ plan.team_state_path, job.worker_id, "error", reason
382
+ )
383
+ if not _update_dispatch_status(
384
+ plan.team_state_path, job, attempt, "error", reason
385
+ ):
386
+ handle = WorkerHandle(
387
+ job,
388
+ "",
389
+ None,
390
+ status_path_for_prompt(job.prompt_path),
391
+ "",
392
+ )
393
+ _record_dispatch(
394
+ plan.team_state_path,
395
+ handle,
396
+ attempt,
397
+ "error",
398
+ reason,
399
+ )
400
+
401
+
228
402
  def await_dispatches(
229
403
  plan: DispatchPlan,
230
404
  *,
@@ -312,7 +486,15 @@ def _select_workers(
312
486
  options,
313
487
  )
314
488
  supported = set(options.supported_worker_wrappers)
315
- selected = list(requested_workers) if requested_workers else [w for w in recommended if w in supported]
489
+ selected = (
490
+ list(requested_workers)
491
+ if requested_workers
492
+ else [
493
+ worker
494
+ for worker in recommended
495
+ if worker in supported and worker != REPORT_WRITER_WORKER_ID
496
+ ]
497
+ )
316
498
  unknown = [worker for worker in selected if worker not in recommended]
317
499
  if unknown:
318
500
  raise DispatchError("requested worker(s) are not in this run roster: " + ", ".join(unknown))
@@ -339,7 +521,11 @@ def _select_cli_wrapper_assignments(
339
521
  if _is_cli_wrapper_assignment(team_state, worker_id, options)
340
522
  }
341
523
  if not requested_workers:
342
- selected = [worker for worker in recommended if worker in dispatchable]
524
+ selected = [
525
+ worker
526
+ for worker in recommended
527
+ if worker in dispatchable and worker != REPORT_WRITER_WORKER_ID
528
+ ]
343
529
  if selected:
344
530
  return selected
345
531
  raise DispatchError("run roster has no cli-wrapper assignments for codex-dispatch")
@@ -400,6 +586,8 @@ def _job_from_roster_worker(
400
586
  options: _BuildOptions,
401
587
  ) -> WorkerJob:
402
588
  result_path = _resolve_project_path(project_root, fact.result_path)
589
+ metadata = _agent_metadata(prompt_path)
590
+ invocation = _invocation_job_fields(metadata, prompt_path)
403
591
  return WorkerJob(
404
592
  worker_id=fact.worker_id,
405
593
  provider=fact.provider,
@@ -419,9 +607,47 @@ def _job_from_roster_worker(
419
607
  role=fact.role,
420
608
  idle_timeout_seconds=options.idle_timeout_seconds,
421
609
  dispatch_kind=options.dispatch_kind,
610
+ **invocation,
422
611
  )
423
612
 
424
613
 
614
+ def _agent_metadata(prompt_path: Path) -> Mapping[str, Any] | None:
615
+ metadata_path = prompt_path.with_name(prompt_path.name + ".meta.json")
616
+ if not metadata_path.is_file():
617
+ return None
618
+ return _load_json_object(metadata_path, "agent invocation metadata")
619
+
620
+
621
+ def _invocation_job_fields(
622
+ metadata: Mapping[str, Any] | None,
623
+ prompt_path: Path,
624
+ ) -> dict[str, Any]:
625
+ if metadata is None:
626
+ return {}
627
+ digests = metadata.get("digests")
628
+ assignment = metadata.get("modelAssignment")
629
+ if not isinstance(digests, Mapping) or not isinstance(assignment, Mapping):
630
+ raise DispatchError("agent invocation metadata is incomplete")
631
+ host_model_value = assignment.get("hostModelValue")
632
+ if host_model_value is not None and not isinstance(host_model_value, str):
633
+ raise DispatchError("agent invocation hostModelValue is invalid")
634
+ return {
635
+ "invocation_id": _require_string(metadata, "invocationId"),
636
+ "audience": _require_string(metadata, "audience"),
637
+ "assignment_ref": _require_string(metadata, "assignmentRef"),
638
+ "prompt_metadata_path": prompt_path.with_name(
639
+ prompt_path.name + ".meta.json"
640
+ ),
641
+ "catalog_digest": _require_string(digests, "catalogDigest"),
642
+ "assignment_digest": _require_string(digests, "assignmentDigest"),
643
+ "duty_digest": _require_string(digests, "dutyDigest"),
644
+ "instruction_digest": _require_string(digests, "instructionDigest"),
645
+ "prompt_digest": _require_string(digests, "promptDigest"),
646
+ "host_model_value": host_model_value,
647
+ "enforcement_mode": "core-pre-dispatch",
648
+ }
649
+
650
+
425
651
  def _materialize_roster_prompts(
426
652
  project_root: Path,
427
653
  workspace_root: Path,
@@ -510,6 +736,29 @@ def _spawn_job(plan: DispatchPlan, job: WorkerJob, attempt: int) -> WorkerHandle
510
736
  return handle
511
737
 
512
738
 
739
+ def _spawn_cli_job_nonblocking(
740
+ plan: DispatchPlan, job: WorkerJob, attempt: int,
741
+ ) -> WorkerHandle:
742
+ _transition_worker_status(
743
+ plan.team_state_path, job.worker_id, "in-progress", "",
744
+ model_execution_value=job.model_execution_value,
745
+ )
746
+ process: subprocess.Popen[str] | None = None
747
+ try:
748
+ process = subprocess.Popen(job.command, cwd=plan.project_root, text=True)
749
+ handle = WorkerHandle(
750
+ job, "", None, status_path_for_prompt(job.prompt_path), "", process
751
+ )
752
+ _record_dispatch(plan.team_state_path, handle, attempt, "running", "")
753
+ _append_event(plan, "worker-dispatched", _attempt_details(job, attempt, handle))
754
+ return handle
755
+ except (OSError, DispatchError):
756
+ if process is not None and process.poll() is None:
757
+ process.terminate()
758
+ process.wait()
759
+ raise
760
+
761
+
513
762
  def _dispatch_job_with_retry(plan: DispatchPlan, job: WorkerJob) -> int:
514
763
  for attempt in range(1, MAX_WORKER_ATTEMPTS + 1):
515
764
  handle = _spawn_job(plan, job, attempt)
@@ -693,13 +942,29 @@ def _retry_from_record(
693
942
  worker_id = _require_string(record, "workerId")
694
943
  attempt = int(record.get("attempt", 1))
695
944
  job = _job_from_record(plan.project_root, record)
696
- _update_dispatch_status(plan.team_state_path, job, attempt, "error", "required worker artifact was not produced")
697
- _append_event(plan, "worker-retry-scheduled", {"workerId": worker_id, "attempt": attempt})
945
+ reason = "required worker artifact was not produced"
946
+ _update_dispatch_status(plan.team_state_path, job, attempt, "error", reason)
947
+ # `team-contract` counts the first attempt's failure as a recorded
948
+ # `cli-failure`; a retry that succeeds settles `completed` and would
949
+ # otherwise leave no trace that anything had to be re-run.
950
+ details: dict[str, Any] = {"workerId": worker_id, "attempt": attempt}
951
+ details["errorLogAppend"] = _record_wrapper_failure(
952
+ plan, job, attempt, outcome, reason
953
+ )
954
+ _append_event(plan, "worker-retry-scheduled", details)
698
955
  _spawn_job(plan, job, attempt + 1)
699
956
 
700
957
 
701
958
  def _finish_attempt(plan: DispatchPlan, job: WorkerJob, attempt: int, outcome: WorkerOutcome) -> None:
702
- if outcome.returncode == 0 and not outcome.missing_completion_paths and not outcome.timeout:
959
+ settlement = _settle(plan, job, attempt, outcome)
960
+ if settlement.completed:
961
+ if job.invocation_id:
962
+ _link_agent_dispatch_result(
963
+ project_root=plan.project_root,
964
+ run_manifest_path=plan.manifest_path,
965
+ dispatch_id=f"{job.invocation_id}:attempt-{attempt}",
966
+ result_path=job.worker_result_path,
967
+ )
703
968
  post_process = _post_process_report_writer_result(plan, job)
704
969
  if not post_process["ok"]:
705
970
  reason = _require_string(post_process, "reason")
@@ -724,16 +989,23 @@ def _finish_attempt(plan: DispatchPlan, job: WorkerJob, attempt: int, outcome: W
724
989
  _transition_worker_status(
725
990
  plan.team_state_path, job.worker_id, "completed", ""
726
991
  )
727
- _update_dispatch_status(plan.team_state_path, job, attempt, "completed", "")
992
+ _update_dispatch_status(
993
+ plan.team_state_path, job, attempt, "completed", settlement.note
994
+ )
728
995
  details = _result_details(job, attempt, outcome)
729
996
  details["postProcessing"] = post_process["steps"]
997
+ if settlement.error_log_append is not None:
998
+ details["errorLogAppend"] = settlement.error_log_append
730
999
  _append_event(plan, "worker-result-collected", details)
731
1000
  return
732
- reason = _failure_reason(outcome)
1001
+ reason = settlement.reason
733
1002
  status = "timeout" if outcome.timeout else "error"
734
1003
  _transition_worker_status(plan.team_state_path, job.worker_id, status, reason)
735
1004
  _update_dispatch_status(plan.team_state_path, job, attempt, status, reason)
736
- _append_event(plan, "worker-failed", _failure_details(job, attempt, outcome, reason))
1005
+ details = _failure_details(job, attempt, outcome, reason)
1006
+ if settlement.error_log_append is not None:
1007
+ details["errorLogAppend"] = settlement.error_log_append
1008
+ _append_event(plan, "worker-failed", details)
737
1009
 
738
1010
 
739
1011
  def _finish_record(plan: DispatchPlan, record: Mapping[str, Any], outcome: WorkerOutcome) -> None:
@@ -741,6 +1013,191 @@ def _finish_record(plan: DispatchPlan, record: Mapping[str, Any], outcome: Worke
741
1013
  _finish_attempt(plan, job, int(record.get("attempt", 1)), outcome)
742
1014
 
743
1015
 
1016
+ @dataclass(frozen=True)
1017
+ class _Settlement:
1018
+ """How one attempt's terminal status was decided."""
1019
+
1020
+ completed: bool
1021
+ reason: str
1022
+ note: str
1023
+ error_log_append: dict[str, Any] | None
1024
+
1025
+
1026
+ def _settle(
1027
+ plan: DispatchPlan, job: WorkerJob, attempt: int, outcome: WorkerOutcome
1028
+ ) -> _Settlement:
1029
+ """Judge an attempt by its artifacts, not by the wrapper's exit code alone.
1030
+
1031
+ A wrapper can die after its worker has already written everything — an
1032
+ observed case is a connection dropped at session teardown, long after the
1033
+ result file and its audit sidecar were on disk. Settling that as `error`
1034
+ discards a complete analysis, and not figuratively: `convergence_engine`
1035
+ admits only dispatches that settled `completed`, so the worker's findings
1036
+ never reach re-verification. The lead's own re-dispatch triggers agree —
1037
+ they name a missing, unparseable, or audit-failing result, never an exit
1038
+ code — but the only signal `team await` gave was the status.
1039
+
1040
+ So a non-zero exit with every completion path present is re-judged by the
1041
+ audit-sidecar contract, the same rules `okstra worker-audit-check` runs. It
1042
+ passes and the dispatch settles `completed`; it fails and the dispatch stays
1043
+ `error` exactly as before. Either way the wrapper's failure is written to the
1044
+ run error log, so a `completed` here is never a swallowed failure.
1045
+ """
1046
+ if outcome.returncode == 0 and not outcome.missing_completion_paths and not outcome.timeout:
1047
+ return _Settlement(True, "", "", None)
1048
+ reason = _failure_reason(outcome)
1049
+ completed = False
1050
+ note = ""
1051
+ if not outcome.timeout and not outcome.missing_completion_paths:
1052
+ audit_failures = _audit_sidecar_failures(job)
1053
+ if audit_failures:
1054
+ reason = (
1055
+ f"{reason}; worker artifacts failed the audit-sidecar "
1056
+ f"contract: {audit_failures[0]}"
1057
+ )
1058
+ else:
1059
+ completed = True
1060
+ note = (
1061
+ f"{reason}, but every completion artifact was written and "
1062
+ f"passed the audit-sidecar contract"
1063
+ )
1064
+ append = _record_wrapper_failure(plan, job, attempt, outcome, note or reason)
1065
+ return _Settlement(completed, "" if completed else reason, note, append)
1066
+
1067
+
1068
+ def _audit_sidecar_failures(job: WorkerJob) -> tuple[str, ...]:
1069
+ """This worker's audit-sidecar contract failures, if the check can run.
1070
+
1071
+ The check's arguments come from the result filename rather than the manifest
1072
+ so the scan cannot widen past the file this job produced: `worker-results/`
1073
+ accumulates every run's artifacts, and the `worker=` filter matches the
1074
+ `-worker`-suffixed role, not the bare provider id. A non-canonical name
1075
+ leaves nothing to enforce, and an unverifiable artifact must not be promoted
1076
+ to `completed`, so that reports one failure rather than an empty tuple.
1077
+ """
1078
+ parsed = parse_worker_result_name(job.worker_result_path.name)
1079
+ if parsed is None:
1080
+ return (
1081
+ f"worker result `{job.worker_result_path.name}` is not a canonical "
1082
+ f"`<role>-worker-<task-type>-<seq>.md` name, so the audit-sidecar "
1083
+ f"contract could not be checked",
1084
+ )
1085
+ return tuple(
1086
+ check_worker_results_audit(
1087
+ job.worker_result_path.parent.parent,
1088
+ parsed.task_type,
1089
+ parsed.seq,
1090
+ worker=parsed.worker_role,
1091
+ )
1092
+ )
1093
+
1094
+
1095
+ def _record_wrapper_failure(
1096
+ plan: DispatchPlan,
1097
+ job: WorkerJob,
1098
+ attempt: int,
1099
+ outcome: WorkerOutcome,
1100
+ message: str,
1101
+ ) -> dict[str, Any]:
1102
+ """Write the wrapper's own failure to the run-level error log.
1103
+
1104
+ `okstra-lead-contract` tells Lead the deterministic dispatcher records this
1105
+ and that Lead does not need to re-record it. Nothing did: no code path
1106
+ anywhere called the error-log writer, so every wrapper failure vanished, and
1107
+ with it the `instruction-set/prior-run-errors.md` digest the next run reads
1108
+ and the `/okstra-inspect errors` report. The dispatcher is the only component
1109
+ that holds the exit code, so it is the one that writes.
1110
+
1111
+ Never raises. Logging is bookkeeping around a dispatch that has already
1112
+ settled; letting a rejected or unwritable record throw here would turn a
1113
+ recorded outcome into an unrecorded crash. What went wrong travels back in
1114
+ the lead event instead.
1115
+ """
1116
+ result: dict[str, Any] = {"ok": False, "reason": "", "path": ""}
1117
+ try:
1118
+ out_path = resolve_errors_log_path(
1119
+ plan.project_root,
1120
+ plan.manifest,
1121
+ _load_optional_json(
1122
+ plan.project_root, plan.manifest.get("activeRunContextPath")
1123
+ ),
1124
+ )
1125
+ result["path"] = str(out_path)
1126
+ append_observed(
1127
+ out_path=out_path,
1128
+ task_key=_string_value(plan.manifest.get("taskKey")),
1129
+ phase=_workflow_phase(plan.manifest),
1130
+ agent=_error_log_agent(job.worker_id),
1131
+ agent_role=(
1132
+ "report-writer"
1133
+ if job.worker_id == REPORT_WRITER_WORKER_ID
1134
+ else "worker"
1135
+ ),
1136
+ model=job.model_execution_value,
1137
+ error_type="cli-failure",
1138
+ command=" ".join(job.command),
1139
+ command_kind="wrapper",
1140
+ exit_code=_WRAPPER_TIMEOUT_EXIT_CODE if outcome.timeout else outcome.returncode,
1141
+ duration_ms=_wrapper_duration_ms(outcome),
1142
+ message=f"attempt {attempt}: {message}",
1143
+ stderr_excerpt=_wrapper_log_tail(job),
1144
+ context=None,
1145
+ )
1146
+ except (OSError, ValueError, TypeError, WorkerPromptHeaderError) as exc:
1147
+ result["reason"] = f"{type(exc).__name__}: {exc}"
1148
+ return result
1149
+ result["ok"] = True
1150
+ return result
1151
+
1152
+
1153
+ def _error_log_agent(worker_id: str) -> str:
1154
+ """The error log's `--agent` enum value for a worker id.
1155
+
1156
+ The log's own allow-list is the authority on what it accepts; a worker whose
1157
+ name is outside it is reported as such by `append_observed` rather than
1158
+ silently rewritten into some other agent's records.
1159
+ """
1160
+ if worker_id == REPORT_WRITER_WORKER_ID:
1161
+ return REPORT_WRITER_WORKER_ID
1162
+ return f"{worker_id}-worker"
1163
+
1164
+
1165
+ def _workflow_phase(manifest: Mapping[str, Any]) -> str:
1166
+ workflow = manifest.get("workflow")
1167
+ if isinstance(workflow, Mapping):
1168
+ return _string_value(workflow.get("currentPhase"))
1169
+ return ""
1170
+
1171
+
1172
+ def _wrapper_duration_ms(outcome: WorkerOutcome) -> int | None:
1173
+ if outcome.status_sidecar_path is None:
1174
+ return None
1175
+ status = read_wrapper_status(outcome.status_sidecar_path)
1176
+ if status is None:
1177
+ return None
1178
+ value = status.raw.get("duration_ms")
1179
+ return value if isinstance(value, int) and not isinstance(value, bool) else None
1180
+
1181
+
1182
+ def _wrapper_log_tail(job: WorkerJob) -> str | None:
1183
+ """The tail of the wrapper transcript, which usually names the real failure.
1184
+
1185
+ The observed case put `API Error: Connection lost mid-response.` in the last
1186
+ two lines and nothing anywhere else; without it the record says only that
1187
+ some process exited 1. Capped well under the writer's own excerpt limit
1188
+ because a whole record must stay inside one atomic `PIPE_BUF` append.
1189
+ """
1190
+ log_path = job.prompt_path.with_suffix(job.prompt_path.suffix + ".log")
1191
+ try:
1192
+ with log_path.open("rb") as handle:
1193
+ handle.seek(0, 2)
1194
+ handle.seek(max(0, handle.tell() - _WRAPPER_LOG_TAIL_BYTES))
1195
+ tail = handle.read()
1196
+ except OSError:
1197
+ return None
1198
+ return tail.decode("utf-8", errors="replace").strip() or None
1199
+
1200
+
744
1201
  def _post_process_report_writer_result(
745
1202
  plan: DispatchPlan,
746
1203
  job: WorkerJob,
@@ -779,18 +1236,23 @@ def _worker_terminal_status(team_state_path: Path, worker_id: str) -> str:
779
1236
  def _record_dispatch(
780
1237
  team_state_path: Path, handle: WorkerHandle, attempt: int, status: str, reason: str
781
1238
  ) -> None:
782
- payload = _load_json_object(team_state_path, "team-state")
783
- dispatches = payload.setdefault("workerDispatches", [])
784
- if not isinstance(dispatches, list):
785
- raise DispatchError(f"team-state workerDispatches must be an array: {team_state_path}")
786
- dispatches.append(_dispatch_record(handle.job, attempt, status, handle.pane_id, handle.degraded_from, reason))
787
- _write_json(team_state_path, payload)
1239
+ _append_worker_dispatch(
1240
+ team_state_path,
1241
+ _dispatch_record(
1242
+ handle.job,
1243
+ attempt,
1244
+ status,
1245
+ handle.pane_id,
1246
+ handle.degraded_from,
1247
+ reason,
1248
+ ),
1249
+ )
788
1250
 
789
1251
 
790
1252
  def _dispatch_record(
791
1253
  job: WorkerJob, attempt: int, status: str, pane_id: str, degraded_from: str, reason: str = ""
792
1254
  ) -> dict[str, Any]:
793
- return {
1255
+ record = {
794
1256
  "workerId": job.worker_id,
795
1257
  "role": job.role,
796
1258
  "kind": job.dispatch_kind,
@@ -814,6 +1276,18 @@ def _dispatch_record(
814
1276
  "completionPaths": [str(path) for path in job.completion_paths],
815
1277
  "worktreePath": job.worktree_path,
816
1278
  }
1279
+ if job.invocation_id:
1280
+ record.update({
1281
+ "dispatchId": f"{job.invocation_id}:attempt-{attempt}",
1282
+ "invocationId": job.invocation_id,
1283
+ "audience": job.audience,
1284
+ "assignmentRef": job.assignment_ref,
1285
+ "promptMetadataPath": str(job.prompt_metadata_path),
1286
+ **job.digests,
1287
+ "hostModelValue": job.host_model_value,
1288
+ "enforcementMode": job.enforcement_mode,
1289
+ })
1290
+ return record
817
1291
 
818
1292
 
819
1293
  def _liveness_mode(backend: str) -> str:
@@ -824,22 +1298,14 @@ def _liveness_mode(backend: str) -> str:
824
1298
 
825
1299
  def _update_dispatch_status(
826
1300
  team_state_path: Path, job: WorkerJob, attempt: int, status: str, reason: str
827
- ) -> None:
828
- payload = _load_json_object(team_state_path, "team-state")
829
- matches = [
830
- record
831
- for record in payload.get("workerDispatches", [])
832
- if dispatch_record_matches(record, job.prompt_path, attempt)
833
- ]
834
- if not matches:
835
- return
836
- # A lead may re-send a round's prompt without advancing `attempt`, which
837
- # appends a second row the prompt cannot tell from the first. The newest is
838
- # the one still running, so an outcome written to the older row would leave
839
- # the live one running for good — the shape that wedged an await loop.
840
- matches[-1]["status"] = status
841
- matches[-1]["reason"] = reason
842
- _write_json(team_state_path, payload)
1301
+ ) -> bool:
1302
+ return _update_worker_dispatch_status(
1303
+ team_state_path,
1304
+ prompt_path=job.prompt_path,
1305
+ attempt=attempt,
1306
+ status=status,
1307
+ reason=reason,
1308
+ )
843
1309
 
844
1310
 
845
1311
  def _teardown_marked_dispatches(team_state_path: Path) -> list[Mapping[str, Any]]:
@@ -884,15 +1350,23 @@ def _mark_roster_skips(
884
1350
  if not reasons:
885
1351
  return
886
1352
  team_state_path = _resolve_required_path(project_root, manifest, "teamStatePath")
887
- current_team_state = _load_json_object(team_state_path, "team-state")
888
- for worker_id, reason in reasons.items():
889
- worker = _worker_state(current_team_state, worker_id)
890
- if (
891
- _string_value(worker.get("status")) != "not-run"
892
- or _string_value(worker.get("reason"))
893
- ):
894
- continue
895
- _transition_worker_status(team_state_path, worker_id, "not-run", reason)
1353
+
1354
+ def mark_pristine_skips(current_team_state: dict[str, Any]) -> bool:
1355
+ changed = False
1356
+ for worker_id, reason in reasons.items():
1357
+ worker = _worker_state(current_team_state, worker_id)
1358
+ if (
1359
+ _string_value(worker.get("status")) != "not-run"
1360
+ or _string_value(worker.get("reason"))
1361
+ ):
1362
+ continue
1363
+ worker["reason"] = reason
1364
+ worker.pop("startedAt", None)
1365
+ worker.pop("endedAt", None)
1366
+ changed = True
1367
+ return changed
1368
+
1369
+ _mutate_team_state(team_state_path, mark_pristine_skips)
896
1370
 
897
1371
 
898
1372
  def _skip_reasons(
@@ -1004,6 +1478,10 @@ def _result_details(job: WorkerJob, attempt: int, outcome: WorkerOutcome) -> dic
1004
1478
  "attempt": attempt,
1005
1479
  "dispatchMode": BACKEND_CLI_WRAPPER if outcome.degraded_from else job.backend,
1006
1480
  "missingCompletionPaths": [],
1481
+ # A collected result can still come from a wrapper that exited non-zero
1482
+ # (`_settle`). The status says the artifacts are good; this says what the
1483
+ # process did, so the trace never loses one fact to the other.
1484
+ "wrapperExitCode": outcome.returncode,
1007
1485
  }
1008
1486
 
1009
1487
 
@@ -1068,6 +1546,23 @@ def _job_from_record(project_root: Path, record: Mapping[str, Any]) -> WorkerJob
1068
1546
  role=_require_string(record, "role"),
1069
1547
  idle_timeout_seconds=600,
1070
1548
  dispatch_kind=_require_string(record, "kind"),
1549
+ invocation_id=_string_value(record.get("invocationId")),
1550
+ audience=_string_value(record.get("audience")),
1551
+ assignment_ref=_string_value(record.get("assignmentRef")),
1552
+ prompt_metadata_path=Path(
1553
+ _string_value(record.get("promptMetadataPath"))
1554
+ ),
1555
+ catalog_digest=_string_value(record.get("catalogDigest")),
1556
+ assignment_digest=_string_value(record.get("assignmentDigest")),
1557
+ duty_digest=_string_value(record.get("dutyDigest")),
1558
+ instruction_digest=_string_value(record.get("instructionDigest")),
1559
+ prompt_digest=_string_value(record.get("promptDigest")),
1560
+ host_model_value=(
1561
+ record.get("hostModelValue")
1562
+ if isinstance(record.get("hostModelValue"), str)
1563
+ else None
1564
+ ),
1565
+ enforcement_mode=_string_value(record.get("enforcementMode")),
1071
1566
  )
1072
1567
 
1073
1568