okstra 0.205.0 → 0.205.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json
CHANGED
package/runtime/BUILD.json
CHANGED
|
@@ -387,7 +387,16 @@ def _carry_unchanged_checklists(
|
|
|
387
387
|
def _recompute_dispatch_queue(
|
|
388
388
|
narrative: Mapping[str, Any],
|
|
389
389
|
state_items: dict[str, dict],
|
|
390
|
+
stage_ledger: Mapping[str, Any] | None = None,
|
|
390
391
|
) -> list[str]:
|
|
392
|
+
"""이월 뒤 남은 디스패치 큐.
|
|
393
|
+
|
|
394
|
+
stage 상태는 디스크 원장이 답한다. 그것 없이 서사의 depends-on 만으로 다시
|
|
395
|
+
세면 이미 구현된 stage 가 `ready` 로 읽혀 큐에 들어가고, 정작 이번 run 이
|
|
396
|
+
추가한 stage 는 의존이 안 풀린 것으로 보여 빠진다 — 큐가 정확히 뒤집힌다
|
|
397
|
+
(2026-09-24, jobs implementation-planning 003: stage 1~8 done + stage 9
|
|
398
|
+
추가인 carry-all run 에서 큐 33 → 60, 내용은 done stage 의 항목뿐).
|
|
399
|
+
"""
|
|
391
400
|
try:
|
|
392
401
|
extracted = extract_plan_items(_planning(dict(narrative)))
|
|
393
402
|
except (CarryError, PlanItemContractError, KeyError, TypeError):
|
|
@@ -404,7 +413,13 @@ def _recompute_dispatch_queue(
|
|
|
404
413
|
}
|
|
405
414
|
if not previous_hashes:
|
|
406
415
|
return []
|
|
407
|
-
ledger = planning_stage_ledger(
|
|
416
|
+
ledger = planning_stage_ledger(
|
|
417
|
+
_planning(dict(narrative)),
|
|
418
|
+
{
|
|
419
|
+
str(stage): str(status)
|
|
420
|
+
for stage, status in (stage_ledger or {}).items()
|
|
421
|
+
},
|
|
422
|
+
)
|
|
408
423
|
return reverify_item_ids(extracted, previous_hashes, ledger)
|
|
409
424
|
|
|
410
425
|
|
|
@@ -479,7 +494,9 @@ def merge_v3_plan_state(
|
|
|
479
494
|
pbv["planItems"] = [
|
|
480
495
|
state_items[item_id] for item_id in sorted(state_items)
|
|
481
496
|
]
|
|
482
|
-
queue = _recompute_dispatch_queue(
|
|
497
|
+
queue = _recompute_dispatch_queue(
|
|
498
|
+
narrative, state_items, pbv.get("stageLedger"),
|
|
499
|
+
)
|
|
483
500
|
if queue:
|
|
484
501
|
pbv["dispatchQueue"] = queue
|
|
485
502
|
return state
|
|
@@ -1579,6 +1579,13 @@ def _reject_uncompleted_round_loss(
|
|
|
1579
1579
|
|
|
1580
1580
|
이력이 없는 데이터(final-report `--data` 경로)는 대상이 아니다 — 라운드
|
|
1581
1581
|
스냅샷을 갖는 것은 convergence 소유 상태 파일뿐이다.
|
|
1582
|
+
|
|
1583
|
+
직전 run 에서 이월된 표도 대상이 아니다. 그 행의 라운드 번호는 **그 run 의**
|
|
1584
|
+
번호라 이번 run 의 이력에는 없고, 그래서 열린 라운드로 읽혔다 — 안내하는
|
|
1585
|
+
`complete-round --round 2` 는 이번 run 에 존재하지도 않는 라운드라 실행할 수
|
|
1586
|
+
없었다(2026-09-24, jobs implementation-planning 003: seed --prior-state 가
|
|
1587
|
+
이월한 P-Dep·P-Var 5건이 round 1 적용을 막음). 그 표의 이력은 직전 run 의
|
|
1588
|
+
상태 파일이 갖고 있다.
|
|
1582
1589
|
"""
|
|
1583
1590
|
if not isinstance(history, list):
|
|
1584
1591
|
return
|
|
@@ -1594,6 +1601,8 @@ def _reject_uncompleted_round_loss(
|
|
|
1594
1601
|
for row in item.get("verdicts") or []:
|
|
1595
1602
|
if not isinstance(row, Mapping):
|
|
1596
1603
|
continue
|
|
1604
|
+
if str(row.get("carriedForwardFromSeq") or "").strip():
|
|
1605
|
+
continue
|
|
1597
1606
|
recorded_round = row.get("round")
|
|
1598
1607
|
if (
|
|
1599
1608
|
isinstance(recorded_round, int)
|
|
@@ -1429,6 +1429,29 @@ def _dispatch_roster_key(row: Mapping[str, Any]) -> str:
|
|
|
1429
1429
|
return ""
|
|
1430
1430
|
|
|
1431
1431
|
|
|
1432
|
+
def _dispatch_row_paths(team_state: Mapping[str, Any], worker_id: str) -> tuple[str, str]:
|
|
1433
|
+
"""이 로스터 워커의 dispatch 행이 기록한 (promptPath, resultPath).
|
|
1434
|
+
|
|
1435
|
+
v2 assignment 로 띄운 워커는 로스터 행의 경로가 빈 채로 남는다.
|
|
1436
|
+
`workers[].promptPath` 는 첫 v1 dispatch 하나만 가리키는 필드이고 v2 행은
|
|
1437
|
+
그것을 채우지 않는다(`okstra_ctl.dispatch_state.worker_dispatch_records` 의
|
|
1438
|
+
주석, 2026-09-08 실측). 그래서 acceptance critic 처럼 v2 로만 띄우는 워커는
|
|
1439
|
+
프롬프트와 결과 파일이 디스크에 그대로 있는데도 "promptPath 가 없다",
|
|
1440
|
+
"결과 파일이 없다" 로 보고됐다(2026-09-24, jobs final-verification 002).
|
|
1441
|
+
사실은 dispatch 행에 있으므로 그리로 폴백한다.
|
|
1442
|
+
"""
|
|
1443
|
+
prompt = ""
|
|
1444
|
+
result = ""
|
|
1445
|
+
for row in team_state.get("workerDispatches") or ():
|
|
1446
|
+
if not isinstance(row, Mapping) or _dispatch_roster_key(row) != worker_id:
|
|
1447
|
+
continue
|
|
1448
|
+
prompt = prompt or str(row.get("promptPath") or "")
|
|
1449
|
+
result = result or str(
|
|
1450
|
+
row.get("workerResultPath") or row.get("resultPath") or ""
|
|
1451
|
+
)
|
|
1452
|
+
return prompt, result
|
|
1453
|
+
|
|
1454
|
+
|
|
1432
1455
|
def _validate_cmux_workers_were_dispatched_by_okstra(
|
|
1433
1456
|
team_state: dict,
|
|
1434
1457
|
workers: list,
|
|
@@ -1628,14 +1651,26 @@ def validate_team_state(
|
|
|
1628
1651
|
f"{role} must use modelExecutionValue `{expected_model_execution_value}`"
|
|
1629
1652
|
)
|
|
1630
1653
|
|
|
1654
|
+
# 경로는 로스터 행이 소유하지만 v2 dispatch 는 그 행을 채우지 않는다.
|
|
1655
|
+
# 비어 있을 때만 dispatch 행에서 읽는다 — 로스터 행에 값이 있으면 그것이
|
|
1656
|
+
# 대조 대상이다.
|
|
1657
|
+
roster_prompt = str(worker.get("promptPath") or "")
|
|
1658
|
+
roster_result = str(worker.get("resultPath") or "")
|
|
1659
|
+
if not roster_prompt or not roster_result:
|
|
1660
|
+
fallback_prompt, fallback_result = _dispatch_row_paths(
|
|
1661
|
+
team_state, str(worker.get("workerId") or "")
|
|
1662
|
+
)
|
|
1663
|
+
roster_prompt = roster_prompt or fallback_prompt
|
|
1664
|
+
roster_result = roster_result or fallback_result
|
|
1665
|
+
|
|
1631
1666
|
expected_result_relative = expected.get("resultPath")
|
|
1632
|
-
result_relative =
|
|
1667
|
+
result_relative = roster_result
|
|
1633
1668
|
if expected_result_relative and result_relative != expected_result_relative:
|
|
1634
1669
|
failures.append(
|
|
1635
1670
|
f"{role} must use resultPath `{expected_result_relative}`"
|
|
1636
1671
|
)
|
|
1637
1672
|
expected_prompt_relative = expected.get("promptPath")
|
|
1638
|
-
prompt_relative =
|
|
1673
|
+
prompt_relative = roster_prompt
|
|
1639
1674
|
if expected_prompt_relative and prompt_relative != expected_prompt_relative:
|
|
1640
1675
|
failures.append(
|
|
1641
1676
|
f"{role} must use promptPath `{expected_prompt_relative}`"
|
|
@@ -383,6 +383,9 @@ def _collect_lead_evidence(
|
|
|
383
383
|
if candidate is not None:
|
|
384
384
|
sessions.setdefault(sid, candidate)
|
|
385
385
|
evidence = _LeadEvidence(window=(since, until))
|
|
386
|
+
ledger_progress = _ledger_progress(
|
|
387
|
+
team_state, run_manifest, project_root, task_type, suffix
|
|
388
|
+
)
|
|
386
389
|
for sid, path in sorted(sessions.items()):
|
|
387
390
|
progress, reads, agent_name = _scan_one_jsonl(path, since, until)
|
|
388
391
|
if agent_name and sid != lead_sid:
|
|
@@ -400,7 +403,10 @@ def _collect_lead_evidence(
|
|
|
400
403
|
"implementation entry-guard conformance cannot be verified, which "
|
|
401
404
|
"fails the run (same principle as the token-usage accuracy contract)."
|
|
402
405
|
)
|
|
403
|
-
|
|
406
|
+
# 전사와 원장은 같은 체크포인트의 두 기록이다. 계약은 둘 다 요구하고
|
|
407
|
+
# (`lead-progress append` 로 기록, 같은 줄을 대화에 raw 로 emit), 검사는
|
|
408
|
+
# 어느 쪽에 남았든 그 체크포인트를 본 것으로 판정한다.
|
|
409
|
+
evidence.progress = _merge_progress(evidence.progress, ledger_progress)
|
|
404
410
|
for ts_list in evidence.sidecar_reads.values():
|
|
405
411
|
ts_list.sort()
|
|
406
412
|
if _is_activity_contract_v1_planning(run_manifest):
|
|
@@ -643,6 +649,69 @@ def _collect_artifact_lead_evidence(
|
|
|
643
649
|
return evidence, None
|
|
644
650
|
|
|
645
651
|
|
|
652
|
+
def _ledger_progress(
|
|
653
|
+
team_state: dict,
|
|
654
|
+
run_manifest: Mapping[str, Any],
|
|
655
|
+
project_root: Path,
|
|
656
|
+
task_type: str,
|
|
657
|
+
suffix: str | None,
|
|
658
|
+
) -> list[tuple[str, str, str]]:
|
|
659
|
+
"""원장에 기록된 이 run 의 PROGRESS 체크포인트. 못 읽으면 빈 목록.
|
|
660
|
+
|
|
661
|
+
`okstra lead-progress append` 는 호스트와 무관하게 체크포인트를
|
|
662
|
+
`leadEventsPath` 에 쓰고, 리드 계약은 모든 체크포인트를 그 명령으로
|
|
663
|
+
기록하라고 요구한다(`prompts/lead/okstra-lead-contract.md` "Progress
|
|
664
|
+
reporting"). 그런데 `claude-jsonl` 증거 경로는 세션 전사만 훑어서, 계약대로
|
|
665
|
+
기록한 run 이 체크포인트 전건 누락으로 보고됐다 — 기본 호스트에서 그 명령의
|
|
666
|
+
출력을 읽는 소비자가 없었다(2026-09-24, jobs final-verification 002:
|
|
667
|
+
원장에 progress 30행, advisory 10건).
|
|
668
|
+
|
|
669
|
+
원장 행은 스크랩한 대화 텍스트보다 약한 증거가 아니다. `--phase` 는 열거된
|
|
670
|
+
phase id 만 받고 `--worker` 는 로스터 역할로 다시 쓰이므로, 그 행은 검증된
|
|
671
|
+
입력으로 okstra 자신이 쓴 것이다.
|
|
672
|
+
"""
|
|
673
|
+
events_path, _error = _resolve_lead_events_path(
|
|
674
|
+
team_state, run_manifest, project_root
|
|
675
|
+
)
|
|
676
|
+
if events_path is None:
|
|
677
|
+
return []
|
|
678
|
+
try:
|
|
679
|
+
events = read_lead_events(events_path)
|
|
680
|
+
except LeadEventParseError:
|
|
681
|
+
return []
|
|
682
|
+
run_seq = _run_sequence(run_manifest, suffix)
|
|
683
|
+
rows: list[tuple[str, str, str]] = []
|
|
684
|
+
for event in events:
|
|
685
|
+
if event.event_type not in ("progress", "progress-checkpoint"):
|
|
686
|
+
continue
|
|
687
|
+
if not _event_matches_run(event, team_state, run_manifest, task_type, run_seq):
|
|
688
|
+
continue
|
|
689
|
+
progress = _progress_line_from_event(event)
|
|
690
|
+
if progress is not None:
|
|
691
|
+
rows.append(progress)
|
|
692
|
+
return rows
|
|
693
|
+
|
|
694
|
+
|
|
695
|
+
def _merge_progress(
|
|
696
|
+
rows: list[tuple[str, str, str]],
|
|
697
|
+
extra: list[tuple[str, str, str]],
|
|
698
|
+
) -> list[tuple[str, str, str]]:
|
|
699
|
+
"""두 증거 출처의 체크포인트를 합친다. 같은 줄은 한 번만 남는다.
|
|
700
|
+
|
|
701
|
+
리드는 계약상 원장에 기록하고 같은 줄을 대화에 내보내므로, 합치면 같은
|
|
702
|
+
체크포인트가 두 번 들어온다. 서술 정확성 검사는 줄 단위로 대조하니 중복은
|
|
703
|
+
판정을 바꾸지 않지만, 보고 문구에 같은 줄이 두 번 실리는 것을 막는다.
|
|
704
|
+
"""
|
|
705
|
+
seen = {(phase, line) for _ts, phase, line in rows}
|
|
706
|
+
for ts, phase, line in extra:
|
|
707
|
+
if (phase, line) in seen:
|
|
708
|
+
continue
|
|
709
|
+
seen.add((phase, line))
|
|
710
|
+
rows.append((ts, phase, line))
|
|
711
|
+
rows.sort()
|
|
712
|
+
return rows
|
|
713
|
+
|
|
714
|
+
|
|
646
715
|
def _conformance_evidence_source(
|
|
647
716
|
team_state: dict,
|
|
648
717
|
) -> tuple[str | None, str | None]:
|