okstra 0.206.1 → 0.207.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -1
- package/dist/cli-registry.mjs +7 -1
- package/dist/cli-registry.mjs.map +1 -1
- package/docs/architecture/storage-model.md +1 -0
- package/docs/architecture.md +28 -4
- package/docs/cli.md +13 -11
- package/docs/project-structure-overview.md +4 -2
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/agents/operations/code-review.json +1 -1
- package/runtime/bin/lib/okstra/usage.sh +3 -3
- package/runtime/bin/okstra-compact-reminder.sh +1 -1
- package/runtime/prompts/duties/direction-selection-worker.json +1 -1
- package/runtime/prompts/launch.template.md +1 -1
- package/runtime/prompts/lead/adapters/cmux.md +4 -3
- package/runtime/prompts/lead/convergence.md +41 -9
- package/runtime/prompts/lead/okstra-lead-contract.md +31 -19
- package/runtime/prompts/lead/report-writer.md +8 -6
- package/runtime/prompts/profiles/_clarification-recommendation.md +4 -4
- package/runtime/prompts/profiles/_common-contract.md +1 -1
- package/runtime/prompts/wizard/prompts.ko.json +2 -1
- package/runtime/python/okstra_ctl/adapters/hosts/antigravity/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +5 -5
- package/runtime/python/okstra_ctl/adapters/hosts/codex/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +3 -2
- package/runtime/python/okstra_ctl/adapters/hosts/grok/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/hosts/kimi/relay.md +1 -1
- package/runtime/python/okstra_ctl/adapters/providers/codex/adapter.py +17 -26
- package/runtime/python/okstra_ctl/agent/prompt_cli/batch.py +183 -0
- package/runtime/python/okstra_ctl/agent/prompt_cli/cli.py +60 -10
- package/runtime/python/okstra_ctl/agent/prompt_cli/jobs.py +21 -4
- package/runtime/python/okstra_ctl/approval_decisions.py +32 -2
- package/runtime/python/okstra_ctl/assignment_resolver.py +8 -0
- package/runtime/python/okstra_ctl/blocking_checks.py +7 -0
- package/runtime/python/okstra_ctl/code_review_target.py +92 -6
- package/runtime/python/okstra_ctl/dispatch_checkpoints.py +121 -0
- package/runtime/python/okstra_ctl/dispatch_core.py +54 -32
- package/runtime/python/okstra_ctl/dispatch_state.py +12 -5
- package/runtime/python/okstra_ctl/domain/provider.py +5 -0
- package/runtime/python/okstra_ctl/domain/worker_presentation.py +21 -2
- package/runtime/python/okstra_ctl/domain/write_policy.py +2 -1
- package/runtime/python/okstra_ctl/execution_mutation_audit.py +19 -8
- package/runtime/python/okstra_ctl/initial_prompt_materialization.py +5 -0
- package/runtime/python/okstra_ctl/lead_progress.py +33 -1
- package/runtime/python/okstra_ctl/manager_view.py +34 -21
- package/runtime/python/okstra_ctl/model_io/lines.py +21 -4
- package/runtime/python/okstra_ctl/models.py +4 -1
- package/runtime/python/okstra_ctl/operation_invocation.py +11 -2
- package/runtime/python/okstra_ctl/phases/final_verification/profile.md +1 -1
- package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-executor.md +1 -1
- package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-verifier.md +14 -3
- package/runtime/python/okstra_ctl/phases/implementation_option_selection/profile.md +1 -1
- package/runtime/python/okstra_ctl/phases/implementation_planning/profile.md +1 -1
- package/runtime/python/okstra_ctl/phases/technical_verification/profile.md +1 -1
- package/runtime/python/okstra_ctl/process_group.py +118 -0
- package/runtime/python/okstra_ctl/render.py +6 -2
- package/runtime/python/okstra_ctl/report_assembly.py +17 -2
- package/runtime/python/okstra_ctl/report_finalize.py +106 -2
- package/runtime/python/okstra_ctl/run.py +1 -1
- package/runtime/python/okstra_ctl/run_artifact_prune.py +200 -0
- package/runtime/python/okstra_ctl/team.py +108 -9
- package/runtime/python/okstra_ctl/wizard/steps_options.py +8 -0
- package/runtime/python/okstra_ctl/worker_dispatch.py +44 -3
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +19 -0
- package/runtime/python/okstra_ctl/worker_runner.py +21 -3
- package/runtime/python/okstra_ctl/write_policy.py +57 -7
- package/runtime/python/okstra_project/dirs.py +14 -0
- package/runtime/python/okstra_project/resolver.py +2 -1
- package/runtime/schemas/execution-manifest-v2.schema.json +2 -1
- package/runtime/skills/okstra-code-review/SKILL.md +70 -32
- package/runtime/skills/okstra-code-review/references/review-calibration.md +26 -6
- package/runtime/skills/okstra-run/SKILL.md +2 -2
- package/runtime/templates/manager/view.template.html +21 -1
|
@@ -41,11 +41,16 @@ def _table(headers: tuple[str, ...], rows: list[list[str]], empty: str) -> str:
|
|
|
41
41
|
if not rows:
|
|
42
42
|
return f'<p class="empty">{_e(empty)}</p>'
|
|
43
43
|
head = "".join(f"<th>{_e(header)}</th>" for header in headers)
|
|
44
|
-
body = "".join(
|
|
44
|
+
body = "".join(
|
|
45
|
+
"<tr>"
|
|
46
|
+
+ "".join(f'<td data-label="{_e(header)}">{cell}</td>' for header, cell in zip(headers, row))
|
|
47
|
+
+ "</tr>"
|
|
48
|
+
for row in rows
|
|
49
|
+
)
|
|
45
50
|
return f'<div class="scroll"><table><thead><tr>{head}</tr></thead><tbody>{body}</tbody></table></div>'
|
|
46
51
|
|
|
47
52
|
|
|
48
|
-
def _file_link(project_root: str, path: str) -> str:
|
|
53
|
+
def _file_link(project_root: str, path: str, label: str) -> str:
|
|
49
54
|
"""하위 프로젝트 기준 상대 경로를 절대 file 링크로 바꾼다. 빈 값은 `-`."""
|
|
50
55
|
if not path:
|
|
51
56
|
return "-"
|
|
@@ -53,8 +58,8 @@ def _file_link(project_root: str, path: str) -> str:
|
|
|
53
58
|
if not target.is_absolute() and project_root:
|
|
54
59
|
target = Path(project_root) / target
|
|
55
60
|
if not target.is_absolute():
|
|
56
|
-
return f"
|
|
57
|
-
return f'<a href="{_e(target.as_uri())}"
|
|
61
|
+
return f'<span title="{_e(path)}">{_e(label)}</span>'
|
|
62
|
+
return f'<a href="{_e(target.as_uri())}" title="{_e(target)}">{_e(label)}</a>'
|
|
58
63
|
|
|
59
64
|
|
|
60
65
|
def _command(*args: str) -> str:
|
|
@@ -81,7 +86,8 @@ def _projects_table(projects: list[dict]) -> str:
|
|
|
81
86
|
]
|
|
82
87
|
for project in projects
|
|
83
88
|
]
|
|
84
|
-
|
|
89
|
+
table = _table(("Project", "Root", "Role", "Tags", "Linked at"), rows, "No projects registered.")
|
|
90
|
+
return f'<div class="projects">{table}</div>'
|
|
85
91
|
|
|
86
92
|
|
|
87
93
|
def _child_rows(children: list[dict], roots: dict[str, str]) -> list[list[str]]:
|
|
@@ -93,23 +99,30 @@ def _child_rows(children: list[dict], roots: dict[str, str]) -> list[list[str]]:
|
|
|
93
99
|
assignment = " — ".join(part for part in (child.get("role"), child.get("assignment")) if part)
|
|
94
100
|
# `task split` 하위 태스크는 assignment 대신 범위 목록을 가진다.
|
|
95
101
|
assignment = assignment or "; ".join(child.get("scope") or [])
|
|
102
|
+
status_fields = (
|
|
103
|
+
("Work status", child.get("workStatus") or "-"),
|
|
104
|
+
("Phase", f"{phase} ({phase_state})" if phase and phase_state else phase or "-"),
|
|
105
|
+
("Verdict", child.get("finalVerdict") or "-"),
|
|
106
|
+
("Launch", launch.get("status") or "-"),
|
|
107
|
+
)
|
|
108
|
+
status = "".join(f"<dt>{_e(label)}</dt><dd>{_e(value)}</dd>" for label, value in status_fields)
|
|
109
|
+
links = "".join(
|
|
110
|
+
_file_link(root, str(path), label)
|
|
111
|
+
for root, path, label in (
|
|
112
|
+
("", child.get("briefPath"), "Brief"),
|
|
113
|
+
(roots.get(str(child.get("projectId") or ""), ""), child.get("latestReportRecordPath"), "Latest report"),
|
|
114
|
+
)
|
|
115
|
+
if path
|
|
116
|
+
)
|
|
117
|
+
error = f'<p class="child-error"><strong>Error:</strong> {_e(child["error"])}</p>' if child.get("error") else ""
|
|
118
|
+
ticket = f'<strong class="child-ticket">{_e(child["ticketId"])}</strong>' if child.get("ticketId") else ""
|
|
96
119
|
rows.append(
|
|
97
120
|
[
|
|
98
|
-
f"<code>{_e(child.get('taskKey'))}</code>"
|
|
121
|
+
ticket + f"<code class=\"child-key\">{_e(child.get('taskKey'))}</code>"
|
|
99
122
|
+ ("" if child.get("projectLinked", True) else " " + _chip("unlinked")),
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
_e(f"{phase} ({phase_state})" if phase and phase_state else phase or "-"),
|
|
104
|
-
_e(child.get("finalVerdict") or "-"),
|
|
105
|
-
_e(launch.get("status") or "-"),
|
|
106
|
-
f'<span class="wrap">{_e(assignment or "-")}</span>',
|
|
107
|
-
_e(child.get("error") or "-"),
|
|
108
|
-
_file_link("", str(child.get("briefPath") or "")),
|
|
109
|
-
_file_link(
|
|
110
|
-
roots.get(str(child.get("projectId") or ""), ""),
|
|
111
|
-
str(child.get("latestReportRecordPath") or ""),
|
|
112
|
-
),
|
|
123
|
+
_chip(str(child.get("summaryBucket") or "planned")) + f'<dl class="child-status">{status}</dl>',
|
|
124
|
+
f'<div class="child-assignment">{_e(assignment or "-")}</div>'
|
|
125
|
+
+ error + (f'<div class="child-links">{links}</div>' if links else ""),
|
|
113
126
|
]
|
|
114
127
|
)
|
|
115
128
|
return rows
|
|
@@ -161,11 +174,11 @@ def _task_panel(home: Path, manager_id: str, task: dict, status: dict, roots: di
|
|
|
161
174
|
f"<code>{_e(sync_command)}</code>, then re-run the view command.</p>"
|
|
162
175
|
)
|
|
163
176
|
children = _table(
|
|
164
|
-
("
|
|
165
|
-
"Brief", "Latest report"),
|
|
177
|
+
("TaskID", "Status", "Assignment"),
|
|
166
178
|
_child_rows(status["children"], roots),
|
|
167
179
|
"No child tasks.",
|
|
168
180
|
)
|
|
181
|
+
children = f'<div class="children">{children}</div>'
|
|
169
182
|
directives = read_jsonl(directives_jsonl_path(home, manager_id, group, task_id))
|
|
170
183
|
events = read_jsonl(events_jsonl_path(home, manager_id, group, task_id))
|
|
171
184
|
details = (
|
|
@@ -75,14 +75,31 @@ def _assignment_line(prefix: str, assignment: Mapping[str, Any], identifier: obj
|
|
|
75
75
|
|
|
76
76
|
def _worker_assignment_lines(run: Mapping[str, Any]) -> str:
|
|
77
77
|
assignments = run.get("workerAssignments")
|
|
78
|
-
if not isinstance(assignments, list):
|
|
79
|
-
return "- None\n"
|
|
80
78
|
lines = [
|
|
81
79
|
_assignment_line("Worker", assignment, assignment.get("workerId"))
|
|
82
|
-
for assignment in assignments
|
|
80
|
+
for assignment in (assignments if isinstance(assignments, list) else ())
|
|
83
81
|
if isinstance(assignment, Mapping)
|
|
84
82
|
]
|
|
85
|
-
return "".join(lines) or "- None\n"
|
|
83
|
+
return "".join(lines + _critic_assignment_lines(run)) or "- None\n"
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def _critic_assignment_lines(run: Mapping[str, Any]) -> list[str]:
|
|
87
|
+
"""Opt-in critics live only in `invocationAssignments`, never in `workerAssignments`.
|
|
88
|
+
|
|
89
|
+
The id is the assignment ref's last segment, the value team dispatch and
|
|
90
|
+
`optionalWorkerRoles[].workerId` use (`render._optional_worker_roles`).
|
|
91
|
+
"""
|
|
92
|
+
assignments = run.get("invocationAssignments")
|
|
93
|
+
if not isinstance(assignments, Mapping):
|
|
94
|
+
return []
|
|
95
|
+
return [
|
|
96
|
+
_assignment_line(
|
|
97
|
+
"Worker", {"role": "critic", **assignment}, ref.rsplit("/", 1)[-1],
|
|
98
|
+
)
|
|
99
|
+
for ref, assignment in sorted(assignments.items())
|
|
100
|
+
if isinstance(ref, str) and ref.startswith("critic/")
|
|
101
|
+
and isinstance(assignment, Mapping)
|
|
102
|
+
]
|
|
86
103
|
|
|
87
104
|
|
|
88
105
|
def _worker_prompt_lines(run: Mapping[str, Any]) -> str:
|
|
@@ -35,7 +35,7 @@ IMPLEMENTATION_ROLES = frozenset({"executor", "verifier"})
|
|
|
35
35
|
# Legacy role-keyed compatibility defaults. Canonical assignments resolve a
|
|
36
36
|
# provider first and then use ProviderSpec.default_models for that role.
|
|
37
37
|
ROLE_DEFAULTS = {
|
|
38
|
-
"lead": "opus", "claude": "opus", "codex": "gpt-6-sol",
|
|
38
|
+
"lead": "opus", "claude": "opus", "codex": "gpt-6.1-sol",
|
|
39
39
|
"antigravity": "gemini-3.1-pro", "report-writer": "sonnet",
|
|
40
40
|
}
|
|
41
41
|
|
|
@@ -65,6 +65,9 @@ def lead_launch_argv(
|
|
|
65
65
|
argv.extend(launch.sandbox_waiver)
|
|
66
66
|
if model:
|
|
67
67
|
argv.extend([launch.model_flag, model])
|
|
68
|
+
spec = provider_spec(provider).models.get(model)
|
|
69
|
+
if spec is not None:
|
|
70
|
+
argv.extend(spec.cli_args)
|
|
68
71
|
if session_id and launch.start_session_id_flag:
|
|
69
72
|
argv.extend([launch.start_session_id_flag, session_id])
|
|
70
73
|
if launch.prompt_flag:
|
|
@@ -42,12 +42,14 @@ def resolve_operation_slots(
|
|
|
42
42
|
*,
|
|
43
43
|
contract_root: Path,
|
|
44
44
|
pool: ModelPool,
|
|
45
|
+
count: int | None = None,
|
|
45
46
|
) -> Sequence[OperationSlot]:
|
|
46
47
|
"""작업 계약이 요구하는 칸과 각 칸의 모델.
|
|
47
48
|
|
|
48
49
|
모델은 런과 같은 기본값 사슬에서 고른다. 서로 다른 모델 참조를 요구한
|
|
49
50
|
수만큼 채우지 못하면 여기서 멈춘다 — 같은 모델을 반복하거나 워커 수를
|
|
50
|
-
줄이면 그 작업이 요구한 교차검증이 조용히 사라진다.
|
|
51
|
+
몰래 줄이면 그 작업이 요구한 교차검증이 조용히 사라진다. `count` 는
|
|
52
|
+
호출자가 계약 상한 안에서 명시적으로 고른 수다.
|
|
51
53
|
"""
|
|
52
54
|
graph_root = ContractGraphRoot.from_base(contract_root)
|
|
53
55
|
try:
|
|
@@ -63,7 +65,14 @@ def resolve_operation_slots(
|
|
|
63
65
|
f"unknown operation: {operation_id} (known: {known})"
|
|
64
66
|
)
|
|
65
67
|
duty_id = str(operation.payload["dutyId"])
|
|
66
|
-
|
|
68
|
+
# 계약의 수는 상한이다. 더 적게 여는 것은 호출자가 명시할 때만이다.
|
|
69
|
+
ceiling = int(operation.payload["count"])
|
|
70
|
+
if count is None:
|
|
71
|
+
count = ceiling
|
|
72
|
+
elif not 1 <= count <= ceiling:
|
|
73
|
+
raise OperationPreparationError(
|
|
74
|
+
f"operation {operation_id!r} takes a count from 1 to {ceiling}, got {count}"
|
|
75
|
+
)
|
|
67
76
|
role_id = graph.duties[duty_id].role_id
|
|
68
77
|
|
|
69
78
|
candidates = pool.default_candidates(role_id)
|
|
@@ -66,7 +66,7 @@
|
|
|
66
66
|
- **Read-only command log**: any pre-existing test/validation command touched during this run MUST be listed with its exact command line and one honest status — `executed` (ran; carries its exit code) / `advisory` (external Tier 3 did not PASS; carries observed/expected results and remains user-owned) / `env-unavailable` (should run but cannot in this environment — missing replica DB, container, or service; carries the reason, never a faked pass) / `not-configured` (no such qa-command tier) / `rejected` (a mutating/denied token — skipped, carries the denied token). A check that could not run locally is recorded as `env-unavailable` or `advisory` according to the external QA policy — never silently dropped and never reported as `executed` with an invented exit code. Mutating-command prohibition is the shared read-only boundary (see Non-goals); it is not restated per row.
|
|
67
67
|
- **Could-not-verify roll-up (§5.8.9)**: the template mechanically aggregates every not-confirmed check into one scannable list — `gap` requirement-coverage rows, `advisory` / `not-configured` / `env-unavailable` / `rejected` command rows, and `blocked` manual tests. You do not hand-author it, but you MUST give those rows their honest status so nothing unverified hides across sections: a check silently recorded as `executed`/`covered` will not surface in the roll-up. This is okstra's answer to "say what could not be verified this run."
|
|
68
68
|
- **Routing recommendation**: this phase records the verdict, the blockers, the conditions, and the repair each blocker needs. The lead chooses the next phase from the `## final-verification` section of `prompts/lead/phase-routing.md`, which lists the allowed targets and the shape of `finalVerification.routingRecommendation`. Read that section before writing the field.
|
|
69
|
-
- **Verified-row recording** (both scopes): when the verdict is release-ready,
|
|
69
|
+
- **Verified-row recording** (both scopes): when the verdict is release-ready, `okstra report-finalize` writes the `verified` rows itself in its `record-verified` step, right before `validate-run`: it runs `okstra handoff record-verified` once per stage in this report's `stageReports`, with the final-report data.json as both report and record. The lead does not run `handoff record-verified` by hand; it needs the data.json that `report-finalize` assembles and `validate-run` reads the rows in the same call, so no lead step fits between them. Without those rows the stage is never offered a pull request, and release-handoff opens one PR per stage. The helper checks the latest verification manifest, task/stage identity, report pointer, prepared target, and recorded implementation commit, and records the captured commit and original verdict, including conditional acceptance conditions. A missing target or mismatched commit fails the step and requires re-verification. Eligibility is granted only after the verification passes validation. **Enforced:** `okstra_ctl.report_finalize._record_verified_stages` runs the step, `okstra_ctl.handoff_verification` validates the evidence, and `validators/validate-run.py` `_validate_verified_row_recorded` fails the run (blocking, `okstra_ctl.blocking_checks`) when a cleared stage has no `verified` row matching this report, captured commit, and verdict.
|
|
70
70
|
- Clarification request policy (phase-specific addendum — shared policy is in `_common-contract.md`):
|
|
71
71
|
- populate `## 1. Clarification Items` only when a blocker hinges on information only the user can supply (deployment intent, intended target environment, business-rule interpretation); use `Blocks=next-phase` for items that gate continuing to release-handoff
|
|
72
72
|
- Self-review pass before finalising the report (the Okstra lead runs this; do not delegate it):
|
package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-executor.md
CHANGED
|
@@ -89,7 +89,7 @@ template's check; that template is gone.
|
|
|
89
89
|
```
|
|
90
90
|
|
|
91
91
|
The file MUST NOT exist before the run starts (overwrite is refused — see `--force-stage` non-goal). **Enforced:** `validators/validate-run.py` `_validate_stage_carry_sidecar_exists` fails a run that declares `stageSidecarEvidence` without the file on disk. Transcribing the JSON into the report is not the same as writing it: `consumers` treats the carry file as the source of truth for marking the stage `done`, so a missing file leaves the stage permanently incomplete and blocks every dependent stage with a `PrepareError` — while this run reports success.
|
|
92
|
-
- **Verifier gates are not yours to run (BLOCKING).** The self-mock detector (`validators/detect_self_mock.py`) belongs to the implementation verifier and is never delegated to you (`_implementation-verifier.md` §"Self-mock detection"). The coding-conventions preflight names it as the enforcement behind the no-self-mocking principle — that names who will check your diff, not a command for you to run. You MUST NOT invoke it and MUST NOT write `<task_root>/qa/self-mock-*.json`; running it early does not pre-satisfy the gate, because the verifier runs it again under its own duty. **Enforced:** that sidecar is not among the paths your attempt's `writePolicy.artifactPolicy.allowedPaths` carries, so the write audit closes the attempt as `error` with `artifact-root change exceeds batch policy union` (`scripts/okstra_ctl/execution_mutation_audit.py`) — a stage whose every gate passed still lands as a failed run. The same holds for the conformance results `<task_root>/qa/result-*.json`: you may run a conformance script to check your work, but the verifier records every result, including an earlier stage's result that your rewrite of its script made stale. When the approved plan tells you to write one, leave it and name it in your result as a plan step the verifier owns.
|
|
92
|
+
- **Verifier gates are not yours to run (BLOCKING).** The self-mock detector (`validators/detect_self_mock.py`) belongs to the implementation verifier and is never delegated to you (`_implementation-verifier.md` §"Self-mock detection"). The coding-conventions preflight names it as the enforcement behind the no-self-mocking principle — that names who will check your diff, not a command for you to run. You MUST NOT invoke it and MUST NOT write `<task_root>/qa/self-mock-*.json`; running it early does not pre-satisfy the gate, because the verifier runs it again under its own duty. **Enforced:** that sidecar is not among the paths your attempt's `writePolicy.artifactPolicy.allowedPaths` carries, so the write audit closes the attempt as `error` with `artifact-root change exceeds batch policy union` (`scripts/okstra_ctl/execution_mutation_audit.py`) — a stage whose every gate passed still lands as a failed run. The same holds for the conformance results `<task_root>/qa/result-*.json`: you may run a conformance script to check your work — whatever it writes (a baseline capture, build logs) goes under `<task_root>/qa/output/`, the one qa directory besides `qa/scripts/` your attempt may write — but the verifier records every result, including an earlier stage's result that your rewrite of its script made stale. When the approved plan tells you to write one, leave it and name it in your result as a plan step the verifier owns.
|
|
93
93
|
- **An external Tier 3 non-PASS does NOT withhold the carry evidence.** A Tier 3 entry whose `requires` include `http`, `external`, or `db` is advisory. Its FAIL, MISSING, no result, startup failure, or credential / network / service absence gets recorded honestly — exact command, exit code, output tail, marked `ADVISORY` in `Validation evidence` — and you emit the carry evidence anyway. Only Tier 1 and Tier 2 failures withhold it. Withholding on an external result is what actually blocks the stage: the carry file is the only thing that can mark a stage `done`, the verifier re-runs that same command from the host (where a call your sandbox could not complete often passes), and a stage the verifier then PASSes can never be closed because its evidence was never written.
|
|
94
94
|
- **Reverse link (BLOCKING).** The runtime already appended a `status:"started"` row for this stage before the run began. The terminal row belongs to the lead's post-stage persistence and is verdict-gated — `status:"done"` with `carry_path` on a non-`FAIL` verdict, `status:"failed"` on `FAIL` (`_implementation-deliverable.md` §"Lead post-stage persistence").
|
|
95
95
|
- **No PR / push in this phase.** This run produces local commits, carry sidecar evidence, verifier results, and the implementation final report only. Push and PR creation belong exclusively to the later `release-handoff` phase after `final-verification` returns `accepted`.
|
package/runtime/python/okstra_ctl/phases/implementation/instructions/_implementation-verifier.md
CHANGED
|
@@ -48,7 +48,15 @@ Verifier obtains the QA command set from exactly two declared sources, in order
|
|
|
48
48
|
|
|
49
49
|
### Execution rule
|
|
50
50
|
|
|
51
|
-
Tier 1 commands run verbatim first. Then every Tier 2 entry runs once. Then the Tier 3 stage conformance script (below) runs once. Then the self-mock detector (below) runs once whenever the diff changed a test file. Each command runs in the worktree cwd, and is recorded in the worker result with its exact command line, exit code, and the tail of stdout/stderr. Substituting or paraphrasing a Tier 1 command is forbidden (see Verifier-specific forbidden actions below).
|
|
51
|
+
Tier 1 commands run verbatim first. Then every Tier 2 entry runs once. Then the Tier 3 stage conformance script (below) runs once. Then the self-mock detector (below) runs once whenever the diff changed a test file. Those last two are run by the stage QA owner only (next section). Each command runs in the worktree cwd, and is recorded in the worker result with its exact command line, exit code, and the tail of stdout/stderr. Substituting or paraphrasing a Tier 1 command is forbidden (see Verifier-specific forbidden actions below).
|
|
52
|
+
|
|
53
|
+
### Stage QA owner
|
|
54
|
+
|
|
55
|
+
Every verifier of this stage runs in the same worktree at the same time. Two checks change that worktree while they run: a Tier 3 conformance script may build into it (`.next/`, `dist/`) or bind fixed ports, and the self-mock detector's mutation probe rewrites production sources in place (cosmic-ray) or replaces a report file inside the worktree (Stryker). Run concurrently, they corrupt each other's result and the other verifiers' test runs (observed 2026-09-26, fontsninja-v3-site dev-11054 stage 1: `next build` + `next start -p 3000` + a proxy on 8889 from three verifiers at once).
|
|
56
|
+
|
|
57
|
+
Your prompt carries a `**Stage QA owner:**` anchor naming one worker id. If that id is yours (the `<id>` in your `**Model:** <id> worker` line), you run Tier 3 and the self-mock detector as the two sections below describe. Otherwise you run neither and write neither sidecar; each section says what to record instead. A prompt with no such anchor predates this rule: run both yourself.
|
|
58
|
+
|
|
59
|
+
**Enforced:** the anchor is generated (`scripts/okstra_ctl/worker_prompt_policy.py` `stage_qa_owner`, the first verifier in roster order). If the owner is never dispatched, no sidecar is written and the existing gates report it: a missing conformance result is BLOCKING for an `io`-only entry and ADVISORY for an external one (`scripts/okstra_ctl/conformance.py` `decide_conformance_gate`), and a missing self-mock sidecar blocks when the diff changed a test file (`validators/validate-run.py` `_validate_selfmock`). Nothing checks which verifier's log holds the commands.
|
|
52
60
|
|
|
53
61
|
### Tier 3 — stage conformance scripts
|
|
54
62
|
|
|
@@ -75,6 +83,7 @@ lock the core, gate, and accepted-report behavior respectively.
|
|
|
75
83
|
An `io`-only non-PASS remains BLOCKING. Manifest/schema/source-mutation defects
|
|
76
84
|
also remain contract violations.
|
|
77
85
|
|
|
86
|
+
- **Owner only.** This section is the stage QA owner's (§ Stage QA owner above). A verifier that is not the owner runs no `runCommand`, writes no `result-*.json`, logs `conformance: owned by <worker-id> — not run`, and judges Tier 3 only from the diff and the manifest entry (declaration, `requires` coverage).
|
|
78
87
|
- **Source.** The conformance manifest is `<task_root>/qa/conformance-manifest.json` (the directory is the `TASK_QA_PATH` token). This run's stage conformance entry is the manifest `entries[]` item whose `stageKey` equals this run's stageKey — `<task-id>-stage-<N>`, where `<N>` is the injected Stage number. Find that one entry. Also run every other entry whose `script` appears in this stage's `plannedPaths`: this stage rewrote that script, so the result its own stage recorded is stale, and you write that entry's result sidecar exactly as below. Ignore the rest (other stages are run by their own implementation runs or by final-verification).
|
|
79
88
|
- **Exemption / waiver → do NOT run.** If the entry carries an `exemption` (or a user `waiver`), the verifier does NOT execute the script. It records the fact and the reason (`exemption.reason` / `waiver.reason` + `waiver.acknowledgedBy`) in the Read-only command log AND writes the result sidecar reflecting the skip. An `exemption` passes outright. An external-advisory waiver is reported as `ADVISORY` with `conditional=false`; only an `io`-only blocking waiver is conditional. An empty `requires` list cannot be waived; it is declaration/contract trouble and remains BLOCKING. No script runs in either permitted waiver case.
|
|
80
89
|
- **Otherwise run `runCommand` in the worktree cwd.** Execute the entry's `runCommand` verbatim from the worktree cwd. Inject env from `<PROJECT_ROOT>/.okstra/project.json`'s `qaEnv` (replica DB DSN / app base URL / env file — declared in Phase 4e). This is a **replica / test environment only** path — never run it against shared / staging / prod, identical to the DB real-execution gate principle above.
|
|
@@ -96,6 +105,7 @@ also remain contract violations.
|
|
|
96
105
|
|
|
97
106
|
A green suite does not prove a test exercises the unit it names — a test that stubs its own SUT passes forever, including after the real implementation is deleted. The static detector is the machine half of the **Self-mocking** blocking check below, and running it is the verifier's own duty: it is never delegated to the executor and never inferred from the executor's evidence.
|
|
98
107
|
|
|
108
|
+
- **Owner only.** The detector is the stage QA owner's (§ Stage QA owner above). A verifier that is not the owner does not run it and writes no `self-mock-*` file; it logs `self-mock detector: owned by <worker-id> — not run` and still performs the **Self-mocking** blocking check below by reading the changed tests.
|
|
99
109
|
- **Trigger.** This run's diff changed at least one **test** file. Enumerate the changed files with `git diff --name-only <base>...HEAD` from the worktree cwd — the same enumeration the static review's Scope rule uses — then keep only the paths the gate itself treats as tests: `*.spec.*`, `*.test.*`, a `test_`-prefixed basename, a `_test.` suffixed basename, or any path segment `test/` or `tests/`. Pass nothing else; non-test files are excluded. Exclude `tests/fixtures/self_mock/**` as well — those are the detector's own deliberately self-mocked fixtures, which `validate-run.py` also excludes from the trigger, so feeding them in would manufacture a `FAIL` the gate then blocks on. No changed test file → no run and no sidecar; the gate is vacuous by design.
|
|
100
110
|
- **Run the detector once, in the worktree cwd**, one `--test-file` per changed test file:
|
|
101
111
|
```bash
|
|
@@ -158,7 +168,7 @@ Tier 3 external-advisory discrepancies are excluded from this promotion: preserv
|
|
|
158
168
|
|
|
159
169
|
### Read-only command log (per verifier)
|
|
160
170
|
|
|
161
|
-
The worker result MUST contain a `Read-only command log` block listing every command executed during the verifier run with its exact invocation and exit code, in execution order — including the Tier 3 conformance `runCommand` (or the exemption/waiver skip note
|
|
171
|
+
The worker result MUST contain a `Read-only command log` block listing every command executed during the verifier run with its exact invocation and exit code, in execution order — including the Tier 3 conformance `runCommand` (or, when no script ran, the exemption/waiver skip note or the not-the-owner note), and the self-mock detector invocation (or the not-the-owner note). No source-mutating command may appear in this block; the only permitted mutations are a Tier 3 conformance script writing to its `qaEnv` replica datastore and the self-mock detector writing its own `<task_root>/qa/self-mock-*.json` sidecar — both are artifact-directory writes, both are logged like any other command, and neither touches the worktree source, so the verifier runs them without hesitation. This log is copied into the final report's verifier result section verbatim.
|
|
162
172
|
|
|
163
173
|
**Enforced:** `_validate_verifier_command_log_is_read_only` in `validators/validate-run.py` scans every `verifierResults[].readOnlyCommandLog` for mutation modes (`--fix`, `--write`, `gofmt -w`, `jest -u`, snapshot/golden updates, `cargo insta accept`, a non-`no` `INSTA_UPDATE`, and a trailing `|| true`) and fails the run. Check-only forms (`--check`, `--check-only`) pass.
|
|
164
174
|
|
|
@@ -207,7 +217,8 @@ diff re-buys wall-clock without new information, so the static scope narrows —
|
|
|
207
217
|
the command re-run does not:
|
|
208
218
|
|
|
209
219
|
- **Command re-run stays full.** Every Tier 1/2/3 command from the plan's
|
|
210
|
-
validation set runs end-to-end exactly as in a first run
|
|
220
|
+
validation set runs end-to-end exactly as in a first run (Tier 3 and the self-mock
|
|
221
|
+
detector by the stage QA owner only, as in a first run). QA-RESULT gating
|
|
211
222
|
(`validate-run.py` Tier 3) is unchanged.
|
|
212
223
|
- **Static design & test-quality sweep narrows to the fix diff.** Enumerate
|
|
213
224
|
`git diff <prev-head>..HEAD` (the `Previous run HEAD` line of the Fix-Run
|
|
@@ -14,7 +14,7 @@
|
|
|
14
14
|
- Apply the shared reporter-confirmation precondition exactly as written. Unresolved `intent-check:` and `conversion-block:` rows use `Blocks=next-phase`.
|
|
15
15
|
- Treat each stable brief end-state ID as a required evaluation target. A missing ID is a preparation failure; do not invent a replacement requirement.
|
|
16
16
|
- Worker direction-selection procedure:
|
|
17
|
-
- In `candidate-comparison` mode, produce candidate, supporting and contradicting evidence, criterion scores, and requirement mappings. For every candidate, state one feasibility verdict — `feasible`, `not-feasible`, or `uncertain` — with a
|
|
17
|
+
- In `candidate-comparison` mode, produce candidate, supporting and contradicting evidence, criterion scores, and requirement mappings. For every candidate, state one feasibility verdict — `feasible`, `not-feasible`, or `uncertain` — with a rationale and the strongest counterevidence, each citing inspected evidence. Write the rationale as long as the argument needs: say why each cited finding supports the verdict, not only which findings do. The report's `feasibilityVotes` row for this worker is built from that statement, so a candidate without one leaves the writer nothing but another worker's words.
|
|
18
18
|
- In `candidate-comparison` mode only, submit at most three candidates. A candidate must be feasible from inspected evidence, not from an assumed future change.
|
|
19
19
|
- In `preselected-validation` mode, receive one preselected direction from the lead and validate its evidence, counterevidence, criterion scores, and requirement mappings, and state the same feasibility verdict with its rationale and counterevidence. The worker must not generate new candidates.
|
|
20
20
|
- Do not produce detailed file lists, stage maps, execution commands, or a plan approval request.
|
|
@@ -163,7 +163,7 @@ Plan for the actual worktree layout before approval. Shared documentation direct
|
|
|
163
163
|
unavailable environment is a user-owned follow-up, never a plan approval or
|
|
164
164
|
later run blocker. `requires=[]` and `requires=[io]` remain blocking.
|
|
165
165
|
Remote IO should also declare `external`.
|
|
166
|
-
Layout split (not this phase's writes): executable scripts (conformance + any real-IO test) live under `<task_root>/qa/scripts/`; data sidecars (`conformance-manifest.json`, `result-*.json`) stay at the `qa/` root. The implementer writes the scripts and the manifest entry. Only the implementation verifier writes `result-*.json`, including the result of an earlier stage's entry whose script this stage rewrites, so a step never assigns a result file to the implementer and never lists one in `plannedPaths`. This declaration is enforced at four layers: `validators/validate-implementation-plan-stages.py` check **S11** forces every stage to carry one of the two lines; at the planning boundary `scripts/okstra_ctl/phases/implementation_planning/plan_body.py` `_validate_planning_conformance_declared` accepts a well-formed `Conformance tests:` line even when the script file and manifest entry are absent (malformed `requires` still fails); the matching `implementation` stage run that inherited `Conformance tests:` fails closed when the script file is missing (`_validate_conformance`); and the manifest JSON structure — including each entry's `script` living under `qa/scripts/` and a `runCommand` that does not change cwd — is enforced by `validate_conformance_manifest` when the implementer writes the entry.
|
|
166
|
+
Layout split (not this phase's writes): executable scripts (conformance + any real-IO test) live under `<task_root>/qa/scripts/`; data sidecars (`conformance-manifest.json`, `result-*.json`) stay at the `qa/` root. A step that runs a script writing its own files (a baseline capture, a `--compare` target, build logs) points those paths at `<task_root>/qa/output/`: it is the only other qa directory the implementer's write audit accepts, and any other qa path discards the attempt (`scripts/okstra_ctl/write_policy.py` `_role_qa_artifact_paths`; observed 2026-09-26, dev-11054 stage 1, `qa/baseline/`). The implementer writes the scripts and the manifest entry. Only the implementation verifier writes `result-*.json`, including the result of an earlier stage's entry whose script this stage rewrites, so a step never assigns a result file to the implementer and never lists one in `plannedPaths`. This declaration is enforced at four layers: `validators/validate-implementation-plan-stages.py` check **S11** forces every stage to carry one of the two lines; at the planning boundary `scripts/okstra_ctl/phases/implementation_planning/plan_body.py` `_validate_planning_conformance_declared` accepts a well-formed `Conformance tests:` line even when the script file and manifest entry are absent (malformed `requires` still fails); the matching `implementation` stage run that inherited `Conformance tests:` fails closed when the script file is missing (`_validate_conformance`); and the manifest JSON structure — including each entry's `script` living under `qa/scripts/` and a `runCommand` that does not change cwd — is enforced by `validate_conformance_manifest` when the implementer writes the entry.
|
|
167
167
|
- `### Stage Exit Contract` — predicted added/modified files, newly exposed identifiers/types/endpoints, downstream-usable resources.
|
|
168
168
|
- `### Stage Validation` — pre / mid / post exact commands or observable outcomes for this stage only.
|
|
169
169
|
- **Run-executable only (BLOCKING).** A `validationChecklist` row that carries `stageRefs` gates that stage's carry, so it MUST be executable by the implementation run itself — no deployment, no manual walk-through, no observation "by hand", no credentials the run does not hold. The implementation phase forbids deploys and holds no deploy credentials, so such a row can never pass and blocks a stage with zero code defects (observed: dev-10341 VC-013/VC-014). Manual or deployed-environment verification belongs in the brief's `External Gates` and final-verification's user-owned external QA — record it there without `stageRefs`. **Enforced:** `okstra_ctl.implementation_direction.validate_selected_direction_plan` (`stage_validation_executability_errors`) rejects stage-gating rows whose observation text requires a manual or deployed-environment step.
|
|
@@ -12,7 +12,7 @@
|
|
|
12
12
|
1. Read the frozen technical input linked in the analysis packet. Preserve the source hash, candidate IDs and fact identities. `resolve_technical_verification_input` enforces same-task source ownership, schema validity, eligible technical facts and unresolved user-decision gates.
|
|
13
13
|
2. Before executing a probe, write its plan under `experiments/<seq>/<worker-id>/`: hypothesis, procedure, commands, confirming and rejecting signals, environment, and inconclusive criteria. Retain the plan with the evidence.
|
|
14
14
|
3. Create a separate source copy in that directory. Record the source commit and any uncommitted input differences. Read the project and task worktree as inputs; perform source edits, installations and builds only in your own copy. Do not share writable dependency directories or use real credentials. Use synthetic local accounts for request-boundary experiments.
|
|
15
|
-
4. Preserve the baseline lockfile. When the hypothesis needs a deliberate dependency change, retain the resulting lockfile and diff, then use a frozen install against that experimental lockfile. Record command, working directory, stdout/stderr and exit code in run-local logs. Distinguish infrastructure failure from a falsified compatibility claim.
|
|
15
|
+
4. Phase 7 removes every `node_modules` and `.next` directory under the run directory, so evidence must not live inside them. Preserve the baseline lockfile. When the hypothesis needs a deliberate dependency change, retain the resulting lockfile and diff, then use a frozen install against that experimental lockfile. Record command, working directory, stdout/stderr and exit code in run-local logs. Distinguish infrastructure failure from a falsified compatibility claim.
|
|
16
16
|
5. Classify each fact as `supported`, `refuted`, `inconclusive` or `not-run` using the declared signals. A zero exit code alone does not establish compatibility. Record limitations and all failed or unavailable probes.
|
|
17
17
|
6. During convergence, inspect plans and logs; rerun disputed probes in a separate copy when feasible. An independent report is not a substitute for command evidence.
|
|
18
18
|
|
|
@@ -0,0 +1,118 @@
|
|
|
1
|
+
"""CLI 워커 프로세스 그룹의 메모리 상한과 종료 뒤 잔여 프로세스 정리.
|
|
2
|
+
|
|
3
|
+
워커가 실행한 빌드는 워커와 같은 프로세스 그룹에 들어간다(`worker_runner` 가
|
|
4
|
+
`start_new_session=True` 로 띄운다). fontsninja-v3-site 에서 Turbopack PostCSS 워커가
|
|
5
|
+
57GB, `next build` 가 78GB 까지 커져 커널 패닉(watchdog timeout)이 났고, 본 프로세스가
|
|
6
|
+
죽은 뒤에도 PostCSS 워커가 남아 계속 커졌다(2026-09-27).
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import os
|
|
11
|
+
import signal
|
|
12
|
+
import subprocess
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
|
|
15
|
+
from okstra_project.dirs import okstra_home, project_json_path
|
|
16
|
+
|
|
17
|
+
from .json_boundary import JsonBoundaryError, load_owned_object
|
|
18
|
+
|
|
19
|
+
CONFIG_KEY = "workerMemoryCapMb"
|
|
20
|
+
# 이 프로젝트의 정상 `next build` 는 5.4–5.6GB 였다. 상한은 그보다 넉넉하되, 물리
|
|
21
|
+
# 메모리가 작은 기기에서는 절반을 넘지 않게 한다.
|
|
22
|
+
_DEFAULT_CEILING_BYTES = 8 * 1024 ** 3
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class MemoryCapConfigError(ValueError):
|
|
26
|
+
pass
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def physical_memory_bytes() -> int:
|
|
30
|
+
return os.sysconf("SC_PAGE_SIZE") * os.sysconf("SC_PHYS_PAGES")
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def memory_cap_bytes(project_root: Path, global_config_path: Path | None = None) -> int:
|
|
34
|
+
"""워커 그룹 RSS 합계 상한. 0 은 감시하지 않음.
|
|
35
|
+
|
|
36
|
+
프로젝트 `.okstra/project.json` → `~/.okstra/config.json` → 기본값 순서다.
|
|
37
|
+
"""
|
|
38
|
+
for label, path in (
|
|
39
|
+
("project", project_json_path(project_root)),
|
|
40
|
+
("global", global_config_path or okstra_home() / "config.json"),
|
|
41
|
+
):
|
|
42
|
+
value = _configured_cap_mb(label, path)
|
|
43
|
+
if value is not None:
|
|
44
|
+
return value * 1024 ** 2
|
|
45
|
+
return min(_DEFAULT_CEILING_BYTES, physical_memory_bytes() // 2)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _configured_cap_mb(label: str, path: Path) -> int | None:
|
|
49
|
+
if not path.is_file():
|
|
50
|
+
return None
|
|
51
|
+
try:
|
|
52
|
+
payload = load_owned_object(path, artifact=f"{label} configuration")
|
|
53
|
+
except JsonBoundaryError as exc:
|
|
54
|
+
raise MemoryCapConfigError(f"{label} config is unreadable: {path}") from exc
|
|
55
|
+
value = payload.get(CONFIG_KEY) if isinstance(payload, dict) else None
|
|
56
|
+
if value is None:
|
|
57
|
+
return None
|
|
58
|
+
if isinstance(value, bool) or not isinstance(value, int) or value < 0:
|
|
59
|
+
raise MemoryCapConfigError(
|
|
60
|
+
f"{CONFIG_KEY} must be a non-negative integer (MB), got {value!r}: {path}")
|
|
61
|
+
return value
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def group_rss_bytes(pgid: int) -> int:
|
|
65
|
+
"""프로세스 그룹 `pgid` 에 속한 프로세스들의 RSS 합계."""
|
|
66
|
+
try:
|
|
67
|
+
output = subprocess.run(
|
|
68
|
+
["ps", "-A", "-o", "pgid=,rss="],
|
|
69
|
+
capture_output=True, text=True, check=True,
|
|
70
|
+
).stdout
|
|
71
|
+
except (OSError, subprocess.CalledProcessError):
|
|
72
|
+
return 0
|
|
73
|
+
total_kb = 0
|
|
74
|
+
for row in output.splitlines():
|
|
75
|
+
fields = row.split()
|
|
76
|
+
if len(fields) == 2 and fields[0] == str(pgid) and fields[1].isdigit():
|
|
77
|
+
total_kb += int(fields[1])
|
|
78
|
+
return total_kb * 1024
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def kill_group(pgid: int) -> bool:
|
|
82
|
+
"""그룹에 남은 프로세스를 SIGKILL 한다. 보낼 대상이 있었으면 True."""
|
|
83
|
+
try:
|
|
84
|
+
os.killpg(pgid, signal.SIGKILL)
|
|
85
|
+
except ProcessLookupError:
|
|
86
|
+
return False
|
|
87
|
+
return True
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
class MemoryWatch:
|
|
91
|
+
"""주기적으로 그룹 RSS 를 재고, 상한을 넘으면 그룹을 죽인다."""
|
|
92
|
+
|
|
93
|
+
interval_seconds = 2.0
|
|
94
|
+
|
|
95
|
+
def __init__(self, cap_bytes: int) -> None:
|
|
96
|
+
self.cap_bytes = cap_bytes
|
|
97
|
+
self.exceeded_bytes = 0
|
|
98
|
+
self._next_sample = 0.0
|
|
99
|
+
|
|
100
|
+
def check(self, pgid: int, now: float) -> bool:
|
|
101
|
+
"""상한을 넘어 그룹을 죽였으면 True."""
|
|
102
|
+
if not self.cap_bytes or self.exceeded_bytes or now < self._next_sample:
|
|
103
|
+
return False
|
|
104
|
+
self._next_sample = now + self.interval_seconds
|
|
105
|
+
rss = group_rss_bytes(pgid)
|
|
106
|
+
if rss <= self.cap_bytes:
|
|
107
|
+
return False
|
|
108
|
+
self.exceeded_bytes = rss
|
|
109
|
+
kill_group(pgid)
|
|
110
|
+
return True
|
|
111
|
+
|
|
112
|
+
def failure(self) -> str:
|
|
113
|
+
mb = 1024 ** 2
|
|
114
|
+
return (
|
|
115
|
+
f"memory-cap exceeded: worker process group used {self.exceeded_bytes // mb} MB, "
|
|
116
|
+
f"cap {self.cap_bytes // mb} MB ({CONFIG_KEY} in .okstra/project.json "
|
|
117
|
+
"or ~/.okstra/config.json)"
|
|
118
|
+
)
|
|
@@ -2787,8 +2787,12 @@ def inject_lead_prompt_computed_tokens(ctx: dict) -> None:
|
|
|
2787
2787
|
"`shutdown_workers`, `record_lead_event`, and `collect_usage` only "
|
|
2788
2788
|
"through its mapping.\n"
|
|
2789
2789
|
"- Do not call a primitive documented by an unselected adapter.\n"
|
|
2790
|
-
"- Record the adapter-required Phase 3 state
|
|
2791
|
-
"`
|
|
2790
|
+
"- Record the adapter-required Phase 3 state. When team-state has "
|
|
2791
|
+
"no `teamCreate` yet, do not append `phase-3-team-create`: the "
|
|
2792
|
+
"first `team dispatch` (cmux-pane run) or `worker-dispatch` writes "
|
|
2793
|
+
"the implicit-team marker, records the checkpoint and prints the "
|
|
2794
|
+
"line to emit. Otherwise append `PROGRESS: phase-3-team-create ...` "
|
|
2795
|
+
"before dispatch.\n"
|
|
2792
2796
|
f"{adapter_setup_block}\n"
|
|
2793
2797
|
"- The run manifest is authoritative for concurrent-run metadata, "
|
|
2794
2798
|
"dispatch backend, and artifact paths."
|
|
@@ -163,12 +163,21 @@ def _clarifications(
|
|
|
163
163
|
_clarification_row(row, activity_by_id, path)
|
|
164
164
|
for row in active
|
|
165
165
|
]
|
|
166
|
-
|
|
166
|
+
superseded = {
|
|
167
|
+
cid
|
|
168
|
+
for row in active
|
|
169
|
+
if isinstance(row.get("resolutionInput"), Mapping)
|
|
170
|
+
for cid in row.get("supersedes") or []
|
|
171
|
+
}
|
|
172
|
+
rows.extend(_carried_clarification_rows(
|
|
173
|
+
ledger, {row.get("id") for row in rows}, path, superseded,
|
|
174
|
+
))
|
|
167
175
|
return rows
|
|
168
176
|
|
|
169
177
|
|
|
170
178
|
def _carried_clarification_rows(
|
|
171
179
|
ledger: Mapping[str, Any], seen_ids: set[object], path: Path,
|
|
180
|
+
superseded: set[str],
|
|
172
181
|
) -> list[dict[str, Any]]:
|
|
173
182
|
"""이월 결정은 active 질문이 아니다. 이번 런 활동 원장을 요구하지 않는다."""
|
|
174
183
|
carried = ledger.get("carriedDecisions")
|
|
@@ -188,7 +197,13 @@ def _carried_clarification_rows(
|
|
|
188
197
|
cid = decision.get("id")
|
|
189
198
|
if cid in seen:
|
|
190
199
|
continue
|
|
191
|
-
|
|
200
|
+
row = _carried_clarification_row(decision)
|
|
201
|
+
if cid in superseded:
|
|
202
|
+
# 다음 run 의 seed 는 `obsolete` 행을 이월하지 않는다. `answered` 로
|
|
203
|
+
# 남기면 대체된 답과 대체한 답이 함께 이월된다.
|
|
204
|
+
row["status"] = "obsolete"
|
|
205
|
+
row.pop("resolution", None)
|
|
206
|
+
rows.append(row)
|
|
192
207
|
seen.add(cid)
|
|
193
208
|
return rows
|
|
194
209
|
|