okstra 0.208.0 → 0.209.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/lead/okstra-lead-contract.md +1 -1
- package/runtime/python/okstra_ctl/cmux.py +21 -16
- package/runtime/python/okstra_ctl/coverage_census.py +8 -1
- package/runtime/python/okstra_ctl/phases/requirements_discovery/profile.md +4 -0
- package/runtime/python/okstra_ctl/report_finalize.py +10 -0
package/package.json
CHANGED
package/runtime/BUILD.json
CHANGED
|
@@ -608,7 +608,7 @@ For every other task type:
|
|
|
608
608
|
|
|
609
609
|
When the host native picker is available and two of those rows could apply, ask with that picker (recommended first). Do not end the turn after the status dump.
|
|
610
610
|
|
|
611
|
-
When the `report-finalize` result carries `censusWarnings`, add one line to the reply with the unverdicted coverage-census cell count per worker (`unverdicted`), or name the `auditProblem`. It is a warning: it changes neither the recommendation nor the next command.
|
|
611
|
+
When the `report-finalize` result carries `censusWarnings`, add one line to the reply with the unverdicted coverage-census cell count per worker (`unverdicted`), or name the `auditProblem`. It is a warning: it changes neither the recommendation nor the next command. When it carries `businessFlowWarnings`, add one line naming each failed business-flow execution's `mode` and `error`; the business-flow step is optional, so this is also a warning only.
|
|
612
612
|
|
|
613
613
|
**Cite run-artifact paths, do not assemble them.** The `report-finalize` result's `reportPaths` carries this run's `humanReport`, `reportRecord`, `teamState`, and `renderFullCopy` command, each already rooted at the project (`.okstra/tasks/<task-group>/<task-id>/runs/...`). Every other run-artifact path the reply cites — the resume command among them — comes from the launch prompt's `## Manifests` / `## Run Paths` lists, which are rooted the same way. A path you compose from a `runs/<task-type>/...` pattern instead is identical across every task of that task-type, so it names no task and does not resolve from the project root either.
|
|
614
614
|
|
|
@@ -379,8 +379,8 @@ def plan_worker_height_resize(
|
|
|
379
379
|
return None
|
|
380
380
|
|
|
381
381
|
|
|
382
|
-
def
|
|
383
|
-
"""The
|
|
382
|
+
def shim_free_worker_path() -> str:
|
|
383
|
+
"""The lead's inherited PATH with cmux's per-surface CLI shims removed.
|
|
384
384
|
|
|
385
385
|
A cmux pane execs its command through `login … bash --noprofile --norc`, so
|
|
386
386
|
none of the user's shell rc runs and PATH is cmux's own short list. What that
|
|
@@ -389,14 +389,23 @@ def shim_free_login_path() -> str:
|
|
|
389
389
|
CLIs — and where a provider cmux does not know has no entry at all, which is
|
|
390
390
|
a plain exit 127 inside the worker wrapper.
|
|
391
391
|
|
|
392
|
-
|
|
393
|
-
|
|
394
|
-
|
|
395
|
-
|
|
392
|
+
The base is the PATH the lead process inherited, because the cli-wrapper
|
|
393
|
+
runtime (and every automatic retry, which always runs there) spawns with it.
|
|
394
|
+
A pane that resolved PATH from the login shell instead could pick another
|
|
395
|
+
install of the same CLI: on 2026-10-04 (fontsninja-v3-site dev-11118) the
|
|
396
|
+
login shell put Homebrew's codex 0.153.4 first, which rejects `gpt-6.1-sol`,
|
|
397
|
+
while the lead's PATH found nvm's 0.160.0, so every first attempt failed and
|
|
398
|
+
every retry passed. The login shell's PATH is the fallback when the inherited
|
|
399
|
+
one is empty after filtering.
|
|
396
400
|
"""
|
|
397
|
-
|
|
401
|
+
path = _without_shims(os.environ.get("PATH", ""))
|
|
402
|
+
return path or _without_shims(_login_shell_path())
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
def _without_shims(path_value: str) -> str:
|
|
398
406
|
return os.pathsep.join(
|
|
399
|
-
entry for entry in
|
|
407
|
+
entry for entry in path_value.split(os.pathsep)
|
|
408
|
+
if entry and SHIM_DIR_MARKER not in entry
|
|
400
409
|
)
|
|
401
410
|
|
|
402
411
|
|
|
@@ -783,7 +792,7 @@ def _surface_from_split_echo(stdout: str) -> str:
|
|
|
783
792
|
|
|
784
793
|
def _exec_worker(surface_uuid: str, *, cwd: Path, command: Sequence[str]) -> None:
|
|
785
794
|
line = worker_command_line(
|
|
786
|
-
cwd=cwd, argv=command, path_value=
|
|
795
|
+
cwd=cwd, argv=command, path_value=shim_free_worker_path()
|
|
787
796
|
)
|
|
788
797
|
started = run_cmux(["respawn-pane", "--surface", surface_uuid, "--command", line])
|
|
789
798
|
if started.returncode != 0:
|
|
@@ -939,11 +948,7 @@ def _pane_geometry(entry: dict[str, Any]) -> PaneGeometry:
|
|
|
939
948
|
|
|
940
949
|
|
|
941
950
|
def _login_shell_path() -> str:
|
|
942
|
-
"""PATH as the user's own login shell resolves it.
|
|
943
|
-
|
|
944
|
-
Falls back to the inherited PATH: a worker started with a shim-filtered
|
|
945
|
-
inherited PATH is still better than one started with cmux's bare list.
|
|
946
|
-
"""
|
|
951
|
+
"""PATH as the user's own login shell resolves it, or '' when it cannot run."""
|
|
947
952
|
shell = os.environ.get("SHELL", "") or "/bin/sh"
|
|
948
953
|
try:
|
|
949
954
|
result = subprocess.run(
|
|
@@ -954,9 +959,9 @@ def _login_shell_path() -> str:
|
|
|
954
959
|
check=False,
|
|
955
960
|
)
|
|
956
961
|
except (OSError, subprocess.SubprocessError):
|
|
957
|
-
return
|
|
962
|
+
return ""
|
|
958
963
|
if result.returncode != 0 or not result.stdout.strip():
|
|
959
|
-
return
|
|
964
|
+
return ""
|
|
960
965
|
return result.stdout.strip()
|
|
961
966
|
|
|
962
967
|
|
|
@@ -242,8 +242,15 @@ def _split_verdict_section(text: str) -> tuple[list[str], str]:
|
|
|
242
242
|
)
|
|
243
243
|
if start is None:
|
|
244
244
|
return [], text
|
|
245
|
+
level = len(lines[start]) - len(lines[start].lstrip("#"))
|
|
246
|
+
# Subheadings (`### EB-001`) group the verdict lines; only a heading at the
|
|
247
|
+
# section's own level or above closes it.
|
|
245
248
|
end = next(
|
|
246
|
-
(
|
|
249
|
+
(
|
|
250
|
+
index for index in range(start + 1, len(lines))
|
|
251
|
+
if _HEADING_RE.match(lines[index])
|
|
252
|
+
and len(lines[index]) - len(lines[index].lstrip("#")) <= level
|
|
253
|
+
),
|
|
247
254
|
len(lines),
|
|
248
255
|
)
|
|
249
256
|
return lines[start + 1:end], "\n".join(lines[:start] + lines[end:])
|
|
@@ -24,6 +24,10 @@
|
|
|
24
24
|
- task rejection-criteria: Which delivered outcome would the reporter reject?
|
|
25
25
|
- task missing-materials: Which missing material blocks reliable routing?
|
|
26
26
|
- task terminology: Which fuzzy or overloaded terms need one canonical form, and what decides it?
|
|
27
|
+
- task existing-reuse: Which existing code path, export, test seam or helper should this work reuse instead of adding a new one?
|
|
28
|
+
- task related-work: Which related, follow-up or blocking task does this work overlap, and which units stay in or out of this split?
|
|
29
|
+
- task input-consistency: Where do the brief, packet and recorded decisions contradict each other or the code?
|
|
30
|
+
- task shared-impact: Which shared type, model, package or consumer outside the stated scope does this change reach?
|
|
27
31
|
- Primary focus areas:
|
|
28
32
|
- classify the work as bugfix, feature, improvement, refactor, or ops
|
|
29
33
|
- capture the reporter's **rejection criteria** — the delivered outcome that would make this work wrong or unacceptable — as a routing input. Consume it from the brief's `Desired Outcome` / `Out of Scope` / `Source Material` when present; when it is absent AND it would change the classification (e.g. bugfix vs feature) or the next-phase choice, raise it as one `decision` clarification row with `Evidence checked: none — reporter intent`. Never infer it — this is a reporter-intent signal, the mirror of improvement-discovery's `Anti-goals`
|
|
@@ -726,6 +726,16 @@ def run_finalize(
|
|
|
726
726
|
for step in steps:
|
|
727
727
|
if step["name"] == "business-flow":
|
|
728
728
|
payload["businessProcess"] = step["result"]
|
|
729
|
+
# 선택 단계라 exitCode 는 0 이다. 실패를 중첩 JSON 에만 두면 리드가 못 본다.
|
|
730
|
+
failed = [
|
|
731
|
+
{"id": row.get("id", ""), "mode": row.get("mode", ""), "error": row.get("error", "")}
|
|
732
|
+
for row in step["result"].get("executions") or []
|
|
733
|
+
if isinstance(row, dict) and row.get("status") == "failed"
|
|
734
|
+
]
|
|
735
|
+
if step["result"].get("error"):
|
|
736
|
+
failed.append({"id": "", "mode": "", "error": str(step["result"]["error"])})
|
|
737
|
+
if failed:
|
|
738
|
+
payload["businessFlowWarnings"] = failed
|
|
729
739
|
census = _census_summary(ctx)
|
|
730
740
|
if census is not None:
|
|
731
741
|
payload["censusWarnings"] = census
|