okstra 0.209.0 → 0.209.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/docs/architecture.md +1 -1
- package/docs/cli.md +9 -1
- package/package.json +1 -1
- package/runtime/BUILD.json +2 -2
- package/runtime/prompts/duties/business-flow-investigator.json +1 -1
- package/runtime/prompts/launch.template.md +1 -1
- package/runtime/prompts/lead/adapters/cmux.md +1 -1
- package/runtime/prompts/lead/convergence.md +2 -2
- package/runtime/prompts/lead/okstra-lead-contract.md +4 -4
- package/runtime/prompts/wizard/prompts.ko.json +15 -14
- package/runtime/python/okstra_ctl/adapters/hosts/capability_adapter.py +15 -4
- package/runtime/python/okstra_ctl/adapters/runtime/assembly.py +7 -3
- package/runtime/python/okstra_ctl/adapters/runtime/cmux.py +8 -4
- package/runtime/python/okstra_ctl/agent/activity.py +2 -2
- package/runtime/python/okstra_ctl/analysis_packet.py +7 -0
- package/runtime/python/okstra_ctl/business_flow/engine.py +2 -2
- package/runtime/python/okstra_ctl/business_flow/source.py +35 -2
- package/runtime/python/okstra_ctl/cmux.py +93 -25
- package/runtime/python/okstra_ctl/convergence.py +29 -10
- package/runtime/python/okstra_ctl/dispatch_checkpoints.py +93 -3
- package/runtime/python/okstra_ctl/dispatch_core.py +6 -1
- package/runtime/python/okstra_ctl/initial_prompt_materialization.py +16 -0
- package/runtime/python/okstra_ctl/phases/implementation_planning/authoring.py +70 -2
- package/runtime/python/okstra_ctl/phases/implementation_planning/instructions/plan-body-verification.md +21 -16
- package/runtime/python/okstra_ctl/phases/implementation_planning/plan_body.py +11 -9
- package/runtime/python/okstra_ctl/phases/implementation_planning/wizard.py +79 -17
- package/runtime/python/okstra_ctl/plan_items.py +4 -2
- package/runtime/python/okstra_ctl/render.py +4 -0
- package/runtime/python/okstra_ctl/report_finalize.py +19 -7
- package/runtime/python/okstra_ctl/run.py +15 -1
- package/runtime/python/okstra_ctl/team.py +6 -4
- package/runtime/python/okstra_ctl/user_response.py +22 -0
- package/runtime/python/okstra_ctl/wizard/cli.py +0 -2
- package/runtime/python/okstra_ctl/wizard/engine.py +33 -46
- package/runtime/python/okstra_ctl/wizard/registry.py +2 -2
- package/runtime/python/okstra_ctl/wizard/roles.py +2 -8
- package/runtime/python/okstra_ctl/wizard/steps_identity.py +7 -5
- package/runtime/python/okstra_ctl/wizard/steps_plan.py +1 -1
- package/runtime/python/okstra_ctl/worker_prompt_body.py +6 -1
- package/runtime/python/okstra_ctl/worker_prompt_policy.py +13 -0
- package/runtime/python/okstra_ctl/write_policy.py +16 -0
- package/runtime/skills/okstra-run/SKILL.md +3 -3
- package/runtime/validators/validate-run.py +9 -2
package/docs/architecture.md
CHANGED
|
@@ -463,7 +463,7 @@ The fourth column is the `workflow.nextRecommendedPhase` pointer Phase 7 leaves
|
|
|
463
463
|
|---|---|---|---|---|
|
|
464
464
|
| `requirements-discovery` | Classify the request as bugfix, feature, refactor, ops, or improvement, then route it to a safe next phase | work category, routing decision, missing-input list, clarification requests | from `requirementsDiscovery.routing.nextTaskType`: `ready` at `error-analysis` or `implementation-option-selection`; `pending` when the run settles on neither | No |
|
|
465
465
|
| `error-analysis` | Analyze the symptoms, causes, and reproduction gaps of a reported error/incident based on evidence | symptom/trigger summary, root-cause hypotheses, reproduction gap, validation path | from `errorAnalysis.routing.nextTaskType`: `ready` at `implementation-option-selection` after a credible cause, or at `error-analysis` for continued investigation | No |
|
|
466
|
-
| `implementation-option-selection` | Compare or validate implementation directions before detailed planning | up to three ranked directions, per-direction `coveragePercent` and `scopePrecisionPercent`, rejected-candidate audit, separate `DIRECTION SELECTION` response | from the `implementationOptionSelection.routing` string enum: `ready` at `implementation-planning` both for a confirmed direction and for `pending-direction-selection` with ranked candidates (the user picks the direction in-session with `/okstra-user-response`, or in the HTML report's `DIRECTION SELECTION` Export saved under `user-responses/`; the planning wizard's `selected_direction_pick` then picks that report; the rationale names the candidates and that path), `pending` when no candidate was ranked, `blocked` on `blocked` | No (strictly read-only; source edits, builds, tests, migrations, and deploys are prohibited) |
|
|
466
|
+
| `implementation-option-selection` | Compare or validate implementation directions before detailed planning | up to three ranked directions, per-direction `coveragePercent` and `scopePrecisionPercent`, rejected-candidate audit, separate `DIRECTION SELECTION` response | from the `implementationOptionSelection.routing` string enum: `ready` at `implementation-planning` both for a confirmed direction and for `pending-direction-selection` with ranked candidates (the user picks the direction in-session with `/okstra-user-response`, or in the HTML report's `DIRECTION SELECTION` Export saved under `user-responses/`; the planning wizard's `selected_direction_pick` then picks that report, and for a report with no recorded selection it lists the ranked candidates and records the user's pick in that sidecar; the rationale names the candidates and that path), `pending` when no candidate was ranked, `blocked` on `blocked` | No (strictly read-only; source edits, builds, tests, migrations, and deploys are prohibited) |
|
|
467
467
|
| `implementation-planning` | Expand one selected direction into an executable plan without changing its mechanism or architecture boundary | selected-direction snapshot/reference, direction realization, affected-file list, Stage Map, validation/rollback, exact plan coverage, YAML frontmatter `approved: false`, **§5.5.9 Plan Body Verification**. Existing plans without `planningContract: selected-direction` retain the legacy option-candidate and `implementation-option:` contract | from `implementationPlanning.outcome`: `ready` at `implementation` on `plan-ready` or on a candidate-comparison plan with no `outcome` (the plan still needs its separate approval before that run starts), `ready` at `implementation-option-selection` on `direction-invalidated` | No |
|
|
468
468
|
| `implementation` | Modify source code according to the approved `implementation-planning` final report. **One run executes exactly one stage** (selected with `--stage <auto\|N>`) | commit list, diff summary, out-of-plan edits block, validation/TDD evidence, rollback verification, verifier results (Antigravity/Codex/Claude), `carry/stage-<N>.json` evidence sidecar | from `implementation.routingRecommendation.target`: `ready` at that phase — `final-verification` on a clean stage, otherwise `error-analysis`, `implementation-planning`, or `implementation` | Yes (limited to the approved plan's file list; `git push`/publish/deploy/real migration prohibited) |
|
|
469
469
|
| `final-verification` | Check completed work for residual defects and regression risk, then make a release judgment | acceptance verdict, residual risk, follow-up routing (`error-analysis`/`implementation-option-selection`/`implementation-planning`/`release-handoff`/`final-verification`) | from `finalVerification.routingRecommendation.target`: `ready` at that value — `release-handoff` only on an `accepted` verdict, otherwise the phase owning the defect (cause, selected direction, or detailed plan). `release-handoff(stage-group)` is a scope qualifier on the same phase, so it projects to `release-handoff`; either verification scope may route there, since a handoff opens one PR per stage. `final-verification` re-runs this phase on the same head, and is for a run whose every remaining blocker is an environment or configuration fault this report already diagnosed. `done` becomes `terminal` | No (read-only tests only) |
|
package/docs/cli.md
CHANGED
|
@@ -725,6 +725,14 @@ the very cmux it is running inside is still open. Passing
|
|
|
725
725
|
The value is recorded as `terminalBackend` in the run manifest, and every later
|
|
726
726
|
consumer reads it from there rather than probing again.
|
|
727
727
|
|
|
728
|
+
A `cmux-pane` run also records where the lead sits, as `cmuxLead`
|
|
729
|
+
(`workspaceId`, `surfaceId`), taken from the terminal that ran preparation. Worker
|
|
730
|
+
placement, lead-pane sizing, sidebar messages and pane reclaim read that record
|
|
731
|
+
before the dispatching process's own `CMUX_*` variables, because the lead process
|
|
732
|
+
may not have inherited them from that terminal: Claude Code can start the lead by
|
|
733
|
+
claiming a pre-warmed process created in another workspace. When the recorded
|
|
734
|
+
surface has closed, workers run as cli-wrapper processes instead.
|
|
735
|
+
|
|
728
736
|
- **Enabled (default)**: Immediately after the report-writer worker drafts its narrative in Phase 6, the lead extracts the synthesized plan into `P-*` items and dispatches them for reverification to every analyzer worker: `claude`, `codex`, and opted-in `antigravity`. A selected-direction plan uses `P-Dir-1` plus its step, dependency, validation, rollback, requirement, preparation, and variation items. A legacy candidate plan retains `P-Opt-*`. Worker verdicts (`AGREE` / `DISAGREE(a-e)` / `SUPPLEMENT`) are aggregated into one of four gate results: `passed`, `passed-with-dissent`, `blocked-by-disagreement`, or `aborted-non-result`. The approval control is available only for `passed` or `passed-with-dissent`. Items with majority DISAGREE become rows with `Blocks=approval` in `## 1. Clarification Items`. There is no automatic revision; the user answers and resumes the same phase.
|
|
729
737
|
- **Disabled (with `--no-plan-verification`)**: The entire Phase 6 substep is skipped and the Approval marker is always rendered at the top of the final report, matching legacy behavior. This is a fast-iteration opt-out and is not recommended for a handoff-ready plan.
|
|
730
738
|
- **Advisory auto-path (not this flag)**: when `designPreparation.mode` is `no-design-inputs` and the Stage Map has exactly one row, `okstra plan-items prepare` sets `convergence.planBodyVerification.gating=false`. Extraction and one verification round still run; the self-fix loop and a sweep batch do not. Two-or-more stages, a PREP item, or non-empty design-preparation items keep `gating=true`.
|
|
@@ -879,7 +887,7 @@ The `okstra` Node CLI (`bin/okstra`) provides both installer/admin commands and
|
|
|
879
887
|
| `okstra agent-prompt materialize --project-root <dir> --run-manifest <path> --batch <file> [--jobs-out <path>] [--json]` | Materialize every invocation of one dispatch batch in one call. The batch file holds `{"invocations": [...]}`; each entry maps the run-mode per-invocation flags (`invocation-id`, `audience`, `instruction`, `prompt`, `worker-id`, `dispatch-kind`, `assignment-ref`, `source-role-execution-ref`, `result`, `corrections`, `audit-source`, and `replace-undispatched: true`) to values. Every entry goes through the same parser and materialization as a single call and fails with the same error. Entries run in order; a failure names the entry and every invocation already published, and an identical rerun after the fix reuses those. An unknown key, a repeated invocation id, a per-invocation flag given at the top level, or mixed dispatch kinds under `--jobs-out` is refused before any entry is written. `--jobs-out` (also accepted by a single run-mode `materialize`) writes the verified jobs file for `okstra team dispatch --jobs-file` exactly as `agent-prompt jobs` would; every entry must share one dispatch kind. Prints each prompt path and then the jobs path; `--json` returns `invocations` (one single-call payload each) and `jobsPath`. |
|
|
880
888
|
| `okstra agent-prompt materialize\|check-corrections\|apply-corrections\|verify\|record-dispatch\|link-result\|reject-result\|abandon-attempt\|materialize-result\|complete\|verify-completion` | Internal invocation-contract CLI. `materialize` composes model assignment, functional duty, and task instructions; `verify` rejects identity, path, snapshot, assignment, source, or digest drift. Every run-branch report-writer prompt gets its `## Output` section (narrative, pointer record, reading audit) rendered by okstra, and an instruction body that writes a `## Output` or `## Corrections` heading is refused. A corrective report-writer round — the narrative at `reportNarrativePath` already exists and its structure parses, value defects included — must pass `--corrections <ledger>` (`schemas/report-writer-corrections-v1.0.schema.json`: `replace` / `remove` / `add` / `move` / `rewrite` entries keyed by the validator's field-path grammar, `baseNarrativePath` naming a preserved copy of the attempt, and optional `baseNarrativeSha256` binding that version): the ledger is applied to that base and checked against the writer-owned schema and the task's semantic validator before dispatch, every defect is reported at once, and okstra renders the prompt's `## Corrections` section from it; a report-writer materialization without a ledger over such a narrative is refused before any prompt is written, while a narrative whose structure does not parse (line grammar, unknown top-level field) is re-authored without one. `check-corrections --run-manifest <path> --corrections <ledger> [--json]` runs the same check without materializing (exit 1 lists the defects; `mechanical: true` means every entry is a validated `replace`, `remove`, `add`, or `move`, including derived planning step counts). `apply-corrections` with the same arguments applies such a mechanical ledger without a writer round: it writes the corrected narrative to `reportNarrativePath` and records a `lead-correction-applied` activity row (`evidenceRefs` = ledger path + correction ids) through the run's activity contract; `--rewrite-results <file>` also accepts hash-bound replacement values for exactly the requested rewrite ids. Correction-only materialization sends those target fields, evidence, and constraints instead of the initial instructions and complete synthesis packet; the runtime merges the submitted values and checks the complete narrative before writing. It refuses unresolved `rewrite` entries or any defect, a stale live narrative, a base that is the live narrative, a run without `activityContractVersion` 1, and a ledger already applied. An empty ledger can preflight the initial writer result and derive `stageMap[].stepCount` from matching execution rows without another writer call. Run-backed calls resolve `assignmentRef` from the manifest, enforce `authorizedPaths`, and reject real-path or symbolic-link escape. `record-dispatch` records a verified host-native specification before dispatch and `link-result` binds the accepted result; one result path belongs to one dispatch, so a corrective round retires the first attempt with `reject-result --dispatch-id <first> --superseded-by <corrective> --reason <text>` before the new link is accepted — the rejected row stays in `agentResultLinks` carrying `supersededBy` and `rejectionReason` rather than being deleted. The corrective dispatch is a new invocation: an invocation whose last attempt finished with a mutation takes no further attempt (`execution_manifest._validate_next_attempt` lets only `failed-no-mutation` be followed), so a retry attempt of the rejected invocation itself is refused by the manifest, and `reject-result` does not make it possible. `abandon-attempt --invocation-ref <ref> --reason <text>` closes a started attempt whose worker died without producing a result — the one case neither `link-result` (which needs the result file) nor the dispatch-failure path covers — so a retry can follow it instead of the run having to be re-rendered. It refuses any attempt whose `writePolicy.sourcePolicy.mode` is not `source-readonly`: closing an attempt records `failed-no-mutation`, which is true by policy for a read-only worker and a guess for a mutating one. Standalone calls are identified by `(purpose, invocationId)` under `.okstra/agent-invocations/<purpose>/`; they publish a canonical result envelope and publish the completion marker last. Consumers use only the `returnedBody` from `verify-completion`. Metadata contains exactly `catalogDigest`, `assignmentDigest`, `dutyDigest`, `instructionDigest`, and `promptDigest`; JSON inputs use UTF-8, sorted keys, compact separators, and no non-finite values, while duty files use versioned sorted-name/byte framing. Instruction sources use `{kind: project\|runtime, path: <relative POSIX path>}` and never persist an installed absolute runtime path. A published prompt is immutable, so re-running `materialize` with an edited instruction file fails as `existing_invocation_conflict`; `--replace-undispatched` is the one exit, for a call that failed a pre-dispatch gate and therefore ran nowhere — it covers a differing prompt and a differing metadata alike, since the two are published together and describe one call. It republishes prompt and metadata together, and it is verified rather than trusted — a row in `agentDispatches` or `workerDispatches` naming this `invocationId` refuses the replacement and names the dispatch that used it. |
|
|
881
889
|
| `okstra agent-prompt refreeze-contracts --project-root <dir> --run-manifest <path> [--json]` | Re-freeze one run's duty contract snapshot in the installed format and stamp `agentContract.catalogDigest` and `contractFormatVersion` on its run manifest. A run freezes its contracts at prepare time and pins their digest; installing a release that changed the contract format leaves that digest unmatchable, so the run can materialize no further prompt and `materialize` stops with the format message naming this command. It replaces the frozen directory's contents (no file of the old format is kept), touches no prompt, result or ledger, and reports `changed: false` when the run is already on the installed format. User-invoked recovery only — nothing runs it automatically, because a run's contracts are frozen on purpose. |
|
|
882
|
-
| `okstra team dispatch --project-root <dir> --run-manifest <path> [--workers <csv>] [--jobs-file <path>] [--dry-run]` / `okstra team await --project-root <dir> --run-manifest <path> [--json]` / `okstra team teardown --project-root <dir> --run-manifest <path> [--dry-run] [--json]` | Read a `leadRuntime=external` run manifest and dispatch, await, or tear down pane-backed workers. On a `cmux-pane` run, dispatch records `phase-3-team-create` when it writes the implicit-team marker, `phase-4-dispatch` for each `initial` job and `phase-6-synthesis` on the first report-writer dispatch; await records `phase-5-poll` and `phase-5-collect` for each `initial` dispatch it settled. Both write the run's lead-events ledger and print the `PROGRESS:` lines to emit (`progressLines` under `--json`). Any other run records nothing there. Default dispatch excludes report writer; Phase 6 selects it explicitly, and mixed analysis/report jobs are rejected. If a pane cannot be opened, gracefully degrade to the CLI wrapper, print a `DEGRADED <role>: cmux-pane -> cli-wrapper (<why>)` line, and record the fallback in `workerDispatches[].degradedFrom` with the reason in `degradedReason`. A degraded worker is not waited for inside the dispatch — it settles through `okstra team await` like a pane worker, so the round still runs concurrently |
|
|
890
|
+
| `okstra team dispatch --project-root <dir> --run-manifest <path> [--workers <csv>] [--jobs-file <path>] [--dry-run]` / `okstra team await --project-root <dir> --run-manifest <path> [--json]` / `okstra team teardown --project-root <dir> --run-manifest <path> [--dry-run] [--json]` | Read a `leadRuntime=external` run manifest and dispatch, await, or tear down pane-backed workers. On a `cmux-pane` run, dispatch records `phase-3-team-create` when it writes the implicit-team marker, `phase-4-dispatch` for each `initial` job and `phase-6-synthesis` on the first report-writer dispatch; await records `phase-5-poll` and `phase-5-collect` for each `initial` dispatch it settled. Both write the run's lead-events ledger and print the `PROGRESS:` lines to emit (`progressLines` under `--json`). Any other run records nothing there. Default dispatch excludes report writer; Phase 6 selects it explicitly, and mixed analysis/report jobs are rejected. If a pane cannot be opened, gracefully degrade to the CLI wrapper, print a `DEGRADED <role>: cmux-pane -> cli-wrapper (<why>)` line to stderr (stdout stays one JSON object), and record the fallback in `workerDispatches[].degradedFrom` with the reason in `degradedReason`. A degraded worker is not waited for inside the dispatch — it settles through `okstra team await` like a pane worker, so the round still runs concurrently |
|
|
883
891
|
| `okstra agent-activity append --project-root <dir> --run-manifest <path> --kind <kind> --agent <assigned-id> (--summary <text>\|--summary-file <markdown>) --outcome <outcome> [--plan-item-id <current-id>]… [--command <text> --command-cwd <dir> --command-exit-code <n> --command-output-file <markdown>] [--request-ref <returned-ref>]` | Append one structured activity after checking the agent against this run's role assignments and every plan item against its current convergence state. Python returns an `activityRequestRef`; supply only that returned value with `--request-ref` to retry idempotently. A new call without it remains a distinct activity even with identical contents. Legacy JSON command records remain automation compatibility only. |
|
|
884
892
|
| `okstra agent-activity project --project-root <dir> --run-manifest <path> --data <data.json>` | Project this run's canonical activity events into `agentActivity[]`. The command preserves event order, rejects duplicate or decreasing activity IDs, and replaces no other report field. A historical manifest without `activityContractVersion: 1` returns an empty projection and leaves data.json unchanged. Normal Phase 7 execution reaches this behavior through `report-finalize`; use the standalone command only for diagnostics. |
|
|
885
893
|
| `okstra lead-progress append --project-root <dir> --run-manifest <path> --phase <phase-id> [--worker <role>] [--field NAME=VALUE]… [--detail <text>]` | Append one `PROGRESS:` checkpoint to the run's `leadEventsPath` and print the line to emit to the user as `progressLine`. The checkpoint is what `validate_session_conformance.py` reads, and on a host whose adapter declares `sessionAccounting: artifact-only` the ledger is the only place it can read one — a conversation line alone is not retained there. `--phase` accepts the phase ids the lead contract's "Progress reporting (BLOCKING)" list defines; the fixed-prose checkpoints render their contract wording without `--detail`. `--worker` is resolved against team-state and rewritten to the roster `workers[].role` the per-worker checks match, so a phase-specific functional label still lands on the right worker; a name that matches no roster row is written through with a note on stderr. |
|
package/package.json
CHANGED
package/runtime/BUILD.json
CHANGED
|
@@ -6,7 +6,7 @@
|
|
|
6
6
|
"requiredConduct": ["Read implementation, callers, configuration, test bodies and available existing verification records. Inspect every registered candidate for inbound as well as outbound relationships. Follow indirect calls, events, queues, shared data, file transfers and batch processing. Record every unscanned or inaccessible segment."],
|
|
7
7
|
"decisionPrinciples": ["Use the preserved project-specific source baseline for before/after comparisons. Separate static inspection, existing execution verification, expected changes and unknowns. Preserve contradictory claims and investigate their evidence before proposing an unresolved business-policy question."],
|
|
8
8
|
"authorityAndBoundaries": ["The request explicitly lists the source roots and existing artifacts you may read. Source, Git state and existing artifacts are read-only. Return your result in the final response; the runtime owns publication."],
|
|
9
|
-
"evidenceStandards": ["Cite projectId, relative path, one-based line range and
|
|
9
|
+
"evidenceStandards": ["Cite projectId, relative path, one-based line range and an excerpt that copies that whole range verbatim, including indentation, with no elision. Do not infer a deployed behavior from source, an execution result from a test body, or absence of impact from missing access. Claims marked execution-verified also cite an existing verification record."],
|
|
10
10
|
"collaborationContract": ["Consume relevant shared facts with their version and conflict status. Contribute only claims grounded in the supplied source state. Never resolve conflict by recency or suppress either original provenance."],
|
|
11
11
|
"completionCriteria": ["Return the requested JSON object with the complete business flow, inspected candidates, structured claims and evidence, source gaps, risks and step-level changes. Empty contributions are valid when no new business facts were observed."],
|
|
12
12
|
"prohibitions": ["Do not edit source or Git state, execute tests, start services, deploy, migrate data, access uncited external material, or publish your own knowledge/report files. Do not invent source evidence, classify expected behavior as current fact, or describe Okstra agent procedure as the product business process."],
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
|
|
10
10
|
Emit one `PROGRESS: <phase-id> <verb-phrase>` line as plain user-facing text at every checkpoint enumerated in the lifecycle core contract (`{{OKSTRA_LEAD_CONTRACT_PATH}}` "Progress reporting (BLOCKING)") — phase-1-intake start/complete, phase-2-prompts, phase-3-team-create, phase-4-dispatch (per worker), phase-5-collect (per worker), phase-5.5-convergence (per round), phase-6-synthesis, phase-7-persist, and final `complete`. One line per checkpoint, never batched, never replaced with prose. This is the only signal the user has during multi-minute silent windows. Record each one with `okstra lead-progress append --project-root <dir> --run-manifest <path> --phase <phase-id>` and emit the `progressLine` it prints — that call is what puts the checkpoint where the validator reads it. The exception is a checkpoint the command performing the step records itself (`team dispatch`, `team await`, `team reclaim`, `worker-dispatch`, `report-finalize`; the contract lists which): emit the `PROGRESS:` line that command prints and do not append it again.
|
|
11
11
|
|
|
12
|
-
When the run manifest declares `activityContractVersion: 1`, call `okstra agent-activity append` before each required activity boundary. Only after the structured append succeeds, emit the matching `PROGRESS:` line and the immediately following `ACTIVITY:` projection from the same fields. If the structured append fails, do not mark that boundary completed. Never reconstruct structured activity by parsing `ACTIVITY:` conversation text.
|
|
12
|
+
When the run manifest declares `activityContractVersion: 1`, call `okstra agent-activity append` before each required activity boundary. Only after the structured append succeeds, emit the matching `PROGRESS:` line and the immediately following `ACTIVITY:` projection from the same fields. If the structured append fails, do not mark that boundary completed. The exception is a `worker-dispatched` / `worker-completed` activity whose checkpoint a command records: that command appends the activity too and prints its `ACTIVITY:` line after the `PROGRESS:` line — emit both and do not append again. Never reconstruct structured activity by parsing `ACTIVITY:` conversation text.
|
|
13
13
|
|
|
14
14
|
For a new `implementation-planning` run, the plan-body sequence is initial verification → one planner self-fix → targeted re-verification → user gate. The initial verification is round 1, the targeted re-verification is round 2, and a second automatic self-fix is a contract violation. A user-directed correction does not consume the automatic self-fix limit, and a verification failure after that correction does not restart the automatic loop.
|
|
15
15
|
|
|
@@ -61,7 +61,7 @@ Use the screen to tell "still working" from "stuck", and to see at a glance whic
|
|
|
61
61
|
- Worker completion is valid only from `workerDispatches[]`, terminal status sidecars, and required Result Paths. Pane creation alone is not completion.
|
|
62
62
|
- Reverify uses a fresh jobs file at `runs/<task-type>/state/reverify-jobs-r<N>-<task-type>-<seq>.json`, sets `dispatchKind: "reverify-r<N>"`, and dispatches with `okstra team dispatch --project-root <root> --run-manifest <path> --dispatch-kind reverify-r<N> --jobs-file <jobs-file>`.
|
|
63
63
|
- Report-writer uses a fresh one-job jobs file with `dispatchKind: "report-writer"` and the same schema, then dispatches through `okstra team dispatch --project-root <root> --run-manifest <path> --jobs-file <jobs-file>`.
|
|
64
|
-
- Build a batch's jobs file in the call that materializes it: one `materialize --batch <file> --jobs-out <jobs-file>` call for every worker of the batch (convergence "Invocation materialization gate"), or one `materialize … --jobs-out <jobs-file>` call for the report writer, chained with `&& okstra team dispatch … --jobs-file <jobs-file>` in the same shell command. One materialize, verify, or jobs call per worker adds one lead turn over the whole context per worker; `team dispatch` verifies every invocation of the jobs file, so no `agent-prompt verify` call precedes it.
|
|
64
|
+
- Build a batch's jobs file in the call that materializes it: one `materialize --batch <file> --jobs-out <jobs-file>` call for every worker of the batch (convergence "Invocation materialization gate"), or one `materialize … --jobs-out <jobs-file>` call for the report writer, chained with `&& okstra team dispatch … --jobs-file <jobs-file>` in the same shell command. Leave `--json` off `materialize` in that chain: with it the output holds two JSON documents and parsing it as one fails. One materialize, verify, or jobs call per worker adds one lead turn over the whole context per worker; `team dispatch` verifies every invocation of the jobs file, so no `agent-prompt verify` call precedes it.
|
|
65
65
|
- For metadata already materialized (a retry), generate v2 jobs files with `okstra agent-prompt jobs --project-root <root> --run-manifest <path> --dispatch-kind <kind> --metadata <prompt-meta.json> [--metadata <prompt-meta.json>] --out <jobs-file>`. Pass only the metadata paths for the core-planned batch. The command derives the `workers` array, canonical role execution, result headers, and five digests, and verifies them through the same consumer used at dispatch. Do not transcribe those fields. Identical output is reused; differing output is preserved and requires a new `--out` path. Existing v1 files remain readable by dispatch.
|
|
66
66
|
- `workerResultPath` is the path the prompt tells the worker to write: the prompt's `**Result Path:**`, or `**Worker Result Path:**` for the report writer. `okstra team dispatch` refuses an entry whose value differs from that anchor, because the collector waits on `workerResultPath` while the worker writes where the anchor says. A reverify result is named `<worker-id>-worker-reverify-r<N>-<task-type>-<seq>.md`: the `-worker-` token is what the audit sidecar name inserts `-audit-` after (a name without it is refused with `worker result path has no canonical -worker- token`), and the round label is what keeps one round's file apart from the next. The report writer's `workerResultPath` is the roster's `resultPath` for `report-writer` and its `**Result Path:**` is the run manifest's `reportNarrativePath` — the report-writer materialization ([report-writer](../report-writer.md)) refuses any other pair. **Enforced:** `_validate_jobs_file_prompt_anchors` in `scripts/okstra_ctl/dispatch_state.py`, `_validate_report_writer_paths` in `scripts/okstra_ctl/agent/prompt_cli/materialize.py`.
|
|
67
67
|
- `role` names the role execution's own role — `verifier` for reverify, `report-writer` for the report writer. It is not a per-round label: `dispatch_state.py` requires the entry's `role` to equal both the role execution's `role` and the duty's role, so a value like `worker-reverify-r<N>` is refused as `jobs file v2 identity does not match role execution authority`. The round lives in `dispatchKind` and in `invocationRef`. The report-writer completion paths include its narrative Markdown, worker-result pointer, and audit sidecar; they do not include the Phase 7 report record.
|
|
@@ -104,7 +104,7 @@ Read the worker result files generated in Phase 4/5 and extract individual findi
|
|
|
104
104
|
- Same semantics but disjoint ticket sets → separate groups (do NOT over-merge across tickets).
|
|
105
105
|
- Only one worker confirms a finding → one single-source group.
|
|
106
106
|
4. When grouping is ambiguous, prefer splitting over merging (avoid over-merging). Semantic matching, ticket-set equality, and evidence interpretation remain lead judgments; the engine does not perform fuzzy matching or decide whether evidence is credible.
|
|
107
|
-
5. Author the fixed grouping Markdown accepted by `okstra convergence prepare-groups --run-manifest <run-manifest> --input <grouping.md>`, then run that command. Python owns the artifact identifier, target path, schema version, task identity, run-manifest reference, and every participant reference. Each Markdown group records ticket IDs, origin worker and evidence, discovering workers, source worker item IDs, and optional captured evidence. An analysis sidetrack with no ticket uses an empty `Tickets:` value, never a placeholder. A group that answers coverage-census cells adds one optional line, `Cells: C-…, C-…`, which the command records as `cellRefs`; leave it out for a finding outside the census. Use the ordered functional roster: finding workers have the `analysis` audience, the report author has `report-writer`, and the lead uses `lead`. A lead source never votes. For `implementation` runs the convergence sources are the verifier-role results only — the executor's result is deliverable evidence, not a convergence source (**Enforced:** `_validate_worker_execution_identity` in `scripts/okstra_ctl/convergence_engine.py` rejects an `implementer` source with `analysis audience source role is not allowed`). Never infer live evidence or functional scope from wording, provider, model, or execution label.
|
|
107
|
+
5. Author the fixed grouping Markdown accepted by `okstra convergence prepare-groups --run-manifest <run-manifest> --input <grouping.md>`, then run that command. Python owns the artifact identifier, target path, schema version, task identity, run-manifest reference, and every participant reference. Each Markdown group records ticket IDs, origin worker and evidence, discovering workers, source worker item IDs, and optional captured evidence. An analysis sidetrack with no ticket uses an empty `Tickets:` value, never a placeholder. A group takes one `Source:` line per worker: when one worker reported two items that fit the same group, the second item becomes its own group — never drop it. **Enforced:** `_group_source_provenance` in `scripts/okstra_ctl/convergence.py` refuses a second `Source:` line for the same worker and names both item ids. A group that answers coverage-census cells adds one optional line, `Cells: C-…, C-…`, which the command records as `cellRefs`; leave it out for a finding outside the census. Use the ordered functional roster: finding workers have the `analysis` audience, the report author has `report-writer`, and the lead uses `lead`. A lead source never votes. For `implementation` runs the convergence sources are the verifier-role results only — the executor's result is deliverable evidence, not a convergence source (**Enforced:** `_validate_worker_execution_identity` in `scripts/okstra_ctl/convergence_engine.py` rejects an `implementer` source with `analysis audience source role is not allowed`). Never infer live evidence or functional scope from wording, provider, model, or execution label.
|
|
108
108
|
|
|
109
109
|
The command sets each worker's paired `participantRef` and `sourceRoleExecutionRef` from the run manifest's canonical role state. It sets `sourceRoleExecutionRef` to the selected source `RoleExecution` row's `roleExecutionRef`, not that row's `sourceRoleExecutionRef` field.
|
|
110
110
|
6. Do not write a queue or classification in this grouped-input artifact. `okstra convergence seed` classifies Round 0 the same way in both modes: a group whose sources are **two or more distinct role executions** becomes `full-consensus` immediately, and only single-source groups enter the working queue. Independent co-derivation is already cross-verification — the adversarial burden of proof targets single-source claims, not a finding two roles reached on their own. A source is counted once per analysis worker, and one analysis worker is exactly one `sourceRoleExecutionRef` — the same identity the reverify roster uses for independence — so two roles held by one provider count as two and no role can count twice. **Enforced:** `_parse_workers` rejects a duplicate `workerId` and `_validate_worker_execution_identity` rejects a duplicate `sourceRoleExecutionRef`, both in `scripts/okstra_ctl/convergence_engine.py`. Semantic grouping merges provenance only; it does not decide a single-source finding is reliable. Section 7 never enters the grouped input.
|
|
@@ -127,7 +127,7 @@ Follow this protocol exactly:
|
|
|
127
127
|
|
|
128
128
|
0. Version-selected schemas describe what the reducer reads: `schemas/convergence-groups-v1.0.schema.json` accepts only legacy groups, `schemas/convergence-groups-v2.0.schema.json` accepts only explicit v2 execution identity, and `schemas/convergence-round-results-v1.0.schema.json` feeds step 4's `apply-round --results`. `schemas/convergence-critic-results-v1.0.schema.json` is a fourth shape but **not** a reducer input — it describes the critic worker's own result document. Step 6's `apply-critic-gaps --results` takes the coverage batch you assemble from those candidates plus each analyser's vote (`{schemaVersion, taskKey, mode, provider, modelExecutionValue, dispatches[], gaps[]}`, spelled out in §"Coverage critic pass" §"State"); feeding the critic document straight in is rejected, by design. `okstra convergence example --kind <groups|round-results|critic-results>` prints a deterministic valid v1 instance of each and writes only JSON to stdout.
|
|
129
129
|
1. Run `okstra convergence seed --groups <groups> --run-manifest <current-run-manifest> --work-state <work> --final-state <final> --migration-dir <state/migrations>` for a v2 worker roster. `--run-manifest` must be the exact current manifest named by the groups document's `runManifestPath`; a previous run from the same task is not interchangeable. `seed` refuses a grouping that cites a worker whose initial attempt the mutation audit discarded (`contract-failed-unattributed`, `mutation-present-unresolved`) — the result file is still on disk, but the ledger says it was never accepted. Re-dispatch that worker as a new invocation and cite the new result, or leave it out of the grouping. **Enforced:** `discarded_worker_errors` in `scripts/okstra_ctl/convergence_provenance.py`, called by `_seed`. Omit the flag for a legacy v1 roster. A `reuse-final` action means validate the existing final and continue to Phase 6. `create-work`, `resume-work`, and `restart-round0` continue with planning.
|
|
130
|
-
2. Run `okstra convergence plan-round --work-state <work> --plan <round-plan>`. This is read-only with respect to the working state.
|
|
130
|
+
2. Run `okstra convergence plan-round --work-state <work> --plan <round-plan>`. This is read-only with respect to the working state. Its stdout JSON carries `round` and `inputQueueSize`; on a `dispatch` plan the round's `PROGRESS: phase-5.5-convergence round=<N> queue=<count>` line copies those two values — never count the queue by hand. **Enforced:** `_plan_round` in `scripts/okstra_ctl/convergence.py`, `tests/run/test_convergence_cli.py`.
|
|
131
131
|
3. Every plan carries `dispatchable`: `true` on an `action: "dispatch"` plan, `false` on an `action: "finalize"` plan. When the plan action is `dispatch`, create exactly one reverify prompt for each `dispatches[]` row and dispatch it through the selected runtime adapter. Its findings are exactly that row's `findingIds`.
|
|
132
132
|
4. Run `okstra convergence collect-results --plan <round-plan> --mode <adversarial|collaborative> --result <worker>=<path>… --run-manifest <current-run-manifest> --output convergence-round-<N>-results-<task-type>-<seq>.json`. Each planned worker's terminal outcome and `durationMs` come from the manifest's attempt ledger, not from the lead: an attempt still `started` refuses the collect — run `okstra team await` first so the attempt closes — and a worker whose attempt did not close `ok` cannot contribute a vote even if its result file exists. **Enforced:** `_dispatch_rows_from_manifest` in `scripts/okstra_ctl/convergence.py`. Then run `okstra convergence apply-round --work-state <work> --plan <round-plan> --results <round-results>`.
|
|
133
133
|
5. Repeat `plan-round` and `apply-round` until the plan action is `finalize`. A `finalize` plan is **never dispatched**: it carries `dispatchable: false`, an empty `dispatches[]`, and a `note` naming the next command. Its `round` is only the `<N>` in `convergence-round-<N>-plan-<task-type>-<seq>.json` — not a round to run — so do not open reverify workers for it. **Enforced:** `okstra convergence apply-round` refuses a non-dispatch plan and returns the gate's closing reason with the remedy (`scripts/okstra_ctl/convergence_engine.py` `apply_round_results`).
|
|
@@ -78,7 +78,7 @@ Every `okstra` command the lead documents cite, grouped by phase, each spelled w
|
|
|
78
78
|
| Command | Use when | Procedure |
|
|
79
79
|
|---|---|---|
|
|
80
80
|
| `okstra agent-prompt materialize --project-root <dir> --invocation-id <id> --audience <audience> --instruction <path> --prompt <path>` | Materialize one worker prompt; the command owns the anchor headers and paths. Add `--jobs-out <jobs-file>` in run mode to also write its verified jobs file (report writer) | `convergence` "Invocation materialization gate"; `report-writer` "Report-writer dispatch" |
|
|
81
|
-
| `okstra agent-prompt materialize --project-root <dir> --run-manifest <path> --batch <file> --jobs-out <jobs-file>` | Materialize every worker of one dispatch batch and write its jobs file in one call — one call per batch, never one per worker. For `runner=cli-wrapper` rows chain the dispatcher the execution surface selects in the same shell command: `&& okstra team dispatch … --jobs-file <jobs-file>` on `terminalBackend: cmux-pane`, `&& okstra worker-dispatch … --jobs-file <jobs-file>` otherwise. `--jobs-out` needs a v2 run manifest; `runner=native-session` rows take the batch without `--jobs-out` and keep their per-row `verify` + `record-dispatch` + host call | `convergence` "Invocation materialization gate" |
|
|
81
|
+
| `okstra agent-prompt materialize --project-root <dir> --run-manifest <path> --batch <file> --jobs-out <jobs-file>` | Materialize every worker of one dispatch batch and write its jobs file in one call — one call per batch, never one per worker. For `runner=cli-wrapper` rows chain the dispatcher the execution surface selects in the same shell command: `&& okstra team dispatch … --jobs-file <jobs-file>` on `terminalBackend: cmux-pane`, `&& okstra worker-dispatch … --jobs-file <jobs-file>` otherwise. In the chained form leave `--json` off `materialize`: it then prints one path per line and the dispatcher's single JSON object follows, whereas with `--json` the output holds two JSON documents and parsing it as one fails. `--jobs-out` needs a v2 run manifest; `runner=native-session` rows take the batch without `--jobs-out` and keep their per-row `verify` + `record-dispatch` + host call | `convergence` "Invocation materialization gate" |
|
|
82
82
|
| `okstra agent-prompt verify --project-root <dir> --metadata <path>` | Immediately before a native-session dispatch (`record-dispatch`). Not needed before a `--jobs-file` dispatch: jobs generation and the dispatcher verify every invocation | `convergence` "Invocation materialization gate" |
|
|
83
83
|
| `okstra agent-prompt record-dispatch --project-root <dir> --run-manifest <path> --metadata <path> --enforcement-mode <mode>` | Record a native-session dispatch | `convergence` "Invocation materialization gate" |
|
|
84
84
|
| `okstra agent-prompt jobs --project-root <dir> --run-manifest <path> --dispatch-kind <kind> --metadata <path> --out <jobs-file>` | Build a verified jobs file for `okstra team dispatch` | cmux adapter "cmux dispatch details" |
|
|
@@ -230,7 +230,7 @@ Some checkpoints are recorded by the command that performs the step, because tha
|
|
|
230
230
|
- The team commands record nothing on any other `terminalBackend`: a run that dispatches with `okstra team dispatch` and waits with `okstra team await` there appends the team commands' checkpoints itself. `okstra worker-dispatch` still records its own, as listed above.
|
|
231
231
|
- `okstra report-finalize` without `--only` records `phase-7-persist` before its first step, so `validate-run` reads it.
|
|
232
232
|
|
|
233
|
-
Emit the `PROGRESS:` lines those commands print (the trailing text lines of `team await` and `team reclaim`, or `progressLines` in the JSON the others print) verbatim, and do not call `lead-progress append` for them — each extra call is one more lead turn over the whole context. Everything else still goes through `lead-progress append`: every other checkpoint; `phase-3-team-create` when the lead itself recorded the marker (the concurrent-run `skipped (concurrent-run)` variant); and `phase-4-dispatch`, `phase-5-collect` and `phase-6-synthesis` for a worker dispatched through a host-native primitive (`record-dispatch` / `link-result`). `phase-5-collect` from a command does not replace the lead's own acceptance check (`okstra worker-audit-check`)
|
|
233
|
+
Emit the `PROGRESS:` lines those commands print (the trailing text lines of `team await` and `team reclaim`, or `progressLines` in the JSON the others print) verbatim, and do not call `lead-progress append` for them — each extra call is one more lead turn over the whole context. Everything else still goes through `lead-progress append`: every other checkpoint; `phase-3-team-create` when the lead itself recorded the marker (the concurrent-run `skipped (concurrent-run)` variant); and `phase-4-dispatch`, `phase-5-collect` and `phase-6-synthesis` for a worker dispatched through a host-native primitive (`record-dispatch` / `link-result`). `phase-5-collect` from a command does not replace the lead's own acceptance check (`okstra worker-audit-check`). On an activity-contract-v1 run the command that records a worker's `phase-4-dispatch` / `phase-5-collect` also appends that worker's `worker-dispatched` / `worker-completed` activity and puts its `ACTIVITY:` line right after the `PROGRESS:` line in what it prints; emit both verbatim and do not append those two activities again. The lead appends them only for a worker whose checkpoint it records itself (host-native primitive, or a `team` command on another backend). **Enforced:** `scripts/okstra_ctl/dispatch_checkpoints.py`, `record_checkpoint` in `scripts/okstra_ctl/lead_progress.py`, `tests/run/test_team_cmux_backend.py`, `tests/run/test_okstra_ctl_dispatcher.py`, `tests/report/test_report_finalize.py`.
|
|
234
234
|
|
|
235
235
|
For an `implementation-planning` run whose run manifest declares `activityContractVersion: 1`, record every required activity boundary with `okstra agent-activity append` against the manifest-provided `leadEventsPath`. Model-facing calls pass prose through `--summary-file <md>` and a command through `--command`, `--command-cwd`, `--command-exit-code`, and `--command-output-file <md>`; do not construct `--command-record` JSON. The ordering is fixed: the structured append succeeds first, the matching `PROGRESS:` line is emitted second, and the immediately following `ACTIVITY:` line projects the same structured fields into the conversation language. Do not reconstruct structured activity from conversation text. If the append fails, do not present that activity boundary as completed.
|
|
236
236
|
|
|
@@ -241,7 +241,7 @@ PROGRESS: phase-4-dispatch worker=codex-worker model=gpt-6.1-sol
|
|
|
241
241
|
ACTIVITY: id=A-001 agent=codex-worker summary="Verify Stage Map paths and commands" items=P-Step-001,P-Step-002 result=runs/.../codex-worker-....md outcome=pending
|
|
242
242
|
```
|
|
243
243
|
|
|
244
|
-
Use the exact CLI projection: `id`, `agent`, quoted `summary`, comma-joined `items` (`<none>` when empty), `result` (`<none>` when empty), and `outcome`. Only prose inside `summary` is localized to the conversation language. Required kinds are `worker-dispatched`, `worker-completed`, `verification-round-completed`, `self-fix-applied`, `user-decision-required`, and `user-decision-evaluated`.
|
|
244
|
+
Use the exact CLI projection: `id`, `agent`, quoted `summary`, comma-joined `items` (`<none>` when empty), `result` (`<none>` when empty), and `outcome`. Only prose inside `summary` is localized to the conversation language; a row an okstra command writes itself (the `worker-dispatched` / `worker-completed` rows above, `lead-correction-applied`) carries a fixed English summary and is exempt. Required kinds are `worker-dispatched`, `worker-completed`, `verification-round-completed`, `self-fix-applied`, `user-decision-required`, and `user-decision-evaluated`.
|
|
245
245
|
|
|
246
246
|
**Enforcement:** `tests/contract/test_host_orchestration_rules.py` keeps this instruction on every lead path. `validators/validate_session_conformance.py` `_check_activity_contract` checks the structured event log and does not treat an `ACTIVITY:` conversation line as evidence.
|
|
247
247
|
|
|
@@ -256,7 +256,7 @@ Required checkpoints:
|
|
|
256
256
|
- `PROGRESS: phase-5-collect worker=<role> status=<terminal-status>` — once per worker, immediately after the result file is verified. `<role>` is the roster role, same rule as `phase-4-dispatch` above. `team await` (cmux-pane run only) and `worker-dispatch` record it for the dispatches they settle.
|
|
257
257
|
- `PROGRESS: phase-5-stage stage=<N> title=<title> steps=<count>` — `implementation` only, immediately before the Executor's `phase-4-dispatch` line, after parsing the approved plan's Stage Map and this run's `**Stage for this implementation run:**` anchor. `<title>` is the stage's Stage Map title and `<count>` is its `stepwiseExecution` row count, both read from the approved plan — this is the line that tells the user WHICH plan stage this run executes. The numbering keeps the line sorted where the work happens — the stage runs in Phase 5.
|
|
258
258
|
- `PROGRESS: phase-5-stage-complete stage=<N> steps=<done>/<count>` — `implementation` only, immediately after the Executor result is verified and its `### Stage Carry Evidence` block is parsed, before the verifier dispatch. `<done>` counts the block's `stepResults[]` rows whose `status` is `done` — read from the emitted block, never recomputed from git. Omitted when the Executor ends without carry evidence (`FAIL` or a non-result); the user then learns the outcome from `phase-5-collect status=<terminal-status>`.
|
|
259
|
-
- `PROGRESS: phase-5.5-convergence round=<N> queue=<count>` — at the start of each convergence round (Phase 5.5).
|
|
259
|
+
- `PROGRESS: phase-5.5-convergence round=<N> queue=<count>` — at the start of each convergence round (Phase 5.5). `<N>` and `<count>` are the `round` and `inputQueueSize` that `okstra convergence plan-round` prints for that round, copied verbatim.
|
|
260
260
|
- `PROGRESS: phase-5.6-critic provider=<provider> gaps=<n>` — after the critic result is collected (Phase 5.6, opt-in; the critic dispatch itself fires concurrently with the first 5.5 reverify round). Omitted when `convergence.critic.enabled == false`.
|
|
261
261
|
- `PROGRESS: phase-batch-cleanup panes=<n>` — immediately after cleaning up the previous batch's panes, at each batch boundary (① just before the first `phase-5.5-convergence` round ② just before the `phase-6-synthesis` report-writer dispatch). `<n>` is the number of panes closed at that boundary — the panes of dispatches this run recorded and that have since finished — counted by `okstra team reclaim --project-root <dir> --run-manifest <path>` as the panes that pass closed, never estimated; no `--dry-run` pass is needed to learn it. A pane the harness opened for its own teammate carries no recorded id, so it is not counted and not closed. Expose only the counts and NEVER expose a raw `paneId` or worker handle. Just before the first batch (analysis-worker dispatch) there is nothing to clean up, so it is a no-op and the marker is omitted.
|
|
262
262
|
- `PROGRESS: phase-6-synthesis dispatching report-writer-worker` — at the start of Phase 6. `team dispatch` (cmux-pane run only) and `worker-dispatch` record it on the first report-writer dispatch; on any other backend a `team dispatch` run appends it with `lead-progress append`.
|
|
@@ -84,9 +84,8 @@
|
|
|
84
84
|
"label": "Task type?",
|
|
85
85
|
"echo_template": "task-type: {value}",
|
|
86
86
|
"options": {
|
|
87
|
-
"
|
|
88
|
-
"
|
|
89
|
-
"_APPROVE_SUFFIX": " (recommended · 계획 승인 후 구현)",
|
|
87
|
+
"_BRIEF_SUFFIX": " (brief 가 지목한 진입 단계)",
|
|
88
|
+
"_APPROVE_SUFFIX": " (계획 승인 후 구현)",
|
|
90
89
|
"_RERUN_SUFFIX": " (현재 phase 재실행)",
|
|
91
90
|
"_BLOCKED_RERUN_SUFFIX": " (현재 phase 재실행 — 열린 명료화에 답한 뒤)",
|
|
92
91
|
"_NEXT_SUFFIX": " (다음 단계)",
|
|
@@ -101,7 +100,7 @@
|
|
|
101
100
|
"echo_template": "task-type: {value}"
|
|
102
101
|
},
|
|
103
102
|
"selected_direction_pick": {
|
|
104
|
-
"label": "상세 계획의
|
|
103
|
+
"label": "상세 계획의 기준이 될 구현 방향을 고르세요 (같은 task 의 최신 비교 보고서 3개). 방향을 아직 고르지 않은 보고서는 후보마다 한 줄로 나오고, 고른 방향은 그 보고서의 답변 파일(user-responses/)에 기록됩니다. 보고서에 딸린 답변 — 고른 방향과 확인 질문(C-NNN) 답 — 이 함께 전달되므로 뒤의 clarification 단계는 건너뜁니다",
|
|
105
104
|
"echo_template": "selected-direction: {value}",
|
|
106
105
|
"errors": {
|
|
107
106
|
"none": "같은 task에서 선택할 implementation-option-selection 최종 보고서를 찾을 수 없습니다.",
|
|
@@ -109,10 +108,14 @@
|
|
|
109
108
|
},
|
|
110
109
|
"labels": {
|
|
111
110
|
"options": "{path} — 후보 {ids}",
|
|
112
|
-
"no_options": "{path} — 후보 없음 (이 리포트로는 계획을 열 수 없습니다)"
|
|
111
|
+
"no_options": "{path} — 후보 없음 (이 리포트로는 계획을 열 수 없습니다)",
|
|
112
|
+
"pending_option": "{id} — {name} (비교 보고서 {seq})",
|
|
113
|
+
"abort": "중단 — 방향을 정하지 않고 끝냅니다"
|
|
113
114
|
},
|
|
114
115
|
"echo_variants": {
|
|
115
|
-
"cleared_clarification": "selected-direction: {value} (새 계획이므로 clarification-response 는 비웁니다 — 답변 사이드카는 이 방향과 함께 전달됩니다)"
|
|
116
|
+
"cleared_clarification": "selected-direction: {value} (새 계획이므로 clarification-response 는 비웁니다 — 답변 사이드카는 이 방향과 함께 전달됩니다)",
|
|
117
|
+
"recorded_direction": " (방향 {id} 를 답변 파일에 기록했습니다: {sidecar})",
|
|
118
|
+
"abort": "selected-direction: 중단"
|
|
116
119
|
}
|
|
117
120
|
},
|
|
118
121
|
"brief_keep": {
|
|
@@ -211,7 +214,7 @@
|
|
|
211
214
|
"label": "이 task 에는 이어받을 brief 가 없습니다. {task_type} 는 앞 단계가 쓴 brief 를 물려받아 도는 단계라 그 brief 없이는 시작할 수 없습니다 — 어떻게 할까요?",
|
|
212
215
|
"echo_template": "brief-carry: {value}",
|
|
213
216
|
"options": {
|
|
214
|
-
"__switch_entry__": "brief 를 처음 쓰는 단계부터 시작
|
|
217
|
+
"__switch_entry__": "brief 를 처음 쓰는 단계부터 시작 — requirements-discovery / error-analysis / improvement-discovery",
|
|
215
218
|
"__free_input__": "brief 경로 직접 입력",
|
|
216
219
|
"__abort__": "중단"
|
|
217
220
|
},
|
|
@@ -224,14 +227,13 @@
|
|
|
224
227
|
"label": "이 task 의 base branch?",
|
|
225
228
|
"echo_template": "base-ref: {value}",
|
|
226
229
|
"options": {
|
|
227
|
-
"_RECOMMENDED_SUFFIX": " (recommended)",
|
|
228
230
|
"__free_input__": "직접 입력"
|
|
229
231
|
}
|
|
230
232
|
},
|
|
231
233
|
"fix_cycle_confirm": {
|
|
232
234
|
"label": "이 task 는 release-handoff 까지 완료된 task 입니다. 이번 재진입을 기존 산출물에 대한 버그 픽스 사이클로 기록할까요?",
|
|
233
235
|
"options": {
|
|
234
|
-
"yes": "버그 픽스 사이클로 기록
|
|
236
|
+
"yes": "버그 픽스 사이클로 기록",
|
|
235
237
|
"no": "일반 후속 작업 (기록 안 함)",
|
|
236
238
|
"abort": "중단"
|
|
237
239
|
},
|
|
@@ -278,7 +280,7 @@
|
|
|
278
280
|
"no": "아니오 — 진행하지 않음"
|
|
279
281
|
},
|
|
280
282
|
"options_html_approval": {
|
|
281
|
-
"yes_apply": "예 — 지난 {plan_label} 보고서, 사용자의 승인 기록으로 진행
|
|
283
|
+
"yes_apply": "예 — 지난 {plan_label} 보고서, 사용자의 승인 기록으로 진행",
|
|
282
284
|
"yes": "예 — 내보낸 기록 무시 후 진행 — 진행 방식은 구현 중 다시 정함",
|
|
283
285
|
"no": "아니오 — 진행하지 않음"
|
|
284
286
|
},
|
|
@@ -423,11 +425,11 @@
|
|
|
423
425
|
"label_unlinked": "답변한 항목 중 직전 리포트의 stage 에 연결되지 않은 것이 있습니다. 다시 볼 stage 번호를 지정하거나, 이번 재실행이 stage 를 추가만 한다면 전부 이월을, 아니면 전체 재검증을 고르세요. 연결되지 않은 id 가 재실행 전체를 full 로 만들지는 않습니다.",
|
|
424
426
|
"echo_template": "reverify-scope: {value}",
|
|
425
427
|
"options": {
|
|
426
|
-
"auto": "관련 stage 만
|
|
428
|
+
"auto": "관련 stage 만 — 답변이 닿는 stage 와 그 하위만 다시 검증하고 나머지는 직전 판정을 그대로 이월",
|
|
427
429
|
"full": "전체 재검증 — stage 전부를 처음부터 다시 교차검증 (시간은 더 들지만 계획 형태가 바뀌었을 때 안전)",
|
|
428
430
|
"__free_input__": "직접 입력 — 다시 볼 stage 번호를 지정",
|
|
429
431
|
"carry-all": "재검증 없이 전부 이월 — 직전 stage 는 모두 완료됐고 이번 재실행은 stage 를 추가만 합니다. 기존 stage 는 직전 판정을 그대로 가져오고 새 stage 만 봅니다 (브랜치가 움직였으면 자동으로 전체 재검증으로 내려갑니다)",
|
|
430
|
-
"full_recommended": "전체 재검증
|
|
432
|
+
"full_recommended": "전체 재검증 — 되짚기가 stage 를 못 냈으므로 좁힐 근거가 없습니다. 시간은 더 들지만 답변의 영향 범위를 놓치지 않습니다"
|
|
431
433
|
},
|
|
432
434
|
"echo_suffixes": {
|
|
433
435
|
"auto": "reverify-scope: auto (좁힐 수 있으면 좁힘)",
|
|
@@ -485,8 +487,7 @@
|
|
|
485
487
|
"label": "{role} 역할의 모델을 선택하세요 (인스턴스 1개)",
|
|
486
488
|
"echo_template": "role-model: {value}",
|
|
487
489
|
"options": {
|
|
488
|
-
"model": "{model_ref} — {display}
|
|
489
|
-
"default_suffix": " (기본 후보)"
|
|
490
|
+
"model": "{model_ref} — {display}"
|
|
490
491
|
}
|
|
491
492
|
},
|
|
492
493
|
"directive": {
|
|
@@ -102,15 +102,23 @@ def native_limits_from_relay(contract: Mapping[str, object]) -> NativePickerLimi
|
|
|
102
102
|
|
|
103
103
|
|
|
104
104
|
class CapabilityInteractionPort:
|
|
105
|
-
def __init__(
|
|
105
|
+
def __init__(
|
|
106
|
+
self,
|
|
107
|
+
*,
|
|
108
|
+
limits: NativePickerLimits | None = None,
|
|
109
|
+
functions: frozenset[str] = INTERACTION_FUNCTIONS,
|
|
110
|
+
) -> None:
|
|
106
111
|
self._limits = limits or NativePickerLimits()
|
|
112
|
+
self._functions = functions
|
|
107
113
|
|
|
108
114
|
def plan(
|
|
109
115
|
self,
|
|
110
116
|
prompt: WizardPrompt,
|
|
111
117
|
context: HostSessionContext,
|
|
112
118
|
) -> InteractionPlan:
|
|
113
|
-
|
|
119
|
+
# 리드가 선언한 기능은 relay 의 `semanticFunctions` 로 거른다. 거르지 않으면
|
|
120
|
+
# relay 에 없는 kind 를 내보내고(예: codex 의 native-multi) 리드는 멈춘다.
|
|
121
|
+
functions = context.available_functions & self._functions
|
|
114
122
|
if prompt.kind == "pick_group":
|
|
115
123
|
if (
|
|
116
124
|
"native_question_group" in functions
|
|
@@ -178,12 +186,15 @@ class CapabilityInteractionPort:
|
|
|
178
186
|
|
|
179
187
|
|
|
180
188
|
def numbered_interaction_port() -> CapabilityInteractionPort:
|
|
181
|
-
return CapabilityInteractionPort()
|
|
189
|
+
return CapabilityInteractionPort(functions=PLAIN_TEXT_FUNCTIONS)
|
|
182
190
|
|
|
183
191
|
|
|
184
192
|
def relay_interaction_port(relay_path: str | Path) -> CapabilityInteractionPort:
|
|
185
193
|
contract = wizard_relay_contract(Path(relay_path))
|
|
186
|
-
return CapabilityInteractionPort(
|
|
194
|
+
return CapabilityInteractionPort(
|
|
195
|
+
limits=native_limits_from_relay(contract),
|
|
196
|
+
functions=frozenset(contract.get("semanticFunctions") or ()) & INTERACTION_FUNCTIONS,
|
|
197
|
+
)
|
|
187
198
|
|
|
188
199
|
|
|
189
200
|
class ProviderLeadSessionPort:
|
|
@@ -1,20 +1,24 @@
|
|
|
1
1
|
"""Build the injected execution-surface chain for one planned surface."""
|
|
2
2
|
from __future__ import annotations
|
|
3
3
|
|
|
4
|
-
from collections.abc import Sequence
|
|
4
|
+
from collections.abc import Mapping, Sequence
|
|
5
|
+
from typing import Any
|
|
5
6
|
|
|
7
|
+
from ...cmux import recorded_lead
|
|
6
8
|
from ...domain.worker_runtime import SURFACE_CLI_WRAPPER, SURFACE_CMUX_PANE
|
|
7
9
|
from ...ports.worker_runtime import WorkerRuntimePort
|
|
8
10
|
from .cli_wrapper import CliWrapperRuntime
|
|
9
11
|
from .cmux import CmuxRuntime
|
|
10
12
|
|
|
11
13
|
|
|
12
|
-
def runtime_chain(
|
|
14
|
+
def runtime_chain(
|
|
15
|
+
planned_surface: str, *, manifest: Mapping[str, Any]
|
|
16
|
+
) -> tuple[WorkerRuntimePort, ...]:
|
|
13
17
|
cli = CliWrapperRuntime()
|
|
14
18
|
if planned_surface == SURFACE_CLI_WRAPPER:
|
|
15
19
|
return (cli,)
|
|
16
20
|
if planned_surface == SURFACE_CMUX_PANE:
|
|
17
|
-
return (CmuxRuntime(), cli)
|
|
21
|
+
return (CmuxRuntime(recorded_lead(manifest)), cli)
|
|
18
22
|
raise ValueError(f"unknown execution surface: {planned_surface}")
|
|
19
23
|
|
|
20
24
|
|
|
@@ -17,10 +17,13 @@ from ...domain.worker_runtime import (
|
|
|
17
17
|
class CmuxRuntime:
|
|
18
18
|
surface = SURFACE_CMUX_PANE
|
|
19
19
|
|
|
20
|
+
def __init__(self, lead: cmux.LeadLocation | None = None) -> None:
|
|
21
|
+
self._lead = lead
|
|
22
|
+
|
|
20
23
|
def spawn(self, request: WorkerSpawnRequest) -> RuntimeHandle:
|
|
21
|
-
workspace = cmux.resolve_lead_workspace()
|
|
24
|
+
workspace = cmux.resolve_lead_workspace(self._lead)
|
|
22
25
|
if not workspace:
|
|
23
|
-
reason = cmux.unreachable_reason()
|
|
26
|
+
reason = cmux.unreachable_reason(self._lead)
|
|
24
27
|
if reason in (cmux.LOST_ENVIRONMENT, cmux.LOST_DENIED):
|
|
25
28
|
# 두 사유는 처방이 다르고, 그 문구는 prepare 의 강등 거부와
|
|
26
29
|
# 공유한다 — cmux.unreachable_advice 가 정본이다.
|
|
@@ -39,6 +42,7 @@ class CmuxRuntime:
|
|
|
39
42
|
command=request.command,
|
|
40
43
|
title=request.title,
|
|
41
44
|
owned_surface_ids=request.owned_surface_ids,
|
|
45
|
+
recorded=self._lead,
|
|
42
46
|
)
|
|
43
47
|
except (RuntimeError, OSError, subprocess.SubprocessError) as exc:
|
|
44
48
|
raise SurfaceUnavailable(str(exc)) from exc
|
|
@@ -58,7 +62,7 @@ class CmuxRuntime:
|
|
|
58
62
|
self.close(handle)
|
|
59
63
|
|
|
60
64
|
def notify(self, event: ProgressEvent) -> None:
|
|
61
|
-
workspace = cmux.resolve_lead_workspace()
|
|
65
|
+
workspace = cmux.resolve_lead_workspace(self._lead)
|
|
62
66
|
cmux.sidebar_log(workspace, event.message, level=event.level)
|
|
63
67
|
if event.notify_title is not None:
|
|
64
68
|
cmux.sidebar_notify(
|
|
@@ -68,4 +72,4 @@ class CmuxRuntime:
|
|
|
68
72
|
)
|
|
69
73
|
|
|
70
74
|
def restore_lead(self) -> None:
|
|
71
|
-
cmux.restore_lead_width()
|
|
75
|
+
cmux.restore_lead_width(self._lead)
|
|
@@ -330,7 +330,7 @@ def _activity_summary(args: argparse.Namespace) -> str:
|
|
|
330
330
|
) from exc
|
|
331
331
|
|
|
332
332
|
|
|
333
|
-
def
|
|
333
|
+
def conversation_activity_line(details: Mapping[str, Any]) -> str:
|
|
334
334
|
plan_items = ",".join(details["planItemIds"]) or "<none>"
|
|
335
335
|
result_path = str(details["resultPath"] or "<none>")
|
|
336
336
|
summary = json.dumps(details["summary"], ensure_ascii=False)
|
|
@@ -364,7 +364,7 @@ def _append(args: argparse.Namespace) -> int:
|
|
|
364
364
|
event = record_activity(args.project_root, args.run_manifest, details)
|
|
365
365
|
payload = dict(event.details)
|
|
366
366
|
payload["ok"] = True
|
|
367
|
-
payload["activityLine"] =
|
|
367
|
+
payload["activityLine"] = conversation_activity_line(payload)
|
|
368
368
|
print(json.dumps(payload, ensure_ascii=False, indent=2))
|
|
369
369
|
return 0
|
|
370
370
|
|
|
@@ -110,6 +110,7 @@ def build_analysis_packet(
|
|
|
110
110
|
direct_work_text: str = "",
|
|
111
111
|
business_knowledge_text: str = "",
|
|
112
112
|
coverage_census_text: str = "",
|
|
113
|
+
has_selected_direction: bool = False,
|
|
113
114
|
) -> str:
|
|
114
115
|
"""Return the primary compact input for Claude/Codex/Antigravity analysers.
|
|
115
116
|
|
|
@@ -143,6 +144,7 @@ def build_analysis_packet(
|
|
|
143
144
|
instruction_set_relative_path,
|
|
144
145
|
bool(clarification_response_path),
|
|
145
146
|
bool(group_context_text),
|
|
147
|
+
has_selected_direction,
|
|
146
148
|
)
|
|
147
149
|
)
|
|
148
150
|
# 그룹 문서는 사람 절과 okstra 메모리 영역으로 갈라 싣는다. 사람 절("왜")은
|
|
@@ -339,6 +341,7 @@ def _intro_block(
|
|
|
339
341
|
instruction_set: str,
|
|
340
342
|
has_clarification: bool,
|
|
341
343
|
has_group_context: bool = False,
|
|
344
|
+
has_selected_direction: bool = False,
|
|
342
345
|
) -> list[str]:
|
|
343
346
|
lines = [
|
|
344
347
|
f"# OKSTRA Analysis Packet - {task_key}",
|
|
@@ -368,6 +371,10 @@ def _intro_block(
|
|
|
368
371
|
lines.append(
|
|
369
372
|
f"- Task-group context: `{instruction_set}/task-group-context.md`"
|
|
370
373
|
)
|
|
374
|
+
if has_selected_direction:
|
|
375
|
+
lines.append(
|
|
376
|
+
f"- Selected direction snapshot: `{instruction_set}/selected-direction.json`"
|
|
377
|
+
)
|
|
371
378
|
return lines
|
|
372
379
|
|
|
373
380
|
|
|
@@ -290,7 +290,7 @@ def _validate_fact_evidence(
|
|
|
290
290
|
"confirmed business claims require inspected code evidence"
|
|
291
291
|
)
|
|
292
292
|
for evidence in fact["evidence"]:
|
|
293
|
-
validate_evidence(evidence, projects)
|
|
293
|
+
evidence["excerpt"] = validate_evidence(evidence, projects)
|
|
294
294
|
if evidence["projectId"] not in result["inspectedProjects"]:
|
|
295
295
|
raise BusinessFlowError(
|
|
296
296
|
"claim evidence references an uninspected candidate"
|
|
@@ -380,7 +380,7 @@ def _instructions(
|
|
|
380
380
|
"Read the listed source roots and source artifacts only. Do not edit files, execute tests, start services, deploy or migrate.",
|
|
381
381
|
"Inspect inbound AND outbound dependencies across all candidates, including indirect call/event/queue/shared-data/file/batch relationships. Name every uninspected candidate or missing branch in gaps.",
|
|
382
382
|
"Explain the product business from start to finish for a new developer: purpose, rules, input/output, states, failure, retry and recovery. Link each step and relationship to factKeys.",
|
|
383
|
-
"Use stable keys in business/subject/attribute form. Facts are source-backed assertions; expected changes remain expected. Cite
|
|
383
|
+
"Use stable keys in business/subject/attribute form. Facts are source-backed assertions; expected changes remain expected. Cite inspected evidence with one-based line/endLine; each excerpt must copy the whole line..endLine range verbatim, including indentation, with no elision or omitted lines. Tests are static evidence unless a supplied existing verification record supports the execution claim.",
|
|
384
384
|
"For post mode, use the supplied project-specific baseline. Never invent missing before-state. Describe the actual delta and distinguish confirmed side effects from risks.",
|
|
385
385
|
"For contribute mode, extract newly observed business facts from the supplied validated task record and inspect source to confirm them. Empty facts are valid. Do not copy the report narrative as established truth.",
|
|
386
386
|
"For reconcile mode, reinspect every conflicting source and condition. Return explicit resolutions only with evidence-backed reasoning and original claimIds/acceptedClaimId. Preserve genuinely unresolved policy decisions in questions. Other modes return resolutions=[].",
|
|
@@ -135,7 +135,7 @@ def capture_projects(candidates: list[ProjectCandidate]) -> list[SourceSnapshot]
|
|
|
135
135
|
|
|
136
136
|
def validate_evidence(
|
|
137
137
|
evidence: BusinessEvidence, projects: list[SourceSnapshot]
|
|
138
|
-
) ->
|
|
138
|
+
) -> str:
|
|
139
139
|
project = next(
|
|
140
140
|
(row for row in projects if row["projectId"] == evidence["projectId"]), None
|
|
141
141
|
)
|
|
@@ -157,10 +157,43 @@ def validate_evidence(
|
|
|
157
157
|
start, end = evidence["line"], evidence["endLine"]
|
|
158
158
|
if end < start or end > len(lines):
|
|
159
159
|
raise BusinessFlowError("evidence line range is invalid")
|
|
160
|
-
|
|
160
|
+
source = lines[start - 1 : end]
|
|
161
|
+
if not _excerpt_matches(evidence["excerpt"], source):
|
|
161
162
|
raise BusinessFlowError(
|
|
162
163
|
f"evidence excerpt does not match source: {evidence['path']}:{start}"
|
|
163
164
|
)
|
|
165
|
+
return "\n".join(source)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
_ELISION_MARKERS = {"...", "\u2026"}
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def _excerpt_matches(excerpt: str, source: list[str]) -> bool:
|
|
172
|
+
actual = [line.strip() for line in source if line.strip()]
|
|
173
|
+
segments: list[list[str]] = [[]]
|
|
174
|
+
for raw in excerpt.splitlines():
|
|
175
|
+
line = raw.strip()
|
|
176
|
+
if line in _ELISION_MARKERS:
|
|
177
|
+
segments.append([])
|
|
178
|
+
elif line:
|
|
179
|
+
segments[-1].append(line)
|
|
180
|
+
segments = [segment for segment in segments if segment]
|
|
181
|
+
if not segments:
|
|
182
|
+
return False
|
|
183
|
+
cursor = 0
|
|
184
|
+
for segment in segments:
|
|
185
|
+
found = next(
|
|
186
|
+
(
|
|
187
|
+
index
|
|
188
|
+
for index in range(cursor, len(actual) - len(segment) + 1)
|
|
189
|
+
if actual[index : index + len(segment)] == segment
|
|
190
|
+
),
|
|
191
|
+
None,
|
|
192
|
+
)
|
|
193
|
+
if found is None:
|
|
194
|
+
return False
|
|
195
|
+
cursor = found + len(segment)
|
|
196
|
+
return True
|
|
164
197
|
|
|
165
198
|
|
|
166
199
|
def changed_projects(
|