codex-flow 2.1.13__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codex_flow/__init__.py +28 -0
- codex_flow/__main__.py +9 -0
- codex_flow/cli.py +242 -0
- codex_flow/data/LICENSE +21 -0
- codex_flow/data/README.en.md +303 -0
- codex_flow/data/README.md +305 -0
- codex_flow/data/VERSION +1 -0
- codex_flow/data/apps/chatgpt-mcp/README.md +86 -0
- codex_flow/data/apps/chatgpt-mcp/__init__.py +1 -0
- codex_flow/data/apps/chatgpt-mcp/adapter.py +458 -0
- codex_flow/data/apps/chatgpt-mcp/server.py +358 -0
- codex_flow/data/apps/chatgpt-mcp/widget.html +927 -0
- codex_flow/data/apps/macos-overlay/README.en.md +121 -0
- codex_flow/data/apps/macos-overlay/README.md +123 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayRuntimeState.swift +126 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayScreenGeometry.swift +82 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayWindowController.swift +1052 -0
- codex_flow/data/apps/macos-overlay/Sources/Localization.swift +197 -0
- codex_flow/data/apps/macos-overlay/Sources/Models/TelemetryData.swift +1557 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/AccountSnapshotService.swift +1101 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/FlowPilotInstanceLock.swift +153 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/IPCServer.swift +298 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryQueryEngine.swift +800 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryWatcher.swift +135 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/UpdateService.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AccountView.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AnalyticsView.swift +566 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AutostartView.swift +293 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/BubbleView.swift +317 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HistoryView.swift +1124 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HoverRevealText.swift +165 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/InspectorSkillsToolsView.swift +121 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/LogoView.swift +182 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SleekSwitch.swift +117 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/StrategyModeView.swift +561 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SummaryView.swift +1273 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/UpdateView.swift +352 -0
- codex_flow/data/apps/macos-overlay/Sources/main.swift +340 -0
- codex_flow/data/apps/macos-overlay/Tests/OverlayScreenGeometryTests.swift +163 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryPhase1ContractTests.swift +357 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQueryEngineConcurrencyTests.swift +221 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQuotaSelectionTests.swift +158 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryWorkerTokenTests.swift +122 -0
- codex_flow/data/apps/macos-overlay/build.sh +75 -0
- codex_flow/data/benchmark/corpus.json +103 -0
- codex_flow/data/benchmark/manifest.example.json +41 -0
- codex_flow/data/benchmark/manifest.schema.json +137 -0
- codex_flow/data/benchmark/prices/gpt-5.6-2026-08-30.json +5 -0
- codex_flow/data/benchmark/profiles.json +90 -0
- codex_flow/data/benchmark/schema.json +77 -0
- codex_flow/data/benchmark/tasks.json +50 -0
- codex_flow/data/completions/codex-flow.bash +34 -0
- codex_flow/data/completions/codex-flow.zsh +52 -0
- codex_flow/data/glama.json +6 -0
- codex_flow/data/install-release.ps1 +126 -0
- codex_flow/data/install-release.sh +155 -0
- codex_flow/data/install.ps1 +349 -0
- codex_flow/data/install.sh +362 -0
- codex_flow/data/policy/benchmark.toml +49 -0
- codex_flow/data/policy/defaults.toml +70 -0
- codex_flow/data/scripts/analyze-benchmark.py +510 -0
- codex_flow/data/scripts/benchmark-local.py +171 -0
- codex_flow/data/scripts/check-recommendation.py +277 -0
- codex_flow/data/scripts/doctor.py +449 -0
- codex_flow/data/scripts/generate-release-manifest.py +74 -0
- codex_flow/data/scripts/localization.py +192 -0
- codex_flow/data/scripts/manage-hooks.py +448 -0
- codex_flow/data/scripts/manage-instructions.py +389 -0
- codex_flow/data/scripts/manage-shell.py +151 -0
- codex_flow/data/scripts/materialize-corpus.py +193 -0
- codex_flow/data/scripts/menu.py +646 -0
- codex_flow/data/scripts/migrations/0001_update_settings.py +80 -0
- codex_flow/data/scripts/package-release.py +132 -0
- codex_flow/data/scripts/render-benchmark-report.py +292 -0
- codex_flow/data/scripts/run-benchmark.py +829 -0
- codex_flow/data/scripts/strategies/__init__.py +28 -0
- codex_flow/data/scripts/strategies/balanced.py +115 -0
- codex_flow/data/scripts/strategies/base.py +363 -0
- codex_flow/data/scripts/strategies/efficient.py +158 -0
- codex_flow/data/scripts/strategies/lifecycle_runtime.py +590 -0
- codex_flow/data/scripts/strategies/quality.py +209 -0
- codex_flow/data/scripts/strategies/speed.py +108 -0
- codex_flow/data/scripts/strategies/task_budget_runtime.py +644 -0
- codex_flow/data/scripts/strategies/task_phase_runtime.py +341 -0
- codex_flow/data/scripts/strategies/work_unit_runtime.py +421 -0
- codex_flow/data/scripts/strategy_runtime.py +1091 -0
- codex_flow/data/scripts/telemetry.py +400 -0
- codex_flow/data/scripts/telemetry_core/__init__.py +192 -0
- codex_flow/data/scripts/telemetry_core/app_server.py +1192 -0
- codex_flow/data/scripts/telemetry_core/collector.py +1247 -0
- codex_flow/data/scripts/telemetry_core/common.py +421 -0
- codex_flow/data/scripts/telemetry_core/latency.py +593 -0
- codex_flow/data/scripts/telemetry_core/query.py +427 -0
- codex_flow/data/scripts/telemetry_core/quota_ledger.py +598 -0
- codex_flow/data/scripts/telemetry_core/render.py +460 -0
- codex_flow/data/scripts/telemetry_core/repair.py +223 -0
- codex_flow/data/scripts/ui.py +266 -0
- codex_flow/data/scripts/update-homebrew-formula.py +146 -0
- codex_flow/data/scripts/update_runtime_config.py +134 -0
- codex_flow/data/scripts/updater.py +1718 -0
- codex_flow/data/smithery.yaml +18 -0
- codex_flow/data/templates/agents/worker-explorer.toml +24 -0
- codex_flow/data/templates/agents/worker-implementer.toml +49 -0
- codex_flow/data/templates/agents/worker-reviewer.toml +25 -0
- codex_flow/data/templates/flow-pilot-instructions.md +35 -0
- codex_flow/data/templates/skills/flow-pilot/SKILL.md +577 -0
- codex_flow/mcp.py +35 -0
- codex_flow-2.1.13.dist-info/METADATA +342 -0
- codex_flow-2.1.13.dist-info/RECORD +113 -0
- codex_flow-2.1.13.dist-info/WHEEL +5 -0
- codex_flow-2.1.13.dist-info/entry_points.txt +3 -0
- codex_flow-2.1.13.dist-info/licenses/LICENSE +21 -0
- codex_flow-2.1.13.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,577 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: flow-pilot
|
|
3
|
+
description: Require the FlowPilot strategy gate for repository technical work, then compile a deterministic ExecutionPlan through codex-flow and execute its worker, lifecycle, task-budget, review, and validation contracts exactly.
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# FlowPilot Strategy Runtime
|
|
7
|
+
|
|
8
|
+
FlowPilot is the semantic profiler and execution runtime for codex-flow. It is **not** a second strategy engine.
|
|
9
|
+
|
|
10
|
+
> **FlowPilot profiles. `strategy_runtime.py` + the strategy registry decide. FlowPilot executes the returned plan.**
|
|
11
|
+
|
|
12
|
+
Do not independently re-implement strategy topology, capability selection, reasoning selection, quota policy, Worker counts, review mode, fan-out, lifecycle policy, local repair budget, task budget, phase admission, or implementation work-unit policy. The installed planner and deterministic runtime helpers are authoritative.
|
|
13
|
+
|
|
14
|
+
Default policy remains `strategy=efficient` with `routing=adaptive`.
|
|
15
|
+
|
|
16
|
+
## 0. Entry caller and receipt
|
|
17
|
+
|
|
18
|
+
The managed global `AGENTS.md` entry instructions call this skill before
|
|
19
|
+
repository exploration, edits, configuration, integration, tests, or other
|
|
20
|
+
technical implementation. Apply that entry rule to every repository
|
|
21
|
+
technical task; do not invent a “non-trivial” escape hatch. Conversation-only
|
|
22
|
+
answers and work explicitly assigned to a subagent are outside this entry
|
|
23
|
+
caller.
|
|
24
|
+
|
|
25
|
+
When the entry caller is present, first read this installed skill completely,
|
|
26
|
+
then run `show --json` and, only when enabled, `consume-bypass` below before
|
|
27
|
+
repository exploration or technical action. Treat the output as a task-local
|
|
28
|
+
receipt. `enabled=false` or a `true` bypass receipt means ordinary execution
|
|
29
|
+
for this task. Never infer enabled state from an earlier turn. Same-task
|
|
30
|
+
follow-ups preserve the current plan and ledger; a bypass is consumed once per
|
|
31
|
+
task. Higher-priority instructions and explicit current-task overrides still
|
|
32
|
+
apply. The prompt entry is a host-dependent caller, not a security boundary
|
|
33
|
+
or a guarantee of model adherence; hooks remain telemetry-only.
|
|
34
|
+
|
|
35
|
+
## 0. Strategy gate and precedence
|
|
36
|
+
|
|
37
|
+
Before automatic FlowPilot delegation, read the installed strategy state:
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
python3 ~/.codex/codex-flow/strategy_runtime.py \
|
|
41
|
+
--policy ~/.codex/codex-flow.toml \
|
|
42
|
+
show --json
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
If enabled, atomically consume any one-shot bypass:
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
python3 ~/.codex/codex-flow/strategy_runtime.py \
|
|
49
|
+
--policy ~/.codex/codex-flow.toml \
|
|
50
|
+
consume-bypass
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
If that prints `true`, bypass FlowPilot for this task only. If `enabled=false`, do not build a FlowPilot TaskProfile or compile an ExecutionPlan; continue with ordinary Codex execution. Repository policy and task overrides cannot re-enable a disabled global switch.
|
|
54
|
+
|
|
55
|
+
Policy precedence is:
|
|
56
|
+
|
|
57
|
+
```text
|
|
58
|
+
hard runtime / safety ceilings
|
|
59
|
+
> explicit current-task overrides
|
|
60
|
+
> repository .codex-flow.toml
|
|
61
|
+
> ~/.codex/codex-flow.toml
|
|
62
|
+
> release defaults
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
Persistent dimensions are independent:
|
|
66
|
+
|
|
67
|
+
- strategy: `efficient | balanced | quality | speed`
|
|
68
|
+
- routing: `adaptive | direct | delegate`
|
|
69
|
+
- review: `auto | standard | strict`
|
|
70
|
+
- fanout: `auto | conservative | aggressive`
|
|
71
|
+
|
|
72
|
+
`direct` — do not spawn or delegate to subagents for this task.
|
|
73
|
+
|
|
74
|
+
`delegate` — use subagent delegation for execution when the runtime supports it and safe scoping is possible.
|
|
75
|
+
|
|
76
|
+
`adaptive` — let the deterministic planner choose from the TaskProfile.
|
|
77
|
+
|
|
78
|
+
Strategy and routing are orthogonal.
|
|
79
|
+
|
|
80
|
+
## 1. Build only the semantic TaskProfile
|
|
81
|
+
|
|
82
|
+
Profile the task before broad execution:
|
|
83
|
+
|
|
84
|
+
```text
|
|
85
|
+
complexity: small | routine | complex | critical
|
|
86
|
+
uncertainty: low | medium | high
|
|
87
|
+
risk: low | medium | high | critical
|
|
88
|
+
scope: local | module | cross-module | repo-wide
|
|
89
|
+
parallelism: none | limited | high
|
|
90
|
+
write_conflict: low | high
|
|
91
|
+
exploration_need: low | medium | high
|
|
92
|
+
verification_cost: low | medium | high
|
|
93
|
+
iteration_intensity: one-shot | iterative | heavy-loop
|
|
94
|
+
writable_workstreams: positive integer
|
|
95
|
+
quality_intent: normal | strong | absolute
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
`writable_workstreams` counts already-proven isolated writable scopes/worktrees. Default to `1`; never invent extra writable streams merely to gain concurrency.
|
|
99
|
+
|
|
100
|
+
`quality_intent` is user intent, not risk. `strong` and `absolute` require explicit preference for correctness/quality over ordinary quota or latency.
|
|
101
|
+
|
|
102
|
+
Re-profile only when material evidence changes scope, risk, uncertainty, isolation, iteration intensity, or explicit quality intent.
|
|
103
|
+
|
|
104
|
+
## 2. Compile the authoritative plan
|
|
105
|
+
|
|
106
|
+
Invoke:
|
|
107
|
+
|
|
108
|
+
```bash
|
|
109
|
+
python3 ~/.codex/codex-flow/strategy_runtime.py \
|
|
110
|
+
--policy ~/.codex/codex-flow.toml \
|
|
111
|
+
plan \
|
|
112
|
+
--complexity <...> \
|
|
113
|
+
--uncertainty <...> \
|
|
114
|
+
--risk <...> \
|
|
115
|
+
--scope <...> \
|
|
116
|
+
--parallelism <...> \
|
|
117
|
+
--write-conflict <...> \
|
|
118
|
+
--exploration-need <...> \
|
|
119
|
+
--verification-cost <...> \
|
|
120
|
+
--iteration-intensity <...> \
|
|
121
|
+
--writable-workstreams <N> \
|
|
122
|
+
--quality-intent normal|strong|absolute
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
Task-only overrides may append:
|
|
126
|
+
|
|
127
|
+
```text
|
|
128
|
+
--profile efficient|balanced|quality|speed
|
|
129
|
+
--routing adaptive|direct|delegate
|
|
130
|
+
--review auto|standard|strict
|
|
131
|
+
--fanout auto|conservative|aggressive
|
|
132
|
+
--efficient-reasoning legacy|shadow|adaptive
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
If the planner/registry is unavailable, treat that as an installation/runtime failure. Do not reconstruct its policy from this skill.
|
|
136
|
+
|
|
137
|
+
## 3. ExecutionPlan is the hard boundary
|
|
138
|
+
|
|
139
|
+
Current contract (schema v11):
|
|
140
|
+
|
|
141
|
+
```text
|
|
142
|
+
ExecutionPlan
|
|
143
|
+
schema_version = 11
|
|
144
|
+
strategy
|
|
145
|
+
routing
|
|
146
|
+
review_modifier
|
|
147
|
+
fanout_modifier
|
|
148
|
+
quality_intent
|
|
149
|
+
parent_capability_policy
|
|
150
|
+
parent_model_floor
|
|
151
|
+
parent_reasoning
|
|
152
|
+
reasoning_rollout | none
|
|
153
|
+
mode
|
|
154
|
+
legacy_worker_reasoning
|
|
155
|
+
proposed_worker_reasoning
|
|
156
|
+
selected_worker_reasoning
|
|
157
|
+
applied
|
|
158
|
+
explorer_capability_policy/model/reasoning | none
|
|
159
|
+
implementer_capability_policy/model/reasoning | none
|
|
160
|
+
reviewer_capability_policy/model/reasoning | none
|
|
161
|
+
worker_budget
|
|
162
|
+
task_budget | none
|
|
163
|
+
soft_timeout_seconds
|
|
164
|
+
hard_timeout_seconds
|
|
165
|
+
max_work_units
|
|
166
|
+
max_implementation_attempts
|
|
167
|
+
max_replans
|
|
168
|
+
max_replacements
|
|
169
|
+
max_review_attempts
|
|
170
|
+
parent_finalization_seconds
|
|
171
|
+
exploration_workers
|
|
172
|
+
implementation_workers
|
|
173
|
+
reviewer_workers
|
|
174
|
+
planned_worker_count
|
|
175
|
+
exploration_stage | none
|
|
176
|
+
implementation_stage | none
|
|
177
|
+
review_stage | none
|
|
178
|
+
join_policy
|
|
179
|
+
min_successful_workers
|
|
180
|
+
idle_timeout_seconds
|
|
181
|
+
hard_timeout_seconds
|
|
182
|
+
soft_timeout_seconds | none
|
|
183
|
+
checkpoint_rearm_seconds | none
|
|
184
|
+
max_worker_repair_attempts | none
|
|
185
|
+
work_unit_mode
|
|
186
|
+
minimum_work_units
|
|
187
|
+
join_between_work_units
|
|
188
|
+
maximum_work_units | none
|
|
189
|
+
require_write_paths
|
|
190
|
+
cancel_if_superseded
|
|
191
|
+
cancel_stragglers_after_quorum
|
|
192
|
+
fallback_policy
|
|
193
|
+
review_mode
|
|
194
|
+
max_repair_cycles
|
|
195
|
+
max_concurrent_threads
|
|
196
|
+
escalate_on_failure
|
|
197
|
+
quota_pressure
|
|
198
|
+
repo_policy | none
|
|
199
|
+
context_mode
|
|
200
|
+
notes
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
Direct plans emit `task_budget=null` and no delegated stages. Delegated built-in strategies emit one canonical `task_budget`; there is no second effective budget.
|
|
204
|
+
|
|
205
|
+
For delegated implementation:
|
|
206
|
+
|
|
207
|
+
```text
|
|
208
|
+
implementation_workers <= task_budget.max_work_units
|
|
209
|
+
implementation_workers <= task_budget.max_implementation_attempts
|
|
210
|
+
implementation_stage.maximum_work_units == task_budget.max_work_units
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
The compiler enforces these invariants. Do not locally raise topology or counters.
|
|
214
|
+
|
|
215
|
+
## 4. Canonical task budget and phase admission
|
|
216
|
+
|
|
217
|
+
Immediately after the **initial** delegated ExecutionPlan is compiled, persist that exact plan JSON as the task's initial budget plan and initialize through the phase helper:
|
|
218
|
+
|
|
219
|
+
```bash
|
|
220
|
+
python3 ~/.codex/codex-flow/strategies/task_phase_runtime.py init \
|
|
221
|
+
--state-file <task-ledger-path> \
|
|
222
|
+
--task-id <task-id> \
|
|
223
|
+
--plan-json '<initial ExecutionPlan JSON>' \
|
|
224
|
+
--now <unix-seconds>
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
The initial budget-plan identity is immutable for the task. A later semantic re-profile may compile another ExecutionPlan for current topology/lifecycle decisions, but it must not reset, replace, or extend the original task ledger. All phase status/reservation calls continue to pass the initial budget plan.
|
|
228
|
+
|
|
229
|
+
The canonical task timeline is:
|
|
230
|
+
|
|
231
|
+
```text
|
|
232
|
+
started_at
|
|
233
|
+
|
|
|
234
|
+
| general exploration / writable implementation
|
|
235
|
+
v
|
|
236
|
+
soft_deadline == general_work_deadline
|
|
237
|
+
|
|
|
238
|
+
| required read-only review, when planned
|
|
239
|
+
v
|
|
240
|
+
review_deadline == hard_deadline - parent_finalization_seconds
|
|
241
|
+
|
|
|
242
|
+
| Parent finalization only
|
|
243
|
+
v
|
|
244
|
+
hard_deadline == absolute task execution stop
|
|
245
|
+
```
|
|
246
|
+
|
|
247
|
+
The compiler has already made `hard_timeout_seconds` large enough to contain any planned review-stage hard window plus `parent_finalization_seconds`; the phase helper validates this rather than deriving another budget.
|
|
248
|
+
|
|
249
|
+
Before exploration or implementation continuation:
|
|
250
|
+
|
|
251
|
+
```bash
|
|
252
|
+
python3 ~/.codex/codex-flow/strategies/task_phase_runtime.py status \
|
|
253
|
+
--state-file <task-ledger-path> \
|
|
254
|
+
--task-id <task-id> \
|
|
255
|
+
--plan-json '<initial ExecutionPlan JSON>' \
|
|
256
|
+
--phase exploration|implementation \
|
|
257
|
+
--now <unix-seconds>
|
|
258
|
+
```
|
|
259
|
+
|
|
260
|
+
Before each logical unit, implementation attempt, replan, or implementation replacement:
|
|
261
|
+
|
|
262
|
+
```bash
|
|
263
|
+
python3 ~/.codex/codex-flow/strategies/task_phase_runtime.py reserve \
|
|
264
|
+
--state-file <task-ledger-path> \
|
|
265
|
+
--task-id <task-id> \
|
|
266
|
+
--plan-json '<initial ExecutionPlan JSON>' \
|
|
267
|
+
--phase implementation \
|
|
268
|
+
--kind work_unit|implementation_attempt|replan|replacement \
|
|
269
|
+
--reservation-id <stable-id> \
|
|
270
|
+
--fingerprint <stable-fingerprint> \
|
|
271
|
+
--now <unix-seconds>
|
|
272
|
+
```
|
|
273
|
+
|
|
274
|
+
At `soft_deadline`, genuinely new general-work reservations stop. Exact durable-ledger replay of an already-recorded general reservation remains idempotent and is not new work. Existing writable work must checkpoint/converge rather than silently opening another attempt.
|
|
275
|
+
|
|
276
|
+
If the plan has reviewers, required completion starts after general-work convergence. Query:
|
|
277
|
+
|
|
278
|
+
```bash
|
|
279
|
+
python3 ~/.codex/codex-flow/strategies/task_phase_runtime.py status \
|
|
280
|
+
--state-file <task-ledger-path> \
|
|
281
|
+
--task-id <task-id> \
|
|
282
|
+
--plan-json '<initial ExecutionPlan JSON>' \
|
|
283
|
+
--phase required_completion \
|
|
284
|
+
--now <unix-seconds>
|
|
285
|
+
```
|
|
286
|
+
|
|
287
|
+
Before every reviewer Worker start or retry, reserve:
|
|
288
|
+
|
|
289
|
+
```bash
|
|
290
|
+
python3 ~/.codex/codex-flow/strategies/task_phase_runtime.py reserve \
|
|
291
|
+
--state-file <task-ledger-path> \
|
|
292
|
+
--task-id <task-id> \
|
|
293
|
+
--plan-json '<initial ExecutionPlan JSON>' \
|
|
294
|
+
--phase required_completion \
|
|
295
|
+
--kind review_attempt \
|
|
296
|
+
--reservation-id <stable-review-attempt-id> \
|
|
297
|
+
--fingerprint <stable-review-fingerprint> \
|
|
298
|
+
--now <unix-seconds>
|
|
299
|
+
```
|
|
300
|
+
|
|
301
|
+
`max_review_attempts` counts reviewer starts/retries independently of implementation replans/replacements.
|
|
302
|
+
|
|
303
|
+
The phase result is binding:
|
|
304
|
+
|
|
305
|
+
- `permits_general_work`: new exploration/writable implementation may start.
|
|
306
|
+
- `permits_review_start`: a new read-only reviewer may start.
|
|
307
|
+
- `permits_parent_finalization`: Parent may continue final reconciliation/verification before hard.
|
|
308
|
+
- `review_deadline`: no new reviewer starts at or after this boundary.
|
|
309
|
+
- `action=finalize_parent`: reviewer admission is closed; use the remaining tail only for Parent finalization. This is an admission transition, **not** evidence that required review succeeded.
|
|
310
|
+
- `action=stop`: task hard deadline/closed state reached; do not start new execution.
|
|
311
|
+
|
|
312
|
+
A soft deadline must never silently skip an independent/strict review already required by the initial plan. A reviewer must never consume the Parent finalization reserve. If `review_stage.join_policy` is required/quorum, successful delivery still requires at least `review_stage.min_successful_workers` accepted reviewer results. Reaching `review_deadline` without satisfying that join is fail-closed: do not silently downgrade to Parent-only review or report task success.
|
|
313
|
+
|
|
314
|
+
Before entering Parent-only finalization, every writable Worker must already be terminal/cancel-confirmed or safely fenced with its latest returned checkpoint harvested. The Parent finalization tail must not be consumed by unresolved writable execution.
|
|
315
|
+
|
|
316
|
+
The raw task-budget helper is the durable atomic ledger. The phase helper is the scheduling admission boundary. Neither schedules Workers by itself.
|
|
317
|
+
|
|
318
|
+
## 5. Worker lifecycle
|
|
319
|
+
|
|
320
|
+
Lifecycle comes from the relevant StagePolicy, not Parent heuristics. Track stable lineage:
|
|
321
|
+
|
|
322
|
+
```text
|
|
323
|
+
(scope_id, generation)
|
|
324
|
+
(scope_id, unit_id, generation) for bounded implementation
|
|
325
|
+
```
|
|
326
|
+
|
|
327
|
+
Replacement/replan increments generation. Checkpoint/continue does not.
|
|
328
|
+
|
|
329
|
+
Track when evidence exists:
|
|
330
|
+
|
|
331
|
+
- `last_progress_at`: liveness/tool/in-flight activity.
|
|
332
|
+
- `last_meaningful_progress_at`: acceptance-relevant delta, validation change, concrete blocker reduction, or bounded-scope completion.
|
|
333
|
+
|
|
334
|
+
Do not mark repeated unchanged reads/tests/heartbeats as meaningful progress.
|
|
335
|
+
|
|
336
|
+
`idle_timeout_seconds` is a renewable liveness lease. `soft_timeout_seconds` is an advisory checkpoint/convergence boundary. `hard_timeout_seconds` is the Worker wall-clock ceiling.
|
|
337
|
+
|
|
338
|
+
**A `wait()` timeout is never a Worker timeout.**
|
|
339
|
+
|
|
340
|
+
Use the deterministic evaluator:
|
|
341
|
+
|
|
342
|
+
```bash
|
|
343
|
+
python3 ~/.codex/codex-flow/strategies/lifecycle_runtime.py \
|
|
344
|
+
--policy-json '<StagePolicy JSON>' \
|
|
345
|
+
--scope-id <scope-id> \
|
|
346
|
+
--stage exploration|implementation|review \
|
|
347
|
+
--started-at <unix-seconds> \
|
|
348
|
+
--last-progress-at <unix-seconds> \
|
|
349
|
+
[--last-meaningful-progress-at <unix-seconds>] \
|
|
350
|
+
[--checkpoint-sequence-json <json-array>] \
|
|
351
|
+
[--generation <non-negative-int>] \
|
|
352
|
+
--now <unix-seconds> \
|
|
353
|
+
[--writable] [--in-flight] \
|
|
354
|
+
[--terminal-success] [--terminal-failure] \
|
|
355
|
+
[--scope-superseded] [--cancel-confirmed] \
|
|
356
|
+
[--replacement-isolated]
|
|
357
|
+
```
|
|
358
|
+
|
|
359
|
+
Use returned `state`, `action`, `cancel_required`, `replacement_allowed`, `fence_required`, `progress_quality`, checkpoint fields, `replan_scope`, `checkpoint_reuse_mode`, and `fallback_policy` exactly.
|
|
360
|
+
|
|
361
|
+
### Checkpoints
|
|
362
|
+
|
|
363
|
+
Checkpoint sequence state is:
|
|
364
|
+
|
|
365
|
+
```text
|
|
366
|
+
not_requested -> requested -> received -> harvested
|
|
367
|
+
```
|
|
368
|
+
|
|
369
|
+
Soft-budget actions:
|
|
370
|
+
|
|
371
|
+
- `request_checkpoint`: ask the same Worker for a non-terminal checkpoint.
|
|
372
|
+
- `await_checkpoint`: do not spam another request.
|
|
373
|
+
- `harvest_checkpoint`: persist returned partial work before any destructive boundary.
|
|
374
|
+
- after a harvest, re-arm only when explicit acceptance-relevant progress occurred after the harvest and `checkpoint_rearm_seconds` elapsed.
|
|
375
|
+
|
|
376
|
+
Implementation checkpoint payload must carry enough evidence to continue safely:
|
|
377
|
+
|
|
378
|
+
```text
|
|
379
|
+
scope_id / unit_id when bounded
|
|
380
|
+
status
|
|
381
|
+
completed
|
|
382
|
+
changed_files/current_patch_state
|
|
383
|
+
validation
|
|
384
|
+
blockers
|
|
385
|
+
remaining_delta
|
|
386
|
+
workspace_state
|
|
387
|
+
```
|
|
388
|
+
|
|
389
|
+
Checkpoint is not completion. Never discard/reset/stash work merely to checkpoint.
|
|
390
|
+
|
|
391
|
+
**Harvest-before-fallback invariant:** a received but unharvested checkpoint outranks terminal failure, idle fallback, hard timeout, cancellation handling, and writer replacement. Harvest it first, then re-evaluate.
|
|
392
|
+
|
|
393
|
+
### Remaining-delta replan
|
|
394
|
+
|
|
395
|
+
Fallback always operates on the missing delta, never by restarting the whole stage.
|
|
396
|
+
|
|
397
|
+
For implementation `fallback_policy=replan`:
|
|
398
|
+
|
|
399
|
+
- `uncovered_scope`: replan only demonstrably uncovered scope.
|
|
400
|
+
- `checkpoint_remaining_delta`: use harvested `remaining_delta` plus completed/patch/validation/blocker evidence.
|
|
401
|
+
- `retained_workspace`: only after the old writer is terminal/cancelled.
|
|
402
|
+
- `harvested_snapshot_only`: isolated replacement consumes the immutable harvested snapshot and fences later old-worker output.
|
|
403
|
+
|
|
404
|
+
Completed accepted work is reopened only with concrete invalidating evidence.
|
|
405
|
+
|
|
406
|
+
Writable fallback still obeys cancellation/fencing. Never let a Parent writer or replacement Worker overlap a live old writer on the same scope.
|
|
407
|
+
|
|
408
|
+
### Read-only review retry
|
|
409
|
+
|
|
410
|
+
Review fallback is `retry_review`, not implementation `replan`.
|
|
411
|
+
|
|
412
|
+
`retry_review`:
|
|
413
|
+
|
|
414
|
+
- is valid only for review stage;
|
|
415
|
+
- is read-only;
|
|
416
|
+
- never produces implementation `replan_scope`;
|
|
417
|
+
- never opens writable scope or requires writer fencing;
|
|
418
|
+
- must consume a `review_attempt` reservation before the new reviewer starts;
|
|
419
|
+
- is forbidden once `review_deadline` is reached.
|
|
420
|
+
|
|
421
|
+
## 6. Exploration
|
|
422
|
+
|
|
423
|
+
Spawn at most `exploration_workers`, using planned role capability/model/reasoning when supported. Give each Explorer a distinct evidence question.
|
|
424
|
+
|
|
425
|
+
Do not kill a Luna `xhigh/max` Explorer merely because a short Parent wait returned no final output. Re-evaluate lifecycle from real progress evidence.
|
|
426
|
+
|
|
427
|
+
Respect join policy and `min_successful_workers`. Cancel stragglers only when the plan/evaluator permits it.
|
|
428
|
+
|
|
429
|
+
## 7. Evidence-based bounded implementation
|
|
430
|
+
|
|
431
|
+
When `implementation_stage.work_unit_mode=bounded`, Parent creates an explicit manifest before implementation spawn. Every unit contains:
|
|
432
|
+
|
|
433
|
+
```text
|
|
434
|
+
unit_id
|
|
435
|
+
scope_id
|
|
436
|
+
generation
|
|
437
|
+
acceptance_delta
|
|
438
|
+
write_scope_id
|
|
439
|
+
validation: non-empty list
|
|
440
|
+
write_paths: normalized repo-relative POSIX paths when required
|
|
441
|
+
depends_on
|
|
442
|
+
parallel_group: optional
|
|
443
|
+
```
|
|
444
|
+
|
|
445
|
+
Do not manufacture meaningless splits solely to satisfy a number. `minimum_work_units` remains the hard minimum; current built-in strategies keep it at `1`. `maximum_work_units` is a hard manifest bound, not a Worker count. Strategy semantic limits may be raised only by already-proven writable topology within the strategy's WorkerBudget; this permits serial waves without inventing extra workstreams.
|
|
446
|
+
|
|
447
|
+
Validate before spawn:
|
|
448
|
+
|
|
449
|
+
```bash
|
|
450
|
+
python3 ~/.codex/codex-flow/strategies/work_unit_runtime.py \
|
|
451
|
+
--policy-json '<implementation_stage JSON>' \
|
|
452
|
+
--manifest-json '<{"units":[...]}>' \
|
|
453
|
+
--implementation-workers <ExecutionPlan implementation_workers> \
|
|
454
|
+
--max-concurrent-threads <ExecutionPlan max_concurrent_threads>
|
|
455
|
+
```
|
|
456
|
+
|
|
457
|
+
Safety rules:
|
|
458
|
+
|
|
459
|
+
- same write scope is serial and dependency ordered;
|
|
460
|
+
- overlapping/ancestor-descendant `write_paths` require dependency ordering;
|
|
461
|
+
- a parallel group requires distinct non-overlapping write scopes/paths and no dependency path inside the group;
|
|
462
|
+
- a parallel group cannot exceed already-planned implementation/thread concurrency;
|
|
463
|
+
- work-unit partitioning never creates new writable workstreams;
|
|
464
|
+
- `write_paths` are lexical preflight evidence, not OS locks or symlink-safe ownership enforcement.
|
|
465
|
+
|
|
466
|
+
`implementation_workers` is concurrent topology, not total logical units. Reuse planned slots across serial waves. `join_between_work_units=true` returns control to Parent after each completed unit.
|
|
467
|
+
|
|
468
|
+
Do not send all bounded units to one Worker as a giant “complete everything” transaction.
|
|
469
|
+
|
|
470
|
+
## 8. Implementation handoff and local repair
|
|
471
|
+
|
|
472
|
+
Every implementation handoff should be bounded to its assigned scope/unit and contain:
|
|
473
|
+
|
|
474
|
+
- scope/unit/generation;
|
|
475
|
+
- acceptance delta;
|
|
476
|
+
- allowed write scope/paths;
|
|
477
|
+
- dependencies and prior harvested evidence;
|
|
478
|
+
- root-cause/design decision;
|
|
479
|
+
- constraints/non-goals;
|
|
480
|
+
- validation;
|
|
481
|
+
- `max_worker_repair_attempts` when present.
|
|
482
|
+
|
|
483
|
+
A Worker changes only its assigned unit and proves it with validation. For bounded mode it must return to Parent rather than beginning the next unit.
|
|
484
|
+
|
|
485
|
+
Local repair semantics:
|
|
486
|
+
|
|
487
|
+
- initial implementation + first validation = zero repairs;
|
|
488
|
+
- one repair = attributable validation failure -> corrective edit -> targeted revalidation;
|
|
489
|
+
- investigation without an edit does not consume an attempt;
|
|
490
|
+
- unrelated pre-existing failures do not count;
|
|
491
|
+
- on exhaustion, preserve patch/evidence and return the smallest remaining delta.
|
|
492
|
+
|
|
493
|
+
Local repair is separate from Parent `max_repair_cycles` and from lifecycle fallback.
|
|
494
|
+
|
|
495
|
+
## 9. Review and Parent finalization
|
|
496
|
+
|
|
497
|
+
Parent always reviews relevant diff, affected call sites, validation evidence, acceptance criteria, architecture consistency, and regression risk.
|
|
498
|
+
|
|
499
|
+
- `review_mode=parent`: no reviewer Worker.
|
|
500
|
+
- `review_mode=independent+parent`: spawn exactly the planned reviewer topology, subject to `review_attempt` reservations and the review window.
|
|
501
|
+
|
|
502
|
+
Review only a terminal workspace or immutable harvested snapshot, never concurrently mutating writable scope.
|
|
503
|
+
|
|
504
|
+
Multiple reviewers should receive complementary scopes. They are read-only and must not silently implement.
|
|
505
|
+
|
|
506
|
+
Track accepted reviewer completions against `review_stage.min_successful_workers`. `action=finalize_parent` only closes reviewer admission; it does not satisfy the review join. If the required/quorum reviewer join is still unsatisfied at `review_deadline`, preserve the available review evidence and fail/escalate rather than silently converting the plan to Parent-only success.
|
|
507
|
+
|
|
508
|
+
Once `action=finalize_parent`, do not start/retry reviewers. Use the remaining tail for Parent reconciliation and final verification. Enter that tail only after writable Workers are terminal/cancel-confirmed or safely fenced and all returned checkpoints have been harvested. At hard deadline, no new Worker or Parent writer starts; only already-collected evidence may be reported.
|
|
509
|
+
|
|
510
|
+
Direct mode still requires Parent self-review.
|
|
511
|
+
|
|
512
|
+
## 10. Parent repair and replan
|
|
513
|
+
|
|
514
|
+
Parent repair receives the smallest defect delta, not the original full task. Never exceed `max_repair_cycles`.
|
|
515
|
+
|
|
516
|
+
A repair requiring writes is general work and therefore cannot open after the task soft/general-work deadline.
|
|
517
|
+
|
|
518
|
+
Re-profile and compile a newer ExecutionPlan when material evidence changes task semantics, but continue using the original task ledger and initial budget plan for cumulative admission. A newer plan cannot reset counters or extend deadlines.
|
|
519
|
+
|
|
520
|
+
Worker stall/failure/checkpoint/replan does not itself consume Parent repair cycles.
|
|
521
|
+
|
|
522
|
+
## 11. Efficient reasoning rollout
|
|
523
|
+
|
|
524
|
+
Reasoning rollout is currently efficient-specific. Modes:
|
|
525
|
+
|
|
526
|
+
- `legacy`: select historical Worker effort.
|
|
527
|
+
- `shadow`: report proposal but select historical effort.
|
|
528
|
+
- `adaptive`: select proposal.
|
|
529
|
+
|
|
530
|
+
Proposal:
|
|
531
|
+
|
|
532
|
+
```text
|
|
533
|
+
max(rollout class target, rollout minimum, parent_reasoning)
|
|
534
|
+
```
|
|
535
|
+
|
|
536
|
+
The planner output is intent, not proof that the active runtime applied a per-spawn override. If runtime override is unsupported, use the installed baseline and report that limitation. Do not fabricate observed effort.
|
|
537
|
+
|
|
538
|
+
## 12. Quota and telemetry discipline
|
|
539
|
+
|
|
540
|
+
Quota is never guessed. Unavailable quota state is `unknown`.
|
|
541
|
+
|
|
542
|
+
Telemetry is observational. Never call a model solely to estimate tokens, quota, or duration.
|
|
543
|
+
|
|
544
|
+
For each Worker, record available deterministic lifecycle/latency evidence with stable IDs and no prompt/transcript/path payload. `observed_effort` must be runtime-confirmed; otherwise record null.
|
|
545
|
+
|
|
546
|
+
Telemetry failures are fail-open and must never block checkpoint harvest, cancellation/fencing, joining, or delivery.
|
|
547
|
+
|
|
548
|
+
Do not auto-tune policy from a handful of runs. Compare latency, completion, failures/censoring, and quota after a sufficiently homogeneous sample.
|
|
549
|
+
|
|
550
|
+
## 13. Concurrency discipline
|
|
551
|
+
|
|
552
|
+
`max_concurrent_threads` is a stage concurrency ceiling. WorkerBudget maxima are envelopes, not mandatory counts.
|
|
553
|
+
|
|
554
|
+
Parallel writable work requires already-proven isolated scopes/worktrees represented in `writable_workstreams`. Replan does not magically create another safe writer.
|
|
555
|
+
|
|
556
|
+
Prefer targeted search, compact context, diff-scoped review, narrow validation, and harvested evidence. Avoid duplicated agents, rereading unchanged files, and Parent reimplementation of Worker work.
|
|
557
|
+
|
|
558
|
+
## 14. Re-plan triggers
|
|
559
|
+
|
|
560
|
+
Re-profile when:
|
|
561
|
+
|
|
562
|
+
- complexity/risk materially changes;
|
|
563
|
+
- root cause is disproven;
|
|
564
|
+
- cross-module dependency appears;
|
|
565
|
+
- writable isolation changes;
|
|
566
|
+
- explicit quality intent changes;
|
|
567
|
+
- a required implementation stage returns `fallback_policy=replan`;
|
|
568
|
+
- Parent repair fails;
|
|
569
|
+
- reliable quota/runtime state materially changes.
|
|
570
|
+
|
|
571
|
+
Replan from evaluator-provided missing scope, never the original task by default. Completed accepted units remain evidence unless specifically invalidated.
|
|
572
|
+
|
|
573
|
+
## Runtime/version invariant
|
|
574
|
+
|
|
575
|
+
Persistent user policy remains schema v4. This FlowPilot runtime consumes ExecutionPlan schema v11 and task-ledger schema v2 for this feature set. These task-ledger semantics were not released with an older persisted-task format, so there is intentionally no grandfather/adoption path for incompatible task-ledger state: mismatched plan/policy/schema fails closed.
|
|
576
|
+
|
|
577
|
+
This does **not** relax ordinary persistent policy precedence or installer configuration preservation; it only means the new v11 task-execution contract has one canonical interpretation.
|
codex_flow/mcp.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""MCP server entry point for codex-flow-mcp."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import sys
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
from typing import List, Optional
|
|
8
|
+
from codex_flow.cli import get_resource_root, run_python_script
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def main(argv: Optional[List[str]] = None) -> int:
|
|
12
|
+
"""Run FlowPilot ChatGPT MCP Server."""
|
|
13
|
+
if argv is None:
|
|
14
|
+
argv = sys.argv[1:]
|
|
15
|
+
|
|
16
|
+
root = get_resource_root()
|
|
17
|
+
candidates = [
|
|
18
|
+
root / "apps" / "chatgpt-mcp" / "server.py",
|
|
19
|
+
root / "server.py",
|
|
20
|
+
]
|
|
21
|
+
server_path = None
|
|
22
|
+
for c in candidates:
|
|
23
|
+
if c.is_file():
|
|
24
|
+
server_path = c
|
|
25
|
+
break
|
|
26
|
+
|
|
27
|
+
if not server_path:
|
|
28
|
+
print("Error: MCP server.py not found in codex-flow package", file=sys.stderr)
|
|
29
|
+
return 1
|
|
30
|
+
|
|
31
|
+
return run_python_script(server_path, argv)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
if __name__ == "__main__":
|
|
35
|
+
sys.exit(main())
|