codex-flow 2.1.13__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codex_flow/__init__.py +28 -0
- codex_flow/__main__.py +9 -0
- codex_flow/cli.py +242 -0
- codex_flow/data/LICENSE +21 -0
- codex_flow/data/README.en.md +303 -0
- codex_flow/data/README.md +305 -0
- codex_flow/data/VERSION +1 -0
- codex_flow/data/apps/chatgpt-mcp/README.md +86 -0
- codex_flow/data/apps/chatgpt-mcp/__init__.py +1 -0
- codex_flow/data/apps/chatgpt-mcp/adapter.py +458 -0
- codex_flow/data/apps/chatgpt-mcp/server.py +358 -0
- codex_flow/data/apps/chatgpt-mcp/widget.html +927 -0
- codex_flow/data/apps/macos-overlay/README.en.md +121 -0
- codex_flow/data/apps/macos-overlay/README.md +123 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayRuntimeState.swift +126 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayScreenGeometry.swift +82 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayWindowController.swift +1052 -0
- codex_flow/data/apps/macos-overlay/Sources/Localization.swift +197 -0
- codex_flow/data/apps/macos-overlay/Sources/Models/TelemetryData.swift +1557 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/AccountSnapshotService.swift +1101 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/FlowPilotInstanceLock.swift +153 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/IPCServer.swift +298 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryQueryEngine.swift +800 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryWatcher.swift +135 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/UpdateService.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AccountView.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AnalyticsView.swift +566 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AutostartView.swift +293 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/BubbleView.swift +317 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HistoryView.swift +1124 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HoverRevealText.swift +165 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/InspectorSkillsToolsView.swift +121 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/LogoView.swift +182 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SleekSwitch.swift +117 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/StrategyModeView.swift +561 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SummaryView.swift +1273 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/UpdateView.swift +352 -0
- codex_flow/data/apps/macos-overlay/Sources/main.swift +340 -0
- codex_flow/data/apps/macos-overlay/Tests/OverlayScreenGeometryTests.swift +163 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryPhase1ContractTests.swift +357 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQueryEngineConcurrencyTests.swift +221 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQuotaSelectionTests.swift +158 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryWorkerTokenTests.swift +122 -0
- codex_flow/data/apps/macos-overlay/build.sh +75 -0
- codex_flow/data/benchmark/corpus.json +103 -0
- codex_flow/data/benchmark/manifest.example.json +41 -0
- codex_flow/data/benchmark/manifest.schema.json +137 -0
- codex_flow/data/benchmark/prices/gpt-5.6-2026-08-30.json +5 -0
- codex_flow/data/benchmark/profiles.json +90 -0
- codex_flow/data/benchmark/schema.json +77 -0
- codex_flow/data/benchmark/tasks.json +50 -0
- codex_flow/data/completions/codex-flow.bash +34 -0
- codex_flow/data/completions/codex-flow.zsh +52 -0
- codex_flow/data/glama.json +6 -0
- codex_flow/data/install-release.ps1 +126 -0
- codex_flow/data/install-release.sh +155 -0
- codex_flow/data/install.ps1 +349 -0
- codex_flow/data/install.sh +362 -0
- codex_flow/data/policy/benchmark.toml +49 -0
- codex_flow/data/policy/defaults.toml +70 -0
- codex_flow/data/scripts/analyze-benchmark.py +510 -0
- codex_flow/data/scripts/benchmark-local.py +171 -0
- codex_flow/data/scripts/check-recommendation.py +277 -0
- codex_flow/data/scripts/doctor.py +449 -0
- codex_flow/data/scripts/generate-release-manifest.py +74 -0
- codex_flow/data/scripts/localization.py +192 -0
- codex_flow/data/scripts/manage-hooks.py +448 -0
- codex_flow/data/scripts/manage-instructions.py +389 -0
- codex_flow/data/scripts/manage-shell.py +151 -0
- codex_flow/data/scripts/materialize-corpus.py +193 -0
- codex_flow/data/scripts/menu.py +646 -0
- codex_flow/data/scripts/migrations/0001_update_settings.py +80 -0
- codex_flow/data/scripts/package-release.py +132 -0
- codex_flow/data/scripts/render-benchmark-report.py +292 -0
- codex_flow/data/scripts/run-benchmark.py +829 -0
- codex_flow/data/scripts/strategies/__init__.py +28 -0
- codex_flow/data/scripts/strategies/balanced.py +115 -0
- codex_flow/data/scripts/strategies/base.py +363 -0
- codex_flow/data/scripts/strategies/efficient.py +158 -0
- codex_flow/data/scripts/strategies/lifecycle_runtime.py +590 -0
- codex_flow/data/scripts/strategies/quality.py +209 -0
- codex_flow/data/scripts/strategies/speed.py +108 -0
- codex_flow/data/scripts/strategies/task_budget_runtime.py +644 -0
- codex_flow/data/scripts/strategies/task_phase_runtime.py +341 -0
- codex_flow/data/scripts/strategies/work_unit_runtime.py +421 -0
- codex_flow/data/scripts/strategy_runtime.py +1091 -0
- codex_flow/data/scripts/telemetry.py +400 -0
- codex_flow/data/scripts/telemetry_core/__init__.py +192 -0
- codex_flow/data/scripts/telemetry_core/app_server.py +1192 -0
- codex_flow/data/scripts/telemetry_core/collector.py +1247 -0
- codex_flow/data/scripts/telemetry_core/common.py +421 -0
- codex_flow/data/scripts/telemetry_core/latency.py +593 -0
- codex_flow/data/scripts/telemetry_core/query.py +427 -0
- codex_flow/data/scripts/telemetry_core/quota_ledger.py +598 -0
- codex_flow/data/scripts/telemetry_core/render.py +460 -0
- codex_flow/data/scripts/telemetry_core/repair.py +223 -0
- codex_flow/data/scripts/ui.py +266 -0
- codex_flow/data/scripts/update-homebrew-formula.py +146 -0
- codex_flow/data/scripts/update_runtime_config.py +134 -0
- codex_flow/data/scripts/updater.py +1718 -0
- codex_flow/data/smithery.yaml +18 -0
- codex_flow/data/templates/agents/worker-explorer.toml +24 -0
- codex_flow/data/templates/agents/worker-implementer.toml +49 -0
- codex_flow/data/templates/agents/worker-reviewer.toml +25 -0
- codex_flow/data/templates/flow-pilot-instructions.md +35 -0
- codex_flow/data/templates/skills/flow-pilot/SKILL.md +577 -0
- codex_flow/mcp.py +35 -0
- codex_flow-2.1.13.dist-info/METADATA +342 -0
- codex_flow-2.1.13.dist-info/RECORD +113 -0
- codex_flow-2.1.13.dist-info/WHEEL +5 -0
- codex_flow-2.1.13.dist-info/entry_points.txt +3 -0
- codex_flow-2.1.13.dist-info/licenses/LICENSE +21 -0
- codex_flow-2.1.13.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,829 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Run reproducible direct-model and codex-flow benchmark strategies.
|
|
3
|
+
|
|
4
|
+
Result schema v2 adds an explicit strategy dimension. Direct strategies use
|
|
5
|
+
one Codex model for implementation and bounded verifier repairs. The flow
|
|
6
|
+
strategy uses a capable parent for read-only planning/review and a cheaper
|
|
7
|
+
worker for all writes. Usage is attributed per role/model so mixed-model cost
|
|
8
|
+
is measurable instead of estimated.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import argparse
|
|
13
|
+
import importlib.util
|
|
14
|
+
import json
|
|
15
|
+
import os
|
|
16
|
+
import re
|
|
17
|
+
import shutil
|
|
18
|
+
import subprocess
|
|
19
|
+
import sys
|
|
20
|
+
import tempfile
|
|
21
|
+
import time
|
|
22
|
+
from pathlib import Path
|
|
23
|
+
from typing import Any
|
|
24
|
+
|
|
25
|
+
VALID_CLASSES = {"routine", "complex", "critical"}
|
|
26
|
+
VALID_EFFORTS = {"high", "xhigh", "max"}
|
|
27
|
+
VALID_STRATEGIES = {"direct", "flow", "runtime"}
|
|
28
|
+
FULL_COMMIT_RE = re.compile(r"^[0-9a-fA-F]{40}$")
|
|
29
|
+
MAX_CAPTURE = 8000
|
|
30
|
+
REVIEW_SCHEMA = {
|
|
31
|
+
"type": "object",
|
|
32
|
+
"properties": {
|
|
33
|
+
"verdict": {"enum": ["pass", "repair"]},
|
|
34
|
+
"feedback": {"type": "string"},
|
|
35
|
+
},
|
|
36
|
+
"required": ["verdict", "feedback"],
|
|
37
|
+
"additionalProperties": False,
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def fail(message: str) -> None:
|
|
42
|
+
raise ValueError(message)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def config_id(config: dict[str, Any]) -> str:
|
|
46
|
+
if isinstance(config.get("id"), str) and config["id"]:
|
|
47
|
+
return config["id"]
|
|
48
|
+
if config.get("strategy", "direct") in {"flow", "runtime"}:
|
|
49
|
+
fail(f"{config['strategy']} strategies require a non-empty id")
|
|
50
|
+
return f"{config['model']}-{config['reasoning_effort']}"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def config_strategy(config: dict[str, Any]) -> str:
|
|
54
|
+
return config.get("strategy", "direct")
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def validate_actor(actor: Any, label: str) -> None:
|
|
58
|
+
if not isinstance(actor, dict):
|
|
59
|
+
fail(f"{label} must be an object")
|
|
60
|
+
if not isinstance(actor.get("model"), str) or not actor["model"]:
|
|
61
|
+
fail(f"{label}.model must be a non-empty string")
|
|
62
|
+
effort = actor.get("reasoning_effort")
|
|
63
|
+
if isinstance(effort, str):
|
|
64
|
+
if effort not in VALID_EFFORTS:
|
|
65
|
+
fail(f"invalid {label}.reasoning_effort: {effort!r}")
|
|
66
|
+
elif isinstance(effort, dict):
|
|
67
|
+
if set(effort) != VALID_CLASSES or any(value not in VALID_EFFORTS for value in effort.values()):
|
|
68
|
+
fail(f"{label}.reasoning_effort must define valid routine/complex/critical efforts")
|
|
69
|
+
else:
|
|
70
|
+
fail(f"invalid {label}.reasoning_effort: {effort!r}")
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def actor_effort(actor: dict[str, Any], task_class: str) -> str:
|
|
74
|
+
effort = actor["reasoning_effort"]
|
|
75
|
+
return effort[task_class] if isinstance(effort, dict) else effort
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def config_models(config: dict[str, Any]) -> set[str]:
|
|
79
|
+
if config_strategy(config) == "direct":
|
|
80
|
+
return {config["model"]}
|
|
81
|
+
return {config["parent"]["model"], config["worker"]["model"]}
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def load_manifest(path: Path) -> dict[str, Any]:
|
|
85
|
+
data = json.loads(path.read_text())
|
|
86
|
+
version = data.get("schema_version")
|
|
87
|
+
if version not in {1, 2}:
|
|
88
|
+
fail("manifest schema_version must be 1 or 2")
|
|
89
|
+
tasks = data.get("tasks")
|
|
90
|
+
matrix = data.get("matrix")
|
|
91
|
+
if not isinstance(tasks, list) or not tasks:
|
|
92
|
+
fail("manifest tasks must be a non-empty array")
|
|
93
|
+
if not isinstance(matrix, list) or not matrix:
|
|
94
|
+
fail("manifest matrix must be a non-empty array")
|
|
95
|
+
allow_floating = data.get("allow_floating_refs", False)
|
|
96
|
+
if not isinstance(allow_floating, bool):
|
|
97
|
+
fail("allow_floating_refs must be boolean")
|
|
98
|
+
|
|
99
|
+
task_ids: set[str] = set()
|
|
100
|
+
for task in tasks:
|
|
101
|
+
required = {"id", "class", "source", "base_ref", "prompt", "verify"}
|
|
102
|
+
missing = required - task.keys()
|
|
103
|
+
if missing:
|
|
104
|
+
fail(f"task missing fields: {sorted(missing)}")
|
|
105
|
+
if task["id"] in task_ids:
|
|
106
|
+
fail(f"duplicate task id: {task['id']}")
|
|
107
|
+
task_ids.add(task["id"])
|
|
108
|
+
if task["class"] not in VALID_CLASSES:
|
|
109
|
+
fail(f"invalid task class for {task['id']}: {task['class']}")
|
|
110
|
+
if not isinstance(task["verify"], list) or not task["verify"] or not all(isinstance(x, str) for x in task["verify"]):
|
|
111
|
+
fail(f"verify for {task['id']} must be a non-empty argv array")
|
|
112
|
+
for field in ("prompt", "source", "base_ref"):
|
|
113
|
+
if not isinstance(task[field], str) or not task[field].strip():
|
|
114
|
+
fail(f"{field} for {task['id']} must be non-empty")
|
|
115
|
+
if not allow_floating and not FULL_COMMIT_RE.fullmatch(task["base_ref"]):
|
|
116
|
+
fail(
|
|
117
|
+
f"task {task['id']} base_ref must be a full 40-character commit SHA; "
|
|
118
|
+
"set allow_floating_refs=true only for exploratory runs"
|
|
119
|
+
)
|
|
120
|
+
|
|
121
|
+
strategy_ids: set[str] = set()
|
|
122
|
+
for config in matrix:
|
|
123
|
+
if not isinstance(config, dict):
|
|
124
|
+
fail("matrix entries must be objects")
|
|
125
|
+
strategy = config_strategy(config)
|
|
126
|
+
if strategy not in VALID_STRATEGIES:
|
|
127
|
+
fail(f"invalid strategy: {strategy!r}")
|
|
128
|
+
identifier = config_id(config)
|
|
129
|
+
if identifier in strategy_ids:
|
|
130
|
+
fail(f"duplicate strategy id: {identifier}")
|
|
131
|
+
strategy_ids.add(identifier)
|
|
132
|
+
if strategy == "direct":
|
|
133
|
+
validate_actor(config, f"matrix[{identifier}]")
|
|
134
|
+
if not isinstance(config["reasoning_effort"], str):
|
|
135
|
+
fail(f"matrix[{identifier}] direct strategies require a fixed reasoning_effort")
|
|
136
|
+
else:
|
|
137
|
+
if version < 2:
|
|
138
|
+
fail(f"{strategy} strategies require manifest schema_version 2")
|
|
139
|
+
validate_actor(config.get("parent"), f"matrix[{identifier}].parent")
|
|
140
|
+
validate_actor(config.get("worker"), f"matrix[{identifier}].worker")
|
|
141
|
+
policy = config.get("reasoning_policy", "fixed")
|
|
142
|
+
if policy not in {"fixed", "adaptive"}:
|
|
143
|
+
fail(f"matrix[{identifier}].reasoning_policy must be fixed or adaptive")
|
|
144
|
+
parent_effort = config["parent"]["reasoning_effort"]
|
|
145
|
+
worker_effort = config["worker"]["reasoning_effort"]
|
|
146
|
+
if policy == "fixed" and not all(isinstance(value, str) for value in (parent_effort, worker_effort)):
|
|
147
|
+
fail(f"matrix[{identifier}] fixed {strategy} requires fixed actor reasoning efforts")
|
|
148
|
+
if policy == "adaptive" and not all(isinstance(value, dict) for value in (parent_effort, worker_effort)):
|
|
149
|
+
fail(f"matrix[{identifier}] adaptive {strategy} requires per-class actor reasoning efforts")
|
|
150
|
+
|
|
151
|
+
repetitions = data.get("repetitions", 1)
|
|
152
|
+
timeout = data.get("timeout_seconds", 1800)
|
|
153
|
+
repairs = data.get("max_repair_cycles", 2)
|
|
154
|
+
if not isinstance(repetitions, int) or repetitions < 1:
|
|
155
|
+
fail("repetitions must be >= 1")
|
|
156
|
+
if not isinstance(timeout, int) or timeout < 1:
|
|
157
|
+
fail("timeout_seconds must be >= 1")
|
|
158
|
+
if not isinstance(repairs, int) or repairs < 0:
|
|
159
|
+
fail("max_repair_cycles must be >= 0")
|
|
160
|
+
return data
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def clone_source(source: str, base_ref: str, destination: Path) -> str:
|
|
164
|
+
subprocess.run(
|
|
165
|
+
["git", "clone", "--quiet", "--no-checkout", source, str(destination)],
|
|
166
|
+
check=True,
|
|
167
|
+
stdout=subprocess.PIPE,
|
|
168
|
+
stderr=subprocess.PIPE,
|
|
169
|
+
text=True,
|
|
170
|
+
)
|
|
171
|
+
subprocess.run(
|
|
172
|
+
["git", "-C", str(destination), "checkout", "--quiet", "--detach", base_ref],
|
|
173
|
+
check=True,
|
|
174
|
+
stdout=subprocess.PIPE,
|
|
175
|
+
stderr=subprocess.PIPE,
|
|
176
|
+
text=True,
|
|
177
|
+
)
|
|
178
|
+
return subprocess.check_output(
|
|
179
|
+
["git", "-C", str(destination), "rev-parse", "HEAD"], text=True
|
|
180
|
+
).strip()
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def parse_usage(jsonl: str) -> dict[str, int]:
|
|
184
|
+
usage = {"input_tokens": 0, "cached_input_tokens": 0, "output_tokens": 0}
|
|
185
|
+
for line in jsonl.splitlines():
|
|
186
|
+
try:
|
|
187
|
+
event = json.loads(line)
|
|
188
|
+
except json.JSONDecodeError:
|
|
189
|
+
continue
|
|
190
|
+
if event.get("type") != "turn.completed" or not isinstance(event.get("usage"), dict):
|
|
191
|
+
continue
|
|
192
|
+
for key in usage:
|
|
193
|
+
value = event["usage"].get(key, 0)
|
|
194
|
+
if isinstance(value, int) and value >= 0:
|
|
195
|
+
usage[key] += value
|
|
196
|
+
return usage
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def run_codex(
|
|
200
|
+
workdir: Path,
|
|
201
|
+
model: str,
|
|
202
|
+
effort: str,
|
|
203
|
+
prompt: str,
|
|
204
|
+
timeout: int,
|
|
205
|
+
*,
|
|
206
|
+
sandbox: str = "workspace-write",
|
|
207
|
+
last_message_path: Path | None = None,
|
|
208
|
+
output_schema_path: Path | None = None,
|
|
209
|
+
ignore_user_config: bool = True,
|
|
210
|
+
env: dict[str, str] | None = None,
|
|
211
|
+
) -> tuple[int, dict[str, int], str, float, str]:
|
|
212
|
+
cmd = [
|
|
213
|
+
"codex", "exec", "--ephemeral", "--json",
|
|
214
|
+
]
|
|
215
|
+
if ignore_user_config:
|
|
216
|
+
cmd.append("--ignore-user-config")
|
|
217
|
+
cmd.extend([
|
|
218
|
+
"--sandbox", sandbox, "--model", model,
|
|
219
|
+
"-c", f'model_reasoning_effort="{effort}"',
|
|
220
|
+
])
|
|
221
|
+
if output_schema_path is not None:
|
|
222
|
+
cmd.extend(["--output-schema", str(output_schema_path)])
|
|
223
|
+
if last_message_path is not None:
|
|
224
|
+
cmd.extend(["--output-last-message", str(last_message_path)])
|
|
225
|
+
cmd.extend(["--cd", str(workdir), prompt])
|
|
226
|
+
started = time.monotonic()
|
|
227
|
+
run_env = os.environ.copy()
|
|
228
|
+
if env:
|
|
229
|
+
run_env.update(env)
|
|
230
|
+
try:
|
|
231
|
+
proc = subprocess.run(cmd, text=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=timeout, env=run_env)
|
|
232
|
+
elapsed = time.monotonic() - started
|
|
233
|
+
last_message = ""
|
|
234
|
+
if last_message_path is not None and last_message_path.exists():
|
|
235
|
+
last_message = last_message_path.read_text()[-MAX_CAPTURE:]
|
|
236
|
+
return proc.returncode, parse_usage(proc.stdout), (proc.stderr or proc.stdout)[-MAX_CAPTURE:], elapsed, last_message
|
|
237
|
+
except subprocess.TimeoutExpired as exc:
|
|
238
|
+
elapsed = time.monotonic() - started
|
|
239
|
+
stdout = exc.stdout if isinstance(exc.stdout, str) else ""
|
|
240
|
+
stderr = exc.stderr if isinstance(exc.stderr, str) else ""
|
|
241
|
+
return 124, parse_usage(stdout), (stderr or stdout)[-MAX_CAPTURE:], elapsed, ""
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def run_verify(workdir: Path, argv: list[str], timeout: int) -> tuple[bool, str, float]:
|
|
245
|
+
started = time.monotonic()
|
|
246
|
+
try:
|
|
247
|
+
proc = subprocess.run(argv, cwd=workdir, text=True, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, timeout=timeout)
|
|
248
|
+
return proc.returncode == 0, proc.stdout[-MAX_CAPTURE:], time.monotonic() - started
|
|
249
|
+
except subprocess.TimeoutExpired as exc:
|
|
250
|
+
output = exc.stdout if isinstance(exc.stdout, str) else ""
|
|
251
|
+
return False, (output + "\nverification timed out")[-MAX_CAPTURE:], time.monotonic() - started
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def empty_usage() -> dict[str, int]:
|
|
255
|
+
return {"input_tokens": 0, "cached_input_tokens": 0, "output_tokens": 0}
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def add_usage(total: dict[str, int], current: dict[str, int]) -> None:
|
|
259
|
+
for key in empty_usage():
|
|
260
|
+
total[key] += current[key]
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
def add_model_usage(
|
|
264
|
+
totals: dict[tuple[str, str, str], dict[str, int]],
|
|
265
|
+
role: str,
|
|
266
|
+
model: str,
|
|
267
|
+
effort: str,
|
|
268
|
+
usage: dict[str, int],
|
|
269
|
+
) -> None:
|
|
270
|
+
key = (role, model, effort)
|
|
271
|
+
if key not in totals:
|
|
272
|
+
totals[key] = {**empty_usage(), "calls": 0}
|
|
273
|
+
add_usage(totals[key], usage)
|
|
274
|
+
totals[key]["calls"] += 1
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def flatten_model_usage(totals: dict[tuple[str, str, str], dict[str, int]]) -> list[dict[str, Any]]:
|
|
278
|
+
return [
|
|
279
|
+
{"role": role, "model": model, "reasoning_effort": effort, **usage}
|
|
280
|
+
for (role, model, effort), usage in sorted(totals.items())
|
|
281
|
+
]
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def build_row(
|
|
285
|
+
task: dict[str, Any],
|
|
286
|
+
config: dict[str, Any],
|
|
287
|
+
repetition: int,
|
|
288
|
+
commit: str,
|
|
289
|
+
passed: bool,
|
|
290
|
+
first_passed: bool,
|
|
291
|
+
repair_cycles: int,
|
|
292
|
+
review_cycles: int,
|
|
293
|
+
total_wall: float,
|
|
294
|
+
codex_exit: int,
|
|
295
|
+
verification: str,
|
|
296
|
+
diagnostic: str,
|
|
297
|
+
usage_by_model: dict[tuple[str, str, str], dict[str, int]],
|
|
298
|
+
) -> dict[str, Any]:
|
|
299
|
+
strategy = config_strategy(config)
|
|
300
|
+
if strategy == "direct":
|
|
301
|
+
model = config["model"]
|
|
302
|
+
effort = actor_effort(config, task["class"])
|
|
303
|
+
worker_model = None
|
|
304
|
+
worker_effort = None
|
|
305
|
+
reasoning_policy = "fixed"
|
|
306
|
+
else:
|
|
307
|
+
model = config["parent"]["model"]
|
|
308
|
+
effort = actor_effort(config["parent"], task["class"])
|
|
309
|
+
worker_model = config["worker"]["model"]
|
|
310
|
+
worker_effort = actor_effort(config["worker"], task["class"])
|
|
311
|
+
reasoning_policy = config.get("reasoning_policy", "fixed")
|
|
312
|
+
model_usage = flatten_model_usage(usage_by_model)
|
|
313
|
+
total_usage = empty_usage()
|
|
314
|
+
for item in model_usage:
|
|
315
|
+
add_usage(total_usage, item)
|
|
316
|
+
return {
|
|
317
|
+
"schema_version": 2,
|
|
318
|
+
"task_id": task["id"],
|
|
319
|
+
"task_class": task["class"],
|
|
320
|
+
"strategy_id": config_id(config),
|
|
321
|
+
"strategy": strategy,
|
|
322
|
+
"reasoning_policy": reasoning_policy,
|
|
323
|
+
"model": model,
|
|
324
|
+
"reasoning_effort": effort,
|
|
325
|
+
"worker_model": worker_model,
|
|
326
|
+
"worker_reasoning_effort": worker_effort,
|
|
327
|
+
"passed": passed,
|
|
328
|
+
"first_passed": first_passed,
|
|
329
|
+
**total_usage,
|
|
330
|
+
"model_usage": model_usage,
|
|
331
|
+
"repair_cycles": repair_cycles,
|
|
332
|
+
"review_cycles": review_cycles,
|
|
333
|
+
"wall_time_seconds": round(total_wall, 3),
|
|
334
|
+
"source_commit": commit,
|
|
335
|
+
"repetition": repetition,
|
|
336
|
+
"codex_exit_code": codex_exit,
|
|
337
|
+
"verification_excerpt": verification[-2000:] if not passed else "",
|
|
338
|
+
"diagnostic_excerpt": diagnostic[-2000:] if codex_exit != 0 else "",
|
|
339
|
+
}
|
|
340
|
+
|
|
341
|
+
|
|
342
|
+
def execute_direct(
|
|
343
|
+
task: dict[str, Any],
|
|
344
|
+
config: dict[str, Any],
|
|
345
|
+
root: Path,
|
|
346
|
+
repetition: int,
|
|
347
|
+
timeout: int,
|
|
348
|
+
max_repairs: int,
|
|
349
|
+
) -> dict[str, Any]:
|
|
350
|
+
workdir = root / "repo"
|
|
351
|
+
commit = clone_source(task["source"], task["base_ref"], workdir)
|
|
352
|
+
model = config["model"]
|
|
353
|
+
effort = actor_effort(config, task["class"])
|
|
354
|
+
usage_by_model: dict[tuple[str, str, str], dict[str, int]] = {}
|
|
355
|
+
total_wall = 0.0
|
|
356
|
+
|
|
357
|
+
codex_exit, usage, diagnostic, elapsed, _ = run_codex(
|
|
358
|
+
workdir, model, effort, task["prompt"], timeout
|
|
359
|
+
)
|
|
360
|
+
total_wall += elapsed
|
|
361
|
+
add_model_usage(usage_by_model, "direct", model, effort, usage)
|
|
362
|
+
repair_cycles = 0
|
|
363
|
+
|
|
364
|
+
if codex_exit == 0:
|
|
365
|
+
passed, verification, verify_elapsed = run_verify(workdir, task["verify"], timeout)
|
|
366
|
+
total_wall += verify_elapsed
|
|
367
|
+
else:
|
|
368
|
+
passed = False
|
|
369
|
+
verification = "benchmark verification skipped because codex exec exited non-zero"
|
|
370
|
+
first_passed = passed
|
|
371
|
+
|
|
372
|
+
while not passed and codex_exit == 0 and repair_cycles < max_repairs:
|
|
373
|
+
repair_cycles += 1
|
|
374
|
+
repair_prompt = (
|
|
375
|
+
"The previous implementation did not pass the fixed benchmark verifier. "
|
|
376
|
+
"Repair only the implementation needed to satisfy the original task; do not weaken, skip, or edit the verifier.\n\n"
|
|
377
|
+
f"Original task:\n{task['prompt']}\n\nVerifier output:\n{verification}"
|
|
378
|
+
)
|
|
379
|
+
codex_exit, usage, diagnostic, elapsed, _ = run_codex(
|
|
380
|
+
workdir, model, effort, repair_prompt, timeout
|
|
381
|
+
)
|
|
382
|
+
total_wall += elapsed
|
|
383
|
+
add_model_usage(usage_by_model, "direct", model, effort, usage)
|
|
384
|
+
if codex_exit != 0:
|
|
385
|
+
passed = False
|
|
386
|
+
verification = "benchmark verification skipped because repair codex exec exited non-zero"
|
|
387
|
+
break
|
|
388
|
+
passed, verification, verify_elapsed = run_verify(workdir, task["verify"], timeout)
|
|
389
|
+
total_wall += verify_elapsed
|
|
390
|
+
|
|
391
|
+
return build_row(
|
|
392
|
+
task, config, repetition, commit, passed, first_passed, repair_cycles, 0,
|
|
393
|
+
total_wall, codex_exit, verification, diagnostic, usage_by_model,
|
|
394
|
+
)
|
|
395
|
+
|
|
396
|
+
|
|
397
|
+
def parse_review(message: str) -> tuple[str, str]:
|
|
398
|
+
try:
|
|
399
|
+
review = json.loads(message)
|
|
400
|
+
except json.JSONDecodeError as exc:
|
|
401
|
+
fail(f"parent review did not return valid JSON: {exc}")
|
|
402
|
+
verdict = review.get("verdict")
|
|
403
|
+
feedback = review.get("feedback")
|
|
404
|
+
if verdict not in {"pass", "repair"} or not isinstance(feedback, str):
|
|
405
|
+
fail("parent review must contain verdict=pass|repair and string feedback")
|
|
406
|
+
return verdict, feedback
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def execute_flow(
|
|
410
|
+
task: dict[str, Any],
|
|
411
|
+
config: dict[str, Any],
|
|
412
|
+
root: Path,
|
|
413
|
+
repetition: int,
|
|
414
|
+
timeout: int,
|
|
415
|
+
max_repairs: int,
|
|
416
|
+
) -> dict[str, Any]:
|
|
417
|
+
workdir = root / "repo"
|
|
418
|
+
commit = clone_source(task["source"], task["base_ref"], workdir)
|
|
419
|
+
parent = config["parent"]
|
|
420
|
+
worker = config["worker"]
|
|
421
|
+
parent_model = parent["model"]
|
|
422
|
+
worker_model = worker["model"]
|
|
423
|
+
parent_effort = actor_effort(parent, task["class"])
|
|
424
|
+
worker_effort = actor_effort(worker, task["class"])
|
|
425
|
+
usage_by_model: dict[tuple[str, str, str], dict[str, int]] = {}
|
|
426
|
+
total_wall = 0.0
|
|
427
|
+
diagnostic = ""
|
|
428
|
+
|
|
429
|
+
plan_path = root / "parent-plan.txt"
|
|
430
|
+
plan_prompt = (
|
|
431
|
+
"You are the high-capability parent in the codex-flow workflow. Inspect the repository read-only. "
|
|
432
|
+
"Do not modify files. Produce a compact implementation handoff containing root cause/design decision, "
|
|
433
|
+
"scope and non-goals, relevant files, ordered steps, compatibility constraints, risks, acceptance criteria, "
|
|
434
|
+
"and required validation. Remove ambiguity so a worker can implement without redesigning.\n\n"
|
|
435
|
+
f"Task:\n{task['prompt']}"
|
|
436
|
+
)
|
|
437
|
+
codex_exit, usage, diagnostic, elapsed, plan = run_codex(
|
|
438
|
+
workdir, parent_model, parent_effort, plan_prompt, timeout,
|
|
439
|
+
sandbox="read-only", last_message_path=plan_path,
|
|
440
|
+
)
|
|
441
|
+
total_wall += elapsed
|
|
442
|
+
add_model_usage(usage_by_model, "parent", parent_model, parent_effort, usage)
|
|
443
|
+
if codex_exit != 0 or not plan.strip():
|
|
444
|
+
if codex_exit == 0:
|
|
445
|
+
codex_exit = 65
|
|
446
|
+
diagnostic = "parent planning completed without a handoff"
|
|
447
|
+
return build_row(
|
|
448
|
+
task, config, repetition, commit, False, False, 0, 0, total_wall,
|
|
449
|
+
codex_exit, "benchmark verification skipped because parent planning failed",
|
|
450
|
+
diagnostic, usage_by_model,
|
|
451
|
+
)
|
|
452
|
+
|
|
453
|
+
worker_prompt = (
|
|
454
|
+
"You are the implementation worker in the codex-flow workflow. Execute the parent's plan without redesigning it. "
|
|
455
|
+
"Stay in scope, implement the complete change, run narrow relevant validation, and do not weaken or edit any external verifier.\n\n"
|
|
456
|
+
f"Original task:\n{task['prompt']}\n\nParent handoff:\n{plan}"
|
|
457
|
+
)
|
|
458
|
+
codex_exit, usage, diagnostic, elapsed, _ = run_codex(
|
|
459
|
+
workdir, worker_model, worker_effort, worker_prompt, timeout
|
|
460
|
+
)
|
|
461
|
+
total_wall += elapsed
|
|
462
|
+
add_model_usage(usage_by_model, "worker", worker_model, worker_effort, usage)
|
|
463
|
+
if codex_exit != 0:
|
|
464
|
+
return build_row(
|
|
465
|
+
task, config, repetition, commit, False, False, 0, 0, total_wall,
|
|
466
|
+
codex_exit, "benchmark verification skipped because worker codex exec exited non-zero",
|
|
467
|
+
diagnostic, usage_by_model,
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
review_schema_path = root / "review-schema.json"
|
|
471
|
+
review_schema_path.write_text(json.dumps(REVIEW_SCHEMA))
|
|
472
|
+
first_passed = False
|
|
473
|
+
passed = False
|
|
474
|
+
repair_cycles = 0
|
|
475
|
+
review_cycles = 0
|
|
476
|
+
verification = ""
|
|
477
|
+
|
|
478
|
+
while True:
|
|
479
|
+
verifier_passed, verification, verify_elapsed = run_verify(workdir, task["verify"], timeout)
|
|
480
|
+
total_wall += verify_elapsed
|
|
481
|
+
review_cycles += 1
|
|
482
|
+
review_path = root / f"parent-review-{review_cycles}.json"
|
|
483
|
+
review_prompt = (
|
|
484
|
+
"You are the high-capability parent performing the codex-flow final review. Work read-only. "
|
|
485
|
+
"Inspect the current git diff and directly affected call sites. Check the original task, architecture, "
|
|
486
|
+
"compatibility, regression risk, and verifier result. Return verdict 'pass' only when the implementation "
|
|
487
|
+
"is complete and the verifier passed. Otherwise return verdict 'repair' with a compact delta instruction "
|
|
488
|
+
"for the worker; do not rewrite the whole plan.\n\n"
|
|
489
|
+
f"Original task:\n{task['prompt']}\n\nVerifier passed: {str(verifier_passed).lower()}\n"
|
|
490
|
+
f"Verifier output:\n{verification}"
|
|
491
|
+
)
|
|
492
|
+
codex_exit, usage, diagnostic, elapsed, review_message = run_codex(
|
|
493
|
+
workdir, parent_model, parent_effort, review_prompt, timeout,
|
|
494
|
+
sandbox="read-only", last_message_path=review_path,
|
|
495
|
+
output_schema_path=review_schema_path,
|
|
496
|
+
)
|
|
497
|
+
total_wall += elapsed
|
|
498
|
+
add_model_usage(usage_by_model, "parent", parent_model, parent_effort, usage)
|
|
499
|
+
if codex_exit != 0:
|
|
500
|
+
passed = False
|
|
501
|
+
break
|
|
502
|
+
try:
|
|
503
|
+
verdict, feedback = parse_review(review_message)
|
|
504
|
+
except ValueError as exc:
|
|
505
|
+
codex_exit = 65
|
|
506
|
+
diagnostic = str(exc)
|
|
507
|
+
passed = False
|
|
508
|
+
break
|
|
509
|
+
passed = verifier_passed and verdict == "pass"
|
|
510
|
+
if review_cycles == 1:
|
|
511
|
+
first_passed = passed
|
|
512
|
+
if passed or repair_cycles >= max_repairs:
|
|
513
|
+
if not passed and feedback:
|
|
514
|
+
verification = (
|
|
515
|
+
f"{verification}\nParent review requested repair:\n{feedback}"
|
|
516
|
+
).strip()
|
|
517
|
+
break
|
|
518
|
+
|
|
519
|
+
repair_cycles += 1
|
|
520
|
+
if not verifier_passed:
|
|
521
|
+
feedback = (
|
|
522
|
+
f"{feedback}\n\nThe fixed external verifier still fails:\n{verification}"
|
|
523
|
+
).strip()
|
|
524
|
+
repair_prompt = (
|
|
525
|
+
"You are the implementation worker repairing a reviewed codex-flow change. Apply only the requested delta, "
|
|
526
|
+
"keep the original acceptance criteria intact, and run narrow validation. Do not weaken or edit the verifier.\n\n"
|
|
527
|
+
f"Original task:\n{task['prompt']}\n\nParent repair instruction:\n{feedback}"
|
|
528
|
+
)
|
|
529
|
+
codex_exit, usage, diagnostic, elapsed, _ = run_codex(
|
|
530
|
+
workdir, worker_model, worker_effort, repair_prompt, timeout
|
|
531
|
+
)
|
|
532
|
+
total_wall += elapsed
|
|
533
|
+
add_model_usage(usage_by_model, "worker", worker_model, worker_effort, usage)
|
|
534
|
+
if codex_exit != 0:
|
|
535
|
+
passed = False
|
|
536
|
+
verification = "benchmark verification skipped because repair worker codex exec exited non-zero"
|
|
537
|
+
break
|
|
538
|
+
|
|
539
|
+
return build_row(
|
|
540
|
+
task, config, repetition, commit, passed, first_passed, repair_cycles,
|
|
541
|
+
review_cycles, total_wall, codex_exit, verification, diagnostic, usage_by_model,
|
|
542
|
+
)
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def _load_manage_hooks() -> Any:
|
|
546
|
+
path = Path(__file__).resolve().parent / "manage-hooks.py"
|
|
547
|
+
spec = importlib.util.spec_from_file_location("manage_hooks", path)
|
|
548
|
+
if spec is None or spec.loader is None:
|
|
549
|
+
return None
|
|
550
|
+
module = importlib.util.module_from_spec(spec)
|
|
551
|
+
spec.loader.exec_module(module)
|
|
552
|
+
return module
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def authorize_hooks(codex_home: Path) -> None:
|
|
556
|
+
hooks_path = codex_home / "hooks.json"
|
|
557
|
+
config_path = codex_home / "config.toml"
|
|
558
|
+
if not hooks_path.exists():
|
|
559
|
+
return
|
|
560
|
+
manage_hooks = _load_manage_hooks()
|
|
561
|
+
if manage_hooks is None:
|
|
562
|
+
return
|
|
563
|
+
entries = manage_hooks._managed_hook_entries(hooks_path)
|
|
564
|
+
if not entries:
|
|
565
|
+
return
|
|
566
|
+
lines = []
|
|
567
|
+
if config_path.exists():
|
|
568
|
+
lines.append(config_path.read_text(encoding="utf-8"))
|
|
569
|
+
lines.append("\n# Pre-authorized FlowPilot benchmark hooks")
|
|
570
|
+
for entry in entries:
|
|
571
|
+
key_repr = json.dumps(entry["key"])
|
|
572
|
+
lines.append(f"[hooks.state.{key_repr}]")
|
|
573
|
+
lines.append(f'trusted_hash = "{entry["current_hash"]}"')
|
|
574
|
+
lines.append("enabled = true\n")
|
|
575
|
+
config_path.write_text("\n".join(lines), encoding="utf-8")
|
|
576
|
+
|
|
577
|
+
|
|
578
|
+
def execute_runtime(
|
|
579
|
+
task: dict[str, Any],
|
|
580
|
+
config: dict[str, Any],
|
|
581
|
+
root: Path,
|
|
582
|
+
repetition: int,
|
|
583
|
+
timeout: int,
|
|
584
|
+
max_repairs: int,
|
|
585
|
+
) -> dict[str, Any]:
|
|
586
|
+
workdir = root / "repo"
|
|
587
|
+
commit = clone_source(task["source"], task["base_ref"], workdir)
|
|
588
|
+
parent = config["parent"]
|
|
589
|
+
worker = config["worker"]
|
|
590
|
+
parent_model = parent["model"]
|
|
591
|
+
worker_model = worker["model"]
|
|
592
|
+
parent_effort = actor_effort(parent, task["class"])
|
|
593
|
+
worker_effort = actor_effort(worker, task["class"])
|
|
594
|
+
usage_by_model: dict[tuple[str, str, str], dict[str, int]] = {}
|
|
595
|
+
total_wall = 0.0
|
|
596
|
+
diagnostic = ""
|
|
597
|
+
|
|
598
|
+
# Provision isolated FlowPilot environment
|
|
599
|
+
codex_home = root / ".codex"
|
|
600
|
+
bin_dir = root / "bin"
|
|
601
|
+
codex_home.mkdir(parents=True, exist_ok=True)
|
|
602
|
+
bin_dir.mkdir(parents=True, exist_ok=True)
|
|
603
|
+
|
|
604
|
+
install_sh = Path(__file__).resolve().parent.parent / "install.sh"
|
|
605
|
+
install_env = os.environ.copy()
|
|
606
|
+
install_env.update({
|
|
607
|
+
"CODEX_HOME": str(codex_home),
|
|
608
|
+
"CODEX_FLOW_BIN_DIR": str(bin_dir),
|
|
609
|
+
"CODEX_FLOW_SHELL": "none",
|
|
610
|
+
"CODEX_FLOW_STRATEGY": config.get("profile", "balanced"),
|
|
611
|
+
"CODEX_FLOW_ROUTING_MODE": config.get("routing_mode", "delegate"),
|
|
612
|
+
"CODEX_FLOW_WORKER_MODEL": worker_model,
|
|
613
|
+
"CODEX_FLOW_WORKER_ROUTINE_EFFORT": actor_effort(worker, "routine"),
|
|
614
|
+
"CODEX_FLOW_WORKER_COMPLEX_EFFORT": actor_effort(worker, "complex"),
|
|
615
|
+
"CODEX_FLOW_WORKER_CRITICAL_EFFORT": actor_effort(worker, "critical"),
|
|
616
|
+
"CODEX_FLOW_PARENT_MIN_EFFORT": actor_effort(parent, "routine"),
|
|
617
|
+
"CODEX_FLOW_PARENT_ROUTINE_EFFORT": actor_effort(parent, "routine"),
|
|
618
|
+
"CODEX_FLOW_PARENT_COMPLEX_EFFORT": actor_effort(parent, "complex"),
|
|
619
|
+
"CODEX_FLOW_PARENT_CRITICAL_EFFORT": actor_effort(parent, "critical"),
|
|
620
|
+
"CODEX_FLOW_TELEMETRY_ENABLED": "true",
|
|
621
|
+
"CODEX_FLOW_TELEMETRY_NOTIFICATIONS": "false",
|
|
622
|
+
})
|
|
623
|
+
try:
|
|
624
|
+
subprocess.run(
|
|
625
|
+
[str(install_sh)],
|
|
626
|
+
env=install_env,
|
|
627
|
+
check=True,
|
|
628
|
+
stdout=subprocess.PIPE,
|
|
629
|
+
stderr=subprocess.PIPE,
|
|
630
|
+
text=True,
|
|
631
|
+
)
|
|
632
|
+
except subprocess.CalledProcessError as exc:
|
|
633
|
+
return build_row(
|
|
634
|
+
task, config, repetition, commit, False, False, 0, 0, 0.0,
|
|
635
|
+
exc.returncode, "FlowPilot runtime installation failed",
|
|
636
|
+
exc.stderr[-MAX_CAPTURE:], usage_by_model,
|
|
637
|
+
)
|
|
638
|
+
|
|
639
|
+
# Pre-authorize hooks for headless non-interactive execution
|
|
640
|
+
authorize_hooks(codex_home)
|
|
641
|
+
|
|
642
|
+
runtime_env = {
|
|
643
|
+
"CODEX_HOME": str(codex_home),
|
|
644
|
+
"PATH": f"{bin_dir}:{os.environ.get('PATH', '')}",
|
|
645
|
+
}
|
|
646
|
+
|
|
647
|
+
runtime_prompt = (
|
|
648
|
+
"You are operating in an environment with the FlowPilot multi-agent runtime installed. "
|
|
649
|
+
"Follow the FlowPilot task entry instructions, use the flow-pilot skill and subagents "
|
|
650
|
+
"to inspect, plan, implement, and verify the task.\n\n"
|
|
651
|
+
f"Task:\n{task['prompt']}"
|
|
652
|
+
)
|
|
653
|
+
|
|
654
|
+
codex_exit, usage, diagnostic, elapsed, _ = run_codex(
|
|
655
|
+
workdir, parent_model, parent_effort, runtime_prompt, timeout,
|
|
656
|
+
ignore_user_config=False, env=runtime_env,
|
|
657
|
+
)
|
|
658
|
+
total_wall += elapsed
|
|
659
|
+
cumulative_codex_usage = empty_usage()
|
|
660
|
+
add_usage(cumulative_codex_usage, usage)
|
|
661
|
+
|
|
662
|
+
repair_cycles = 0
|
|
663
|
+
if codex_exit == 0:
|
|
664
|
+
passed, verification, verify_elapsed = run_verify(workdir, task["verify"], timeout)
|
|
665
|
+
total_wall += verify_elapsed
|
|
666
|
+
else:
|
|
667
|
+
passed = False
|
|
668
|
+
verification = "benchmark verification skipped because codex exec exited non-zero"
|
|
669
|
+
first_passed = passed
|
|
670
|
+
|
|
671
|
+
while not passed and codex_exit == 0 and repair_cycles < max_repairs:
|
|
672
|
+
repair_cycles += 1
|
|
673
|
+
repair_prompt = (
|
|
674
|
+
"The previous implementation did not pass the fixed benchmark verifier. "
|
|
675
|
+
"Use FlowPilot to repair only the implementation needed to satisfy the original task; do not weaken, skip, or edit the verifier.\n\n"
|
|
676
|
+
f"Original task:\n{task['prompt']}\n\nVerifier output:\n{verification}"
|
|
677
|
+
)
|
|
678
|
+
codex_exit, usage, diagnostic, elapsed, _ = run_codex(
|
|
679
|
+
workdir, parent_model, parent_effort, repair_prompt, timeout,
|
|
680
|
+
ignore_user_config=False, env=runtime_env,
|
|
681
|
+
)
|
|
682
|
+
total_wall += elapsed
|
|
683
|
+
add_usage(cumulative_codex_usage, usage)
|
|
684
|
+
if codex_exit != 0:
|
|
685
|
+
passed = False
|
|
686
|
+
verification = "benchmark verification skipped because repair codex exec exited non-zero"
|
|
687
|
+
break
|
|
688
|
+
passed, verification, verify_elapsed = run_verify(workdir, task["verify"], timeout)
|
|
689
|
+
total_wall += verify_elapsed
|
|
690
|
+
|
|
691
|
+
# Attribute tokens from FlowPilot telemetry runs if present
|
|
692
|
+
runs_dir = codex_home / "codex-flow" / "telemetry" / "runs"
|
|
693
|
+
review_cycles = 0
|
|
694
|
+
if runs_dir.exists():
|
|
695
|
+
for run_file in sorted(runs_dir.glob("*.json")):
|
|
696
|
+
try:
|
|
697
|
+
run_data = json.loads(run_file.read_text(encoding="utf-8"))
|
|
698
|
+
except (OSError, json.JSONDecodeError):
|
|
699
|
+
continue
|
|
700
|
+
p = run_data.get("parent")
|
|
701
|
+
if isinstance(p, dict):
|
|
702
|
+
p_u = p.get("usage_delta") or p.get("usage")
|
|
703
|
+
if isinstance(p_u, dict) and any(p_u.get(k, 0) > 0 for k in ("input_tokens", "cached_input_tokens", "output_tokens")):
|
|
704
|
+
add_model_usage(usage_by_model, "parent", parent_model, parent_effort, p_u)
|
|
705
|
+
workers = run_data.get("workers")
|
|
706
|
+
if isinstance(workers, dict):
|
|
707
|
+
for w_name, w in workers.items():
|
|
708
|
+
if "review" in w_name.lower():
|
|
709
|
+
review_cycles += 1
|
|
710
|
+
if isinstance(w, dict) and isinstance(w.get("usage"), dict):
|
|
711
|
+
w_u = w["usage"]
|
|
712
|
+
if any(w_u.get(k, 0) > 0 for k in ("input_tokens", "cached_input_tokens", "output_tokens")):
|
|
713
|
+
add_model_usage(usage_by_model, "worker", worker_model, worker_effort, w_u)
|
|
714
|
+
|
|
715
|
+
# Fallback to top-level codex exec usage attributed to parent if telemetry was not populated
|
|
716
|
+
if not usage_by_model:
|
|
717
|
+
add_model_usage(usage_by_model, "parent", parent_model, parent_effort, cumulative_codex_usage)
|
|
718
|
+
|
|
719
|
+
return build_row(
|
|
720
|
+
task, config, repetition, commit, passed, first_passed, repair_cycles,
|
|
721
|
+
review_cycles, total_wall, codex_exit, verification, diagnostic, usage_by_model,
|
|
722
|
+
)
|
|
723
|
+
|
|
724
|
+
|
|
725
|
+
def execute_run(
|
|
726
|
+
task: dict[str, Any],
|
|
727
|
+
config: dict[str, Any],
|
|
728
|
+
root: Path,
|
|
729
|
+
repetition: int,
|
|
730
|
+
timeout: int,
|
|
731
|
+
max_repairs: int,
|
|
732
|
+
) -> dict[str, Any]:
|
|
733
|
+
strategy = config_strategy(config)
|
|
734
|
+
if strategy == "flow":
|
|
735
|
+
return execute_flow(task, config, root, repetition, timeout, max_repairs)
|
|
736
|
+
if strategy == "runtime":
|
|
737
|
+
return execute_runtime(task, config, root, repetition, timeout, max_repairs)
|
|
738
|
+
return execute_direct(task, config, root, repetition, timeout, max_repairs)
|
|
739
|
+
|
|
740
|
+
|
|
741
|
+
def no_usage(row: dict[str, Any]) -> bool:
|
|
742
|
+
return (
|
|
743
|
+
row["input_tokens"] == 0
|
|
744
|
+
and row["cached_input_tokens"] == 0
|
|
745
|
+
and row["output_tokens"] == 0
|
|
746
|
+
)
|
|
747
|
+
|
|
748
|
+
|
|
749
|
+
def main() -> int:
|
|
750
|
+
parser = argparse.ArgumentParser()
|
|
751
|
+
parser.add_argument("--manifest", required=True)
|
|
752
|
+
parser.add_argument("--output")
|
|
753
|
+
parser.add_argument("--only-task")
|
|
754
|
+
parser.add_argument("--only-model")
|
|
755
|
+
parser.add_argument("--only-strategy")
|
|
756
|
+
parser.add_argument("--dry-run", action="store_true")
|
|
757
|
+
parser.add_argument(
|
|
758
|
+
"--fail-fast-infrastructure",
|
|
759
|
+
action="store_true",
|
|
760
|
+
help="stop the batch when codex exits non-zero before reporting any token usage",
|
|
761
|
+
)
|
|
762
|
+
args = parser.parse_args()
|
|
763
|
+
|
|
764
|
+
if not shutil.which("git"):
|
|
765
|
+
fail("git is required")
|
|
766
|
+
if not args.dry_run and not shutil.which("codex"):
|
|
767
|
+
fail("codex CLI is required")
|
|
768
|
+
|
|
769
|
+
manifest = load_manifest(Path(args.manifest))
|
|
770
|
+
tasks = [t for t in manifest["tasks"] if not args.only_task or t["id"] == args.only_task]
|
|
771
|
+
matrix = [
|
|
772
|
+
m for m in manifest["matrix"]
|
|
773
|
+
if (not args.only_model or args.only_model in config_models(m))
|
|
774
|
+
and (not args.only_strategy or args.only_strategy == config_id(m))
|
|
775
|
+
]
|
|
776
|
+
if not tasks:
|
|
777
|
+
fail("task filter matched nothing")
|
|
778
|
+
if not matrix:
|
|
779
|
+
fail("strategy/model filter matched nothing")
|
|
780
|
+
|
|
781
|
+
planned = len(tasks) * len(matrix) * manifest.get("repetitions", 1)
|
|
782
|
+
if args.dry_run:
|
|
783
|
+
print(json.dumps({
|
|
784
|
+
"planned_runs": planned,
|
|
785
|
+
"tasks": [t["id"] for t in tasks],
|
|
786
|
+
"strategies": [config_id(m) for m in matrix],
|
|
787
|
+
"matrix": matrix,
|
|
788
|
+
}, sort_keys=True))
|
|
789
|
+
return 0
|
|
790
|
+
|
|
791
|
+
if not args.output:
|
|
792
|
+
fail("--output is required unless --dry-run is used")
|
|
793
|
+
|
|
794
|
+
output = Path(args.output)
|
|
795
|
+
output.parent.mkdir(parents=True, exist_ok=True)
|
|
796
|
+
with output.open("a", encoding="utf-8") as sink:
|
|
797
|
+
for task in tasks:
|
|
798
|
+
for config in matrix:
|
|
799
|
+
for repetition in range(1, manifest.get("repetitions", 1) + 1):
|
|
800
|
+
with tempfile.TemporaryDirectory(prefix="codex-flow-bench-") as tmp:
|
|
801
|
+
row = execute_run(
|
|
802
|
+
task, config, Path(tmp), repetition,
|
|
803
|
+
manifest.get("timeout_seconds", 1800),
|
|
804
|
+
task.get("max_repair_cycles", manifest.get("max_repair_cycles", 2)),
|
|
805
|
+
)
|
|
806
|
+
sink.write(json.dumps(row, sort_keys=True) + "\n")
|
|
807
|
+
sink.flush()
|
|
808
|
+
print(
|
|
809
|
+
f"{row['task_id']} {row['strategy_id']} "
|
|
810
|
+
f"rep={repetition} pass={row['passed']} first={row['first_passed']} "
|
|
811
|
+
f"repairs={row['repair_cycles']} reviews={row['review_cycles']} "
|
|
812
|
+
f"tokens={row['input_tokens'] + row['output_tokens']}",
|
|
813
|
+
file=sys.stderr,
|
|
814
|
+
)
|
|
815
|
+
if args.fail_fast_infrastructure and row["codex_exit_code"] != 0 and no_usage(row):
|
|
816
|
+
fail(
|
|
817
|
+
"codex exited non-zero without reporting usage; stopping benchmark "
|
|
818
|
+
f"after {row['task_id']} {row['strategy_id']} "
|
|
819
|
+
f"(exit={row['codex_exit_code']})"
|
|
820
|
+
)
|
|
821
|
+
return 0
|
|
822
|
+
|
|
823
|
+
|
|
824
|
+
if __name__ == "__main__":
|
|
825
|
+
try:
|
|
826
|
+
raise SystemExit(main())
|
|
827
|
+
except Exception as exc:
|
|
828
|
+
print(f"benchmark runner failed: {exc}", file=sys.stderr)
|
|
829
|
+
raise SystemExit(1)
|