codex-flow 2.1.13__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. codex_flow/__init__.py +28 -0
  2. codex_flow/__main__.py +9 -0
  3. codex_flow/cli.py +242 -0
  4. codex_flow/data/LICENSE +21 -0
  5. codex_flow/data/README.en.md +303 -0
  6. codex_flow/data/README.md +305 -0
  7. codex_flow/data/VERSION +1 -0
  8. codex_flow/data/apps/chatgpt-mcp/README.md +86 -0
  9. codex_flow/data/apps/chatgpt-mcp/__init__.py +1 -0
  10. codex_flow/data/apps/chatgpt-mcp/adapter.py +458 -0
  11. codex_flow/data/apps/chatgpt-mcp/server.py +358 -0
  12. codex_flow/data/apps/chatgpt-mcp/widget.html +927 -0
  13. codex_flow/data/apps/macos-overlay/README.en.md +121 -0
  14. codex_flow/data/apps/macos-overlay/README.md +123 -0
  15. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayRuntimeState.swift +126 -0
  16. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayScreenGeometry.swift +82 -0
  17. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayWindowController.swift +1052 -0
  18. codex_flow/data/apps/macos-overlay/Sources/Localization.swift +197 -0
  19. codex_flow/data/apps/macos-overlay/Sources/Models/TelemetryData.swift +1557 -0
  20. codex_flow/data/apps/macos-overlay/Sources/Services/AccountSnapshotService.swift +1101 -0
  21. codex_flow/data/apps/macos-overlay/Sources/Services/FlowPilotInstanceLock.swift +153 -0
  22. codex_flow/data/apps/macos-overlay/Sources/Services/IPCServer.swift +298 -0
  23. codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryQueryEngine.swift +800 -0
  24. codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryWatcher.swift +135 -0
  25. codex_flow/data/apps/macos-overlay/Sources/Services/UpdateService.swift +610 -0
  26. codex_flow/data/apps/macos-overlay/Sources/Views/AccountView.swift +610 -0
  27. codex_flow/data/apps/macos-overlay/Sources/Views/AnalyticsView.swift +566 -0
  28. codex_flow/data/apps/macos-overlay/Sources/Views/AutostartView.swift +293 -0
  29. codex_flow/data/apps/macos-overlay/Sources/Views/BubbleView.swift +317 -0
  30. codex_flow/data/apps/macos-overlay/Sources/Views/HistoryView.swift +1124 -0
  31. codex_flow/data/apps/macos-overlay/Sources/Views/HoverRevealText.swift +165 -0
  32. codex_flow/data/apps/macos-overlay/Sources/Views/InspectorSkillsToolsView.swift +121 -0
  33. codex_flow/data/apps/macos-overlay/Sources/Views/LogoView.swift +182 -0
  34. codex_flow/data/apps/macos-overlay/Sources/Views/SleekSwitch.swift +117 -0
  35. codex_flow/data/apps/macos-overlay/Sources/Views/StrategyModeView.swift +561 -0
  36. codex_flow/data/apps/macos-overlay/Sources/Views/SummaryView.swift +1273 -0
  37. codex_flow/data/apps/macos-overlay/Sources/Views/UpdateView.swift +352 -0
  38. codex_flow/data/apps/macos-overlay/Sources/main.swift +340 -0
  39. codex_flow/data/apps/macos-overlay/Tests/OverlayScreenGeometryTests.swift +163 -0
  40. codex_flow/data/apps/macos-overlay/Tests/TelemetryPhase1ContractTests.swift +357 -0
  41. codex_flow/data/apps/macos-overlay/Tests/TelemetryQueryEngineConcurrencyTests.swift +221 -0
  42. codex_flow/data/apps/macos-overlay/Tests/TelemetryQuotaSelectionTests.swift +158 -0
  43. codex_flow/data/apps/macos-overlay/Tests/TelemetryWorkerTokenTests.swift +122 -0
  44. codex_flow/data/apps/macos-overlay/build.sh +75 -0
  45. codex_flow/data/benchmark/corpus.json +103 -0
  46. codex_flow/data/benchmark/manifest.example.json +41 -0
  47. codex_flow/data/benchmark/manifest.schema.json +137 -0
  48. codex_flow/data/benchmark/prices/gpt-5.6-2026-08-30.json +5 -0
  49. codex_flow/data/benchmark/profiles.json +90 -0
  50. codex_flow/data/benchmark/schema.json +77 -0
  51. codex_flow/data/benchmark/tasks.json +50 -0
  52. codex_flow/data/completions/codex-flow.bash +34 -0
  53. codex_flow/data/completions/codex-flow.zsh +52 -0
  54. codex_flow/data/glama.json +6 -0
  55. codex_flow/data/install-release.ps1 +126 -0
  56. codex_flow/data/install-release.sh +155 -0
  57. codex_flow/data/install.ps1 +349 -0
  58. codex_flow/data/install.sh +362 -0
  59. codex_flow/data/policy/benchmark.toml +49 -0
  60. codex_flow/data/policy/defaults.toml +70 -0
  61. codex_flow/data/scripts/analyze-benchmark.py +510 -0
  62. codex_flow/data/scripts/benchmark-local.py +171 -0
  63. codex_flow/data/scripts/check-recommendation.py +277 -0
  64. codex_flow/data/scripts/doctor.py +449 -0
  65. codex_flow/data/scripts/generate-release-manifest.py +74 -0
  66. codex_flow/data/scripts/localization.py +192 -0
  67. codex_flow/data/scripts/manage-hooks.py +448 -0
  68. codex_flow/data/scripts/manage-instructions.py +389 -0
  69. codex_flow/data/scripts/manage-shell.py +151 -0
  70. codex_flow/data/scripts/materialize-corpus.py +193 -0
  71. codex_flow/data/scripts/menu.py +646 -0
  72. codex_flow/data/scripts/migrations/0001_update_settings.py +80 -0
  73. codex_flow/data/scripts/package-release.py +132 -0
  74. codex_flow/data/scripts/render-benchmark-report.py +292 -0
  75. codex_flow/data/scripts/run-benchmark.py +829 -0
  76. codex_flow/data/scripts/strategies/__init__.py +28 -0
  77. codex_flow/data/scripts/strategies/balanced.py +115 -0
  78. codex_flow/data/scripts/strategies/base.py +363 -0
  79. codex_flow/data/scripts/strategies/efficient.py +158 -0
  80. codex_flow/data/scripts/strategies/lifecycle_runtime.py +590 -0
  81. codex_flow/data/scripts/strategies/quality.py +209 -0
  82. codex_flow/data/scripts/strategies/speed.py +108 -0
  83. codex_flow/data/scripts/strategies/task_budget_runtime.py +644 -0
  84. codex_flow/data/scripts/strategies/task_phase_runtime.py +341 -0
  85. codex_flow/data/scripts/strategies/work_unit_runtime.py +421 -0
  86. codex_flow/data/scripts/strategy_runtime.py +1091 -0
  87. codex_flow/data/scripts/telemetry.py +400 -0
  88. codex_flow/data/scripts/telemetry_core/__init__.py +192 -0
  89. codex_flow/data/scripts/telemetry_core/app_server.py +1192 -0
  90. codex_flow/data/scripts/telemetry_core/collector.py +1247 -0
  91. codex_flow/data/scripts/telemetry_core/common.py +421 -0
  92. codex_flow/data/scripts/telemetry_core/latency.py +593 -0
  93. codex_flow/data/scripts/telemetry_core/query.py +427 -0
  94. codex_flow/data/scripts/telemetry_core/quota_ledger.py +598 -0
  95. codex_flow/data/scripts/telemetry_core/render.py +460 -0
  96. codex_flow/data/scripts/telemetry_core/repair.py +223 -0
  97. codex_flow/data/scripts/ui.py +266 -0
  98. codex_flow/data/scripts/update-homebrew-formula.py +146 -0
  99. codex_flow/data/scripts/update_runtime_config.py +134 -0
  100. codex_flow/data/scripts/updater.py +1718 -0
  101. codex_flow/data/smithery.yaml +18 -0
  102. codex_flow/data/templates/agents/worker-explorer.toml +24 -0
  103. codex_flow/data/templates/agents/worker-implementer.toml +49 -0
  104. codex_flow/data/templates/agents/worker-reviewer.toml +25 -0
  105. codex_flow/data/templates/flow-pilot-instructions.md +35 -0
  106. codex_flow/data/templates/skills/flow-pilot/SKILL.md +577 -0
  107. codex_flow/mcp.py +35 -0
  108. codex_flow-2.1.13.dist-info/METADATA +342 -0
  109. codex_flow-2.1.13.dist-info/RECORD +113 -0
  110. codex_flow-2.1.13.dist-info/WHEEL +5 -0
  111. codex_flow-2.1.13.dist-info/entry_points.txt +3 -0
  112. codex_flow-2.1.13.dist-info/licenses/LICENSE +21 -0
  113. codex_flow-2.1.13.dist-info/top_level.txt +1 -0
@@ -0,0 +1,829 @@
1
+ #!/usr/bin/env python3
2
+ """Run reproducible direct-model and codex-flow benchmark strategies.
3
+
4
+ Result schema v2 adds an explicit strategy dimension. Direct strategies use
5
+ one Codex model for implementation and bounded verifier repairs. The flow
6
+ strategy uses a capable parent for read-only planning/review and a cheaper
7
+ worker for all writes. Usage is attributed per role/model so mixed-model cost
8
+ is measurable instead of estimated.
9
+ """
10
+ from __future__ import annotations
11
+
12
+ import argparse
13
+ import importlib.util
14
+ import json
15
+ import os
16
+ import re
17
+ import shutil
18
+ import subprocess
19
+ import sys
20
+ import tempfile
21
+ import time
22
+ from pathlib import Path
23
+ from typing import Any
24
+
25
+ VALID_CLASSES = {"routine", "complex", "critical"}
26
+ VALID_EFFORTS = {"high", "xhigh", "max"}
27
+ VALID_STRATEGIES = {"direct", "flow", "runtime"}
28
+ FULL_COMMIT_RE = re.compile(r"^[0-9a-fA-F]{40}$")
29
+ MAX_CAPTURE = 8000
30
+ REVIEW_SCHEMA = {
31
+ "type": "object",
32
+ "properties": {
33
+ "verdict": {"enum": ["pass", "repair"]},
34
+ "feedback": {"type": "string"},
35
+ },
36
+ "required": ["verdict", "feedback"],
37
+ "additionalProperties": False,
38
+ }
39
+
40
+
41
+ def fail(message: str) -> None:
42
+ raise ValueError(message)
43
+
44
+
45
+ def config_id(config: dict[str, Any]) -> str:
46
+ if isinstance(config.get("id"), str) and config["id"]:
47
+ return config["id"]
48
+ if config.get("strategy", "direct") in {"flow", "runtime"}:
49
+ fail(f"{config['strategy']} strategies require a non-empty id")
50
+ return f"{config['model']}-{config['reasoning_effort']}"
51
+
52
+
53
+ def config_strategy(config: dict[str, Any]) -> str:
54
+ return config.get("strategy", "direct")
55
+
56
+
57
+ def validate_actor(actor: Any, label: str) -> None:
58
+ if not isinstance(actor, dict):
59
+ fail(f"{label} must be an object")
60
+ if not isinstance(actor.get("model"), str) or not actor["model"]:
61
+ fail(f"{label}.model must be a non-empty string")
62
+ effort = actor.get("reasoning_effort")
63
+ if isinstance(effort, str):
64
+ if effort not in VALID_EFFORTS:
65
+ fail(f"invalid {label}.reasoning_effort: {effort!r}")
66
+ elif isinstance(effort, dict):
67
+ if set(effort) != VALID_CLASSES or any(value not in VALID_EFFORTS for value in effort.values()):
68
+ fail(f"{label}.reasoning_effort must define valid routine/complex/critical efforts")
69
+ else:
70
+ fail(f"invalid {label}.reasoning_effort: {effort!r}")
71
+
72
+
73
+ def actor_effort(actor: dict[str, Any], task_class: str) -> str:
74
+ effort = actor["reasoning_effort"]
75
+ return effort[task_class] if isinstance(effort, dict) else effort
76
+
77
+
78
+ def config_models(config: dict[str, Any]) -> set[str]:
79
+ if config_strategy(config) == "direct":
80
+ return {config["model"]}
81
+ return {config["parent"]["model"], config["worker"]["model"]}
82
+
83
+
84
+ def load_manifest(path: Path) -> dict[str, Any]:
85
+ data = json.loads(path.read_text())
86
+ version = data.get("schema_version")
87
+ if version not in {1, 2}:
88
+ fail("manifest schema_version must be 1 or 2")
89
+ tasks = data.get("tasks")
90
+ matrix = data.get("matrix")
91
+ if not isinstance(tasks, list) or not tasks:
92
+ fail("manifest tasks must be a non-empty array")
93
+ if not isinstance(matrix, list) or not matrix:
94
+ fail("manifest matrix must be a non-empty array")
95
+ allow_floating = data.get("allow_floating_refs", False)
96
+ if not isinstance(allow_floating, bool):
97
+ fail("allow_floating_refs must be boolean")
98
+
99
+ task_ids: set[str] = set()
100
+ for task in tasks:
101
+ required = {"id", "class", "source", "base_ref", "prompt", "verify"}
102
+ missing = required - task.keys()
103
+ if missing:
104
+ fail(f"task missing fields: {sorted(missing)}")
105
+ if task["id"] in task_ids:
106
+ fail(f"duplicate task id: {task['id']}")
107
+ task_ids.add(task["id"])
108
+ if task["class"] not in VALID_CLASSES:
109
+ fail(f"invalid task class for {task['id']}: {task['class']}")
110
+ if not isinstance(task["verify"], list) or not task["verify"] or not all(isinstance(x, str) for x in task["verify"]):
111
+ fail(f"verify for {task['id']} must be a non-empty argv array")
112
+ for field in ("prompt", "source", "base_ref"):
113
+ if not isinstance(task[field], str) or not task[field].strip():
114
+ fail(f"{field} for {task['id']} must be non-empty")
115
+ if not allow_floating and not FULL_COMMIT_RE.fullmatch(task["base_ref"]):
116
+ fail(
117
+ f"task {task['id']} base_ref must be a full 40-character commit SHA; "
118
+ "set allow_floating_refs=true only for exploratory runs"
119
+ )
120
+
121
+ strategy_ids: set[str] = set()
122
+ for config in matrix:
123
+ if not isinstance(config, dict):
124
+ fail("matrix entries must be objects")
125
+ strategy = config_strategy(config)
126
+ if strategy not in VALID_STRATEGIES:
127
+ fail(f"invalid strategy: {strategy!r}")
128
+ identifier = config_id(config)
129
+ if identifier in strategy_ids:
130
+ fail(f"duplicate strategy id: {identifier}")
131
+ strategy_ids.add(identifier)
132
+ if strategy == "direct":
133
+ validate_actor(config, f"matrix[{identifier}]")
134
+ if not isinstance(config["reasoning_effort"], str):
135
+ fail(f"matrix[{identifier}] direct strategies require a fixed reasoning_effort")
136
+ else:
137
+ if version < 2:
138
+ fail(f"{strategy} strategies require manifest schema_version 2")
139
+ validate_actor(config.get("parent"), f"matrix[{identifier}].parent")
140
+ validate_actor(config.get("worker"), f"matrix[{identifier}].worker")
141
+ policy = config.get("reasoning_policy", "fixed")
142
+ if policy not in {"fixed", "adaptive"}:
143
+ fail(f"matrix[{identifier}].reasoning_policy must be fixed or adaptive")
144
+ parent_effort = config["parent"]["reasoning_effort"]
145
+ worker_effort = config["worker"]["reasoning_effort"]
146
+ if policy == "fixed" and not all(isinstance(value, str) for value in (parent_effort, worker_effort)):
147
+ fail(f"matrix[{identifier}] fixed {strategy} requires fixed actor reasoning efforts")
148
+ if policy == "adaptive" and not all(isinstance(value, dict) for value in (parent_effort, worker_effort)):
149
+ fail(f"matrix[{identifier}] adaptive {strategy} requires per-class actor reasoning efforts")
150
+
151
+ repetitions = data.get("repetitions", 1)
152
+ timeout = data.get("timeout_seconds", 1800)
153
+ repairs = data.get("max_repair_cycles", 2)
154
+ if not isinstance(repetitions, int) or repetitions < 1:
155
+ fail("repetitions must be >= 1")
156
+ if not isinstance(timeout, int) or timeout < 1:
157
+ fail("timeout_seconds must be >= 1")
158
+ if not isinstance(repairs, int) or repairs < 0:
159
+ fail("max_repair_cycles must be >= 0")
160
+ return data
161
+
162
+
163
+ def clone_source(source: str, base_ref: str, destination: Path) -> str:
164
+ subprocess.run(
165
+ ["git", "clone", "--quiet", "--no-checkout", source, str(destination)],
166
+ check=True,
167
+ stdout=subprocess.PIPE,
168
+ stderr=subprocess.PIPE,
169
+ text=True,
170
+ )
171
+ subprocess.run(
172
+ ["git", "-C", str(destination), "checkout", "--quiet", "--detach", base_ref],
173
+ check=True,
174
+ stdout=subprocess.PIPE,
175
+ stderr=subprocess.PIPE,
176
+ text=True,
177
+ )
178
+ return subprocess.check_output(
179
+ ["git", "-C", str(destination), "rev-parse", "HEAD"], text=True
180
+ ).strip()
181
+
182
+
183
+ def parse_usage(jsonl: str) -> dict[str, int]:
184
+ usage = {"input_tokens": 0, "cached_input_tokens": 0, "output_tokens": 0}
185
+ for line in jsonl.splitlines():
186
+ try:
187
+ event = json.loads(line)
188
+ except json.JSONDecodeError:
189
+ continue
190
+ if event.get("type") != "turn.completed" or not isinstance(event.get("usage"), dict):
191
+ continue
192
+ for key in usage:
193
+ value = event["usage"].get(key, 0)
194
+ if isinstance(value, int) and value >= 0:
195
+ usage[key] += value
196
+ return usage
197
+
198
+
199
+ def run_codex(
200
+ workdir: Path,
201
+ model: str,
202
+ effort: str,
203
+ prompt: str,
204
+ timeout: int,
205
+ *,
206
+ sandbox: str = "workspace-write",
207
+ last_message_path: Path | None = None,
208
+ output_schema_path: Path | None = None,
209
+ ignore_user_config: bool = True,
210
+ env: dict[str, str] | None = None,
211
+ ) -> tuple[int, dict[str, int], str, float, str]:
212
+ cmd = [
213
+ "codex", "exec", "--ephemeral", "--json",
214
+ ]
215
+ if ignore_user_config:
216
+ cmd.append("--ignore-user-config")
217
+ cmd.extend([
218
+ "--sandbox", sandbox, "--model", model,
219
+ "-c", f'model_reasoning_effort="{effort}"',
220
+ ])
221
+ if output_schema_path is not None:
222
+ cmd.extend(["--output-schema", str(output_schema_path)])
223
+ if last_message_path is not None:
224
+ cmd.extend(["--output-last-message", str(last_message_path)])
225
+ cmd.extend(["--cd", str(workdir), prompt])
226
+ started = time.monotonic()
227
+ run_env = os.environ.copy()
228
+ if env:
229
+ run_env.update(env)
230
+ try:
231
+ proc = subprocess.run(cmd, text=True, stdout=subprocess.PIPE, stderr=subprocess.PIPE, timeout=timeout, env=run_env)
232
+ elapsed = time.monotonic() - started
233
+ last_message = ""
234
+ if last_message_path is not None and last_message_path.exists():
235
+ last_message = last_message_path.read_text()[-MAX_CAPTURE:]
236
+ return proc.returncode, parse_usage(proc.stdout), (proc.stderr or proc.stdout)[-MAX_CAPTURE:], elapsed, last_message
237
+ except subprocess.TimeoutExpired as exc:
238
+ elapsed = time.monotonic() - started
239
+ stdout = exc.stdout if isinstance(exc.stdout, str) else ""
240
+ stderr = exc.stderr if isinstance(exc.stderr, str) else ""
241
+ return 124, parse_usage(stdout), (stderr or stdout)[-MAX_CAPTURE:], elapsed, ""
242
+
243
+
244
+ def run_verify(workdir: Path, argv: list[str], timeout: int) -> tuple[bool, str, float]:
245
+ started = time.monotonic()
246
+ try:
247
+ proc = subprocess.run(argv, cwd=workdir, text=True, stdout=subprocess.PIPE, stderr=subprocess.STDOUT, timeout=timeout)
248
+ return proc.returncode == 0, proc.stdout[-MAX_CAPTURE:], time.monotonic() - started
249
+ except subprocess.TimeoutExpired as exc:
250
+ output = exc.stdout if isinstance(exc.stdout, str) else ""
251
+ return False, (output + "\nverification timed out")[-MAX_CAPTURE:], time.monotonic() - started
252
+
253
+
254
+ def empty_usage() -> dict[str, int]:
255
+ return {"input_tokens": 0, "cached_input_tokens": 0, "output_tokens": 0}
256
+
257
+
258
+ def add_usage(total: dict[str, int], current: dict[str, int]) -> None:
259
+ for key in empty_usage():
260
+ total[key] += current[key]
261
+
262
+
263
+ def add_model_usage(
264
+ totals: dict[tuple[str, str, str], dict[str, int]],
265
+ role: str,
266
+ model: str,
267
+ effort: str,
268
+ usage: dict[str, int],
269
+ ) -> None:
270
+ key = (role, model, effort)
271
+ if key not in totals:
272
+ totals[key] = {**empty_usage(), "calls": 0}
273
+ add_usage(totals[key], usage)
274
+ totals[key]["calls"] += 1
275
+
276
+
277
+ def flatten_model_usage(totals: dict[tuple[str, str, str], dict[str, int]]) -> list[dict[str, Any]]:
278
+ return [
279
+ {"role": role, "model": model, "reasoning_effort": effort, **usage}
280
+ for (role, model, effort), usage in sorted(totals.items())
281
+ ]
282
+
283
+
284
+ def build_row(
285
+ task: dict[str, Any],
286
+ config: dict[str, Any],
287
+ repetition: int,
288
+ commit: str,
289
+ passed: bool,
290
+ first_passed: bool,
291
+ repair_cycles: int,
292
+ review_cycles: int,
293
+ total_wall: float,
294
+ codex_exit: int,
295
+ verification: str,
296
+ diagnostic: str,
297
+ usage_by_model: dict[tuple[str, str, str], dict[str, int]],
298
+ ) -> dict[str, Any]:
299
+ strategy = config_strategy(config)
300
+ if strategy == "direct":
301
+ model = config["model"]
302
+ effort = actor_effort(config, task["class"])
303
+ worker_model = None
304
+ worker_effort = None
305
+ reasoning_policy = "fixed"
306
+ else:
307
+ model = config["parent"]["model"]
308
+ effort = actor_effort(config["parent"], task["class"])
309
+ worker_model = config["worker"]["model"]
310
+ worker_effort = actor_effort(config["worker"], task["class"])
311
+ reasoning_policy = config.get("reasoning_policy", "fixed")
312
+ model_usage = flatten_model_usage(usage_by_model)
313
+ total_usage = empty_usage()
314
+ for item in model_usage:
315
+ add_usage(total_usage, item)
316
+ return {
317
+ "schema_version": 2,
318
+ "task_id": task["id"],
319
+ "task_class": task["class"],
320
+ "strategy_id": config_id(config),
321
+ "strategy": strategy,
322
+ "reasoning_policy": reasoning_policy,
323
+ "model": model,
324
+ "reasoning_effort": effort,
325
+ "worker_model": worker_model,
326
+ "worker_reasoning_effort": worker_effort,
327
+ "passed": passed,
328
+ "first_passed": first_passed,
329
+ **total_usage,
330
+ "model_usage": model_usage,
331
+ "repair_cycles": repair_cycles,
332
+ "review_cycles": review_cycles,
333
+ "wall_time_seconds": round(total_wall, 3),
334
+ "source_commit": commit,
335
+ "repetition": repetition,
336
+ "codex_exit_code": codex_exit,
337
+ "verification_excerpt": verification[-2000:] if not passed else "",
338
+ "diagnostic_excerpt": diagnostic[-2000:] if codex_exit != 0 else "",
339
+ }
340
+
341
+
342
+ def execute_direct(
343
+ task: dict[str, Any],
344
+ config: dict[str, Any],
345
+ root: Path,
346
+ repetition: int,
347
+ timeout: int,
348
+ max_repairs: int,
349
+ ) -> dict[str, Any]:
350
+ workdir = root / "repo"
351
+ commit = clone_source(task["source"], task["base_ref"], workdir)
352
+ model = config["model"]
353
+ effort = actor_effort(config, task["class"])
354
+ usage_by_model: dict[tuple[str, str, str], dict[str, int]] = {}
355
+ total_wall = 0.0
356
+
357
+ codex_exit, usage, diagnostic, elapsed, _ = run_codex(
358
+ workdir, model, effort, task["prompt"], timeout
359
+ )
360
+ total_wall += elapsed
361
+ add_model_usage(usage_by_model, "direct", model, effort, usage)
362
+ repair_cycles = 0
363
+
364
+ if codex_exit == 0:
365
+ passed, verification, verify_elapsed = run_verify(workdir, task["verify"], timeout)
366
+ total_wall += verify_elapsed
367
+ else:
368
+ passed = False
369
+ verification = "benchmark verification skipped because codex exec exited non-zero"
370
+ first_passed = passed
371
+
372
+ while not passed and codex_exit == 0 and repair_cycles < max_repairs:
373
+ repair_cycles += 1
374
+ repair_prompt = (
375
+ "The previous implementation did not pass the fixed benchmark verifier. "
376
+ "Repair only the implementation needed to satisfy the original task; do not weaken, skip, or edit the verifier.\n\n"
377
+ f"Original task:\n{task['prompt']}\n\nVerifier output:\n{verification}"
378
+ )
379
+ codex_exit, usage, diagnostic, elapsed, _ = run_codex(
380
+ workdir, model, effort, repair_prompt, timeout
381
+ )
382
+ total_wall += elapsed
383
+ add_model_usage(usage_by_model, "direct", model, effort, usage)
384
+ if codex_exit != 0:
385
+ passed = False
386
+ verification = "benchmark verification skipped because repair codex exec exited non-zero"
387
+ break
388
+ passed, verification, verify_elapsed = run_verify(workdir, task["verify"], timeout)
389
+ total_wall += verify_elapsed
390
+
391
+ return build_row(
392
+ task, config, repetition, commit, passed, first_passed, repair_cycles, 0,
393
+ total_wall, codex_exit, verification, diagnostic, usage_by_model,
394
+ )
395
+
396
+
397
+ def parse_review(message: str) -> tuple[str, str]:
398
+ try:
399
+ review = json.loads(message)
400
+ except json.JSONDecodeError as exc:
401
+ fail(f"parent review did not return valid JSON: {exc}")
402
+ verdict = review.get("verdict")
403
+ feedback = review.get("feedback")
404
+ if verdict not in {"pass", "repair"} or not isinstance(feedback, str):
405
+ fail("parent review must contain verdict=pass|repair and string feedback")
406
+ return verdict, feedback
407
+
408
+
409
+ def execute_flow(
410
+ task: dict[str, Any],
411
+ config: dict[str, Any],
412
+ root: Path,
413
+ repetition: int,
414
+ timeout: int,
415
+ max_repairs: int,
416
+ ) -> dict[str, Any]:
417
+ workdir = root / "repo"
418
+ commit = clone_source(task["source"], task["base_ref"], workdir)
419
+ parent = config["parent"]
420
+ worker = config["worker"]
421
+ parent_model = parent["model"]
422
+ worker_model = worker["model"]
423
+ parent_effort = actor_effort(parent, task["class"])
424
+ worker_effort = actor_effort(worker, task["class"])
425
+ usage_by_model: dict[tuple[str, str, str], dict[str, int]] = {}
426
+ total_wall = 0.0
427
+ diagnostic = ""
428
+
429
+ plan_path = root / "parent-plan.txt"
430
+ plan_prompt = (
431
+ "You are the high-capability parent in the codex-flow workflow. Inspect the repository read-only. "
432
+ "Do not modify files. Produce a compact implementation handoff containing root cause/design decision, "
433
+ "scope and non-goals, relevant files, ordered steps, compatibility constraints, risks, acceptance criteria, "
434
+ "and required validation. Remove ambiguity so a worker can implement without redesigning.\n\n"
435
+ f"Task:\n{task['prompt']}"
436
+ )
437
+ codex_exit, usage, diagnostic, elapsed, plan = run_codex(
438
+ workdir, parent_model, parent_effort, plan_prompt, timeout,
439
+ sandbox="read-only", last_message_path=plan_path,
440
+ )
441
+ total_wall += elapsed
442
+ add_model_usage(usage_by_model, "parent", parent_model, parent_effort, usage)
443
+ if codex_exit != 0 or not plan.strip():
444
+ if codex_exit == 0:
445
+ codex_exit = 65
446
+ diagnostic = "parent planning completed without a handoff"
447
+ return build_row(
448
+ task, config, repetition, commit, False, False, 0, 0, total_wall,
449
+ codex_exit, "benchmark verification skipped because parent planning failed",
450
+ diagnostic, usage_by_model,
451
+ )
452
+
453
+ worker_prompt = (
454
+ "You are the implementation worker in the codex-flow workflow. Execute the parent's plan without redesigning it. "
455
+ "Stay in scope, implement the complete change, run narrow relevant validation, and do not weaken or edit any external verifier.\n\n"
456
+ f"Original task:\n{task['prompt']}\n\nParent handoff:\n{plan}"
457
+ )
458
+ codex_exit, usage, diagnostic, elapsed, _ = run_codex(
459
+ workdir, worker_model, worker_effort, worker_prompt, timeout
460
+ )
461
+ total_wall += elapsed
462
+ add_model_usage(usage_by_model, "worker", worker_model, worker_effort, usage)
463
+ if codex_exit != 0:
464
+ return build_row(
465
+ task, config, repetition, commit, False, False, 0, 0, total_wall,
466
+ codex_exit, "benchmark verification skipped because worker codex exec exited non-zero",
467
+ diagnostic, usage_by_model,
468
+ )
469
+
470
+ review_schema_path = root / "review-schema.json"
471
+ review_schema_path.write_text(json.dumps(REVIEW_SCHEMA))
472
+ first_passed = False
473
+ passed = False
474
+ repair_cycles = 0
475
+ review_cycles = 0
476
+ verification = ""
477
+
478
+ while True:
479
+ verifier_passed, verification, verify_elapsed = run_verify(workdir, task["verify"], timeout)
480
+ total_wall += verify_elapsed
481
+ review_cycles += 1
482
+ review_path = root / f"parent-review-{review_cycles}.json"
483
+ review_prompt = (
484
+ "You are the high-capability parent performing the codex-flow final review. Work read-only. "
485
+ "Inspect the current git diff and directly affected call sites. Check the original task, architecture, "
486
+ "compatibility, regression risk, and verifier result. Return verdict 'pass' only when the implementation "
487
+ "is complete and the verifier passed. Otherwise return verdict 'repair' with a compact delta instruction "
488
+ "for the worker; do not rewrite the whole plan.\n\n"
489
+ f"Original task:\n{task['prompt']}\n\nVerifier passed: {str(verifier_passed).lower()}\n"
490
+ f"Verifier output:\n{verification}"
491
+ )
492
+ codex_exit, usage, diagnostic, elapsed, review_message = run_codex(
493
+ workdir, parent_model, parent_effort, review_prompt, timeout,
494
+ sandbox="read-only", last_message_path=review_path,
495
+ output_schema_path=review_schema_path,
496
+ )
497
+ total_wall += elapsed
498
+ add_model_usage(usage_by_model, "parent", parent_model, parent_effort, usage)
499
+ if codex_exit != 0:
500
+ passed = False
501
+ break
502
+ try:
503
+ verdict, feedback = parse_review(review_message)
504
+ except ValueError as exc:
505
+ codex_exit = 65
506
+ diagnostic = str(exc)
507
+ passed = False
508
+ break
509
+ passed = verifier_passed and verdict == "pass"
510
+ if review_cycles == 1:
511
+ first_passed = passed
512
+ if passed or repair_cycles >= max_repairs:
513
+ if not passed and feedback:
514
+ verification = (
515
+ f"{verification}\nParent review requested repair:\n{feedback}"
516
+ ).strip()
517
+ break
518
+
519
+ repair_cycles += 1
520
+ if not verifier_passed:
521
+ feedback = (
522
+ f"{feedback}\n\nThe fixed external verifier still fails:\n{verification}"
523
+ ).strip()
524
+ repair_prompt = (
525
+ "You are the implementation worker repairing a reviewed codex-flow change. Apply only the requested delta, "
526
+ "keep the original acceptance criteria intact, and run narrow validation. Do not weaken or edit the verifier.\n\n"
527
+ f"Original task:\n{task['prompt']}\n\nParent repair instruction:\n{feedback}"
528
+ )
529
+ codex_exit, usage, diagnostic, elapsed, _ = run_codex(
530
+ workdir, worker_model, worker_effort, repair_prompt, timeout
531
+ )
532
+ total_wall += elapsed
533
+ add_model_usage(usage_by_model, "worker", worker_model, worker_effort, usage)
534
+ if codex_exit != 0:
535
+ passed = False
536
+ verification = "benchmark verification skipped because repair worker codex exec exited non-zero"
537
+ break
538
+
539
+ return build_row(
540
+ task, config, repetition, commit, passed, first_passed, repair_cycles,
541
+ review_cycles, total_wall, codex_exit, verification, diagnostic, usage_by_model,
542
+ )
543
+
544
+
545
+ def _load_manage_hooks() -> Any:
546
+ path = Path(__file__).resolve().parent / "manage-hooks.py"
547
+ spec = importlib.util.spec_from_file_location("manage_hooks", path)
548
+ if spec is None or spec.loader is None:
549
+ return None
550
+ module = importlib.util.module_from_spec(spec)
551
+ spec.loader.exec_module(module)
552
+ return module
553
+
554
+
555
+ def authorize_hooks(codex_home: Path) -> None:
556
+ hooks_path = codex_home / "hooks.json"
557
+ config_path = codex_home / "config.toml"
558
+ if not hooks_path.exists():
559
+ return
560
+ manage_hooks = _load_manage_hooks()
561
+ if manage_hooks is None:
562
+ return
563
+ entries = manage_hooks._managed_hook_entries(hooks_path)
564
+ if not entries:
565
+ return
566
+ lines = []
567
+ if config_path.exists():
568
+ lines.append(config_path.read_text(encoding="utf-8"))
569
+ lines.append("\n# Pre-authorized FlowPilot benchmark hooks")
570
+ for entry in entries:
571
+ key_repr = json.dumps(entry["key"])
572
+ lines.append(f"[hooks.state.{key_repr}]")
573
+ lines.append(f'trusted_hash = "{entry["current_hash"]}"')
574
+ lines.append("enabled = true\n")
575
+ config_path.write_text("\n".join(lines), encoding="utf-8")
576
+
577
+
578
+ def execute_runtime(
579
+ task: dict[str, Any],
580
+ config: dict[str, Any],
581
+ root: Path,
582
+ repetition: int,
583
+ timeout: int,
584
+ max_repairs: int,
585
+ ) -> dict[str, Any]:
586
+ workdir = root / "repo"
587
+ commit = clone_source(task["source"], task["base_ref"], workdir)
588
+ parent = config["parent"]
589
+ worker = config["worker"]
590
+ parent_model = parent["model"]
591
+ worker_model = worker["model"]
592
+ parent_effort = actor_effort(parent, task["class"])
593
+ worker_effort = actor_effort(worker, task["class"])
594
+ usage_by_model: dict[tuple[str, str, str], dict[str, int]] = {}
595
+ total_wall = 0.0
596
+ diagnostic = ""
597
+
598
+ # Provision isolated FlowPilot environment
599
+ codex_home = root / ".codex"
600
+ bin_dir = root / "bin"
601
+ codex_home.mkdir(parents=True, exist_ok=True)
602
+ bin_dir.mkdir(parents=True, exist_ok=True)
603
+
604
+ install_sh = Path(__file__).resolve().parent.parent / "install.sh"
605
+ install_env = os.environ.copy()
606
+ install_env.update({
607
+ "CODEX_HOME": str(codex_home),
608
+ "CODEX_FLOW_BIN_DIR": str(bin_dir),
609
+ "CODEX_FLOW_SHELL": "none",
610
+ "CODEX_FLOW_STRATEGY": config.get("profile", "balanced"),
611
+ "CODEX_FLOW_ROUTING_MODE": config.get("routing_mode", "delegate"),
612
+ "CODEX_FLOW_WORKER_MODEL": worker_model,
613
+ "CODEX_FLOW_WORKER_ROUTINE_EFFORT": actor_effort(worker, "routine"),
614
+ "CODEX_FLOW_WORKER_COMPLEX_EFFORT": actor_effort(worker, "complex"),
615
+ "CODEX_FLOW_WORKER_CRITICAL_EFFORT": actor_effort(worker, "critical"),
616
+ "CODEX_FLOW_PARENT_MIN_EFFORT": actor_effort(parent, "routine"),
617
+ "CODEX_FLOW_PARENT_ROUTINE_EFFORT": actor_effort(parent, "routine"),
618
+ "CODEX_FLOW_PARENT_COMPLEX_EFFORT": actor_effort(parent, "complex"),
619
+ "CODEX_FLOW_PARENT_CRITICAL_EFFORT": actor_effort(parent, "critical"),
620
+ "CODEX_FLOW_TELEMETRY_ENABLED": "true",
621
+ "CODEX_FLOW_TELEMETRY_NOTIFICATIONS": "false",
622
+ })
623
+ try:
624
+ subprocess.run(
625
+ [str(install_sh)],
626
+ env=install_env,
627
+ check=True,
628
+ stdout=subprocess.PIPE,
629
+ stderr=subprocess.PIPE,
630
+ text=True,
631
+ )
632
+ except subprocess.CalledProcessError as exc:
633
+ return build_row(
634
+ task, config, repetition, commit, False, False, 0, 0, 0.0,
635
+ exc.returncode, "FlowPilot runtime installation failed",
636
+ exc.stderr[-MAX_CAPTURE:], usage_by_model,
637
+ )
638
+
639
+ # Pre-authorize hooks for headless non-interactive execution
640
+ authorize_hooks(codex_home)
641
+
642
+ runtime_env = {
643
+ "CODEX_HOME": str(codex_home),
644
+ "PATH": f"{bin_dir}:{os.environ.get('PATH', '')}",
645
+ }
646
+
647
+ runtime_prompt = (
648
+ "You are operating in an environment with the FlowPilot multi-agent runtime installed. "
649
+ "Follow the FlowPilot task entry instructions, use the flow-pilot skill and subagents "
650
+ "to inspect, plan, implement, and verify the task.\n\n"
651
+ f"Task:\n{task['prompt']}"
652
+ )
653
+
654
+ codex_exit, usage, diagnostic, elapsed, _ = run_codex(
655
+ workdir, parent_model, parent_effort, runtime_prompt, timeout,
656
+ ignore_user_config=False, env=runtime_env,
657
+ )
658
+ total_wall += elapsed
659
+ cumulative_codex_usage = empty_usage()
660
+ add_usage(cumulative_codex_usage, usage)
661
+
662
+ repair_cycles = 0
663
+ if codex_exit == 0:
664
+ passed, verification, verify_elapsed = run_verify(workdir, task["verify"], timeout)
665
+ total_wall += verify_elapsed
666
+ else:
667
+ passed = False
668
+ verification = "benchmark verification skipped because codex exec exited non-zero"
669
+ first_passed = passed
670
+
671
+ while not passed and codex_exit == 0 and repair_cycles < max_repairs:
672
+ repair_cycles += 1
673
+ repair_prompt = (
674
+ "The previous implementation did not pass the fixed benchmark verifier. "
675
+ "Use FlowPilot to repair only the implementation needed to satisfy the original task; do not weaken, skip, or edit the verifier.\n\n"
676
+ f"Original task:\n{task['prompt']}\n\nVerifier output:\n{verification}"
677
+ )
678
+ codex_exit, usage, diagnostic, elapsed, _ = run_codex(
679
+ workdir, parent_model, parent_effort, repair_prompt, timeout,
680
+ ignore_user_config=False, env=runtime_env,
681
+ )
682
+ total_wall += elapsed
683
+ add_usage(cumulative_codex_usage, usage)
684
+ if codex_exit != 0:
685
+ passed = False
686
+ verification = "benchmark verification skipped because repair codex exec exited non-zero"
687
+ break
688
+ passed, verification, verify_elapsed = run_verify(workdir, task["verify"], timeout)
689
+ total_wall += verify_elapsed
690
+
691
+ # Attribute tokens from FlowPilot telemetry runs if present
692
+ runs_dir = codex_home / "codex-flow" / "telemetry" / "runs"
693
+ review_cycles = 0
694
+ if runs_dir.exists():
695
+ for run_file in sorted(runs_dir.glob("*.json")):
696
+ try:
697
+ run_data = json.loads(run_file.read_text(encoding="utf-8"))
698
+ except (OSError, json.JSONDecodeError):
699
+ continue
700
+ p = run_data.get("parent")
701
+ if isinstance(p, dict):
702
+ p_u = p.get("usage_delta") or p.get("usage")
703
+ if isinstance(p_u, dict) and any(p_u.get(k, 0) > 0 for k in ("input_tokens", "cached_input_tokens", "output_tokens")):
704
+ add_model_usage(usage_by_model, "parent", parent_model, parent_effort, p_u)
705
+ workers = run_data.get("workers")
706
+ if isinstance(workers, dict):
707
+ for w_name, w in workers.items():
708
+ if "review" in w_name.lower():
709
+ review_cycles += 1
710
+ if isinstance(w, dict) and isinstance(w.get("usage"), dict):
711
+ w_u = w["usage"]
712
+ if any(w_u.get(k, 0) > 0 for k in ("input_tokens", "cached_input_tokens", "output_tokens")):
713
+ add_model_usage(usage_by_model, "worker", worker_model, worker_effort, w_u)
714
+
715
+ # Fallback to top-level codex exec usage attributed to parent if telemetry was not populated
716
+ if not usage_by_model:
717
+ add_model_usage(usage_by_model, "parent", parent_model, parent_effort, cumulative_codex_usage)
718
+
719
+ return build_row(
720
+ task, config, repetition, commit, passed, first_passed, repair_cycles,
721
+ review_cycles, total_wall, codex_exit, verification, diagnostic, usage_by_model,
722
+ )
723
+
724
+
725
+ def execute_run(
726
+ task: dict[str, Any],
727
+ config: dict[str, Any],
728
+ root: Path,
729
+ repetition: int,
730
+ timeout: int,
731
+ max_repairs: int,
732
+ ) -> dict[str, Any]:
733
+ strategy = config_strategy(config)
734
+ if strategy == "flow":
735
+ return execute_flow(task, config, root, repetition, timeout, max_repairs)
736
+ if strategy == "runtime":
737
+ return execute_runtime(task, config, root, repetition, timeout, max_repairs)
738
+ return execute_direct(task, config, root, repetition, timeout, max_repairs)
739
+
740
+
741
+ def no_usage(row: dict[str, Any]) -> bool:
742
+ return (
743
+ row["input_tokens"] == 0
744
+ and row["cached_input_tokens"] == 0
745
+ and row["output_tokens"] == 0
746
+ )
747
+
748
+
749
+ def main() -> int:
750
+ parser = argparse.ArgumentParser()
751
+ parser.add_argument("--manifest", required=True)
752
+ parser.add_argument("--output")
753
+ parser.add_argument("--only-task")
754
+ parser.add_argument("--only-model")
755
+ parser.add_argument("--only-strategy")
756
+ parser.add_argument("--dry-run", action="store_true")
757
+ parser.add_argument(
758
+ "--fail-fast-infrastructure",
759
+ action="store_true",
760
+ help="stop the batch when codex exits non-zero before reporting any token usage",
761
+ )
762
+ args = parser.parse_args()
763
+
764
+ if not shutil.which("git"):
765
+ fail("git is required")
766
+ if not args.dry_run and not shutil.which("codex"):
767
+ fail("codex CLI is required")
768
+
769
+ manifest = load_manifest(Path(args.manifest))
770
+ tasks = [t for t in manifest["tasks"] if not args.only_task or t["id"] == args.only_task]
771
+ matrix = [
772
+ m for m in manifest["matrix"]
773
+ if (not args.only_model or args.only_model in config_models(m))
774
+ and (not args.only_strategy or args.only_strategy == config_id(m))
775
+ ]
776
+ if not tasks:
777
+ fail("task filter matched nothing")
778
+ if not matrix:
779
+ fail("strategy/model filter matched nothing")
780
+
781
+ planned = len(tasks) * len(matrix) * manifest.get("repetitions", 1)
782
+ if args.dry_run:
783
+ print(json.dumps({
784
+ "planned_runs": planned,
785
+ "tasks": [t["id"] for t in tasks],
786
+ "strategies": [config_id(m) for m in matrix],
787
+ "matrix": matrix,
788
+ }, sort_keys=True))
789
+ return 0
790
+
791
+ if not args.output:
792
+ fail("--output is required unless --dry-run is used")
793
+
794
+ output = Path(args.output)
795
+ output.parent.mkdir(parents=True, exist_ok=True)
796
+ with output.open("a", encoding="utf-8") as sink:
797
+ for task in tasks:
798
+ for config in matrix:
799
+ for repetition in range(1, manifest.get("repetitions", 1) + 1):
800
+ with tempfile.TemporaryDirectory(prefix="codex-flow-bench-") as tmp:
801
+ row = execute_run(
802
+ task, config, Path(tmp), repetition,
803
+ manifest.get("timeout_seconds", 1800),
804
+ task.get("max_repair_cycles", manifest.get("max_repair_cycles", 2)),
805
+ )
806
+ sink.write(json.dumps(row, sort_keys=True) + "\n")
807
+ sink.flush()
808
+ print(
809
+ f"{row['task_id']} {row['strategy_id']} "
810
+ f"rep={repetition} pass={row['passed']} first={row['first_passed']} "
811
+ f"repairs={row['repair_cycles']} reviews={row['review_cycles']} "
812
+ f"tokens={row['input_tokens'] + row['output_tokens']}",
813
+ file=sys.stderr,
814
+ )
815
+ if args.fail_fast_infrastructure and row["codex_exit_code"] != 0 and no_usage(row):
816
+ fail(
817
+ "codex exited non-zero without reporting usage; stopping benchmark "
818
+ f"after {row['task_id']} {row['strategy_id']} "
819
+ f"(exit={row['codex_exit_code']})"
820
+ )
821
+ return 0
822
+
823
+
824
+ if __name__ == "__main__":
825
+ try:
826
+ raise SystemExit(main())
827
+ except Exception as exc:
828
+ print(f"benchmark runner failed: {exc}", file=sys.stderr)
829
+ raise SystemExit(1)