codex-flow 2.1.13__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. codex_flow/__init__.py +28 -0
  2. codex_flow/__main__.py +9 -0
  3. codex_flow/cli.py +242 -0
  4. codex_flow/data/LICENSE +21 -0
  5. codex_flow/data/README.en.md +303 -0
  6. codex_flow/data/README.md +305 -0
  7. codex_flow/data/VERSION +1 -0
  8. codex_flow/data/apps/chatgpt-mcp/README.md +86 -0
  9. codex_flow/data/apps/chatgpt-mcp/__init__.py +1 -0
  10. codex_flow/data/apps/chatgpt-mcp/adapter.py +458 -0
  11. codex_flow/data/apps/chatgpt-mcp/server.py +358 -0
  12. codex_flow/data/apps/chatgpt-mcp/widget.html +927 -0
  13. codex_flow/data/apps/macos-overlay/README.en.md +121 -0
  14. codex_flow/data/apps/macos-overlay/README.md +123 -0
  15. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayRuntimeState.swift +126 -0
  16. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayScreenGeometry.swift +82 -0
  17. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayWindowController.swift +1052 -0
  18. codex_flow/data/apps/macos-overlay/Sources/Localization.swift +197 -0
  19. codex_flow/data/apps/macos-overlay/Sources/Models/TelemetryData.swift +1557 -0
  20. codex_flow/data/apps/macos-overlay/Sources/Services/AccountSnapshotService.swift +1101 -0
  21. codex_flow/data/apps/macos-overlay/Sources/Services/FlowPilotInstanceLock.swift +153 -0
  22. codex_flow/data/apps/macos-overlay/Sources/Services/IPCServer.swift +298 -0
  23. codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryQueryEngine.swift +800 -0
  24. codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryWatcher.swift +135 -0
  25. codex_flow/data/apps/macos-overlay/Sources/Services/UpdateService.swift +610 -0
  26. codex_flow/data/apps/macos-overlay/Sources/Views/AccountView.swift +610 -0
  27. codex_flow/data/apps/macos-overlay/Sources/Views/AnalyticsView.swift +566 -0
  28. codex_flow/data/apps/macos-overlay/Sources/Views/AutostartView.swift +293 -0
  29. codex_flow/data/apps/macos-overlay/Sources/Views/BubbleView.swift +317 -0
  30. codex_flow/data/apps/macos-overlay/Sources/Views/HistoryView.swift +1124 -0
  31. codex_flow/data/apps/macos-overlay/Sources/Views/HoverRevealText.swift +165 -0
  32. codex_flow/data/apps/macos-overlay/Sources/Views/InspectorSkillsToolsView.swift +121 -0
  33. codex_flow/data/apps/macos-overlay/Sources/Views/LogoView.swift +182 -0
  34. codex_flow/data/apps/macos-overlay/Sources/Views/SleekSwitch.swift +117 -0
  35. codex_flow/data/apps/macos-overlay/Sources/Views/StrategyModeView.swift +561 -0
  36. codex_flow/data/apps/macos-overlay/Sources/Views/SummaryView.swift +1273 -0
  37. codex_flow/data/apps/macos-overlay/Sources/Views/UpdateView.swift +352 -0
  38. codex_flow/data/apps/macos-overlay/Sources/main.swift +340 -0
  39. codex_flow/data/apps/macos-overlay/Tests/OverlayScreenGeometryTests.swift +163 -0
  40. codex_flow/data/apps/macos-overlay/Tests/TelemetryPhase1ContractTests.swift +357 -0
  41. codex_flow/data/apps/macos-overlay/Tests/TelemetryQueryEngineConcurrencyTests.swift +221 -0
  42. codex_flow/data/apps/macos-overlay/Tests/TelemetryQuotaSelectionTests.swift +158 -0
  43. codex_flow/data/apps/macos-overlay/Tests/TelemetryWorkerTokenTests.swift +122 -0
  44. codex_flow/data/apps/macos-overlay/build.sh +75 -0
  45. codex_flow/data/benchmark/corpus.json +103 -0
  46. codex_flow/data/benchmark/manifest.example.json +41 -0
  47. codex_flow/data/benchmark/manifest.schema.json +137 -0
  48. codex_flow/data/benchmark/prices/gpt-5.6-2026-08-30.json +5 -0
  49. codex_flow/data/benchmark/profiles.json +90 -0
  50. codex_flow/data/benchmark/schema.json +77 -0
  51. codex_flow/data/benchmark/tasks.json +50 -0
  52. codex_flow/data/completions/codex-flow.bash +34 -0
  53. codex_flow/data/completions/codex-flow.zsh +52 -0
  54. codex_flow/data/glama.json +6 -0
  55. codex_flow/data/install-release.ps1 +126 -0
  56. codex_flow/data/install-release.sh +155 -0
  57. codex_flow/data/install.ps1 +349 -0
  58. codex_flow/data/install.sh +362 -0
  59. codex_flow/data/policy/benchmark.toml +49 -0
  60. codex_flow/data/policy/defaults.toml +70 -0
  61. codex_flow/data/scripts/analyze-benchmark.py +510 -0
  62. codex_flow/data/scripts/benchmark-local.py +171 -0
  63. codex_flow/data/scripts/check-recommendation.py +277 -0
  64. codex_flow/data/scripts/doctor.py +449 -0
  65. codex_flow/data/scripts/generate-release-manifest.py +74 -0
  66. codex_flow/data/scripts/localization.py +192 -0
  67. codex_flow/data/scripts/manage-hooks.py +448 -0
  68. codex_flow/data/scripts/manage-instructions.py +389 -0
  69. codex_flow/data/scripts/manage-shell.py +151 -0
  70. codex_flow/data/scripts/materialize-corpus.py +193 -0
  71. codex_flow/data/scripts/menu.py +646 -0
  72. codex_flow/data/scripts/migrations/0001_update_settings.py +80 -0
  73. codex_flow/data/scripts/package-release.py +132 -0
  74. codex_flow/data/scripts/render-benchmark-report.py +292 -0
  75. codex_flow/data/scripts/run-benchmark.py +829 -0
  76. codex_flow/data/scripts/strategies/__init__.py +28 -0
  77. codex_flow/data/scripts/strategies/balanced.py +115 -0
  78. codex_flow/data/scripts/strategies/base.py +363 -0
  79. codex_flow/data/scripts/strategies/efficient.py +158 -0
  80. codex_flow/data/scripts/strategies/lifecycle_runtime.py +590 -0
  81. codex_flow/data/scripts/strategies/quality.py +209 -0
  82. codex_flow/data/scripts/strategies/speed.py +108 -0
  83. codex_flow/data/scripts/strategies/task_budget_runtime.py +644 -0
  84. codex_flow/data/scripts/strategies/task_phase_runtime.py +341 -0
  85. codex_flow/data/scripts/strategies/work_unit_runtime.py +421 -0
  86. codex_flow/data/scripts/strategy_runtime.py +1091 -0
  87. codex_flow/data/scripts/telemetry.py +400 -0
  88. codex_flow/data/scripts/telemetry_core/__init__.py +192 -0
  89. codex_flow/data/scripts/telemetry_core/app_server.py +1192 -0
  90. codex_flow/data/scripts/telemetry_core/collector.py +1247 -0
  91. codex_flow/data/scripts/telemetry_core/common.py +421 -0
  92. codex_flow/data/scripts/telemetry_core/latency.py +593 -0
  93. codex_flow/data/scripts/telemetry_core/query.py +427 -0
  94. codex_flow/data/scripts/telemetry_core/quota_ledger.py +598 -0
  95. codex_flow/data/scripts/telemetry_core/render.py +460 -0
  96. codex_flow/data/scripts/telemetry_core/repair.py +223 -0
  97. codex_flow/data/scripts/ui.py +266 -0
  98. codex_flow/data/scripts/update-homebrew-formula.py +146 -0
  99. codex_flow/data/scripts/update_runtime_config.py +134 -0
  100. codex_flow/data/scripts/updater.py +1718 -0
  101. codex_flow/data/smithery.yaml +18 -0
  102. codex_flow/data/templates/agents/worker-explorer.toml +24 -0
  103. codex_flow/data/templates/agents/worker-implementer.toml +49 -0
  104. codex_flow/data/templates/agents/worker-reviewer.toml +25 -0
  105. codex_flow/data/templates/flow-pilot-instructions.md +35 -0
  106. codex_flow/data/templates/skills/flow-pilot/SKILL.md +577 -0
  107. codex_flow/mcp.py +35 -0
  108. codex_flow-2.1.13.dist-info/METADATA +342 -0
  109. codex_flow-2.1.13.dist-info/RECORD +113 -0
  110. codex_flow-2.1.13.dist-info/WHEEL +5 -0
  111. codex_flow-2.1.13.dist-info/entry_points.txt +3 -0
  112. codex_flow-2.1.13.dist-info/licenses/LICENSE +21 -0
  113. codex_flow-2.1.13.dist-info/top_level.txt +1 -0
@@ -0,0 +1,510 @@
1
+ #!/usr/bin/env python3
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import json
6
+ import sys
7
+ from collections import defaultdict
8
+ from pathlib import Path
9
+ from typing import Any
10
+
11
+ def _parse_toml_value(raw: str) -> Any:
12
+ raw = raw.strip()
13
+ if raw.lower() == "true":
14
+ return True
15
+ if raw.lower() == "false":
16
+ return False
17
+ if (raw.startswith('"') and raw.endswith('"')) or (raw.startswith("'") and raw.endswith("'")):
18
+ return raw[1:-1]
19
+ try:
20
+ if "." in raw or "e" in raw.lower():
21
+ return float(raw)
22
+ return int(raw)
23
+ except ValueError:
24
+ return raw
25
+
26
+
27
+ def _fallback_load_toml(text: str) -> dict[str, Any]:
28
+ result: dict[str, Any] = {}
29
+ current_section: dict[str, Any] = result
30
+ for line in text.splitlines():
31
+ line = line.split("#", 1)[0].strip()
32
+ if not line:
33
+ continue
34
+ if line.startswith("[") and line.endswith("]"):
35
+ section_name = line[1:-1].strip()
36
+ current_section = result.setdefault(section_name, {})
37
+ continue
38
+ if "=" in line:
39
+ key, val = line.split("=", 1)
40
+ key = key.strip()
41
+ val = _parse_toml_value(val)
42
+ current_section[key] = val
43
+ return result
44
+
45
+
46
+ def load_policy(path: Path) -> dict[str, Any]:
47
+ text = path.read_text(encoding="utf-8")
48
+ try:
49
+ import tomllib
50
+ return tomllib.loads(text)
51
+ except ImportError:
52
+ try:
53
+ import tomli
54
+ return tomli.loads(text)
55
+ except ImportError:
56
+ return _fallback_load_toml(text)
57
+
58
+ EFFORT_RANK = {"high": 0, "xhigh": 1, "max": 2}
59
+ TASK_CLASSES = ("routine", "complex", "critical")
60
+ NONNEGATIVE_INTS = {"input_tokens", "cached_input_tokens", "output_tokens", "repair_cycles"}
61
+
62
+
63
+ def legacy_strategy_id(row: dict[str, Any]) -> str:
64
+ suffix = row["model"].rsplit("-", 1)[-1]
65
+ if suffix in {"luna", "terra", "sol"}:
66
+ base = f"{suffix}-direct"
67
+ return base if row["reasoning_effort"] == "high" else f"{base}-{row['reasoning_effort']}"
68
+ return f"direct:{row['model']}:{row['reasoning_effort']}"
69
+
70
+
71
+ def validate_usage(usage: dict[str, Any], label: str) -> None:
72
+ for field in ("input_tokens", "cached_input_tokens", "output_tokens"):
73
+ value = usage.get(field)
74
+ if type(value) is not int or value < 0:
75
+ raise ValueError(f"{label}: {field} must be a non-negative integer")
76
+ if usage["cached_input_tokens"] > usage["input_tokens"]:
77
+ raise ValueError(f"{label}: cached_input_tokens exceeds input_tokens")
78
+
79
+
80
+ def load_jsonl(path: Path) -> list[dict[str, Any]]:
81
+ rows: list[dict[str, Any]] = []
82
+ for n, line in enumerate(path.read_text().splitlines(), 1):
83
+ if not line.strip():
84
+ continue
85
+ row = json.loads(line)
86
+ required = {
87
+ "schema_version", "task_id", "task_class", "model", "reasoning_effort",
88
+ "passed", "input_tokens", "cached_input_tokens", "output_tokens", "repair_cycles",
89
+ }
90
+ missing = required - row.keys()
91
+ if missing:
92
+ raise ValueError(f"{path}:{n}: missing {sorted(missing)}")
93
+ if row["schema_version"] not in {1, 2}:
94
+ raise ValueError(f"{path}:{n}: unsupported schema_version {row['schema_version']!r}")
95
+ if not isinstance(row["task_id"], str) or not row["task_id"].strip():
96
+ raise ValueError(f"{path}:{n}: invalid task_id")
97
+ if row["task_class"] not in TASK_CLASSES:
98
+ raise ValueError(f"{path}:{n}: invalid task_class {row['task_class']!r}")
99
+ if not isinstance(row["model"], str) or not row["model"].strip():
100
+ raise ValueError(f"{path}:{n}: invalid model")
101
+ if row["reasoning_effort"] not in EFFORT_RANK:
102
+ raise ValueError(f"{path}:{n}: invalid reasoning_effort {row['reasoning_effort']!r}")
103
+ if type(row["passed"]) is not bool:
104
+ raise ValueError(f"{path}:{n}: passed must be boolean")
105
+ for field in NONNEGATIVE_INTS:
106
+ value = row[field]
107
+ if type(value) is not int or value < 0:
108
+ raise ValueError(f"{path}:{n}: {field} must be a non-negative integer")
109
+ validate_usage(row, f"{path}:{n}")
110
+
111
+ if row["schema_version"] == 1:
112
+ row["strategy_id"] = legacy_strategy_id(row)
113
+ row["strategy"] = "direct"
114
+ row["reasoning_policy"] = "fixed"
115
+ row["worker_model"] = None
116
+ row["worker_reasoning_effort"] = None
117
+ row["first_passed"] = row["passed"] and row["repair_cycles"] == 0
118
+ row["review_cycles"] = 0
119
+ row["model_usage"] = [{
120
+ "role": "direct",
121
+ "model": row["model"],
122
+ "reasoning_effort": row["reasoning_effort"],
123
+ "calls": row["repair_cycles"] + 1,
124
+ "input_tokens": row["input_tokens"],
125
+ "cached_input_tokens": row["cached_input_tokens"],
126
+ "output_tokens": row["output_tokens"],
127
+ }]
128
+ else:
129
+ for field in ("strategy_id", "strategy", "reasoning_policy", "first_passed", "review_cycles", "model_usage"):
130
+ if field not in row:
131
+ raise ValueError(f"{path}:{n}: missing {field}")
132
+ if not isinstance(row["strategy_id"], str) or not row["strategy_id"]:
133
+ raise ValueError(f"{path}:{n}: invalid strategy_id")
134
+ if row["strategy"] not in {"direct", "flow", "runtime"}:
135
+ raise ValueError(f"{path}:{n}: invalid strategy")
136
+ if row["reasoning_policy"] not in {"fixed", "adaptive"}:
137
+ raise ValueError(f"{path}:{n}: invalid reasoning_policy")
138
+ if type(row["first_passed"]) is not bool:
139
+ raise ValueError(f"{path}:{n}: first_passed must be boolean")
140
+ if type(row["review_cycles"]) is not int or row["review_cycles"] < 0:
141
+ raise ValueError(f"{path}:{n}: review_cycles must be a non-negative integer")
142
+ if not isinstance(row["model_usage"], list) or not row["model_usage"]:
143
+ raise ValueError(f"{path}:{n}: model_usage must be a non-empty array")
144
+ summed = {"input_tokens": 0, "cached_input_tokens": 0, "output_tokens": 0}
145
+ for index, usage in enumerate(row["model_usage"]):
146
+ if not isinstance(usage, dict):
147
+ raise ValueError(f"{path}:{n}: model_usage[{index}] must be an object")
148
+ if usage.get("role") not in {"direct", "parent", "worker"}:
149
+ raise ValueError(f"{path}:{n}: invalid model_usage role")
150
+ if not isinstance(usage.get("model"), str) or not usage["model"]:
151
+ raise ValueError(f"{path}:{n}: invalid model_usage model")
152
+ if usage.get("reasoning_effort") not in EFFORT_RANK:
153
+ raise ValueError(f"{path}:{n}: invalid model_usage reasoning effort")
154
+ if type(usage.get("calls")) is not int or usage["calls"] < 1:
155
+ raise ValueError(f"{path}:{n}: model_usage calls must be >= 1")
156
+ validate_usage(usage, f"{path}:{n}:model_usage[{index}]")
157
+ for key in summed:
158
+ summed[key] += usage[key]
159
+ if any(row[key] != value for key, value in summed.items()):
160
+ raise ValueError(f"{path}:{n}: top-level usage does not equal model_usage sum")
161
+ if row["strategy"] == "direct":
162
+ if row["reasoning_policy"] != "fixed" or row.get("worker_model") is not None or row.get("worker_reasoning_effort") is not None:
163
+ raise ValueError(f"{path}:{n}: direct strategy has invalid policy or worker metadata")
164
+ if any(
165
+ usage["role"] != "direct"
166
+ or usage["model"] != row["model"]
167
+ or usage["reasoning_effort"] != row["reasoning_effort"]
168
+ for usage in row["model_usage"]
169
+ ):
170
+ raise ValueError(f"{path}:{n}: direct model_usage does not match strategy metadata")
171
+ else:
172
+ if not isinstance(row.get("worker_model"), str) or not row["worker_model"]:
173
+ raise ValueError(f"{path}:{n}: {row['strategy']} strategy requires worker_model")
174
+ if row.get("worker_reasoning_effort") not in EFFORT_RANK:
175
+ raise ValueError(f"{path}:{n}: {row['strategy']} strategy requires worker_reasoning_effort")
176
+ for usage in row["model_usage"]:
177
+ if usage["role"] == "direct":
178
+ raise ValueError(f"{path}:{n}: {row['strategy']} model_usage cannot use direct role")
179
+ expected_model = row["model"] if usage["role"] == "parent" else row["worker_model"]
180
+ expected_effort = row["reasoning_effort"] if usage["role"] == "parent" else row["worker_reasoning_effort"]
181
+ if usage["model"] != expected_model or usage["reasoning_effort"] != expected_effort:
182
+ raise ValueError(f"{path}:{n}: flow model_usage does not match actor metadata")
183
+ rows.append(row)
184
+ if not rows:
185
+ raise ValueError(f"{path}: no benchmark rows")
186
+ return rows
187
+
188
+
189
+ def validate_prices(prices: dict[str, Any]) -> None:
190
+ for model, price in prices.items():
191
+ for key in ("input", "cached_input", "output"):
192
+ if key not in price or not isinstance(price[key], (int, float)) or price[key] < 0:
193
+ raise ValueError(f"invalid {key} price for {model}")
194
+
195
+
196
+ def usage_cost(usage: dict[str, Any], prices: dict[str, Any]) -> float:
197
+ model = usage["model"]
198
+ if model not in prices:
199
+ raise ValueError(f"missing price snapshot for {model}")
200
+ price = prices[model]
201
+ uncached = usage["input_tokens"] - usage["cached_input_tokens"]
202
+ return (
203
+ uncached * price["input"]
204
+ + usage["cached_input_tokens"] * price["cached_input"]
205
+ + usage["output_tokens"] * price["output"]
206
+ ) / 1_000_000
207
+
208
+
209
+ def run_cost(row: dict[str, Any], prices: dict[str, Any]) -> float:
210
+ return sum(usage_cost(usage, prices) for usage in row["model_usage"])
211
+
212
+
213
+ def summarize(items: list[dict[str, Any]], prices: dict[str, Any], threshold: float, min_samples: int, max_repairs: float) -> dict[str, Any]:
214
+ first = items[0]
215
+ samples = len(items)
216
+ identity_fields = (
217
+ "task_class", "strategy_id", "strategy", "reasoning_policy", "model",
218
+ "reasoning_effort", "worker_model", "worker_reasoning_effort",
219
+ )
220
+ identities = {tuple(item.get(field) for field in identity_fields) for item in items}
221
+ if len(identities) != 1:
222
+ raise ValueError(f"strategy {first['strategy_id']} has inconsistent model, policy, or effort metadata")
223
+ sample_keys = {(item["task_id"], item.get("repetition", 1)) for item in items}
224
+ if len(sample_keys) != samples:
225
+ raise ValueError(f"strategy {first['strategy_id']} contains duplicate task/repetition samples")
226
+ pass_rate = sum(1 for item in items if item["passed"]) / samples
227
+ first_pass_rate = sum(1 for item in items if item["first_passed"]) / samples
228
+ avg_repairs = sum(item["repair_cycles"] for item in items) / samples
229
+ avg_reviews = sum(item["review_cycles"] for item in items) / samples
230
+ avg_cost = sum(run_cost(item, prices) for item in items) / samples
231
+ avg_wall = sum(item.get("wall_time_seconds", 0.0) for item in items) / samples
232
+ quality_ok = samples >= min_samples and pass_rate >= threshold and avg_repairs <= max_repairs
233
+ return {
234
+ "task_class": first["task_class"],
235
+ "strategy_id": first["strategy_id"],
236
+ "strategy": first["strategy"],
237
+ "reasoning_policy": first["reasoning_policy"],
238
+ "model": first["model"],
239
+ "reasoning_effort": first["reasoning_effort"],
240
+ "worker_model": first.get("worker_model"),
241
+ "worker_reasoning_effort": first.get("worker_reasoning_effort"),
242
+ "samples": samples,
243
+ "pass_rate": round(pass_rate, 4),
244
+ "first_pass_rate": round(first_pass_rate, 4),
245
+ "average_repair_cycles": round(avg_repairs, 4),
246
+ "average_review_cycles": round(avg_reviews, 4),
247
+ "average_cost_usd": round(avg_cost, 6),
248
+ "average_wall_time_seconds": round(avg_wall, 3),
249
+ "quality_gate": quality_ok,
250
+ "_sample_keys": sample_keys,
251
+ }
252
+
253
+
254
+ def delta(value: float, reference: float) -> float:
255
+ return round(value - reference, 4)
256
+
257
+
258
+ def capability_comparison(task_class: str, by_id: dict[str, dict[str, Any]], comparison: dict[str, Any]) -> dict[str, Any] | None:
259
+ sol_id = comparison["sol_strategy_id"]
260
+ sol = by_id.get(sol_id)
261
+ competitors = [
262
+ item for item in by_id.values()
263
+ if item["strategy"] == "direct"
264
+ and item["strategy_id"] != sol_id
265
+ and item["reasoning_policy"] == "fixed"
266
+ and sol is not None
267
+ and item["reasoning_effort"] == sol["reasoning_effort"]
268
+ ]
269
+ if sol is None or not competitors:
270
+ return None
271
+ competitor = max(
272
+ competitors,
273
+ key=lambda item: (item["pass_rate"], item["first_pass_rate"], -item["average_repair_cycles"]),
274
+ )
275
+ paired = sol["_sample_keys"] == competitor["_sample_keys"]
276
+ controlled = sol["strategy"] == "direct" and sol["reasoning_policy"] == "fixed"
277
+ enough = (
278
+ sol["samples"] >= comparison["min_samples"]
279
+ and competitor["samples"] >= comparison["min_samples"]
280
+ and paired
281
+ and controlled
282
+ )
283
+ pass_gain = delta(sol["pass_rate"], competitor["pass_rate"])
284
+ first_gain = delta(sol["first_pass_rate"], competitor["first_pass_rate"])
285
+ repair_reduction = delta(competitor["average_repair_cycles"], sol["average_repair_cycles"])
286
+ not_worse = sol["pass_rate"] >= competitor["pass_rate"]
287
+ meaningful_gain = (
288
+ pass_gain >= comparison["sol_min_pass_rate_gain"]
289
+ or (pass_gain >= 0 and first_gain >= comparison["sol_min_first_pass_rate_gain"])
290
+ or (pass_gain >= 0 and first_gain >= 0 and repair_reduction >= comparison["sol_min_repair_reduction"])
291
+ )
292
+ return {
293
+ "task_class": task_class,
294
+ "sol_strategy_id": sol_id,
295
+ "competitor_strategy_id": competitor["strategy_id"],
296
+ "samples": min(sol["samples"], competitor["samples"]),
297
+ "evidence_sufficient": enough,
298
+ "paired_samples": paired,
299
+ "controlled_reasoning_effort": sol["reasoning_effort"],
300
+ "pass_rate_gain": pass_gain,
301
+ "first_pass_rate_gain": first_gain,
302
+ "average_repair_reduction": repair_reduction,
303
+ "advantage_demonstrated": enough and not_worse and meaningful_gain,
304
+ }
305
+
306
+
307
+ def flow_comparison(task_class: str, by_id: dict[str, dict[str, Any]], comparison: dict[str, Any]) -> dict[str, Any] | None:
308
+ flow = by_id.get(comparison["flow_strategy_id"])
309
+ sol = by_id.get(comparison["sol_strategy_id"])
310
+ worker = by_id.get(comparison["worker_strategy_id"])
311
+ if flow is None or sol is None or worker is None:
312
+ return None
313
+ paired = flow["_sample_keys"] == sol["_sample_keys"] == worker["_sample_keys"]
314
+ controlled_effort = sol["reasoning_effort"]
315
+ controlled = (
316
+ flow["strategy"] == "flow"
317
+ and flow["reasoning_policy"] == "fixed"
318
+ and sol["strategy"] == "direct"
319
+ and sol["reasoning_policy"] == "fixed"
320
+ and worker["strategy"] == "direct"
321
+ and worker["reasoning_policy"] == "fixed"
322
+ and flow["reasoning_effort"] == controlled_effort
323
+ and flow["worker_reasoning_effort"] == controlled_effort
324
+ and worker["reasoning_effort"] == controlled_effort
325
+ )
326
+ enough = (
327
+ min(flow["samples"], sol["samples"], worker["samples"]) >= comparison["min_samples"]
328
+ and paired
329
+ and controlled
330
+ )
331
+ quality_delta = delta(flow["pass_rate"], sol["pass_rate"])
332
+ cost_reduction = round(1 - flow["average_cost_usd"] / sol["average_cost_usd"], 4) if sol["average_cost_usd"] else 0.0
333
+ worker_pass_gain = delta(flow["pass_rate"], worker["pass_rate"])
334
+ worker_first_gain = delta(flow["first_pass_rate"], worker["first_pass_rate"])
335
+ worker_repair_reduction = delta(worker["average_repair_cycles"], flow["average_repair_cycles"])
336
+ quality_noninferior = quality_delta >= -comparison["flow_max_pass_rate_regression_vs_sol"]
337
+ cost_ok = cost_reduction >= comparison["flow_min_cost_reduction_vs_sol"]
338
+ worker_gain = (
339
+ worker_pass_gain >= comparison["flow_min_pass_rate_gain_vs_worker"]
340
+ or (worker_pass_gain >= 0 and worker_first_gain >= comparison["flow_min_first_pass_rate_gain_vs_worker"])
341
+ or (worker_pass_gain >= 0 and worker_first_gain >= 0 and worker_repair_reduction >= comparison["flow_min_repair_reduction_vs_worker"])
342
+ )
343
+ return {
344
+ "task_class": task_class,
345
+ "flow_strategy_id": flow["strategy_id"],
346
+ "quality_reference_strategy_id": sol["strategy_id"],
347
+ "worker_reference_strategy_id": worker["strategy_id"],
348
+ "samples": min(flow["samples"], sol["samples"], worker["samples"]),
349
+ "evidence_sufficient": enough,
350
+ "paired_samples": paired,
351
+ "controlled_reasoning_effort": controlled_effort if controlled else None,
352
+ "pass_rate_delta_vs_sol": quality_delta,
353
+ "cost_reduction_vs_sol": cost_reduction,
354
+ "pass_rate_gain_vs_worker": worker_pass_gain,
355
+ "first_pass_rate_gain_vs_worker": worker_first_gain,
356
+ "average_repair_reduction_vs_worker": worker_repair_reduction,
357
+ "quality_noninferior_to_sol": quality_noninferior,
358
+ "worker_quality_gain": worker_gain,
359
+ "advantage_demonstrated": enough and quality_noninferior and cost_ok and worker_gain,
360
+ }
361
+
362
+
363
+ def adaptive_comparison(task_class: str, by_id: dict[str, dict[str, Any]], comparison: dict[str, Any]) -> dict[str, Any] | None:
364
+ fixed = by_id.get(comparison["flow_strategy_id"])
365
+ adaptive = by_id.get(comparison["adaptive_flow_strategy_id"])
366
+ if fixed is None or adaptive is None:
367
+ return None
368
+ paired = fixed["_sample_keys"] == adaptive["_sample_keys"]
369
+ policies_match = (
370
+ fixed["strategy"] == "flow"
371
+ and fixed["reasoning_policy"] == "fixed"
372
+ and adaptive["strategy"] == "flow"
373
+ and adaptive["reasoning_policy"] == "adaptive"
374
+ )
375
+ enough = (
376
+ min(fixed["samples"], adaptive["samples"]) >= comparison["min_samples"]
377
+ and paired
378
+ and policies_match
379
+ )
380
+ pass_gain = delta(adaptive["pass_rate"], fixed["pass_rate"])
381
+ first_gain = delta(adaptive["first_pass_rate"], fixed["first_pass_rate"])
382
+ repair_reduction = delta(fixed["average_repair_cycles"], adaptive["average_repair_cycles"])
383
+ cost_change = round(adaptive["average_cost_usd"] / fixed["average_cost_usd"] - 1, 4) if fixed["average_cost_usd"] else 0.0
384
+ wall_change = round(adaptive["average_wall_time_seconds"] / fixed["average_wall_time_seconds"] - 1, 4) if fixed["average_wall_time_seconds"] else 0.0
385
+ quality_gain = (
386
+ pass_gain >= comparison["adaptive_min_pass_rate_gain"]
387
+ or (pass_gain >= 0 and first_gain >= comparison["adaptive_min_first_pass_rate_gain"])
388
+ or (pass_gain >= 0 and first_gain >= 0 and repair_reduction >= comparison["adaptive_min_repair_reduction"])
389
+ )
390
+ return {
391
+ "task_class": task_class,
392
+ "fixed_strategy_id": fixed["strategy_id"],
393
+ "adaptive_strategy_id": adaptive["strategy_id"],
394
+ "samples": min(fixed["samples"], adaptive["samples"]),
395
+ "evidence_sufficient": enough,
396
+ "paired_samples": paired,
397
+ "reasoning_policies_match": policies_match,
398
+ "pass_rate_gain": pass_gain,
399
+ "first_pass_rate_gain": first_gain,
400
+ "average_repair_reduction": repair_reduction,
401
+ "cost_change": cost_change,
402
+ "wall_time_change": wall_change,
403
+ "quality_gain": quality_gain,
404
+ "value_demonstrated": enough and quality_gain and cost_change <= comparison["adaptive_max_cost_increase"],
405
+ }
406
+
407
+
408
+ def main() -> int:
409
+ ap = argparse.ArgumentParser()
410
+ ap.add_argument("--results", required=True)
411
+ ap.add_argument("--prices", required=True)
412
+ ap.add_argument("--policy", default="policy/benchmark.toml")
413
+ ap.add_argument("--min-samples", type=int, default=None, help="override minimum required samples")
414
+ ap.add_argument("--json", action="store_true")
415
+ args = ap.parse_args()
416
+
417
+ policy = load_policy(Path(args.policy))
418
+ if policy.get("schema_version") != 2:
419
+ raise ValueError("benchmark policy schema_version must be 2")
420
+ quality = policy["quality"]
421
+ comparison = policy["comparison"]
422
+
423
+ min_samples_override = args.min_samples
424
+ if min_samples_override is not None and min_samples_override < 1:
425
+ raise ValueError("--min-samples must be >= 1")
426
+
427
+ quality_min_samples = min_samples_override if min_samples_override is not None else quality["min_samples_per_configuration"]
428
+ comparison_config = dict(comparison)
429
+ if min_samples_override is not None:
430
+ comparison_config["min_samples"] = min_samples_override
431
+
432
+ prices = json.loads(Path(args.prices).read_text())
433
+ validate_prices(prices)
434
+ rows = load_jsonl(Path(args.results))
435
+
436
+ grouped: dict[tuple[str, str], list[dict[str, Any]]] = defaultdict(list)
437
+ for row in rows:
438
+ grouped[(row["task_class"], row["strategy_id"])].append(row)
439
+
440
+ thresholds = {
441
+ "routine": quality["routine_min_pass_rate"],
442
+ "complex": quality["complex_min_pass_rate"],
443
+ "critical": quality["critical_min_pass_rate"],
444
+ }
445
+ summaries = [
446
+ summarize(items, prices, thresholds[task_class], quality_min_samples, quality["max_average_repair_cycles"])
447
+ for (task_class, _), items in sorted(grouped.items())
448
+ ]
449
+
450
+ recommendations: dict[str, Any] = {}
451
+ sol_evidence: dict[str, Any] = {}
452
+ flow_evidence: dict[str, Any] = {}
453
+ adaptive_evidence: dict[str, Any] = {}
454
+ for task_class in TASK_CLASSES:
455
+ class_items = [item for item in summaries if item["task_class"] == task_class]
456
+ eligible = [item for item in class_items if item["quality_gate"]]
457
+ if eligible:
458
+ best = min(
459
+ eligible,
460
+ key=lambda item: (item["average_cost_usd"], EFFORT_RANK[item["reasoning_effort"]], item["strategy_id"]),
461
+ )
462
+ recommendations[task_class] = {
463
+ "strategy_id": best["strategy_id"],
464
+ "strategy": best["strategy"],
465
+ "model": best["model"],
466
+ "reasoning_effort": best["reasoning_effort"],
467
+ "worker_model": best["worker_model"],
468
+ "average_cost_usd": best["average_cost_usd"],
469
+ "pass_rate": best["pass_rate"],
470
+ "samples": best["samples"],
471
+ }
472
+ else:
473
+ recommendations[task_class] = None
474
+ by_id = {item["strategy_id"]: item for item in class_items}
475
+ sol_evidence[task_class] = capability_comparison(task_class, by_id, comparison_config)
476
+ flow_evidence[task_class] = flow_comparison(task_class, by_id, comparison_config)
477
+ adaptive_evidence[task_class] = adaptive_comparison(task_class, by_id, comparison_config)
478
+
479
+ result = {
480
+ "recommendations": recommendations,
481
+ "configurations": [
482
+ {key: value for key, value in summary.items() if not key.startswith("_")}
483
+ for summary in summaries
484
+ ],
485
+ "sol_capability_evidence": sol_evidence,
486
+ "flow_advantage_evidence": flow_evidence,
487
+ "adaptive_reasoning_evidence": adaptive_evidence,
488
+ "advisory_only": True,
489
+ }
490
+ if args.json:
491
+ print(json.dumps(result, sort_keys=True))
492
+ else:
493
+ for task_class in TASK_CLASSES:
494
+ sol = sol_evidence[task_class]
495
+ flow = flow_evidence[task_class]
496
+ adaptive = adaptive_evidence[task_class]
497
+ print(
498
+ f"{task_class}: sol advantage={sol and sol['advantage_demonstrated']}; "
499
+ f"flow advantage={flow and flow['advantage_demonstrated']}; "
500
+ f"adaptive value={adaptive and adaptive['value_demonstrated']}"
501
+ )
502
+ return 0
503
+
504
+
505
+ if __name__ == "__main__":
506
+ try:
507
+ raise SystemExit(main())
508
+ except Exception as exc:
509
+ print(f"benchmark analysis failed: {exc}", file=sys.stderr)
510
+ raise SystemExit(1)
@@ -0,0 +1,171 @@
1
+ #!/usr/bin/env python3
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import json
6
+ import os
7
+ import shutil
8
+ import subprocess
9
+ import sys
10
+ from datetime import datetime, timezone
11
+ from pathlib import Path
12
+
13
+ ROOT = Path(__file__).resolve().parents[1]
14
+ DEFAULT_PRICE = ROOT / "benchmark/prices/gpt-5.6-2026-08-30.json"
15
+
16
+
17
+ def run(cmd: list[str], *, check: bool = True, capture: bool = False) -> subprocess.CompletedProcess[str]:
18
+ return subprocess.run(
19
+ cmd,
20
+ check=check,
21
+ text=True,
22
+ stdout=subprocess.PIPE if capture else None,
23
+ stderr=subprocess.PIPE if capture else None,
24
+ )
25
+
26
+
27
+ def require(name: str) -> str:
28
+ path = shutil.which(name)
29
+ if not path:
30
+ raise RuntimeError(f"{name} is required")
31
+ return path
32
+
33
+
34
+ def main() -> int:
35
+ ap = argparse.ArgumentParser(
36
+ description="Run the built-in codex-flow benchmark using the local Codex login session."
37
+ )
38
+ ap.add_argument("profile", nargs="?", choices=["quick", "full"], default="quick")
39
+ ap.add_argument("--workspace", default=".codex-flow-benchmark")
40
+ ap.add_argument("--output")
41
+ ap.add_argument("--prices", default=str(DEFAULT_PRICE))
42
+ ap.add_argument("--yes", action="store_true", help="skip the paid-run confirmation prompt")
43
+ args = ap.parse_args()
44
+
45
+ require("git")
46
+ codex = require("codex")
47
+ require("python3")
48
+
49
+ version = run([codex, "--version"], capture=True).stdout.strip()
50
+
51
+ workspace = Path(args.workspace).resolve()
52
+ manifest = workspace / "manifest.json"
53
+ results_dir = ROOT / "benchmark/results"
54
+ results_dir.mkdir(parents=True, exist_ok=True)
55
+ stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
56
+ output = Path(args.output).resolve() if args.output else (results_dir / f"{args.profile}-{stamp}.jsonl")
57
+
58
+ materialize = run([
59
+ sys.executable,
60
+ str(ROOT / "scripts/materialize-corpus.py"),
61
+ "--profile", args.profile,
62
+ "--output-dir", str(workspace / "corpus"),
63
+ "--manifest", str(manifest),
64
+ ], capture=True)
65
+ plan_meta = json.loads(materialize.stdout)
66
+
67
+ dry = run([
68
+ sys.executable,
69
+ str(ROOT / "scripts/run-benchmark.py"),
70
+ "--manifest", str(manifest),
71
+ "--output", str(output),
72
+ "--dry-run",
73
+ ], capture=True)
74
+ plan = json.loads(dry.stdout)
75
+ planned = int(plan["planned_runs"])
76
+
77
+ disp_output = str(output)
78
+ home_str = str(Path.home())
79
+ if disp_output.startswith(home_str):
80
+ disp_output = "~" + disp_output[len(home_str):]
81
+
82
+ budget = "~15M tokens (planning + repairs may increase)" if args.profile == "quick" else "high (90 runs, substantial Codex quota)"
83
+
84
+ def pad_line(content: str, width: int = 68) -> str:
85
+ pad = max(0, width - len(content))
86
+ return f" │ {content}{' ' * pad} │"
87
+
88
+ print(f"\n⚡ codex-flow benchmark-local\n")
89
+ print(" ╭─ Benchmark Plan ──────────────────────────────────────────────────╮")
90
+ print(pad_line(f"• Profile: {args.profile} ({planned} real model runs)"))
91
+ print(pad_line(f"• Codex CLI: {version} (local auth session)"))
92
+ print(pad_line(f"• Est. Budget: {budget}"))
93
+ print(pad_line(f"• Output File: {disp_output}"))
94
+ print(" ╰───────────────────────────────────────────────────────────────────╯\n")
95
+
96
+ print("┌─ ⚠️ REAL MODEL EXECUTION CONFIRMATION ────────────────────────────┐")
97
+ print("│ This will execute tasks using your authenticated Codex account │")
98
+ print("│ and will consume real quota/credits. │")
99
+ print("└───────────────────────────────────────────────────────────────────┘\n")
100
+
101
+ if not args.yes:
102
+ expected = f"RUN {args.profile.upper()} {planned}"
103
+ typed = input(f"Type '{expected}' to proceed: ").strip()
104
+ if typed != expected:
105
+ print("Cancelled; no model run was started.")
106
+ return 2
107
+
108
+ output.parent.mkdir(parents=True, exist_ok=True)
109
+ run([
110
+ sys.executable,
111
+ str(ROOT / "scripts/run-benchmark.py"),
112
+ "--manifest", str(manifest),
113
+ "--output", str(output),
114
+ "--fail-fast-infrastructure",
115
+ ])
116
+
117
+ analysis = output.with_suffix(".analysis.json")
118
+ report = output.with_suffix(".report.md")
119
+ with analysis.open("w", encoding="utf-8") as fh:
120
+ proc = subprocess.run([
121
+ sys.executable,
122
+ str(ROOT / "scripts/analyze-benchmark.py"),
123
+ "--results", str(output),
124
+ "--prices", str(Path(args.prices).resolve()),
125
+ "--json",
126
+ ], check=True, text=True, stdout=fh)
127
+
128
+ run([
129
+ sys.executable,
130
+ str(ROOT / "scripts/render-benchmark-report.py"),
131
+ "--results", str(output),
132
+ "--prices", str(Path(args.prices).resolve()),
133
+ "--analysis", str(analysis),
134
+ "--output", str(report),
135
+ "--title", f"Codex {args.profile} local benchmark",
136
+ ])
137
+
138
+ meta = output.with_suffix(".meta.json")
139
+ meta.write_text(json.dumps({
140
+ "schema_version": 2,
141
+ "profile": args.profile,
142
+ "planned_runs": planned,
143
+ "codex_cli_version": version,
144
+ "codex_flow_commit": subprocess.check_output(["git", "-C", str(ROOT), "rev-parse", "HEAD"], text=True).strip(),
145
+ "authentication_mode": "local-codex-session",
146
+ "cost_semantics": "API-equivalent reference cost only; not the user's ChatGPT subscription charge",
147
+ "manifest": str(manifest),
148
+ "results": str(output),
149
+ "analysis": str(analysis),
150
+ "report": str(report),
151
+ "materialization": plan_meta,
152
+ }, indent=2) + "\n")
153
+
154
+ print("\nBenchmark complete.")
155
+ print(f"Results: {output}")
156
+ print(f"Analysis: {analysis}")
157
+ print(f"Report: {report}")
158
+ print(f"Metadata: {meta}")
159
+ print("Dollar values in the report are API-equivalent reference costs, not actual ChatGPT subscription charges.")
160
+ return 0
161
+
162
+
163
+ if __name__ == "__main__":
164
+ try:
165
+ raise SystemExit(main())
166
+ except KeyboardInterrupt:
167
+ print("\nCancelled.", file=sys.stderr)
168
+ raise SystemExit(130)
169
+ except Exception as exc:
170
+ print(f"local benchmark failed: {exc}", file=sys.stderr)
171
+ raise SystemExit(1)