codex-flow 2.1.13__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codex_flow/__init__.py +28 -0
- codex_flow/__main__.py +9 -0
- codex_flow/cli.py +242 -0
- codex_flow/data/LICENSE +21 -0
- codex_flow/data/README.en.md +303 -0
- codex_flow/data/README.md +305 -0
- codex_flow/data/VERSION +1 -0
- codex_flow/data/apps/chatgpt-mcp/README.md +86 -0
- codex_flow/data/apps/chatgpt-mcp/__init__.py +1 -0
- codex_flow/data/apps/chatgpt-mcp/adapter.py +458 -0
- codex_flow/data/apps/chatgpt-mcp/server.py +358 -0
- codex_flow/data/apps/chatgpt-mcp/widget.html +927 -0
- codex_flow/data/apps/macos-overlay/README.en.md +121 -0
- codex_flow/data/apps/macos-overlay/README.md +123 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayRuntimeState.swift +126 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayScreenGeometry.swift +82 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayWindowController.swift +1052 -0
- codex_flow/data/apps/macos-overlay/Sources/Localization.swift +197 -0
- codex_flow/data/apps/macos-overlay/Sources/Models/TelemetryData.swift +1557 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/AccountSnapshotService.swift +1101 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/FlowPilotInstanceLock.swift +153 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/IPCServer.swift +298 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryQueryEngine.swift +800 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryWatcher.swift +135 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/UpdateService.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AccountView.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AnalyticsView.swift +566 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AutostartView.swift +293 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/BubbleView.swift +317 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HistoryView.swift +1124 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HoverRevealText.swift +165 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/InspectorSkillsToolsView.swift +121 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/LogoView.swift +182 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SleekSwitch.swift +117 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/StrategyModeView.swift +561 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SummaryView.swift +1273 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/UpdateView.swift +352 -0
- codex_flow/data/apps/macos-overlay/Sources/main.swift +340 -0
- codex_flow/data/apps/macos-overlay/Tests/OverlayScreenGeometryTests.swift +163 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryPhase1ContractTests.swift +357 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQueryEngineConcurrencyTests.swift +221 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQuotaSelectionTests.swift +158 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryWorkerTokenTests.swift +122 -0
- codex_flow/data/apps/macos-overlay/build.sh +75 -0
- codex_flow/data/benchmark/corpus.json +103 -0
- codex_flow/data/benchmark/manifest.example.json +41 -0
- codex_flow/data/benchmark/manifest.schema.json +137 -0
- codex_flow/data/benchmark/prices/gpt-5.6-2026-08-30.json +5 -0
- codex_flow/data/benchmark/profiles.json +90 -0
- codex_flow/data/benchmark/schema.json +77 -0
- codex_flow/data/benchmark/tasks.json +50 -0
- codex_flow/data/completions/codex-flow.bash +34 -0
- codex_flow/data/completions/codex-flow.zsh +52 -0
- codex_flow/data/glama.json +6 -0
- codex_flow/data/install-release.ps1 +126 -0
- codex_flow/data/install-release.sh +155 -0
- codex_flow/data/install.ps1 +349 -0
- codex_flow/data/install.sh +362 -0
- codex_flow/data/policy/benchmark.toml +49 -0
- codex_flow/data/policy/defaults.toml +70 -0
- codex_flow/data/scripts/analyze-benchmark.py +510 -0
- codex_flow/data/scripts/benchmark-local.py +171 -0
- codex_flow/data/scripts/check-recommendation.py +277 -0
- codex_flow/data/scripts/doctor.py +449 -0
- codex_flow/data/scripts/generate-release-manifest.py +74 -0
- codex_flow/data/scripts/localization.py +192 -0
- codex_flow/data/scripts/manage-hooks.py +448 -0
- codex_flow/data/scripts/manage-instructions.py +389 -0
- codex_flow/data/scripts/manage-shell.py +151 -0
- codex_flow/data/scripts/materialize-corpus.py +193 -0
- codex_flow/data/scripts/menu.py +646 -0
- codex_flow/data/scripts/migrations/0001_update_settings.py +80 -0
- codex_flow/data/scripts/package-release.py +132 -0
- codex_flow/data/scripts/render-benchmark-report.py +292 -0
- codex_flow/data/scripts/run-benchmark.py +829 -0
- codex_flow/data/scripts/strategies/__init__.py +28 -0
- codex_flow/data/scripts/strategies/balanced.py +115 -0
- codex_flow/data/scripts/strategies/base.py +363 -0
- codex_flow/data/scripts/strategies/efficient.py +158 -0
- codex_flow/data/scripts/strategies/lifecycle_runtime.py +590 -0
- codex_flow/data/scripts/strategies/quality.py +209 -0
- codex_flow/data/scripts/strategies/speed.py +108 -0
- codex_flow/data/scripts/strategies/task_budget_runtime.py +644 -0
- codex_flow/data/scripts/strategies/task_phase_runtime.py +341 -0
- codex_flow/data/scripts/strategies/work_unit_runtime.py +421 -0
- codex_flow/data/scripts/strategy_runtime.py +1091 -0
- codex_flow/data/scripts/telemetry.py +400 -0
- codex_flow/data/scripts/telemetry_core/__init__.py +192 -0
- codex_flow/data/scripts/telemetry_core/app_server.py +1192 -0
- codex_flow/data/scripts/telemetry_core/collector.py +1247 -0
- codex_flow/data/scripts/telemetry_core/common.py +421 -0
- codex_flow/data/scripts/telemetry_core/latency.py +593 -0
- codex_flow/data/scripts/telemetry_core/query.py +427 -0
- codex_flow/data/scripts/telemetry_core/quota_ledger.py +598 -0
- codex_flow/data/scripts/telemetry_core/render.py +460 -0
- codex_flow/data/scripts/telemetry_core/repair.py +223 -0
- codex_flow/data/scripts/ui.py +266 -0
- codex_flow/data/scripts/update-homebrew-formula.py +146 -0
- codex_flow/data/scripts/update_runtime_config.py +134 -0
- codex_flow/data/scripts/updater.py +1718 -0
- codex_flow/data/smithery.yaml +18 -0
- codex_flow/data/templates/agents/worker-explorer.toml +24 -0
- codex_flow/data/templates/agents/worker-implementer.toml +49 -0
- codex_flow/data/templates/agents/worker-reviewer.toml +25 -0
- codex_flow/data/templates/flow-pilot-instructions.md +35 -0
- codex_flow/data/templates/skills/flow-pilot/SKILL.md +577 -0
- codex_flow/mcp.py +35 -0
- codex_flow-2.1.13.dist-info/METADATA +342 -0
- codex_flow-2.1.13.dist-info/RECORD +113 -0
- codex_flow-2.1.13.dist-info/WHEEL +5 -0
- codex_flow-2.1.13.dist-info/entry_points.txt +3 -0
- codex_flow-2.1.13.dist-info/licenses/LICENSE +21 -0
- codex_flow-2.1.13.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,510 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
import sys
|
|
7
|
+
from collections import defaultdict
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import Any
|
|
10
|
+
|
|
11
|
+
def _parse_toml_value(raw: str) -> Any:
|
|
12
|
+
raw = raw.strip()
|
|
13
|
+
if raw.lower() == "true":
|
|
14
|
+
return True
|
|
15
|
+
if raw.lower() == "false":
|
|
16
|
+
return False
|
|
17
|
+
if (raw.startswith('"') and raw.endswith('"')) or (raw.startswith("'") and raw.endswith("'")):
|
|
18
|
+
return raw[1:-1]
|
|
19
|
+
try:
|
|
20
|
+
if "." in raw or "e" in raw.lower():
|
|
21
|
+
return float(raw)
|
|
22
|
+
return int(raw)
|
|
23
|
+
except ValueError:
|
|
24
|
+
return raw
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _fallback_load_toml(text: str) -> dict[str, Any]:
|
|
28
|
+
result: dict[str, Any] = {}
|
|
29
|
+
current_section: dict[str, Any] = result
|
|
30
|
+
for line in text.splitlines():
|
|
31
|
+
line = line.split("#", 1)[0].strip()
|
|
32
|
+
if not line:
|
|
33
|
+
continue
|
|
34
|
+
if line.startswith("[") and line.endswith("]"):
|
|
35
|
+
section_name = line[1:-1].strip()
|
|
36
|
+
current_section = result.setdefault(section_name, {})
|
|
37
|
+
continue
|
|
38
|
+
if "=" in line:
|
|
39
|
+
key, val = line.split("=", 1)
|
|
40
|
+
key = key.strip()
|
|
41
|
+
val = _parse_toml_value(val)
|
|
42
|
+
current_section[key] = val
|
|
43
|
+
return result
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def load_policy(path: Path) -> dict[str, Any]:
|
|
47
|
+
text = path.read_text(encoding="utf-8")
|
|
48
|
+
try:
|
|
49
|
+
import tomllib
|
|
50
|
+
return tomllib.loads(text)
|
|
51
|
+
except ImportError:
|
|
52
|
+
try:
|
|
53
|
+
import tomli
|
|
54
|
+
return tomli.loads(text)
|
|
55
|
+
except ImportError:
|
|
56
|
+
return _fallback_load_toml(text)
|
|
57
|
+
|
|
58
|
+
EFFORT_RANK = {"high": 0, "xhigh": 1, "max": 2}
|
|
59
|
+
TASK_CLASSES = ("routine", "complex", "critical")
|
|
60
|
+
NONNEGATIVE_INTS = {"input_tokens", "cached_input_tokens", "output_tokens", "repair_cycles"}
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def legacy_strategy_id(row: dict[str, Any]) -> str:
|
|
64
|
+
suffix = row["model"].rsplit("-", 1)[-1]
|
|
65
|
+
if suffix in {"luna", "terra", "sol"}:
|
|
66
|
+
base = f"{suffix}-direct"
|
|
67
|
+
return base if row["reasoning_effort"] == "high" else f"{base}-{row['reasoning_effort']}"
|
|
68
|
+
return f"direct:{row['model']}:{row['reasoning_effort']}"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def validate_usage(usage: dict[str, Any], label: str) -> None:
|
|
72
|
+
for field in ("input_tokens", "cached_input_tokens", "output_tokens"):
|
|
73
|
+
value = usage.get(field)
|
|
74
|
+
if type(value) is not int or value < 0:
|
|
75
|
+
raise ValueError(f"{label}: {field} must be a non-negative integer")
|
|
76
|
+
if usage["cached_input_tokens"] > usage["input_tokens"]:
|
|
77
|
+
raise ValueError(f"{label}: cached_input_tokens exceeds input_tokens")
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def load_jsonl(path: Path) -> list[dict[str, Any]]:
|
|
81
|
+
rows: list[dict[str, Any]] = []
|
|
82
|
+
for n, line in enumerate(path.read_text().splitlines(), 1):
|
|
83
|
+
if not line.strip():
|
|
84
|
+
continue
|
|
85
|
+
row = json.loads(line)
|
|
86
|
+
required = {
|
|
87
|
+
"schema_version", "task_id", "task_class", "model", "reasoning_effort",
|
|
88
|
+
"passed", "input_tokens", "cached_input_tokens", "output_tokens", "repair_cycles",
|
|
89
|
+
}
|
|
90
|
+
missing = required - row.keys()
|
|
91
|
+
if missing:
|
|
92
|
+
raise ValueError(f"{path}:{n}: missing {sorted(missing)}")
|
|
93
|
+
if row["schema_version"] not in {1, 2}:
|
|
94
|
+
raise ValueError(f"{path}:{n}: unsupported schema_version {row['schema_version']!r}")
|
|
95
|
+
if not isinstance(row["task_id"], str) or not row["task_id"].strip():
|
|
96
|
+
raise ValueError(f"{path}:{n}: invalid task_id")
|
|
97
|
+
if row["task_class"] not in TASK_CLASSES:
|
|
98
|
+
raise ValueError(f"{path}:{n}: invalid task_class {row['task_class']!r}")
|
|
99
|
+
if not isinstance(row["model"], str) or not row["model"].strip():
|
|
100
|
+
raise ValueError(f"{path}:{n}: invalid model")
|
|
101
|
+
if row["reasoning_effort"] not in EFFORT_RANK:
|
|
102
|
+
raise ValueError(f"{path}:{n}: invalid reasoning_effort {row['reasoning_effort']!r}")
|
|
103
|
+
if type(row["passed"]) is not bool:
|
|
104
|
+
raise ValueError(f"{path}:{n}: passed must be boolean")
|
|
105
|
+
for field in NONNEGATIVE_INTS:
|
|
106
|
+
value = row[field]
|
|
107
|
+
if type(value) is not int or value < 0:
|
|
108
|
+
raise ValueError(f"{path}:{n}: {field} must be a non-negative integer")
|
|
109
|
+
validate_usage(row, f"{path}:{n}")
|
|
110
|
+
|
|
111
|
+
if row["schema_version"] == 1:
|
|
112
|
+
row["strategy_id"] = legacy_strategy_id(row)
|
|
113
|
+
row["strategy"] = "direct"
|
|
114
|
+
row["reasoning_policy"] = "fixed"
|
|
115
|
+
row["worker_model"] = None
|
|
116
|
+
row["worker_reasoning_effort"] = None
|
|
117
|
+
row["first_passed"] = row["passed"] and row["repair_cycles"] == 0
|
|
118
|
+
row["review_cycles"] = 0
|
|
119
|
+
row["model_usage"] = [{
|
|
120
|
+
"role": "direct",
|
|
121
|
+
"model": row["model"],
|
|
122
|
+
"reasoning_effort": row["reasoning_effort"],
|
|
123
|
+
"calls": row["repair_cycles"] + 1,
|
|
124
|
+
"input_tokens": row["input_tokens"],
|
|
125
|
+
"cached_input_tokens": row["cached_input_tokens"],
|
|
126
|
+
"output_tokens": row["output_tokens"],
|
|
127
|
+
}]
|
|
128
|
+
else:
|
|
129
|
+
for field in ("strategy_id", "strategy", "reasoning_policy", "first_passed", "review_cycles", "model_usage"):
|
|
130
|
+
if field not in row:
|
|
131
|
+
raise ValueError(f"{path}:{n}: missing {field}")
|
|
132
|
+
if not isinstance(row["strategy_id"], str) or not row["strategy_id"]:
|
|
133
|
+
raise ValueError(f"{path}:{n}: invalid strategy_id")
|
|
134
|
+
if row["strategy"] not in {"direct", "flow", "runtime"}:
|
|
135
|
+
raise ValueError(f"{path}:{n}: invalid strategy")
|
|
136
|
+
if row["reasoning_policy"] not in {"fixed", "adaptive"}:
|
|
137
|
+
raise ValueError(f"{path}:{n}: invalid reasoning_policy")
|
|
138
|
+
if type(row["first_passed"]) is not bool:
|
|
139
|
+
raise ValueError(f"{path}:{n}: first_passed must be boolean")
|
|
140
|
+
if type(row["review_cycles"]) is not int or row["review_cycles"] < 0:
|
|
141
|
+
raise ValueError(f"{path}:{n}: review_cycles must be a non-negative integer")
|
|
142
|
+
if not isinstance(row["model_usage"], list) or not row["model_usage"]:
|
|
143
|
+
raise ValueError(f"{path}:{n}: model_usage must be a non-empty array")
|
|
144
|
+
summed = {"input_tokens": 0, "cached_input_tokens": 0, "output_tokens": 0}
|
|
145
|
+
for index, usage in enumerate(row["model_usage"]):
|
|
146
|
+
if not isinstance(usage, dict):
|
|
147
|
+
raise ValueError(f"{path}:{n}: model_usage[{index}] must be an object")
|
|
148
|
+
if usage.get("role") not in {"direct", "parent", "worker"}:
|
|
149
|
+
raise ValueError(f"{path}:{n}: invalid model_usage role")
|
|
150
|
+
if not isinstance(usage.get("model"), str) or not usage["model"]:
|
|
151
|
+
raise ValueError(f"{path}:{n}: invalid model_usage model")
|
|
152
|
+
if usage.get("reasoning_effort") not in EFFORT_RANK:
|
|
153
|
+
raise ValueError(f"{path}:{n}: invalid model_usage reasoning effort")
|
|
154
|
+
if type(usage.get("calls")) is not int or usage["calls"] < 1:
|
|
155
|
+
raise ValueError(f"{path}:{n}: model_usage calls must be >= 1")
|
|
156
|
+
validate_usage(usage, f"{path}:{n}:model_usage[{index}]")
|
|
157
|
+
for key in summed:
|
|
158
|
+
summed[key] += usage[key]
|
|
159
|
+
if any(row[key] != value for key, value in summed.items()):
|
|
160
|
+
raise ValueError(f"{path}:{n}: top-level usage does not equal model_usage sum")
|
|
161
|
+
if row["strategy"] == "direct":
|
|
162
|
+
if row["reasoning_policy"] != "fixed" or row.get("worker_model") is not None or row.get("worker_reasoning_effort") is not None:
|
|
163
|
+
raise ValueError(f"{path}:{n}: direct strategy has invalid policy or worker metadata")
|
|
164
|
+
if any(
|
|
165
|
+
usage["role"] != "direct"
|
|
166
|
+
or usage["model"] != row["model"]
|
|
167
|
+
or usage["reasoning_effort"] != row["reasoning_effort"]
|
|
168
|
+
for usage in row["model_usage"]
|
|
169
|
+
):
|
|
170
|
+
raise ValueError(f"{path}:{n}: direct model_usage does not match strategy metadata")
|
|
171
|
+
else:
|
|
172
|
+
if not isinstance(row.get("worker_model"), str) or not row["worker_model"]:
|
|
173
|
+
raise ValueError(f"{path}:{n}: {row['strategy']} strategy requires worker_model")
|
|
174
|
+
if row.get("worker_reasoning_effort") not in EFFORT_RANK:
|
|
175
|
+
raise ValueError(f"{path}:{n}: {row['strategy']} strategy requires worker_reasoning_effort")
|
|
176
|
+
for usage in row["model_usage"]:
|
|
177
|
+
if usage["role"] == "direct":
|
|
178
|
+
raise ValueError(f"{path}:{n}: {row['strategy']} model_usage cannot use direct role")
|
|
179
|
+
expected_model = row["model"] if usage["role"] == "parent" else row["worker_model"]
|
|
180
|
+
expected_effort = row["reasoning_effort"] if usage["role"] == "parent" else row["worker_reasoning_effort"]
|
|
181
|
+
if usage["model"] != expected_model or usage["reasoning_effort"] != expected_effort:
|
|
182
|
+
raise ValueError(f"{path}:{n}: flow model_usage does not match actor metadata")
|
|
183
|
+
rows.append(row)
|
|
184
|
+
if not rows:
|
|
185
|
+
raise ValueError(f"{path}: no benchmark rows")
|
|
186
|
+
return rows
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def validate_prices(prices: dict[str, Any]) -> None:
|
|
190
|
+
for model, price in prices.items():
|
|
191
|
+
for key in ("input", "cached_input", "output"):
|
|
192
|
+
if key not in price or not isinstance(price[key], (int, float)) or price[key] < 0:
|
|
193
|
+
raise ValueError(f"invalid {key} price for {model}")
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def usage_cost(usage: dict[str, Any], prices: dict[str, Any]) -> float:
|
|
197
|
+
model = usage["model"]
|
|
198
|
+
if model not in prices:
|
|
199
|
+
raise ValueError(f"missing price snapshot for {model}")
|
|
200
|
+
price = prices[model]
|
|
201
|
+
uncached = usage["input_tokens"] - usage["cached_input_tokens"]
|
|
202
|
+
return (
|
|
203
|
+
uncached * price["input"]
|
|
204
|
+
+ usage["cached_input_tokens"] * price["cached_input"]
|
|
205
|
+
+ usage["output_tokens"] * price["output"]
|
|
206
|
+
) / 1_000_000
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def run_cost(row: dict[str, Any], prices: dict[str, Any]) -> float:
|
|
210
|
+
return sum(usage_cost(usage, prices) for usage in row["model_usage"])
|
|
211
|
+
|
|
212
|
+
|
|
213
|
+
def summarize(items: list[dict[str, Any]], prices: dict[str, Any], threshold: float, min_samples: int, max_repairs: float) -> dict[str, Any]:
|
|
214
|
+
first = items[0]
|
|
215
|
+
samples = len(items)
|
|
216
|
+
identity_fields = (
|
|
217
|
+
"task_class", "strategy_id", "strategy", "reasoning_policy", "model",
|
|
218
|
+
"reasoning_effort", "worker_model", "worker_reasoning_effort",
|
|
219
|
+
)
|
|
220
|
+
identities = {tuple(item.get(field) for field in identity_fields) for item in items}
|
|
221
|
+
if len(identities) != 1:
|
|
222
|
+
raise ValueError(f"strategy {first['strategy_id']} has inconsistent model, policy, or effort metadata")
|
|
223
|
+
sample_keys = {(item["task_id"], item.get("repetition", 1)) for item in items}
|
|
224
|
+
if len(sample_keys) != samples:
|
|
225
|
+
raise ValueError(f"strategy {first['strategy_id']} contains duplicate task/repetition samples")
|
|
226
|
+
pass_rate = sum(1 for item in items if item["passed"]) / samples
|
|
227
|
+
first_pass_rate = sum(1 for item in items if item["first_passed"]) / samples
|
|
228
|
+
avg_repairs = sum(item["repair_cycles"] for item in items) / samples
|
|
229
|
+
avg_reviews = sum(item["review_cycles"] for item in items) / samples
|
|
230
|
+
avg_cost = sum(run_cost(item, prices) for item in items) / samples
|
|
231
|
+
avg_wall = sum(item.get("wall_time_seconds", 0.0) for item in items) / samples
|
|
232
|
+
quality_ok = samples >= min_samples and pass_rate >= threshold and avg_repairs <= max_repairs
|
|
233
|
+
return {
|
|
234
|
+
"task_class": first["task_class"],
|
|
235
|
+
"strategy_id": first["strategy_id"],
|
|
236
|
+
"strategy": first["strategy"],
|
|
237
|
+
"reasoning_policy": first["reasoning_policy"],
|
|
238
|
+
"model": first["model"],
|
|
239
|
+
"reasoning_effort": first["reasoning_effort"],
|
|
240
|
+
"worker_model": first.get("worker_model"),
|
|
241
|
+
"worker_reasoning_effort": first.get("worker_reasoning_effort"),
|
|
242
|
+
"samples": samples,
|
|
243
|
+
"pass_rate": round(pass_rate, 4),
|
|
244
|
+
"first_pass_rate": round(first_pass_rate, 4),
|
|
245
|
+
"average_repair_cycles": round(avg_repairs, 4),
|
|
246
|
+
"average_review_cycles": round(avg_reviews, 4),
|
|
247
|
+
"average_cost_usd": round(avg_cost, 6),
|
|
248
|
+
"average_wall_time_seconds": round(avg_wall, 3),
|
|
249
|
+
"quality_gate": quality_ok,
|
|
250
|
+
"_sample_keys": sample_keys,
|
|
251
|
+
}
|
|
252
|
+
|
|
253
|
+
|
|
254
|
+
def delta(value: float, reference: float) -> float:
|
|
255
|
+
return round(value - reference, 4)
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def capability_comparison(task_class: str, by_id: dict[str, dict[str, Any]], comparison: dict[str, Any]) -> dict[str, Any] | None:
|
|
259
|
+
sol_id = comparison["sol_strategy_id"]
|
|
260
|
+
sol = by_id.get(sol_id)
|
|
261
|
+
competitors = [
|
|
262
|
+
item for item in by_id.values()
|
|
263
|
+
if item["strategy"] == "direct"
|
|
264
|
+
and item["strategy_id"] != sol_id
|
|
265
|
+
and item["reasoning_policy"] == "fixed"
|
|
266
|
+
and sol is not None
|
|
267
|
+
and item["reasoning_effort"] == sol["reasoning_effort"]
|
|
268
|
+
]
|
|
269
|
+
if sol is None or not competitors:
|
|
270
|
+
return None
|
|
271
|
+
competitor = max(
|
|
272
|
+
competitors,
|
|
273
|
+
key=lambda item: (item["pass_rate"], item["first_pass_rate"], -item["average_repair_cycles"]),
|
|
274
|
+
)
|
|
275
|
+
paired = sol["_sample_keys"] == competitor["_sample_keys"]
|
|
276
|
+
controlled = sol["strategy"] == "direct" and sol["reasoning_policy"] == "fixed"
|
|
277
|
+
enough = (
|
|
278
|
+
sol["samples"] >= comparison["min_samples"]
|
|
279
|
+
and competitor["samples"] >= comparison["min_samples"]
|
|
280
|
+
and paired
|
|
281
|
+
and controlled
|
|
282
|
+
)
|
|
283
|
+
pass_gain = delta(sol["pass_rate"], competitor["pass_rate"])
|
|
284
|
+
first_gain = delta(sol["first_pass_rate"], competitor["first_pass_rate"])
|
|
285
|
+
repair_reduction = delta(competitor["average_repair_cycles"], sol["average_repair_cycles"])
|
|
286
|
+
not_worse = sol["pass_rate"] >= competitor["pass_rate"]
|
|
287
|
+
meaningful_gain = (
|
|
288
|
+
pass_gain >= comparison["sol_min_pass_rate_gain"]
|
|
289
|
+
or (pass_gain >= 0 and first_gain >= comparison["sol_min_first_pass_rate_gain"])
|
|
290
|
+
or (pass_gain >= 0 and first_gain >= 0 and repair_reduction >= comparison["sol_min_repair_reduction"])
|
|
291
|
+
)
|
|
292
|
+
return {
|
|
293
|
+
"task_class": task_class,
|
|
294
|
+
"sol_strategy_id": sol_id,
|
|
295
|
+
"competitor_strategy_id": competitor["strategy_id"],
|
|
296
|
+
"samples": min(sol["samples"], competitor["samples"]),
|
|
297
|
+
"evidence_sufficient": enough,
|
|
298
|
+
"paired_samples": paired,
|
|
299
|
+
"controlled_reasoning_effort": sol["reasoning_effort"],
|
|
300
|
+
"pass_rate_gain": pass_gain,
|
|
301
|
+
"first_pass_rate_gain": first_gain,
|
|
302
|
+
"average_repair_reduction": repair_reduction,
|
|
303
|
+
"advantage_demonstrated": enough and not_worse and meaningful_gain,
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def flow_comparison(task_class: str, by_id: dict[str, dict[str, Any]], comparison: dict[str, Any]) -> dict[str, Any] | None:
|
|
308
|
+
flow = by_id.get(comparison["flow_strategy_id"])
|
|
309
|
+
sol = by_id.get(comparison["sol_strategy_id"])
|
|
310
|
+
worker = by_id.get(comparison["worker_strategy_id"])
|
|
311
|
+
if flow is None or sol is None or worker is None:
|
|
312
|
+
return None
|
|
313
|
+
paired = flow["_sample_keys"] == sol["_sample_keys"] == worker["_sample_keys"]
|
|
314
|
+
controlled_effort = sol["reasoning_effort"]
|
|
315
|
+
controlled = (
|
|
316
|
+
flow["strategy"] == "flow"
|
|
317
|
+
and flow["reasoning_policy"] == "fixed"
|
|
318
|
+
and sol["strategy"] == "direct"
|
|
319
|
+
and sol["reasoning_policy"] == "fixed"
|
|
320
|
+
and worker["strategy"] == "direct"
|
|
321
|
+
and worker["reasoning_policy"] == "fixed"
|
|
322
|
+
and flow["reasoning_effort"] == controlled_effort
|
|
323
|
+
and flow["worker_reasoning_effort"] == controlled_effort
|
|
324
|
+
and worker["reasoning_effort"] == controlled_effort
|
|
325
|
+
)
|
|
326
|
+
enough = (
|
|
327
|
+
min(flow["samples"], sol["samples"], worker["samples"]) >= comparison["min_samples"]
|
|
328
|
+
and paired
|
|
329
|
+
and controlled
|
|
330
|
+
)
|
|
331
|
+
quality_delta = delta(flow["pass_rate"], sol["pass_rate"])
|
|
332
|
+
cost_reduction = round(1 - flow["average_cost_usd"] / sol["average_cost_usd"], 4) if sol["average_cost_usd"] else 0.0
|
|
333
|
+
worker_pass_gain = delta(flow["pass_rate"], worker["pass_rate"])
|
|
334
|
+
worker_first_gain = delta(flow["first_pass_rate"], worker["first_pass_rate"])
|
|
335
|
+
worker_repair_reduction = delta(worker["average_repair_cycles"], flow["average_repair_cycles"])
|
|
336
|
+
quality_noninferior = quality_delta >= -comparison["flow_max_pass_rate_regression_vs_sol"]
|
|
337
|
+
cost_ok = cost_reduction >= comparison["flow_min_cost_reduction_vs_sol"]
|
|
338
|
+
worker_gain = (
|
|
339
|
+
worker_pass_gain >= comparison["flow_min_pass_rate_gain_vs_worker"]
|
|
340
|
+
or (worker_pass_gain >= 0 and worker_first_gain >= comparison["flow_min_first_pass_rate_gain_vs_worker"])
|
|
341
|
+
or (worker_pass_gain >= 0 and worker_first_gain >= 0 and worker_repair_reduction >= comparison["flow_min_repair_reduction_vs_worker"])
|
|
342
|
+
)
|
|
343
|
+
return {
|
|
344
|
+
"task_class": task_class,
|
|
345
|
+
"flow_strategy_id": flow["strategy_id"],
|
|
346
|
+
"quality_reference_strategy_id": sol["strategy_id"],
|
|
347
|
+
"worker_reference_strategy_id": worker["strategy_id"],
|
|
348
|
+
"samples": min(flow["samples"], sol["samples"], worker["samples"]),
|
|
349
|
+
"evidence_sufficient": enough,
|
|
350
|
+
"paired_samples": paired,
|
|
351
|
+
"controlled_reasoning_effort": controlled_effort if controlled else None,
|
|
352
|
+
"pass_rate_delta_vs_sol": quality_delta,
|
|
353
|
+
"cost_reduction_vs_sol": cost_reduction,
|
|
354
|
+
"pass_rate_gain_vs_worker": worker_pass_gain,
|
|
355
|
+
"first_pass_rate_gain_vs_worker": worker_first_gain,
|
|
356
|
+
"average_repair_reduction_vs_worker": worker_repair_reduction,
|
|
357
|
+
"quality_noninferior_to_sol": quality_noninferior,
|
|
358
|
+
"worker_quality_gain": worker_gain,
|
|
359
|
+
"advantage_demonstrated": enough and quality_noninferior and cost_ok and worker_gain,
|
|
360
|
+
}
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
def adaptive_comparison(task_class: str, by_id: dict[str, dict[str, Any]], comparison: dict[str, Any]) -> dict[str, Any] | None:
|
|
364
|
+
fixed = by_id.get(comparison["flow_strategy_id"])
|
|
365
|
+
adaptive = by_id.get(comparison["adaptive_flow_strategy_id"])
|
|
366
|
+
if fixed is None or adaptive is None:
|
|
367
|
+
return None
|
|
368
|
+
paired = fixed["_sample_keys"] == adaptive["_sample_keys"]
|
|
369
|
+
policies_match = (
|
|
370
|
+
fixed["strategy"] == "flow"
|
|
371
|
+
and fixed["reasoning_policy"] == "fixed"
|
|
372
|
+
and adaptive["strategy"] == "flow"
|
|
373
|
+
and adaptive["reasoning_policy"] == "adaptive"
|
|
374
|
+
)
|
|
375
|
+
enough = (
|
|
376
|
+
min(fixed["samples"], adaptive["samples"]) >= comparison["min_samples"]
|
|
377
|
+
and paired
|
|
378
|
+
and policies_match
|
|
379
|
+
)
|
|
380
|
+
pass_gain = delta(adaptive["pass_rate"], fixed["pass_rate"])
|
|
381
|
+
first_gain = delta(adaptive["first_pass_rate"], fixed["first_pass_rate"])
|
|
382
|
+
repair_reduction = delta(fixed["average_repair_cycles"], adaptive["average_repair_cycles"])
|
|
383
|
+
cost_change = round(adaptive["average_cost_usd"] / fixed["average_cost_usd"] - 1, 4) if fixed["average_cost_usd"] else 0.0
|
|
384
|
+
wall_change = round(adaptive["average_wall_time_seconds"] / fixed["average_wall_time_seconds"] - 1, 4) if fixed["average_wall_time_seconds"] else 0.0
|
|
385
|
+
quality_gain = (
|
|
386
|
+
pass_gain >= comparison["adaptive_min_pass_rate_gain"]
|
|
387
|
+
or (pass_gain >= 0 and first_gain >= comparison["adaptive_min_first_pass_rate_gain"])
|
|
388
|
+
or (pass_gain >= 0 and first_gain >= 0 and repair_reduction >= comparison["adaptive_min_repair_reduction"])
|
|
389
|
+
)
|
|
390
|
+
return {
|
|
391
|
+
"task_class": task_class,
|
|
392
|
+
"fixed_strategy_id": fixed["strategy_id"],
|
|
393
|
+
"adaptive_strategy_id": adaptive["strategy_id"],
|
|
394
|
+
"samples": min(fixed["samples"], adaptive["samples"]),
|
|
395
|
+
"evidence_sufficient": enough,
|
|
396
|
+
"paired_samples": paired,
|
|
397
|
+
"reasoning_policies_match": policies_match,
|
|
398
|
+
"pass_rate_gain": pass_gain,
|
|
399
|
+
"first_pass_rate_gain": first_gain,
|
|
400
|
+
"average_repair_reduction": repair_reduction,
|
|
401
|
+
"cost_change": cost_change,
|
|
402
|
+
"wall_time_change": wall_change,
|
|
403
|
+
"quality_gain": quality_gain,
|
|
404
|
+
"value_demonstrated": enough and quality_gain and cost_change <= comparison["adaptive_max_cost_increase"],
|
|
405
|
+
}
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def main() -> int:
|
|
409
|
+
ap = argparse.ArgumentParser()
|
|
410
|
+
ap.add_argument("--results", required=True)
|
|
411
|
+
ap.add_argument("--prices", required=True)
|
|
412
|
+
ap.add_argument("--policy", default="policy/benchmark.toml")
|
|
413
|
+
ap.add_argument("--min-samples", type=int, default=None, help="override minimum required samples")
|
|
414
|
+
ap.add_argument("--json", action="store_true")
|
|
415
|
+
args = ap.parse_args()
|
|
416
|
+
|
|
417
|
+
policy = load_policy(Path(args.policy))
|
|
418
|
+
if policy.get("schema_version") != 2:
|
|
419
|
+
raise ValueError("benchmark policy schema_version must be 2")
|
|
420
|
+
quality = policy["quality"]
|
|
421
|
+
comparison = policy["comparison"]
|
|
422
|
+
|
|
423
|
+
min_samples_override = args.min_samples
|
|
424
|
+
if min_samples_override is not None and min_samples_override < 1:
|
|
425
|
+
raise ValueError("--min-samples must be >= 1")
|
|
426
|
+
|
|
427
|
+
quality_min_samples = min_samples_override if min_samples_override is not None else quality["min_samples_per_configuration"]
|
|
428
|
+
comparison_config = dict(comparison)
|
|
429
|
+
if min_samples_override is not None:
|
|
430
|
+
comparison_config["min_samples"] = min_samples_override
|
|
431
|
+
|
|
432
|
+
prices = json.loads(Path(args.prices).read_text())
|
|
433
|
+
validate_prices(prices)
|
|
434
|
+
rows = load_jsonl(Path(args.results))
|
|
435
|
+
|
|
436
|
+
grouped: dict[tuple[str, str], list[dict[str, Any]]] = defaultdict(list)
|
|
437
|
+
for row in rows:
|
|
438
|
+
grouped[(row["task_class"], row["strategy_id"])].append(row)
|
|
439
|
+
|
|
440
|
+
thresholds = {
|
|
441
|
+
"routine": quality["routine_min_pass_rate"],
|
|
442
|
+
"complex": quality["complex_min_pass_rate"],
|
|
443
|
+
"critical": quality["critical_min_pass_rate"],
|
|
444
|
+
}
|
|
445
|
+
summaries = [
|
|
446
|
+
summarize(items, prices, thresholds[task_class], quality_min_samples, quality["max_average_repair_cycles"])
|
|
447
|
+
for (task_class, _), items in sorted(grouped.items())
|
|
448
|
+
]
|
|
449
|
+
|
|
450
|
+
recommendations: dict[str, Any] = {}
|
|
451
|
+
sol_evidence: dict[str, Any] = {}
|
|
452
|
+
flow_evidence: dict[str, Any] = {}
|
|
453
|
+
adaptive_evidence: dict[str, Any] = {}
|
|
454
|
+
for task_class in TASK_CLASSES:
|
|
455
|
+
class_items = [item for item in summaries if item["task_class"] == task_class]
|
|
456
|
+
eligible = [item for item in class_items if item["quality_gate"]]
|
|
457
|
+
if eligible:
|
|
458
|
+
best = min(
|
|
459
|
+
eligible,
|
|
460
|
+
key=lambda item: (item["average_cost_usd"], EFFORT_RANK[item["reasoning_effort"]], item["strategy_id"]),
|
|
461
|
+
)
|
|
462
|
+
recommendations[task_class] = {
|
|
463
|
+
"strategy_id": best["strategy_id"],
|
|
464
|
+
"strategy": best["strategy"],
|
|
465
|
+
"model": best["model"],
|
|
466
|
+
"reasoning_effort": best["reasoning_effort"],
|
|
467
|
+
"worker_model": best["worker_model"],
|
|
468
|
+
"average_cost_usd": best["average_cost_usd"],
|
|
469
|
+
"pass_rate": best["pass_rate"],
|
|
470
|
+
"samples": best["samples"],
|
|
471
|
+
}
|
|
472
|
+
else:
|
|
473
|
+
recommendations[task_class] = None
|
|
474
|
+
by_id = {item["strategy_id"]: item for item in class_items}
|
|
475
|
+
sol_evidence[task_class] = capability_comparison(task_class, by_id, comparison_config)
|
|
476
|
+
flow_evidence[task_class] = flow_comparison(task_class, by_id, comparison_config)
|
|
477
|
+
adaptive_evidence[task_class] = adaptive_comparison(task_class, by_id, comparison_config)
|
|
478
|
+
|
|
479
|
+
result = {
|
|
480
|
+
"recommendations": recommendations,
|
|
481
|
+
"configurations": [
|
|
482
|
+
{key: value for key, value in summary.items() if not key.startswith("_")}
|
|
483
|
+
for summary in summaries
|
|
484
|
+
],
|
|
485
|
+
"sol_capability_evidence": sol_evidence,
|
|
486
|
+
"flow_advantage_evidence": flow_evidence,
|
|
487
|
+
"adaptive_reasoning_evidence": adaptive_evidence,
|
|
488
|
+
"advisory_only": True,
|
|
489
|
+
}
|
|
490
|
+
if args.json:
|
|
491
|
+
print(json.dumps(result, sort_keys=True))
|
|
492
|
+
else:
|
|
493
|
+
for task_class in TASK_CLASSES:
|
|
494
|
+
sol = sol_evidence[task_class]
|
|
495
|
+
flow = flow_evidence[task_class]
|
|
496
|
+
adaptive = adaptive_evidence[task_class]
|
|
497
|
+
print(
|
|
498
|
+
f"{task_class}: sol advantage={sol and sol['advantage_demonstrated']}; "
|
|
499
|
+
f"flow advantage={flow and flow['advantage_demonstrated']}; "
|
|
500
|
+
f"adaptive value={adaptive and adaptive['value_demonstrated']}"
|
|
501
|
+
)
|
|
502
|
+
return 0
|
|
503
|
+
|
|
504
|
+
|
|
505
|
+
if __name__ == "__main__":
|
|
506
|
+
try:
|
|
507
|
+
raise SystemExit(main())
|
|
508
|
+
except Exception as exc:
|
|
509
|
+
print(f"benchmark analysis failed: {exc}", file=sys.stderr)
|
|
510
|
+
raise SystemExit(1)
|
|
@@ -0,0 +1,171 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
import os
|
|
7
|
+
import shutil
|
|
8
|
+
import subprocess
|
|
9
|
+
import sys
|
|
10
|
+
from datetime import datetime, timezone
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
ROOT = Path(__file__).resolve().parents[1]
|
|
14
|
+
DEFAULT_PRICE = ROOT / "benchmark/prices/gpt-5.6-2026-08-30.json"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def run(cmd: list[str], *, check: bool = True, capture: bool = False) -> subprocess.CompletedProcess[str]:
|
|
18
|
+
return subprocess.run(
|
|
19
|
+
cmd,
|
|
20
|
+
check=check,
|
|
21
|
+
text=True,
|
|
22
|
+
stdout=subprocess.PIPE if capture else None,
|
|
23
|
+
stderr=subprocess.PIPE if capture else None,
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def require(name: str) -> str:
|
|
28
|
+
path = shutil.which(name)
|
|
29
|
+
if not path:
|
|
30
|
+
raise RuntimeError(f"{name} is required")
|
|
31
|
+
return path
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def main() -> int:
|
|
35
|
+
ap = argparse.ArgumentParser(
|
|
36
|
+
description="Run the built-in codex-flow benchmark using the local Codex login session."
|
|
37
|
+
)
|
|
38
|
+
ap.add_argument("profile", nargs="?", choices=["quick", "full"], default="quick")
|
|
39
|
+
ap.add_argument("--workspace", default=".codex-flow-benchmark")
|
|
40
|
+
ap.add_argument("--output")
|
|
41
|
+
ap.add_argument("--prices", default=str(DEFAULT_PRICE))
|
|
42
|
+
ap.add_argument("--yes", action="store_true", help="skip the paid-run confirmation prompt")
|
|
43
|
+
args = ap.parse_args()
|
|
44
|
+
|
|
45
|
+
require("git")
|
|
46
|
+
codex = require("codex")
|
|
47
|
+
require("python3")
|
|
48
|
+
|
|
49
|
+
version = run([codex, "--version"], capture=True).stdout.strip()
|
|
50
|
+
|
|
51
|
+
workspace = Path(args.workspace).resolve()
|
|
52
|
+
manifest = workspace / "manifest.json"
|
|
53
|
+
results_dir = ROOT / "benchmark/results"
|
|
54
|
+
results_dir.mkdir(parents=True, exist_ok=True)
|
|
55
|
+
stamp = datetime.now(timezone.utc).strftime("%Y%m%dT%H%M%SZ")
|
|
56
|
+
output = Path(args.output).resolve() if args.output else (results_dir / f"{args.profile}-{stamp}.jsonl")
|
|
57
|
+
|
|
58
|
+
materialize = run([
|
|
59
|
+
sys.executable,
|
|
60
|
+
str(ROOT / "scripts/materialize-corpus.py"),
|
|
61
|
+
"--profile", args.profile,
|
|
62
|
+
"--output-dir", str(workspace / "corpus"),
|
|
63
|
+
"--manifest", str(manifest),
|
|
64
|
+
], capture=True)
|
|
65
|
+
plan_meta = json.loads(materialize.stdout)
|
|
66
|
+
|
|
67
|
+
dry = run([
|
|
68
|
+
sys.executable,
|
|
69
|
+
str(ROOT / "scripts/run-benchmark.py"),
|
|
70
|
+
"--manifest", str(manifest),
|
|
71
|
+
"--output", str(output),
|
|
72
|
+
"--dry-run",
|
|
73
|
+
], capture=True)
|
|
74
|
+
plan = json.loads(dry.stdout)
|
|
75
|
+
planned = int(plan["planned_runs"])
|
|
76
|
+
|
|
77
|
+
disp_output = str(output)
|
|
78
|
+
home_str = str(Path.home())
|
|
79
|
+
if disp_output.startswith(home_str):
|
|
80
|
+
disp_output = "~" + disp_output[len(home_str):]
|
|
81
|
+
|
|
82
|
+
budget = "~15M tokens (planning + repairs may increase)" if args.profile == "quick" else "high (90 runs, substantial Codex quota)"
|
|
83
|
+
|
|
84
|
+
def pad_line(content: str, width: int = 68) -> str:
|
|
85
|
+
pad = max(0, width - len(content))
|
|
86
|
+
return f" │ {content}{' ' * pad} │"
|
|
87
|
+
|
|
88
|
+
print(f"\n⚡ codex-flow benchmark-local\n")
|
|
89
|
+
print(" ╭─ Benchmark Plan ──────────────────────────────────────────────────╮")
|
|
90
|
+
print(pad_line(f"• Profile: {args.profile} ({planned} real model runs)"))
|
|
91
|
+
print(pad_line(f"• Codex CLI: {version} (local auth session)"))
|
|
92
|
+
print(pad_line(f"• Est. Budget: {budget}"))
|
|
93
|
+
print(pad_line(f"• Output File: {disp_output}"))
|
|
94
|
+
print(" ╰───────────────────────────────────────────────────────────────────╯\n")
|
|
95
|
+
|
|
96
|
+
print("┌─ ⚠️ REAL MODEL EXECUTION CONFIRMATION ────────────────────────────┐")
|
|
97
|
+
print("│ This will execute tasks using your authenticated Codex account │")
|
|
98
|
+
print("│ and will consume real quota/credits. │")
|
|
99
|
+
print("└───────────────────────────────────────────────────────────────────┘\n")
|
|
100
|
+
|
|
101
|
+
if not args.yes:
|
|
102
|
+
expected = f"RUN {args.profile.upper()} {planned}"
|
|
103
|
+
typed = input(f"Type '{expected}' to proceed: ").strip()
|
|
104
|
+
if typed != expected:
|
|
105
|
+
print("Cancelled; no model run was started.")
|
|
106
|
+
return 2
|
|
107
|
+
|
|
108
|
+
output.parent.mkdir(parents=True, exist_ok=True)
|
|
109
|
+
run([
|
|
110
|
+
sys.executable,
|
|
111
|
+
str(ROOT / "scripts/run-benchmark.py"),
|
|
112
|
+
"--manifest", str(manifest),
|
|
113
|
+
"--output", str(output),
|
|
114
|
+
"--fail-fast-infrastructure",
|
|
115
|
+
])
|
|
116
|
+
|
|
117
|
+
analysis = output.with_suffix(".analysis.json")
|
|
118
|
+
report = output.with_suffix(".report.md")
|
|
119
|
+
with analysis.open("w", encoding="utf-8") as fh:
|
|
120
|
+
proc = subprocess.run([
|
|
121
|
+
sys.executable,
|
|
122
|
+
str(ROOT / "scripts/analyze-benchmark.py"),
|
|
123
|
+
"--results", str(output),
|
|
124
|
+
"--prices", str(Path(args.prices).resolve()),
|
|
125
|
+
"--json",
|
|
126
|
+
], check=True, text=True, stdout=fh)
|
|
127
|
+
|
|
128
|
+
run([
|
|
129
|
+
sys.executable,
|
|
130
|
+
str(ROOT / "scripts/render-benchmark-report.py"),
|
|
131
|
+
"--results", str(output),
|
|
132
|
+
"--prices", str(Path(args.prices).resolve()),
|
|
133
|
+
"--analysis", str(analysis),
|
|
134
|
+
"--output", str(report),
|
|
135
|
+
"--title", f"Codex {args.profile} local benchmark",
|
|
136
|
+
])
|
|
137
|
+
|
|
138
|
+
meta = output.with_suffix(".meta.json")
|
|
139
|
+
meta.write_text(json.dumps({
|
|
140
|
+
"schema_version": 2,
|
|
141
|
+
"profile": args.profile,
|
|
142
|
+
"planned_runs": planned,
|
|
143
|
+
"codex_cli_version": version,
|
|
144
|
+
"codex_flow_commit": subprocess.check_output(["git", "-C", str(ROOT), "rev-parse", "HEAD"], text=True).strip(),
|
|
145
|
+
"authentication_mode": "local-codex-session",
|
|
146
|
+
"cost_semantics": "API-equivalent reference cost only; not the user's ChatGPT subscription charge",
|
|
147
|
+
"manifest": str(manifest),
|
|
148
|
+
"results": str(output),
|
|
149
|
+
"analysis": str(analysis),
|
|
150
|
+
"report": str(report),
|
|
151
|
+
"materialization": plan_meta,
|
|
152
|
+
}, indent=2) + "\n")
|
|
153
|
+
|
|
154
|
+
print("\nBenchmark complete.")
|
|
155
|
+
print(f"Results: {output}")
|
|
156
|
+
print(f"Analysis: {analysis}")
|
|
157
|
+
print(f"Report: {report}")
|
|
158
|
+
print(f"Metadata: {meta}")
|
|
159
|
+
print("Dollar values in the report are API-equivalent reference costs, not actual ChatGPT subscription charges.")
|
|
160
|
+
return 0
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
if __name__ == "__main__":
|
|
164
|
+
try:
|
|
165
|
+
raise SystemExit(main())
|
|
166
|
+
except KeyboardInterrupt:
|
|
167
|
+
print("\nCancelled.", file=sys.stderr)
|
|
168
|
+
raise SystemExit(130)
|
|
169
|
+
except Exception as exc:
|
|
170
|
+
print(f"local benchmark failed: {exc}", file=sys.stderr)
|
|
171
|
+
raise SystemExit(1)
|