codex-flow 2.1.13__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codex_flow/__init__.py +28 -0
- codex_flow/__main__.py +9 -0
- codex_flow/cli.py +242 -0
- codex_flow/data/LICENSE +21 -0
- codex_flow/data/README.en.md +303 -0
- codex_flow/data/README.md +305 -0
- codex_flow/data/VERSION +1 -0
- codex_flow/data/apps/chatgpt-mcp/README.md +86 -0
- codex_flow/data/apps/chatgpt-mcp/__init__.py +1 -0
- codex_flow/data/apps/chatgpt-mcp/adapter.py +458 -0
- codex_flow/data/apps/chatgpt-mcp/server.py +358 -0
- codex_flow/data/apps/chatgpt-mcp/widget.html +927 -0
- codex_flow/data/apps/macos-overlay/README.en.md +121 -0
- codex_flow/data/apps/macos-overlay/README.md +123 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayRuntimeState.swift +126 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayScreenGeometry.swift +82 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayWindowController.swift +1052 -0
- codex_flow/data/apps/macos-overlay/Sources/Localization.swift +197 -0
- codex_flow/data/apps/macos-overlay/Sources/Models/TelemetryData.swift +1557 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/AccountSnapshotService.swift +1101 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/FlowPilotInstanceLock.swift +153 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/IPCServer.swift +298 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryQueryEngine.swift +800 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryWatcher.swift +135 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/UpdateService.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AccountView.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AnalyticsView.swift +566 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AutostartView.swift +293 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/BubbleView.swift +317 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HistoryView.swift +1124 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HoverRevealText.swift +165 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/InspectorSkillsToolsView.swift +121 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/LogoView.swift +182 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SleekSwitch.swift +117 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/StrategyModeView.swift +561 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SummaryView.swift +1273 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/UpdateView.swift +352 -0
- codex_flow/data/apps/macos-overlay/Sources/main.swift +340 -0
- codex_flow/data/apps/macos-overlay/Tests/OverlayScreenGeometryTests.swift +163 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryPhase1ContractTests.swift +357 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQueryEngineConcurrencyTests.swift +221 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQuotaSelectionTests.swift +158 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryWorkerTokenTests.swift +122 -0
- codex_flow/data/apps/macos-overlay/build.sh +75 -0
- codex_flow/data/benchmark/corpus.json +103 -0
- codex_flow/data/benchmark/manifest.example.json +41 -0
- codex_flow/data/benchmark/manifest.schema.json +137 -0
- codex_flow/data/benchmark/prices/gpt-5.6-2026-08-30.json +5 -0
- codex_flow/data/benchmark/profiles.json +90 -0
- codex_flow/data/benchmark/schema.json +77 -0
- codex_flow/data/benchmark/tasks.json +50 -0
- codex_flow/data/completions/codex-flow.bash +34 -0
- codex_flow/data/completions/codex-flow.zsh +52 -0
- codex_flow/data/glama.json +6 -0
- codex_flow/data/install-release.ps1 +126 -0
- codex_flow/data/install-release.sh +155 -0
- codex_flow/data/install.ps1 +349 -0
- codex_flow/data/install.sh +362 -0
- codex_flow/data/policy/benchmark.toml +49 -0
- codex_flow/data/policy/defaults.toml +70 -0
- codex_flow/data/scripts/analyze-benchmark.py +510 -0
- codex_flow/data/scripts/benchmark-local.py +171 -0
- codex_flow/data/scripts/check-recommendation.py +277 -0
- codex_flow/data/scripts/doctor.py +449 -0
- codex_flow/data/scripts/generate-release-manifest.py +74 -0
- codex_flow/data/scripts/localization.py +192 -0
- codex_flow/data/scripts/manage-hooks.py +448 -0
- codex_flow/data/scripts/manage-instructions.py +389 -0
- codex_flow/data/scripts/manage-shell.py +151 -0
- codex_flow/data/scripts/materialize-corpus.py +193 -0
- codex_flow/data/scripts/menu.py +646 -0
- codex_flow/data/scripts/migrations/0001_update_settings.py +80 -0
- codex_flow/data/scripts/package-release.py +132 -0
- codex_flow/data/scripts/render-benchmark-report.py +292 -0
- codex_flow/data/scripts/run-benchmark.py +829 -0
- codex_flow/data/scripts/strategies/__init__.py +28 -0
- codex_flow/data/scripts/strategies/balanced.py +115 -0
- codex_flow/data/scripts/strategies/base.py +363 -0
- codex_flow/data/scripts/strategies/efficient.py +158 -0
- codex_flow/data/scripts/strategies/lifecycle_runtime.py +590 -0
- codex_flow/data/scripts/strategies/quality.py +209 -0
- codex_flow/data/scripts/strategies/speed.py +108 -0
- codex_flow/data/scripts/strategies/task_budget_runtime.py +644 -0
- codex_flow/data/scripts/strategies/task_phase_runtime.py +341 -0
- codex_flow/data/scripts/strategies/work_unit_runtime.py +421 -0
- codex_flow/data/scripts/strategy_runtime.py +1091 -0
- codex_flow/data/scripts/telemetry.py +400 -0
- codex_flow/data/scripts/telemetry_core/__init__.py +192 -0
- codex_flow/data/scripts/telemetry_core/app_server.py +1192 -0
- codex_flow/data/scripts/telemetry_core/collector.py +1247 -0
- codex_flow/data/scripts/telemetry_core/common.py +421 -0
- codex_flow/data/scripts/telemetry_core/latency.py +593 -0
- codex_flow/data/scripts/telemetry_core/query.py +427 -0
- codex_flow/data/scripts/telemetry_core/quota_ledger.py +598 -0
- codex_flow/data/scripts/telemetry_core/render.py +460 -0
- codex_flow/data/scripts/telemetry_core/repair.py +223 -0
- codex_flow/data/scripts/ui.py +266 -0
- codex_flow/data/scripts/update-homebrew-formula.py +146 -0
- codex_flow/data/scripts/update_runtime_config.py +134 -0
- codex_flow/data/scripts/updater.py +1718 -0
- codex_flow/data/smithery.yaml +18 -0
- codex_flow/data/templates/agents/worker-explorer.toml +24 -0
- codex_flow/data/templates/agents/worker-implementer.toml +49 -0
- codex_flow/data/templates/agents/worker-reviewer.toml +25 -0
- codex_flow/data/templates/flow-pilot-instructions.md +35 -0
- codex_flow/data/templates/skills/flow-pilot/SKILL.md +577 -0
- codex_flow/mcp.py +35 -0
- codex_flow-2.1.13.dist-info/METADATA +342 -0
- codex_flow-2.1.13.dist-info/RECORD +113 -0
- codex_flow-2.1.13.dist-info/WHEEL +5 -0
- codex_flow-2.1.13.dist-info/entry_points.txt +3 -0
- codex_flow-2.1.13.dist-info/licenses/LICENSE +21 -0
- codex_flow-2.1.13.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Add OTA update preferences without rewriting unrelated policy fields."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import os
|
|
8
|
+
import re
|
|
9
|
+
import tempfile
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
DEFAULTS = {
|
|
13
|
+
"channel": '"stable"',
|
|
14
|
+
"check": "true",
|
|
15
|
+
"check_interval_hours": "24",
|
|
16
|
+
"notify_cli": "true",
|
|
17
|
+
"notify_app": "true",
|
|
18
|
+
"auto_install": "false",
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def atomic_write(path: Path, text: str) -> None:
|
|
23
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
24
|
+
fd, name = tempfile.mkstemp(prefix=f".{path.name}.", dir=str(path.parent))
|
|
25
|
+
try:
|
|
26
|
+
with os.fdopen(fd, "w", encoding="utf-8", newline="\n") as handle:
|
|
27
|
+
handle.write(text)
|
|
28
|
+
handle.flush()
|
|
29
|
+
os.fsync(handle.fileno())
|
|
30
|
+
os.replace(name, path)
|
|
31
|
+
finally:
|
|
32
|
+
try:
|
|
33
|
+
os.unlink(name)
|
|
34
|
+
except FileNotFoundError:
|
|
35
|
+
pass
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def migrate(policy: Path) -> bool:
|
|
39
|
+
try:
|
|
40
|
+
text = policy.read_text(encoding="utf-8-sig")
|
|
41
|
+
except OSError:
|
|
42
|
+
return False
|
|
43
|
+
|
|
44
|
+
section_re = re.compile(r"(?ms)^\[update\]\s*\n(.*?)(?=^\[[^\n]+\]\s*$|\Z)")
|
|
45
|
+
match = section_re.search(text)
|
|
46
|
+
changed = False
|
|
47
|
+
if not match:
|
|
48
|
+
if text and not text.endswith("\n"):
|
|
49
|
+
text += "\n"
|
|
50
|
+
if text and not text.endswith("\n\n"):
|
|
51
|
+
text += "\n"
|
|
52
|
+
text += "[update]\n" + "".join(f"{key} = {value}\n" for key, value in DEFAULTS.items())
|
|
53
|
+
changed = True
|
|
54
|
+
else:
|
|
55
|
+
body = match.group(1)
|
|
56
|
+
for key, value in DEFAULTS.items():
|
|
57
|
+
if re.search(rf"(?m)^\s*{re.escape(key)}\s*=", body):
|
|
58
|
+
continue
|
|
59
|
+
if body and not body.endswith("\n"):
|
|
60
|
+
body += "\n"
|
|
61
|
+
body += f"{key} = {value}\n"
|
|
62
|
+
changed = True
|
|
63
|
+
if changed:
|
|
64
|
+
text = text[: match.start(1)] + body + text[match.end(1) :]
|
|
65
|
+
|
|
66
|
+
if changed:
|
|
67
|
+
atomic_write(policy, text)
|
|
68
|
+
return changed
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def main() -> int:
|
|
72
|
+
parser = argparse.ArgumentParser()
|
|
73
|
+
parser.add_argument("--policy", required=True, type=Path)
|
|
74
|
+
args = parser.parse_args()
|
|
75
|
+
migrate(args.policy)
|
|
76
|
+
return 0
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
if __name__ == "__main__":
|
|
80
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Build codex-flow OTA release archives."""
|
|
3
|
+
|
|
4
|
+
from __future__ import annotations
|
|
5
|
+
|
|
6
|
+
import argparse
|
|
7
|
+
import gzip
|
|
8
|
+
import hashlib
|
|
9
|
+
import os
|
|
10
|
+
import tarfile
|
|
11
|
+
import zipfile
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
ROOT = Path(__file__).resolve().parents[1]
|
|
15
|
+
|
|
16
|
+
# Everything required after OTA switches STATE_DIR/source to the version package.
|
|
17
|
+
# Keep this list runtime-focused: CI/tests/.github stay source-only, while
|
|
18
|
+
# benchmark data and ChatGPT MCP must remain available after the original git
|
|
19
|
+
# checkout is no longer involved.
|
|
20
|
+
INCLUDE_ROOTS = (
|
|
21
|
+
"VERSION",
|
|
22
|
+
"LICENSE",
|
|
23
|
+
"README.md",
|
|
24
|
+
"README.en.md",
|
|
25
|
+
"install.sh",
|
|
26
|
+
"install.ps1",
|
|
27
|
+
"bin",
|
|
28
|
+
"scripts",
|
|
29
|
+
"templates",
|
|
30
|
+
"completions",
|
|
31
|
+
"policy",
|
|
32
|
+
"benchmark",
|
|
33
|
+
"apps/chatgpt-mcp",
|
|
34
|
+
)
|
|
35
|
+
DARWIN_ROOTS = (
|
|
36
|
+
"apps/macos-overlay",
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def version() -> str:
|
|
41
|
+
return (ROOT / "VERSION").read_text(encoding="utf-8-sig").strip().lstrip("v")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _collect(path: Path, seen: set[Path]) -> None:
|
|
45
|
+
if path.is_file() and not path.is_symlink():
|
|
46
|
+
seen.add(path)
|
|
47
|
+
return
|
|
48
|
+
if not path.is_dir():
|
|
49
|
+
return
|
|
50
|
+
for child in path.rglob("*"):
|
|
51
|
+
if (
|
|
52
|
+
child.is_file()
|
|
53
|
+
and not child.is_symlink()
|
|
54
|
+
and "__pycache__" not in child.parts
|
|
55
|
+
and child.suffix not in {".pyc", ".pyo"}
|
|
56
|
+
and child.name != ".DS_Store"
|
|
57
|
+
):
|
|
58
|
+
seen.add(child)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def iter_files(platform_name: str) -> list[Path]:
|
|
62
|
+
seen: set[Path] = set()
|
|
63
|
+
for item in INCLUDE_ROOTS:
|
|
64
|
+
_collect(ROOT / item, seen)
|
|
65
|
+
if platform_name.startswith("darwin-"):
|
|
66
|
+
for item in DARWIN_ROOTS:
|
|
67
|
+
_collect(ROOT / item, seen)
|
|
68
|
+
return sorted(seen, key=lambda p: p.relative_to(ROOT).as_posix())
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def sha256(path: Path) -> str:
|
|
72
|
+
digest = hashlib.sha256()
|
|
73
|
+
with path.open("rb") as handle:
|
|
74
|
+
for chunk in iter(lambda: handle.read(1024 * 1024), b""):
|
|
75
|
+
digest.update(chunk)
|
|
76
|
+
return digest.hexdigest()
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def _mode(src: Path) -> int:
|
|
80
|
+
return 0o755 if os.access(src, os.X_OK) or src.suffix in {".sh", ".py"} else 0o644
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def build(platform_name: str, output_dir: Path) -> Path:
|
|
84
|
+
output_dir.mkdir(parents=True, exist_ok=True)
|
|
85
|
+
ver = version()
|
|
86
|
+
prefix = f"codex-flow-{ver}"
|
|
87
|
+
files = iter_files(platform_name)
|
|
88
|
+
if platform_name.startswith("windows-"):
|
|
89
|
+
archive = output_dir / f"codex-flow-{ver}-{platform_name}.zip"
|
|
90
|
+
with zipfile.ZipFile(archive, "w", compression=zipfile.ZIP_DEFLATED, compresslevel=9) as zf:
|
|
91
|
+
for src in files:
|
|
92
|
+
rel = src.relative_to(ROOT).as_posix()
|
|
93
|
+
info = zipfile.ZipInfo(f"{prefix}/{rel}")
|
|
94
|
+
info.date_time = (2020, 1, 1, 0, 0, 0)
|
|
95
|
+
info.external_attr = (_mode(src) & 0xFFFF) << 16
|
|
96
|
+
info.compress_type = zipfile.ZIP_DEFLATED
|
|
97
|
+
zf.writestr(info, src.read_bytes())
|
|
98
|
+
else:
|
|
99
|
+
archive = output_dir / f"codex-flow-{ver}-{platform_name}.tar.gz"
|
|
100
|
+
# Pin the gzip header timestamp as well as tar metadata so rebuilds of
|
|
101
|
+
# identical inputs produce stable bytes/checksums.
|
|
102
|
+
with archive.open("wb") as raw, gzip.GzipFile(fileobj=raw, mode="wb", mtime=0) as gz:
|
|
103
|
+
with tarfile.open(fileobj=gz, mode="w", format=tarfile.PAX_FORMAT) as tf:
|
|
104
|
+
for src in files:
|
|
105
|
+
rel = src.relative_to(ROOT).as_posix()
|
|
106
|
+
arcname = f"{prefix}/{rel}"
|
|
107
|
+
info = tf.gettarinfo(str(src), arcname=arcname)
|
|
108
|
+
info.uid = info.gid = 0
|
|
109
|
+
info.uname = info.gname = "root"
|
|
110
|
+
info.mtime = 0
|
|
111
|
+
info.mode = _mode(src)
|
|
112
|
+
with src.open("rb") as handle:
|
|
113
|
+
tf.addfile(info, handle)
|
|
114
|
+
digest = sha256(archive)
|
|
115
|
+
(archive.with_suffix(archive.suffix + ".sha256")).write_text(
|
|
116
|
+
f"{digest} {archive.name}\n", encoding="utf-8"
|
|
117
|
+
)
|
|
118
|
+
print(archive)
|
|
119
|
+
return archive
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def main() -> int:
|
|
123
|
+
parser = argparse.ArgumentParser()
|
|
124
|
+
parser.add_argument("--platform", required=True)
|
|
125
|
+
parser.add_argument("--output-dir", default="dist", type=Path)
|
|
126
|
+
args = parser.parse_args()
|
|
127
|
+
build(args.platform, args.output_dir)
|
|
128
|
+
return 0
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
if __name__ == "__main__":
|
|
132
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,292 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
import argparse
|
|
5
|
+
import json
|
|
6
|
+
from collections import defaultdict
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def load_jsonl(path: Path) -> list[dict[str, Any]]:
|
|
12
|
+
rows = [json.loads(line) for line in path.read_text().splitlines() if line.strip()]
|
|
13
|
+
if not rows:
|
|
14
|
+
raise ValueError("benchmark results are empty")
|
|
15
|
+
for row in rows:
|
|
16
|
+
if row.get("schema_version") == 1:
|
|
17
|
+
suffix = row["model"].rsplit("-", 1)[-1]
|
|
18
|
+
if suffix in {"luna", "terra", "sol"}:
|
|
19
|
+
base = f"{suffix}-direct"
|
|
20
|
+
row["strategy_id"] = base if row["reasoning_effort"] == "high" else f"{base}-{row['reasoning_effort']}"
|
|
21
|
+
else:
|
|
22
|
+
row["strategy_id"] = f"direct:{row['model']}:{row['reasoning_effort']}"
|
|
23
|
+
row["strategy"] = "direct"
|
|
24
|
+
row["reasoning_policy"] = "fixed"
|
|
25
|
+
row["worker_model"] = None
|
|
26
|
+
row["worker_reasoning_effort"] = None
|
|
27
|
+
row["first_passed"] = row["passed"] and row["repair_cycles"] == 0
|
|
28
|
+
row["review_cycles"] = 0
|
|
29
|
+
row["model_usage"] = [{
|
|
30
|
+
"role": "direct",
|
|
31
|
+
"model": row["model"],
|
|
32
|
+
"reasoning_effort": row["reasoning_effort"],
|
|
33
|
+
"calls": row["repair_cycles"] + 1,
|
|
34
|
+
"input_tokens": row["input_tokens"],
|
|
35
|
+
"cached_input_tokens": row["cached_input_tokens"],
|
|
36
|
+
"output_tokens": row["output_tokens"],
|
|
37
|
+
}]
|
|
38
|
+
return rows
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def usage_cost(usage: dict[str, Any], prices: dict[str, Any]) -> float:
|
|
42
|
+
model = usage["model"]
|
|
43
|
+
if model not in prices:
|
|
44
|
+
raise ValueError(f"missing price snapshot for {model}")
|
|
45
|
+
price = prices[model]
|
|
46
|
+
cached = usage["cached_input_tokens"]
|
|
47
|
+
uncached = max(0, usage["input_tokens"] - cached)
|
|
48
|
+
return (
|
|
49
|
+
uncached * price["input"]
|
|
50
|
+
+ cached * price["cached_input"]
|
|
51
|
+
+ usage["output_tokens"] * price["output"]
|
|
52
|
+
) / 1_000_000
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def cost(row: dict[str, Any], prices: dict[str, Any]) -> float:
|
|
56
|
+
return sum(usage_cost(usage, prices) for usage in row["model_usage"])
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def pct(n: float, d: float) -> str:
|
|
60
|
+
return f"{(100 * n / d):.1f}%" if d else "n/a"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def signed_pct(value: float) -> str:
|
|
64
|
+
return f"{value:+.1%}"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def yes_no(value: bool) -> str:
|
|
68
|
+
return "yes" if value else "no"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def composition(row: dict[str, Any]) -> str:
|
|
72
|
+
if row["strategy"] == "direct":
|
|
73
|
+
return f"{row['model']} / {row['reasoning_effort']}"
|
|
74
|
+
policy = row.get("reasoning_policy", "fixed")
|
|
75
|
+
if policy == "adaptive":
|
|
76
|
+
return f"{row['model']} parent → {row['worker_model']} worker / adaptive"
|
|
77
|
+
return f"{row['model']} parent → {row['worker_model']} worker / {row['reasoning_effort']}"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def token_metrics(items: list[dict[str, Any]]) -> dict[str, float]:
|
|
81
|
+
count = len(items)
|
|
82
|
+
total_input = sum(item["input_tokens"] for item in items)
|
|
83
|
+
total_cached = sum(item["cached_input_tokens"] for item in items)
|
|
84
|
+
total_output = sum(item["output_tokens"] for item in items)
|
|
85
|
+
parent_tokens = 0
|
|
86
|
+
worker_tokens = 0
|
|
87
|
+
for item in items:
|
|
88
|
+
for usage in item.get("model_usage", []):
|
|
89
|
+
tokens = usage.get("input_tokens", 0) + usage.get("output_tokens", 0)
|
|
90
|
+
if usage.get("role") == "parent":
|
|
91
|
+
parent_tokens += tokens
|
|
92
|
+
elif usage.get("role") == "worker":
|
|
93
|
+
worker_tokens += tokens
|
|
94
|
+
return {
|
|
95
|
+
"total_tokens": (total_input + total_output) / count,
|
|
96
|
+
"parent_tokens": parent_tokens / count,
|
|
97
|
+
"worker_tokens": worker_tokens / count,
|
|
98
|
+
"cached_input_tokens": total_cached / count,
|
|
99
|
+
"net_new_input_tokens": (total_input - total_cached) / count,
|
|
100
|
+
"cache_rate": total_cached / total_input if total_input else 0.0,
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def relative_change(value: float, reference: float) -> float | None:
|
|
105
|
+
return value / reference - 1 if reference else None
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def fmt_avg_tokens(value: float) -> str:
|
|
109
|
+
return f"{round(value):,}"
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def fmt_change(value: float | None) -> str:
|
|
113
|
+
return signed_pct(value) if value is not None else "n/a"
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def main() -> int:
|
|
117
|
+
ap = argparse.ArgumentParser()
|
|
118
|
+
ap.add_argument("--results", required=True)
|
|
119
|
+
ap.add_argument("--prices", required=True)
|
|
120
|
+
ap.add_argument("--analysis", required=True)
|
|
121
|
+
ap.add_argument("--output", required=True)
|
|
122
|
+
ap.add_argument("--title", default="Codex strategy benchmark report")
|
|
123
|
+
args = ap.parse_args()
|
|
124
|
+
|
|
125
|
+
rows = load_jsonl(Path(args.results))
|
|
126
|
+
prices = json.loads(Path(args.prices).read_text())
|
|
127
|
+
analysis = json.loads(Path(args.analysis).read_text())
|
|
128
|
+
|
|
129
|
+
total_cost = sum(cost(row, prices) for row in rows)
|
|
130
|
+
total_input = sum(row["input_tokens"] for row in rows)
|
|
131
|
+
total_cached = sum(row["cached_input_tokens"] for row in rows)
|
|
132
|
+
total_output = sum(row["output_tokens"] for row in rows)
|
|
133
|
+
total_tokens = total_input + total_output
|
|
134
|
+
net_new_input = total_input - total_cached
|
|
135
|
+
passed = sum(1 for row in rows if row["passed"])
|
|
136
|
+
first_passed = sum(1 for row in rows if row["first_passed"])
|
|
137
|
+
infra_failures = sum(1 for row in rows if row.get("codex_exit_code", 0) != 0)
|
|
138
|
+
repairs = sum(row["repair_cycles"] for row in rows)
|
|
139
|
+
reviews = sum(row["review_cycles"] for row in rows)
|
|
140
|
+
|
|
141
|
+
grouped: dict[str, list[dict[str, Any]]] = defaultdict(list)
|
|
142
|
+
for row in rows:
|
|
143
|
+
grouped[row["strategy_id"]].append(row)
|
|
144
|
+
strategy_tokens = {strategy_id: token_metrics(items) for strategy_id, items in grouped.items()}
|
|
145
|
+
sol_tokens = strategy_tokens.get("sol-direct")
|
|
146
|
+
|
|
147
|
+
lines = [
|
|
148
|
+
f"# {args.title}",
|
|
149
|
+
"",
|
|
150
|
+
"> Dollar figures are **API-equivalent reference costs** calculated from the pinned API price snapshot. Flow cost includes parent and worker usage. They are not ChatGPT subscription charges.",
|
|
151
|
+
"",
|
|
152
|
+
"## Overall",
|
|
153
|
+
"",
|
|
154
|
+
f"- Runs completed: **{len(rows)}**",
|
|
155
|
+
f"- Final pass: **{passed}/{len(rows)} ({pct(passed, len(rows))})**",
|
|
156
|
+
f"- First pass: **{first_passed}/{len(rows)} ({pct(first_passed, len(rows))})**",
|
|
157
|
+
f"- Infrastructure/CLI failures: **{infra_failures}**",
|
|
158
|
+
f"- Repair cycles: **{repairs}**",
|
|
159
|
+
f"- Parent review cycles: **{reviews}**",
|
|
160
|
+
f"- API-equivalent reference cost: **${total_cost:.6f}**",
|
|
161
|
+
f"- Total tokens: **{total_tokens:,}**",
|
|
162
|
+
f"- Input tokens: **{total_input:,}** — cached **{total_cached:,}**, net-new **{net_new_input:,}**",
|
|
163
|
+
f"- Output tokens: **{total_output:,}**",
|
|
164
|
+
"",
|
|
165
|
+
"## Strategy results",
|
|
166
|
+
"",
|
|
167
|
+
"| Strategy | Composition | Runs | Final pass | First pass | Repairs | Reviews | Avg wall | Avg reference cost |",
|
|
168
|
+
"| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |",
|
|
169
|
+
]
|
|
170
|
+
|
|
171
|
+
for strategy_id, items in sorted(grouped.items()):
|
|
172
|
+
item_passed = sum(1 for item in items if item["passed"])
|
|
173
|
+
item_first = sum(1 for item in items if item["first_passed"])
|
|
174
|
+
item_repairs = sum(item["repair_cycles"] for item in items)
|
|
175
|
+
item_reviews = sum(item["review_cycles"] for item in items)
|
|
176
|
+
item_cost = sum(cost(item, prices) for item in items)
|
|
177
|
+
item_wall = sum(item.get("wall_time_seconds", 0) for item in items) / len(items)
|
|
178
|
+
lines.append(
|
|
179
|
+
f"| {strategy_id} | {composition(items[0])} | {len(items)} | {pct(item_passed, len(items))} | "
|
|
180
|
+
f"{pct(item_first, len(items))} | {item_repairs} | {item_reviews} | {item_wall:.1f}s | ${item_cost / len(items):.6f} |"
|
|
181
|
+
)
|
|
182
|
+
|
|
183
|
+
lines.extend([
|
|
184
|
+
"",
|
|
185
|
+
"## Token efficiency",
|
|
186
|
+
"",
|
|
187
|
+
"This table separates raw token volume from where the work ran. Runtime Parent/Worker values come from FlowPilot role-attributed telemetry; net-new input excludes cached input. Deltas use `sol-direct` as the paired high-capability baseline when present.",
|
|
188
|
+
"",
|
|
189
|
+
"| Strategy | Avg total tokens | Avg Parent tokens | Avg Worker tokens | Avg cached input | Avg net-new input | Cache rate | Total Δ vs Sol | Net-new Δ vs Sol | Parent reduction vs Sol |",
|
|
190
|
+
"| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |",
|
|
191
|
+
])
|
|
192
|
+
for strategy_id, items in sorted(grouped.items()):
|
|
193
|
+
metrics = strategy_tokens[strategy_id]
|
|
194
|
+
if sol_tokens is not None:
|
|
195
|
+
total_delta = relative_change(metrics["total_tokens"], sol_tokens["total_tokens"])
|
|
196
|
+
net_new_delta = relative_change(metrics["net_new_input_tokens"], sol_tokens["net_new_input_tokens"])
|
|
197
|
+
else:
|
|
198
|
+
total_delta = net_new_delta = None
|
|
199
|
+
if items[0]["strategy"] != "direct" and sol_tokens is not None:
|
|
200
|
+
parent_reduction = 1 - metrics["parent_tokens"] / sol_tokens["total_tokens"] if sol_tokens["total_tokens"] else None
|
|
201
|
+
else:
|
|
202
|
+
parent_reduction = None
|
|
203
|
+
lines.append(
|
|
204
|
+
f"| {strategy_id} | {fmt_avg_tokens(metrics['total_tokens'])} | {fmt_avg_tokens(metrics['parent_tokens'])} | "
|
|
205
|
+
f"{fmt_avg_tokens(metrics['worker_tokens'])} | {fmt_avg_tokens(metrics['cached_input_tokens'])} | "
|
|
206
|
+
f"{fmt_avg_tokens(metrics['net_new_input_tokens'])} | {metrics['cache_rate']:.1%} | "
|
|
207
|
+
f"{fmt_change(total_delta)} | {fmt_change(net_new_delta)} | {fmt_change(parent_reduction)} |"
|
|
208
|
+
)
|
|
209
|
+
|
|
210
|
+
lines.extend([
|
|
211
|
+
"",
|
|
212
|
+
"## Sol capability evidence",
|
|
213
|
+
"",
|
|
214
|
+
"Sol is compared only with other direct strategies at the same reasoning effort.",
|
|
215
|
+
"",
|
|
216
|
+
"| Class | Comparator | Pass gain | First-pass gain | Repair reduction | Enough evidence | Advantage |",
|
|
217
|
+
"| --- | --- | ---: | ---: | ---: | ---: | ---: |",
|
|
218
|
+
])
|
|
219
|
+
for task_class in ("routine", "complex", "critical"):
|
|
220
|
+
item = analysis.get("sol_capability_evidence", {}).get(task_class)
|
|
221
|
+
if not item:
|
|
222
|
+
lines.append(f"| {task_class} | n/a | n/a | n/a | n/a | no | no |")
|
|
223
|
+
continue
|
|
224
|
+
lines.append(
|
|
225
|
+
f"| {task_class} | {item['competitor_strategy_id']} | {signed_pct(item['pass_rate_gain'])} | "
|
|
226
|
+
f"{signed_pct(item['first_pass_rate_gain'])} | {item['average_repair_reduction']:+.2f} | "
|
|
227
|
+
f"{yes_no(item['evidence_sufficient'])} | {yes_no(item['advantage_demonstrated'])} |"
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
lines.extend([
|
|
231
|
+
"",
|
|
232
|
+
"## Fixed-high flow evidence",
|
|
233
|
+
"",
|
|
234
|
+
"Flow must preserve Sol quality, reduce total parent+worker cost versus Sol, and improve over Luna direct.",
|
|
235
|
+
"",
|
|
236
|
+
"| Class | Pass Δ vs Sol | Cost reduction vs Sol | Pass gain vs Luna | First-pass gain vs Luna | Enough evidence | Advantage |",
|
|
237
|
+
"| --- | ---: | ---: | ---: | ---: | ---: | ---: |",
|
|
238
|
+
])
|
|
239
|
+
for task_class in ("routine", "complex", "critical"):
|
|
240
|
+
item = analysis.get("flow_advantage_evidence", {}).get(task_class)
|
|
241
|
+
if not item:
|
|
242
|
+
lines.append(f"| {task_class} | n/a | n/a | n/a | n/a | no | no |")
|
|
243
|
+
continue
|
|
244
|
+
lines.append(
|
|
245
|
+
f"| {task_class} | {signed_pct(item['pass_rate_delta_vs_sol'])} | {item['cost_reduction_vs_sol']:.1%} | "
|
|
246
|
+
f"{signed_pct(item['pass_rate_gain_vs_worker'])} | {signed_pct(item['first_pass_rate_gain_vs_worker'])} | "
|
|
247
|
+
f"{yes_no(item['evidence_sufficient'])} | {yes_no(item['advantage_demonstrated'])} |"
|
|
248
|
+
)
|
|
249
|
+
|
|
250
|
+
lines.extend([
|
|
251
|
+
"",
|
|
252
|
+
"## Adaptive reasoning evidence",
|
|
253
|
+
"",
|
|
254
|
+
"Adaptive flow is compared only with fixed-high flow and is excluded from the same-effort Sol comparison.",
|
|
255
|
+
"",
|
|
256
|
+
"| Class | Pass gain | First-pass gain | Repair reduction | Cost change | Wall-time change | Enough evidence | Value |",
|
|
257
|
+
"| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |",
|
|
258
|
+
])
|
|
259
|
+
for task_class in ("routine", "complex", "critical"):
|
|
260
|
+
item = analysis.get("adaptive_reasoning_evidence", {}).get(task_class)
|
|
261
|
+
if not item:
|
|
262
|
+
lines.append(f"| {task_class} | n/a | n/a | n/a | n/a | n/a | no | no |")
|
|
263
|
+
continue
|
|
264
|
+
lines.append(
|
|
265
|
+
f"| {task_class} | {signed_pct(item['pass_rate_gain'])} | {signed_pct(item['first_pass_rate_gain'])} | "
|
|
266
|
+
f"{item['average_repair_reduction']:+.2f} | {signed_pct(item['cost_change'])} | {signed_pct(item['wall_time_change'])} | "
|
|
267
|
+
f"{yes_no(item['evidence_sufficient'])} | {yes_no(item['value_demonstrated'])} |"
|
|
268
|
+
)
|
|
269
|
+
|
|
270
|
+
lines.extend(["", "## Advisory routing", ""])
|
|
271
|
+
recommendations = analysis.get("recommendations", {})
|
|
272
|
+
for task_class in ("routine", "complex", "critical"):
|
|
273
|
+
rec = recommendations.get(task_class)
|
|
274
|
+
if rec:
|
|
275
|
+
lines.append(
|
|
276
|
+
f"- **{task_class}**: `{rec['strategy_id']}` — pass {rec['pass_rate']:.1%}, "
|
|
277
|
+
f"API-equivalent ${rec['average_cost_usd']:.6f}/run, n={rec['samples']}"
|
|
278
|
+
)
|
|
279
|
+
else:
|
|
280
|
+
lines.append(f"- **{task_class}**: no tested strategy passed the evidence gate")
|
|
281
|
+
|
|
282
|
+
lines.extend([
|
|
283
|
+
"",
|
|
284
|
+
"> Conclusions are advisory and pre-registered thresholds are applied before cost comparison. Benchmark policy is not modified automatically.",
|
|
285
|
+
"",
|
|
286
|
+
])
|
|
287
|
+
Path(args.output).write_text("\n".join(lines))
|
|
288
|
+
return 0
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
if __name__ == "__main__":
|
|
292
|
+
raise SystemExit(main())
|