codex-flow 2.1.13__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. codex_flow/__init__.py +28 -0
  2. codex_flow/__main__.py +9 -0
  3. codex_flow/cli.py +242 -0
  4. codex_flow/data/LICENSE +21 -0
  5. codex_flow/data/README.en.md +303 -0
  6. codex_flow/data/README.md +305 -0
  7. codex_flow/data/VERSION +1 -0
  8. codex_flow/data/apps/chatgpt-mcp/README.md +86 -0
  9. codex_flow/data/apps/chatgpt-mcp/__init__.py +1 -0
  10. codex_flow/data/apps/chatgpt-mcp/adapter.py +458 -0
  11. codex_flow/data/apps/chatgpt-mcp/server.py +358 -0
  12. codex_flow/data/apps/chatgpt-mcp/widget.html +927 -0
  13. codex_flow/data/apps/macos-overlay/README.en.md +121 -0
  14. codex_flow/data/apps/macos-overlay/README.md +123 -0
  15. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayRuntimeState.swift +126 -0
  16. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayScreenGeometry.swift +82 -0
  17. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayWindowController.swift +1052 -0
  18. codex_flow/data/apps/macos-overlay/Sources/Localization.swift +197 -0
  19. codex_flow/data/apps/macos-overlay/Sources/Models/TelemetryData.swift +1557 -0
  20. codex_flow/data/apps/macos-overlay/Sources/Services/AccountSnapshotService.swift +1101 -0
  21. codex_flow/data/apps/macos-overlay/Sources/Services/FlowPilotInstanceLock.swift +153 -0
  22. codex_flow/data/apps/macos-overlay/Sources/Services/IPCServer.swift +298 -0
  23. codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryQueryEngine.swift +800 -0
  24. codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryWatcher.swift +135 -0
  25. codex_flow/data/apps/macos-overlay/Sources/Services/UpdateService.swift +610 -0
  26. codex_flow/data/apps/macos-overlay/Sources/Views/AccountView.swift +610 -0
  27. codex_flow/data/apps/macos-overlay/Sources/Views/AnalyticsView.swift +566 -0
  28. codex_flow/data/apps/macos-overlay/Sources/Views/AutostartView.swift +293 -0
  29. codex_flow/data/apps/macos-overlay/Sources/Views/BubbleView.swift +317 -0
  30. codex_flow/data/apps/macos-overlay/Sources/Views/HistoryView.swift +1124 -0
  31. codex_flow/data/apps/macos-overlay/Sources/Views/HoverRevealText.swift +165 -0
  32. codex_flow/data/apps/macos-overlay/Sources/Views/InspectorSkillsToolsView.swift +121 -0
  33. codex_flow/data/apps/macos-overlay/Sources/Views/LogoView.swift +182 -0
  34. codex_flow/data/apps/macos-overlay/Sources/Views/SleekSwitch.swift +117 -0
  35. codex_flow/data/apps/macos-overlay/Sources/Views/StrategyModeView.swift +561 -0
  36. codex_flow/data/apps/macos-overlay/Sources/Views/SummaryView.swift +1273 -0
  37. codex_flow/data/apps/macos-overlay/Sources/Views/UpdateView.swift +352 -0
  38. codex_flow/data/apps/macos-overlay/Sources/main.swift +340 -0
  39. codex_flow/data/apps/macos-overlay/Tests/OverlayScreenGeometryTests.swift +163 -0
  40. codex_flow/data/apps/macos-overlay/Tests/TelemetryPhase1ContractTests.swift +357 -0
  41. codex_flow/data/apps/macos-overlay/Tests/TelemetryQueryEngineConcurrencyTests.swift +221 -0
  42. codex_flow/data/apps/macos-overlay/Tests/TelemetryQuotaSelectionTests.swift +158 -0
  43. codex_flow/data/apps/macos-overlay/Tests/TelemetryWorkerTokenTests.swift +122 -0
  44. codex_flow/data/apps/macos-overlay/build.sh +75 -0
  45. codex_flow/data/benchmark/corpus.json +103 -0
  46. codex_flow/data/benchmark/manifest.example.json +41 -0
  47. codex_flow/data/benchmark/manifest.schema.json +137 -0
  48. codex_flow/data/benchmark/prices/gpt-5.6-2026-08-30.json +5 -0
  49. codex_flow/data/benchmark/profiles.json +90 -0
  50. codex_flow/data/benchmark/schema.json +77 -0
  51. codex_flow/data/benchmark/tasks.json +50 -0
  52. codex_flow/data/completions/codex-flow.bash +34 -0
  53. codex_flow/data/completions/codex-flow.zsh +52 -0
  54. codex_flow/data/glama.json +6 -0
  55. codex_flow/data/install-release.ps1 +126 -0
  56. codex_flow/data/install-release.sh +155 -0
  57. codex_flow/data/install.ps1 +349 -0
  58. codex_flow/data/install.sh +362 -0
  59. codex_flow/data/policy/benchmark.toml +49 -0
  60. codex_flow/data/policy/defaults.toml +70 -0
  61. codex_flow/data/scripts/analyze-benchmark.py +510 -0
  62. codex_flow/data/scripts/benchmark-local.py +171 -0
  63. codex_flow/data/scripts/check-recommendation.py +277 -0
  64. codex_flow/data/scripts/doctor.py +449 -0
  65. codex_flow/data/scripts/generate-release-manifest.py +74 -0
  66. codex_flow/data/scripts/localization.py +192 -0
  67. codex_flow/data/scripts/manage-hooks.py +448 -0
  68. codex_flow/data/scripts/manage-instructions.py +389 -0
  69. codex_flow/data/scripts/manage-shell.py +151 -0
  70. codex_flow/data/scripts/materialize-corpus.py +193 -0
  71. codex_flow/data/scripts/menu.py +646 -0
  72. codex_flow/data/scripts/migrations/0001_update_settings.py +80 -0
  73. codex_flow/data/scripts/package-release.py +132 -0
  74. codex_flow/data/scripts/render-benchmark-report.py +292 -0
  75. codex_flow/data/scripts/run-benchmark.py +829 -0
  76. codex_flow/data/scripts/strategies/__init__.py +28 -0
  77. codex_flow/data/scripts/strategies/balanced.py +115 -0
  78. codex_flow/data/scripts/strategies/base.py +363 -0
  79. codex_flow/data/scripts/strategies/efficient.py +158 -0
  80. codex_flow/data/scripts/strategies/lifecycle_runtime.py +590 -0
  81. codex_flow/data/scripts/strategies/quality.py +209 -0
  82. codex_flow/data/scripts/strategies/speed.py +108 -0
  83. codex_flow/data/scripts/strategies/task_budget_runtime.py +644 -0
  84. codex_flow/data/scripts/strategies/task_phase_runtime.py +341 -0
  85. codex_flow/data/scripts/strategies/work_unit_runtime.py +421 -0
  86. codex_flow/data/scripts/strategy_runtime.py +1091 -0
  87. codex_flow/data/scripts/telemetry.py +400 -0
  88. codex_flow/data/scripts/telemetry_core/__init__.py +192 -0
  89. codex_flow/data/scripts/telemetry_core/app_server.py +1192 -0
  90. codex_flow/data/scripts/telemetry_core/collector.py +1247 -0
  91. codex_flow/data/scripts/telemetry_core/common.py +421 -0
  92. codex_flow/data/scripts/telemetry_core/latency.py +593 -0
  93. codex_flow/data/scripts/telemetry_core/query.py +427 -0
  94. codex_flow/data/scripts/telemetry_core/quota_ledger.py +598 -0
  95. codex_flow/data/scripts/telemetry_core/render.py +460 -0
  96. codex_flow/data/scripts/telemetry_core/repair.py +223 -0
  97. codex_flow/data/scripts/ui.py +266 -0
  98. codex_flow/data/scripts/update-homebrew-formula.py +146 -0
  99. codex_flow/data/scripts/update_runtime_config.py +134 -0
  100. codex_flow/data/scripts/updater.py +1718 -0
  101. codex_flow/data/smithery.yaml +18 -0
  102. codex_flow/data/templates/agents/worker-explorer.toml +24 -0
  103. codex_flow/data/templates/agents/worker-implementer.toml +49 -0
  104. codex_flow/data/templates/agents/worker-reviewer.toml +25 -0
  105. codex_flow/data/templates/flow-pilot-instructions.md +35 -0
  106. codex_flow/data/templates/skills/flow-pilot/SKILL.md +577 -0
  107. codex_flow/mcp.py +35 -0
  108. codex_flow-2.1.13.dist-info/METADATA +342 -0
  109. codex_flow-2.1.13.dist-info/RECORD +113 -0
  110. codex_flow-2.1.13.dist-info/WHEEL +5 -0
  111. codex_flow-2.1.13.dist-info/entry_points.txt +3 -0
  112. codex_flow-2.1.13.dist-info/licenses/LICENSE +21 -0
  113. codex_flow-2.1.13.dist-info/top_level.txt +1 -0
@@ -0,0 +1,80 @@
1
+ #!/usr/bin/env python3
2
+ """Add OTA update preferences without rewriting unrelated policy fields."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import argparse
7
+ import os
8
+ import re
9
+ import tempfile
10
+ from pathlib import Path
11
+
12
+ DEFAULTS = {
13
+ "channel": '"stable"',
14
+ "check": "true",
15
+ "check_interval_hours": "24",
16
+ "notify_cli": "true",
17
+ "notify_app": "true",
18
+ "auto_install": "false",
19
+ }
20
+
21
+
22
+ def atomic_write(path: Path, text: str) -> None:
23
+ path.parent.mkdir(parents=True, exist_ok=True)
24
+ fd, name = tempfile.mkstemp(prefix=f".{path.name}.", dir=str(path.parent))
25
+ try:
26
+ with os.fdopen(fd, "w", encoding="utf-8", newline="\n") as handle:
27
+ handle.write(text)
28
+ handle.flush()
29
+ os.fsync(handle.fileno())
30
+ os.replace(name, path)
31
+ finally:
32
+ try:
33
+ os.unlink(name)
34
+ except FileNotFoundError:
35
+ pass
36
+
37
+
38
+ def migrate(policy: Path) -> bool:
39
+ try:
40
+ text = policy.read_text(encoding="utf-8-sig")
41
+ except OSError:
42
+ return False
43
+
44
+ section_re = re.compile(r"(?ms)^\[update\]\s*\n(.*?)(?=^\[[^\n]+\]\s*$|\Z)")
45
+ match = section_re.search(text)
46
+ changed = False
47
+ if not match:
48
+ if text and not text.endswith("\n"):
49
+ text += "\n"
50
+ if text and not text.endswith("\n\n"):
51
+ text += "\n"
52
+ text += "[update]\n" + "".join(f"{key} = {value}\n" for key, value in DEFAULTS.items())
53
+ changed = True
54
+ else:
55
+ body = match.group(1)
56
+ for key, value in DEFAULTS.items():
57
+ if re.search(rf"(?m)^\s*{re.escape(key)}\s*=", body):
58
+ continue
59
+ if body and not body.endswith("\n"):
60
+ body += "\n"
61
+ body += f"{key} = {value}\n"
62
+ changed = True
63
+ if changed:
64
+ text = text[: match.start(1)] + body + text[match.end(1) :]
65
+
66
+ if changed:
67
+ atomic_write(policy, text)
68
+ return changed
69
+
70
+
71
+ def main() -> int:
72
+ parser = argparse.ArgumentParser()
73
+ parser.add_argument("--policy", required=True, type=Path)
74
+ args = parser.parse_args()
75
+ migrate(args.policy)
76
+ return 0
77
+
78
+
79
+ if __name__ == "__main__":
80
+ raise SystemExit(main())
@@ -0,0 +1,132 @@
1
+ #!/usr/bin/env python3
2
+ """Build codex-flow OTA release archives."""
3
+
4
+ from __future__ import annotations
5
+
6
+ import argparse
7
+ import gzip
8
+ import hashlib
9
+ import os
10
+ import tarfile
11
+ import zipfile
12
+ from pathlib import Path
13
+
14
+ ROOT = Path(__file__).resolve().parents[1]
15
+
16
+ # Everything required after OTA switches STATE_DIR/source to the version package.
17
+ # Keep this list runtime-focused: CI/tests/.github stay source-only, while
18
+ # benchmark data and ChatGPT MCP must remain available after the original git
19
+ # checkout is no longer involved.
20
+ INCLUDE_ROOTS = (
21
+ "VERSION",
22
+ "LICENSE",
23
+ "README.md",
24
+ "README.en.md",
25
+ "install.sh",
26
+ "install.ps1",
27
+ "bin",
28
+ "scripts",
29
+ "templates",
30
+ "completions",
31
+ "policy",
32
+ "benchmark",
33
+ "apps/chatgpt-mcp",
34
+ )
35
+ DARWIN_ROOTS = (
36
+ "apps/macos-overlay",
37
+ )
38
+
39
+
40
+ def version() -> str:
41
+ return (ROOT / "VERSION").read_text(encoding="utf-8-sig").strip().lstrip("v")
42
+
43
+
44
+ def _collect(path: Path, seen: set[Path]) -> None:
45
+ if path.is_file() and not path.is_symlink():
46
+ seen.add(path)
47
+ return
48
+ if not path.is_dir():
49
+ return
50
+ for child in path.rglob("*"):
51
+ if (
52
+ child.is_file()
53
+ and not child.is_symlink()
54
+ and "__pycache__" not in child.parts
55
+ and child.suffix not in {".pyc", ".pyo"}
56
+ and child.name != ".DS_Store"
57
+ ):
58
+ seen.add(child)
59
+
60
+
61
+ def iter_files(platform_name: str) -> list[Path]:
62
+ seen: set[Path] = set()
63
+ for item in INCLUDE_ROOTS:
64
+ _collect(ROOT / item, seen)
65
+ if platform_name.startswith("darwin-"):
66
+ for item in DARWIN_ROOTS:
67
+ _collect(ROOT / item, seen)
68
+ return sorted(seen, key=lambda p: p.relative_to(ROOT).as_posix())
69
+
70
+
71
+ def sha256(path: Path) -> str:
72
+ digest = hashlib.sha256()
73
+ with path.open("rb") as handle:
74
+ for chunk in iter(lambda: handle.read(1024 * 1024), b""):
75
+ digest.update(chunk)
76
+ return digest.hexdigest()
77
+
78
+
79
+ def _mode(src: Path) -> int:
80
+ return 0o755 if os.access(src, os.X_OK) or src.suffix in {".sh", ".py"} else 0o644
81
+
82
+
83
+ def build(platform_name: str, output_dir: Path) -> Path:
84
+ output_dir.mkdir(parents=True, exist_ok=True)
85
+ ver = version()
86
+ prefix = f"codex-flow-{ver}"
87
+ files = iter_files(platform_name)
88
+ if platform_name.startswith("windows-"):
89
+ archive = output_dir / f"codex-flow-{ver}-{platform_name}.zip"
90
+ with zipfile.ZipFile(archive, "w", compression=zipfile.ZIP_DEFLATED, compresslevel=9) as zf:
91
+ for src in files:
92
+ rel = src.relative_to(ROOT).as_posix()
93
+ info = zipfile.ZipInfo(f"{prefix}/{rel}")
94
+ info.date_time = (2020, 1, 1, 0, 0, 0)
95
+ info.external_attr = (_mode(src) & 0xFFFF) << 16
96
+ info.compress_type = zipfile.ZIP_DEFLATED
97
+ zf.writestr(info, src.read_bytes())
98
+ else:
99
+ archive = output_dir / f"codex-flow-{ver}-{platform_name}.tar.gz"
100
+ # Pin the gzip header timestamp as well as tar metadata so rebuilds of
101
+ # identical inputs produce stable bytes/checksums.
102
+ with archive.open("wb") as raw, gzip.GzipFile(fileobj=raw, mode="wb", mtime=0) as gz:
103
+ with tarfile.open(fileobj=gz, mode="w", format=tarfile.PAX_FORMAT) as tf:
104
+ for src in files:
105
+ rel = src.relative_to(ROOT).as_posix()
106
+ arcname = f"{prefix}/{rel}"
107
+ info = tf.gettarinfo(str(src), arcname=arcname)
108
+ info.uid = info.gid = 0
109
+ info.uname = info.gname = "root"
110
+ info.mtime = 0
111
+ info.mode = _mode(src)
112
+ with src.open("rb") as handle:
113
+ tf.addfile(info, handle)
114
+ digest = sha256(archive)
115
+ (archive.with_suffix(archive.suffix + ".sha256")).write_text(
116
+ f"{digest} {archive.name}\n", encoding="utf-8"
117
+ )
118
+ print(archive)
119
+ return archive
120
+
121
+
122
+ def main() -> int:
123
+ parser = argparse.ArgumentParser()
124
+ parser.add_argument("--platform", required=True)
125
+ parser.add_argument("--output-dir", default="dist", type=Path)
126
+ args = parser.parse_args()
127
+ build(args.platform, args.output_dir)
128
+ return 0
129
+
130
+
131
+ if __name__ == "__main__":
132
+ raise SystemExit(main())
@@ -0,0 +1,292 @@
1
+ #!/usr/bin/env python3
2
+ from __future__ import annotations
3
+
4
+ import argparse
5
+ import json
6
+ from collections import defaultdict
7
+ from pathlib import Path
8
+ from typing import Any
9
+
10
+
11
+ def load_jsonl(path: Path) -> list[dict[str, Any]]:
12
+ rows = [json.loads(line) for line in path.read_text().splitlines() if line.strip()]
13
+ if not rows:
14
+ raise ValueError("benchmark results are empty")
15
+ for row in rows:
16
+ if row.get("schema_version") == 1:
17
+ suffix = row["model"].rsplit("-", 1)[-1]
18
+ if suffix in {"luna", "terra", "sol"}:
19
+ base = f"{suffix}-direct"
20
+ row["strategy_id"] = base if row["reasoning_effort"] == "high" else f"{base}-{row['reasoning_effort']}"
21
+ else:
22
+ row["strategy_id"] = f"direct:{row['model']}:{row['reasoning_effort']}"
23
+ row["strategy"] = "direct"
24
+ row["reasoning_policy"] = "fixed"
25
+ row["worker_model"] = None
26
+ row["worker_reasoning_effort"] = None
27
+ row["first_passed"] = row["passed"] and row["repair_cycles"] == 0
28
+ row["review_cycles"] = 0
29
+ row["model_usage"] = [{
30
+ "role": "direct",
31
+ "model": row["model"],
32
+ "reasoning_effort": row["reasoning_effort"],
33
+ "calls": row["repair_cycles"] + 1,
34
+ "input_tokens": row["input_tokens"],
35
+ "cached_input_tokens": row["cached_input_tokens"],
36
+ "output_tokens": row["output_tokens"],
37
+ }]
38
+ return rows
39
+
40
+
41
+ def usage_cost(usage: dict[str, Any], prices: dict[str, Any]) -> float:
42
+ model = usage["model"]
43
+ if model not in prices:
44
+ raise ValueError(f"missing price snapshot for {model}")
45
+ price = prices[model]
46
+ cached = usage["cached_input_tokens"]
47
+ uncached = max(0, usage["input_tokens"] - cached)
48
+ return (
49
+ uncached * price["input"]
50
+ + cached * price["cached_input"]
51
+ + usage["output_tokens"] * price["output"]
52
+ ) / 1_000_000
53
+
54
+
55
+ def cost(row: dict[str, Any], prices: dict[str, Any]) -> float:
56
+ return sum(usage_cost(usage, prices) for usage in row["model_usage"])
57
+
58
+
59
+ def pct(n: float, d: float) -> str:
60
+ return f"{(100 * n / d):.1f}%" if d else "n/a"
61
+
62
+
63
+ def signed_pct(value: float) -> str:
64
+ return f"{value:+.1%}"
65
+
66
+
67
+ def yes_no(value: bool) -> str:
68
+ return "yes" if value else "no"
69
+
70
+
71
+ def composition(row: dict[str, Any]) -> str:
72
+ if row["strategy"] == "direct":
73
+ return f"{row['model']} / {row['reasoning_effort']}"
74
+ policy = row.get("reasoning_policy", "fixed")
75
+ if policy == "adaptive":
76
+ return f"{row['model']} parent → {row['worker_model']} worker / adaptive"
77
+ return f"{row['model']} parent → {row['worker_model']} worker / {row['reasoning_effort']}"
78
+
79
+
80
+ def token_metrics(items: list[dict[str, Any]]) -> dict[str, float]:
81
+ count = len(items)
82
+ total_input = sum(item["input_tokens"] for item in items)
83
+ total_cached = sum(item["cached_input_tokens"] for item in items)
84
+ total_output = sum(item["output_tokens"] for item in items)
85
+ parent_tokens = 0
86
+ worker_tokens = 0
87
+ for item in items:
88
+ for usage in item.get("model_usage", []):
89
+ tokens = usage.get("input_tokens", 0) + usage.get("output_tokens", 0)
90
+ if usage.get("role") == "parent":
91
+ parent_tokens += tokens
92
+ elif usage.get("role") == "worker":
93
+ worker_tokens += tokens
94
+ return {
95
+ "total_tokens": (total_input + total_output) / count,
96
+ "parent_tokens": parent_tokens / count,
97
+ "worker_tokens": worker_tokens / count,
98
+ "cached_input_tokens": total_cached / count,
99
+ "net_new_input_tokens": (total_input - total_cached) / count,
100
+ "cache_rate": total_cached / total_input if total_input else 0.0,
101
+ }
102
+
103
+
104
+ def relative_change(value: float, reference: float) -> float | None:
105
+ return value / reference - 1 if reference else None
106
+
107
+
108
+ def fmt_avg_tokens(value: float) -> str:
109
+ return f"{round(value):,}"
110
+
111
+
112
+ def fmt_change(value: float | None) -> str:
113
+ return signed_pct(value) if value is not None else "n/a"
114
+
115
+
116
+ def main() -> int:
117
+ ap = argparse.ArgumentParser()
118
+ ap.add_argument("--results", required=True)
119
+ ap.add_argument("--prices", required=True)
120
+ ap.add_argument("--analysis", required=True)
121
+ ap.add_argument("--output", required=True)
122
+ ap.add_argument("--title", default="Codex strategy benchmark report")
123
+ args = ap.parse_args()
124
+
125
+ rows = load_jsonl(Path(args.results))
126
+ prices = json.loads(Path(args.prices).read_text())
127
+ analysis = json.loads(Path(args.analysis).read_text())
128
+
129
+ total_cost = sum(cost(row, prices) for row in rows)
130
+ total_input = sum(row["input_tokens"] for row in rows)
131
+ total_cached = sum(row["cached_input_tokens"] for row in rows)
132
+ total_output = sum(row["output_tokens"] for row in rows)
133
+ total_tokens = total_input + total_output
134
+ net_new_input = total_input - total_cached
135
+ passed = sum(1 for row in rows if row["passed"])
136
+ first_passed = sum(1 for row in rows if row["first_passed"])
137
+ infra_failures = sum(1 for row in rows if row.get("codex_exit_code", 0) != 0)
138
+ repairs = sum(row["repair_cycles"] for row in rows)
139
+ reviews = sum(row["review_cycles"] for row in rows)
140
+
141
+ grouped: dict[str, list[dict[str, Any]]] = defaultdict(list)
142
+ for row in rows:
143
+ grouped[row["strategy_id"]].append(row)
144
+ strategy_tokens = {strategy_id: token_metrics(items) for strategy_id, items in grouped.items()}
145
+ sol_tokens = strategy_tokens.get("sol-direct")
146
+
147
+ lines = [
148
+ f"# {args.title}",
149
+ "",
150
+ "> Dollar figures are **API-equivalent reference costs** calculated from the pinned API price snapshot. Flow cost includes parent and worker usage. They are not ChatGPT subscription charges.",
151
+ "",
152
+ "## Overall",
153
+ "",
154
+ f"- Runs completed: **{len(rows)}**",
155
+ f"- Final pass: **{passed}/{len(rows)} ({pct(passed, len(rows))})**",
156
+ f"- First pass: **{first_passed}/{len(rows)} ({pct(first_passed, len(rows))})**",
157
+ f"- Infrastructure/CLI failures: **{infra_failures}**",
158
+ f"- Repair cycles: **{repairs}**",
159
+ f"- Parent review cycles: **{reviews}**",
160
+ f"- API-equivalent reference cost: **${total_cost:.6f}**",
161
+ f"- Total tokens: **{total_tokens:,}**",
162
+ f"- Input tokens: **{total_input:,}** — cached **{total_cached:,}**, net-new **{net_new_input:,}**",
163
+ f"- Output tokens: **{total_output:,}**",
164
+ "",
165
+ "## Strategy results",
166
+ "",
167
+ "| Strategy | Composition | Runs | Final pass | First pass | Repairs | Reviews | Avg wall | Avg reference cost |",
168
+ "| --- | --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |",
169
+ ]
170
+
171
+ for strategy_id, items in sorted(grouped.items()):
172
+ item_passed = sum(1 for item in items if item["passed"])
173
+ item_first = sum(1 for item in items if item["first_passed"])
174
+ item_repairs = sum(item["repair_cycles"] for item in items)
175
+ item_reviews = sum(item["review_cycles"] for item in items)
176
+ item_cost = sum(cost(item, prices) for item in items)
177
+ item_wall = sum(item.get("wall_time_seconds", 0) for item in items) / len(items)
178
+ lines.append(
179
+ f"| {strategy_id} | {composition(items[0])} | {len(items)} | {pct(item_passed, len(items))} | "
180
+ f"{pct(item_first, len(items))} | {item_repairs} | {item_reviews} | {item_wall:.1f}s | ${item_cost / len(items):.6f} |"
181
+ )
182
+
183
+ lines.extend([
184
+ "",
185
+ "## Token efficiency",
186
+ "",
187
+ "This table separates raw token volume from where the work ran. Runtime Parent/Worker values come from FlowPilot role-attributed telemetry; net-new input excludes cached input. Deltas use `sol-direct` as the paired high-capability baseline when present.",
188
+ "",
189
+ "| Strategy | Avg total tokens | Avg Parent tokens | Avg Worker tokens | Avg cached input | Avg net-new input | Cache rate | Total Δ vs Sol | Net-new Δ vs Sol | Parent reduction vs Sol |",
190
+ "| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: | ---: |",
191
+ ])
192
+ for strategy_id, items in sorted(grouped.items()):
193
+ metrics = strategy_tokens[strategy_id]
194
+ if sol_tokens is not None:
195
+ total_delta = relative_change(metrics["total_tokens"], sol_tokens["total_tokens"])
196
+ net_new_delta = relative_change(metrics["net_new_input_tokens"], sol_tokens["net_new_input_tokens"])
197
+ else:
198
+ total_delta = net_new_delta = None
199
+ if items[0]["strategy"] != "direct" and sol_tokens is not None:
200
+ parent_reduction = 1 - metrics["parent_tokens"] / sol_tokens["total_tokens"] if sol_tokens["total_tokens"] else None
201
+ else:
202
+ parent_reduction = None
203
+ lines.append(
204
+ f"| {strategy_id} | {fmt_avg_tokens(metrics['total_tokens'])} | {fmt_avg_tokens(metrics['parent_tokens'])} | "
205
+ f"{fmt_avg_tokens(metrics['worker_tokens'])} | {fmt_avg_tokens(metrics['cached_input_tokens'])} | "
206
+ f"{fmt_avg_tokens(metrics['net_new_input_tokens'])} | {metrics['cache_rate']:.1%} | "
207
+ f"{fmt_change(total_delta)} | {fmt_change(net_new_delta)} | {fmt_change(parent_reduction)} |"
208
+ )
209
+
210
+ lines.extend([
211
+ "",
212
+ "## Sol capability evidence",
213
+ "",
214
+ "Sol is compared only with other direct strategies at the same reasoning effort.",
215
+ "",
216
+ "| Class | Comparator | Pass gain | First-pass gain | Repair reduction | Enough evidence | Advantage |",
217
+ "| --- | --- | ---: | ---: | ---: | ---: | ---: |",
218
+ ])
219
+ for task_class in ("routine", "complex", "critical"):
220
+ item = analysis.get("sol_capability_evidence", {}).get(task_class)
221
+ if not item:
222
+ lines.append(f"| {task_class} | n/a | n/a | n/a | n/a | no | no |")
223
+ continue
224
+ lines.append(
225
+ f"| {task_class} | {item['competitor_strategy_id']} | {signed_pct(item['pass_rate_gain'])} | "
226
+ f"{signed_pct(item['first_pass_rate_gain'])} | {item['average_repair_reduction']:+.2f} | "
227
+ f"{yes_no(item['evidence_sufficient'])} | {yes_no(item['advantage_demonstrated'])} |"
228
+ )
229
+
230
+ lines.extend([
231
+ "",
232
+ "## Fixed-high flow evidence",
233
+ "",
234
+ "Flow must preserve Sol quality, reduce total parent+worker cost versus Sol, and improve over Luna direct.",
235
+ "",
236
+ "| Class | Pass Δ vs Sol | Cost reduction vs Sol | Pass gain vs Luna | First-pass gain vs Luna | Enough evidence | Advantage |",
237
+ "| --- | ---: | ---: | ---: | ---: | ---: | ---: |",
238
+ ])
239
+ for task_class in ("routine", "complex", "critical"):
240
+ item = analysis.get("flow_advantage_evidence", {}).get(task_class)
241
+ if not item:
242
+ lines.append(f"| {task_class} | n/a | n/a | n/a | n/a | no | no |")
243
+ continue
244
+ lines.append(
245
+ f"| {task_class} | {signed_pct(item['pass_rate_delta_vs_sol'])} | {item['cost_reduction_vs_sol']:.1%} | "
246
+ f"{signed_pct(item['pass_rate_gain_vs_worker'])} | {signed_pct(item['first_pass_rate_gain_vs_worker'])} | "
247
+ f"{yes_no(item['evidence_sufficient'])} | {yes_no(item['advantage_demonstrated'])} |"
248
+ )
249
+
250
+ lines.extend([
251
+ "",
252
+ "## Adaptive reasoning evidence",
253
+ "",
254
+ "Adaptive flow is compared only with fixed-high flow and is excluded from the same-effort Sol comparison.",
255
+ "",
256
+ "| Class | Pass gain | First-pass gain | Repair reduction | Cost change | Wall-time change | Enough evidence | Value |",
257
+ "| --- | ---: | ---: | ---: | ---: | ---: | ---: | ---: |",
258
+ ])
259
+ for task_class in ("routine", "complex", "critical"):
260
+ item = analysis.get("adaptive_reasoning_evidence", {}).get(task_class)
261
+ if not item:
262
+ lines.append(f"| {task_class} | n/a | n/a | n/a | n/a | n/a | no | no |")
263
+ continue
264
+ lines.append(
265
+ f"| {task_class} | {signed_pct(item['pass_rate_gain'])} | {signed_pct(item['first_pass_rate_gain'])} | "
266
+ f"{item['average_repair_reduction']:+.2f} | {signed_pct(item['cost_change'])} | {signed_pct(item['wall_time_change'])} | "
267
+ f"{yes_no(item['evidence_sufficient'])} | {yes_no(item['value_demonstrated'])} |"
268
+ )
269
+
270
+ lines.extend(["", "## Advisory routing", ""])
271
+ recommendations = analysis.get("recommendations", {})
272
+ for task_class in ("routine", "complex", "critical"):
273
+ rec = recommendations.get(task_class)
274
+ if rec:
275
+ lines.append(
276
+ f"- **{task_class}**: `{rec['strategy_id']}` — pass {rec['pass_rate']:.1%}, "
277
+ f"API-equivalent ${rec['average_cost_usd']:.6f}/run, n={rec['samples']}"
278
+ )
279
+ else:
280
+ lines.append(f"- **{task_class}**: no tested strategy passed the evidence gate")
281
+
282
+ lines.extend([
283
+ "",
284
+ "> Conclusions are advisory and pre-registered thresholds are applied before cost comparison. Benchmark policy is not modified automatically.",
285
+ "",
286
+ ])
287
+ Path(args.output).write_text("\n".join(lines))
288
+ return 0
289
+
290
+
291
+ if __name__ == "__main__":
292
+ raise SystemExit(main())