codex-flow 2.1.13__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (113) hide show
  1. codex_flow/__init__.py +28 -0
  2. codex_flow/__main__.py +9 -0
  3. codex_flow/cli.py +242 -0
  4. codex_flow/data/LICENSE +21 -0
  5. codex_flow/data/README.en.md +303 -0
  6. codex_flow/data/README.md +305 -0
  7. codex_flow/data/VERSION +1 -0
  8. codex_flow/data/apps/chatgpt-mcp/README.md +86 -0
  9. codex_flow/data/apps/chatgpt-mcp/__init__.py +1 -0
  10. codex_flow/data/apps/chatgpt-mcp/adapter.py +458 -0
  11. codex_flow/data/apps/chatgpt-mcp/server.py +358 -0
  12. codex_flow/data/apps/chatgpt-mcp/widget.html +927 -0
  13. codex_flow/data/apps/macos-overlay/README.en.md +121 -0
  14. codex_flow/data/apps/macos-overlay/README.md +123 -0
  15. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayRuntimeState.swift +126 -0
  16. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayScreenGeometry.swift +82 -0
  17. codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayWindowController.swift +1052 -0
  18. codex_flow/data/apps/macos-overlay/Sources/Localization.swift +197 -0
  19. codex_flow/data/apps/macos-overlay/Sources/Models/TelemetryData.swift +1557 -0
  20. codex_flow/data/apps/macos-overlay/Sources/Services/AccountSnapshotService.swift +1101 -0
  21. codex_flow/data/apps/macos-overlay/Sources/Services/FlowPilotInstanceLock.swift +153 -0
  22. codex_flow/data/apps/macos-overlay/Sources/Services/IPCServer.swift +298 -0
  23. codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryQueryEngine.swift +800 -0
  24. codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryWatcher.swift +135 -0
  25. codex_flow/data/apps/macos-overlay/Sources/Services/UpdateService.swift +610 -0
  26. codex_flow/data/apps/macos-overlay/Sources/Views/AccountView.swift +610 -0
  27. codex_flow/data/apps/macos-overlay/Sources/Views/AnalyticsView.swift +566 -0
  28. codex_flow/data/apps/macos-overlay/Sources/Views/AutostartView.swift +293 -0
  29. codex_flow/data/apps/macos-overlay/Sources/Views/BubbleView.swift +317 -0
  30. codex_flow/data/apps/macos-overlay/Sources/Views/HistoryView.swift +1124 -0
  31. codex_flow/data/apps/macos-overlay/Sources/Views/HoverRevealText.swift +165 -0
  32. codex_flow/data/apps/macos-overlay/Sources/Views/InspectorSkillsToolsView.swift +121 -0
  33. codex_flow/data/apps/macos-overlay/Sources/Views/LogoView.swift +182 -0
  34. codex_flow/data/apps/macos-overlay/Sources/Views/SleekSwitch.swift +117 -0
  35. codex_flow/data/apps/macos-overlay/Sources/Views/StrategyModeView.swift +561 -0
  36. codex_flow/data/apps/macos-overlay/Sources/Views/SummaryView.swift +1273 -0
  37. codex_flow/data/apps/macos-overlay/Sources/Views/UpdateView.swift +352 -0
  38. codex_flow/data/apps/macos-overlay/Sources/main.swift +340 -0
  39. codex_flow/data/apps/macos-overlay/Tests/OverlayScreenGeometryTests.swift +163 -0
  40. codex_flow/data/apps/macos-overlay/Tests/TelemetryPhase1ContractTests.swift +357 -0
  41. codex_flow/data/apps/macos-overlay/Tests/TelemetryQueryEngineConcurrencyTests.swift +221 -0
  42. codex_flow/data/apps/macos-overlay/Tests/TelemetryQuotaSelectionTests.swift +158 -0
  43. codex_flow/data/apps/macos-overlay/Tests/TelemetryWorkerTokenTests.swift +122 -0
  44. codex_flow/data/apps/macos-overlay/build.sh +75 -0
  45. codex_flow/data/benchmark/corpus.json +103 -0
  46. codex_flow/data/benchmark/manifest.example.json +41 -0
  47. codex_flow/data/benchmark/manifest.schema.json +137 -0
  48. codex_flow/data/benchmark/prices/gpt-5.6-2026-08-30.json +5 -0
  49. codex_flow/data/benchmark/profiles.json +90 -0
  50. codex_flow/data/benchmark/schema.json +77 -0
  51. codex_flow/data/benchmark/tasks.json +50 -0
  52. codex_flow/data/completions/codex-flow.bash +34 -0
  53. codex_flow/data/completions/codex-flow.zsh +52 -0
  54. codex_flow/data/glama.json +6 -0
  55. codex_flow/data/install-release.ps1 +126 -0
  56. codex_flow/data/install-release.sh +155 -0
  57. codex_flow/data/install.ps1 +349 -0
  58. codex_flow/data/install.sh +362 -0
  59. codex_flow/data/policy/benchmark.toml +49 -0
  60. codex_flow/data/policy/defaults.toml +70 -0
  61. codex_flow/data/scripts/analyze-benchmark.py +510 -0
  62. codex_flow/data/scripts/benchmark-local.py +171 -0
  63. codex_flow/data/scripts/check-recommendation.py +277 -0
  64. codex_flow/data/scripts/doctor.py +449 -0
  65. codex_flow/data/scripts/generate-release-manifest.py +74 -0
  66. codex_flow/data/scripts/localization.py +192 -0
  67. codex_flow/data/scripts/manage-hooks.py +448 -0
  68. codex_flow/data/scripts/manage-instructions.py +389 -0
  69. codex_flow/data/scripts/manage-shell.py +151 -0
  70. codex_flow/data/scripts/materialize-corpus.py +193 -0
  71. codex_flow/data/scripts/menu.py +646 -0
  72. codex_flow/data/scripts/migrations/0001_update_settings.py +80 -0
  73. codex_flow/data/scripts/package-release.py +132 -0
  74. codex_flow/data/scripts/render-benchmark-report.py +292 -0
  75. codex_flow/data/scripts/run-benchmark.py +829 -0
  76. codex_flow/data/scripts/strategies/__init__.py +28 -0
  77. codex_flow/data/scripts/strategies/balanced.py +115 -0
  78. codex_flow/data/scripts/strategies/base.py +363 -0
  79. codex_flow/data/scripts/strategies/efficient.py +158 -0
  80. codex_flow/data/scripts/strategies/lifecycle_runtime.py +590 -0
  81. codex_flow/data/scripts/strategies/quality.py +209 -0
  82. codex_flow/data/scripts/strategies/speed.py +108 -0
  83. codex_flow/data/scripts/strategies/task_budget_runtime.py +644 -0
  84. codex_flow/data/scripts/strategies/task_phase_runtime.py +341 -0
  85. codex_flow/data/scripts/strategies/work_unit_runtime.py +421 -0
  86. codex_flow/data/scripts/strategy_runtime.py +1091 -0
  87. codex_flow/data/scripts/telemetry.py +400 -0
  88. codex_flow/data/scripts/telemetry_core/__init__.py +192 -0
  89. codex_flow/data/scripts/telemetry_core/app_server.py +1192 -0
  90. codex_flow/data/scripts/telemetry_core/collector.py +1247 -0
  91. codex_flow/data/scripts/telemetry_core/common.py +421 -0
  92. codex_flow/data/scripts/telemetry_core/latency.py +593 -0
  93. codex_flow/data/scripts/telemetry_core/query.py +427 -0
  94. codex_flow/data/scripts/telemetry_core/quota_ledger.py +598 -0
  95. codex_flow/data/scripts/telemetry_core/render.py +460 -0
  96. codex_flow/data/scripts/telemetry_core/repair.py +223 -0
  97. codex_flow/data/scripts/ui.py +266 -0
  98. codex_flow/data/scripts/update-homebrew-formula.py +146 -0
  99. codex_flow/data/scripts/update_runtime_config.py +134 -0
  100. codex_flow/data/scripts/updater.py +1718 -0
  101. codex_flow/data/smithery.yaml +18 -0
  102. codex_flow/data/templates/agents/worker-explorer.toml +24 -0
  103. codex_flow/data/templates/agents/worker-implementer.toml +49 -0
  104. codex_flow/data/templates/agents/worker-reviewer.toml +25 -0
  105. codex_flow/data/templates/flow-pilot-instructions.md +35 -0
  106. codex_flow/data/templates/skills/flow-pilot/SKILL.md +577 -0
  107. codex_flow/mcp.py +35 -0
  108. codex_flow-2.1.13.dist-info/METADATA +342 -0
  109. codex_flow-2.1.13.dist-info/RECORD +113 -0
  110. codex_flow-2.1.13.dist-info/WHEEL +5 -0
  111. codex_flow-2.1.13.dist-info/entry_points.txt +3 -0
  112. codex_flow-2.1.13.dist-info/licenses/LICENSE +21 -0
  113. codex_flow-2.1.13.dist-info/top_level.txt +1 -0
@@ -0,0 +1,103 @@
1
+ {
2
+ "schema_version": 2,
3
+ "tasks": [
4
+ {
5
+ "id": "routine-query-normalization",
6
+ "class": "routine",
7
+ "description": "Localized bug fix with edge-case normalization and no API change.",
8
+ "prompt": "Fix search.normalize_query so leading/trailing whitespace is removed, internal whitespace runs collapse to one space, and Unicode text is preserved. Keep the public function name/signature unchanged. Add or update tests only if useful; do not modify anything outside the repository.",
9
+ "files": {
10
+ "search.py": "import re\n\ndef normalize_query(value: str) -> str:\n # BUG: strips only ASCII spaces and does not collapse runs.\n return value.strip(' ')\n",
11
+ "README.md": "# Query normalization\n\n`normalize_query` prepares user-entered search text while preserving Unicode content.\n"
12
+ },
13
+ "verifier": "from pathlib import Path\nimport importlib.util\nroot=Path.cwd()\nspec=importlib.util.spec_from_file_location('search', root/'search.py')\nm=importlib.util.module_from_spec(spec); spec.loader.exec_module(m)\ncases={\n ' hello world ':'hello world',\n '\\t hello\\nworld \\r':'hello world',\n ' \u5fae\u4fe1 \u8bed\u97f3 ':'\u5fae\u4fe1 \u8bed\u97f3',\n '':'',\n ' ':'',\n}\nfor raw,want in cases.items():\n got=m.normalize_query(raw)\n assert got==want,(raw,got,want)\nassert m.normalize_query.__name__=='normalize_query'\n"
14
+ },
15
+ {
16
+ "id": "routine-env-precedence",
17
+ "class": "routine",
18
+ "description": "Configuration precedence bug requiring exact fallback semantics.",
19
+ "prompt": "Fix config.resolve_endpoint. Explicit argument must win when non-empty; otherwise use SERVICE_ENDPOINT from the supplied env mapping when non-empty; otherwise return DEFAULT_ENDPOINT. Treat whitespace-only values as empty. Do not read os.environ implicitly and keep the existing public signature.",
20
+ "files": {
21
+ "config.py": "DEFAULT_ENDPOINT = 'https://api.example.test'\n\ndef resolve_endpoint(explicit, env):\n # BUG: env always wins and whitespace values leak through.\n return env.get('SERVICE_ENDPOINT') or explicit or DEFAULT_ENDPOINT\n"
22
+ },
23
+ "verifier": "from pathlib import Path\nimport importlib.util\nspec=importlib.util.spec_from_file_location('config',Path.cwd()/'config.py')\nm=importlib.util.module_from_spec(spec); spec.loader.exec_module(m)\nassert m.resolve_endpoint('https://explicit', {'SERVICE_ENDPOINT':'https://env'})=='https://explicit'\nassert m.resolve_endpoint(None, {'SERVICE_ENDPOINT':'https://env'})=='https://env'\nassert m.resolve_endpoint('', {'SERVICE_ENDPOINT':'https://env'})=='https://env'\nassert m.resolve_endpoint(' ', {'SERVICE_ENDPOINT':' '})==m.DEFAULT_ENDPOINT\nassert m.resolve_endpoint(None, {})==m.DEFAULT_ENDPOINT\n"
24
+ },
25
+ {
26
+ "id": "routine-header-sanitization",
27
+ "class": "routine",
28
+ "description": "HTTP header sanitization, case normalization, and injection protection.",
29
+ "prompt": "Fix headers.sanitize_headers(headers: dict). Header names must be lowercase stripped strings containing only valid RFC 7230 token characters (alphanumerics and !#$%&'*+-.^_`|~); reject invalid or empty names with ValueError. Header values must be stripped strings; reject any value containing CR (\\r) or LF (\\n) with ValueError; convert integer and float values to strings; and drop headers whose sanitized value is empty. Return a new dict with sanitized headers. Do not modify the input dict.",
30
+ "files": {
31
+ "headers.py": "def sanitize_headers(headers):\n # BUG: does not lower-case names, allows invalid tokens/CRLF injection, mutates input\n return {k: str(v) for k, v in headers.items()}\n"
32
+ },
33
+ "verifier": "from pathlib import Path\nimport importlib.util\nspec = importlib.util.spec_from_file_location('headers', Path.cwd() / 'headers.py')\nm = importlib.util.module_from_spec(spec)\nspec.loader.exec_module(m)\norig = {' Content-Type ': ' text/plain ', 'X-Rate-Limit': 100, 'X-Score': 3.14}\nout = m.sanitize_headers(orig)\nassert out == {'content-type': 'text/plain', 'x-rate-limit': '100', 'x-score': '3.14'}\nassert orig[' Content-Type '] == ' text/plain '\nassert m.sanitize_headers({'X-Empty': '', 'X-Space': ' ', 'Keep': 'val'}) == {'keep': 'val'}\nfor bad_name in ('', ' ', 'Bad Name', 'Header:Colon', 'Foo@Bar', 'Invalid/Slash'):\n try:\n m.sanitize_headers({bad_name: 'val'})\n except ValueError:\n pass\n else:\n raise AssertionError(f'invalid header name accepted: {bad_name!r}')\nfor bad_val in ('evil\\r\\nInjected: true', 'foo\\nbar', 'bar\\rbaz'):\n try:\n m.sanitize_headers({'x-safe': bad_val})\n except ValueError:\n pass\n else:\n raise AssertionError(f'CRLF injection accepted: {bad_val!r}')\n"
34
+ },
35
+ {
36
+ "id": "complex-renew-provider-refactor",
37
+ "class": "complex",
38
+ "description": "Multi-file registry refactor with isolation, validation, normalization, and legacy compatibility.",
39
+ "prompt": "Refactor the renewal package around a reusable ProviderRegistry without breaking the existing renew(name) API. ProviderRegistry.register(name, handler, *, replace=False) must trim and case-fold non-empty provider names, reject non-callables, reject duplicates unless replace=True, and keep registry instances isolated. ProviderRegistry.dispatch(provider, name) must use the same normalization, raise a clear ValueError containing the requested provider when unknown, and propagate handler exceptions unchanged. Keep a module-level registry in renew.service seeded with the legacy handler. Expose register_provider and renew_for through both renew.service and renew.__init__; renew(name) and consumer.run() must retain legacy behavior. Avoid hidden registration side effects outside the explicit module-level legacy seed.",
40
+ "files": {
41
+ "renew/__init__.py": "from .service import renew\n\n__all__ = ['renew']\n",
42
+ "renew/registry.py": "class ProviderRegistry:\n def __init__(self):\n self._handlers = {}\n\n def register(self, name, handler):\n self._handlers[name] = handler\n\n def dispatch(self, provider, name):\n return self._handlers[provider](name)\n",
43
+ "renew/service.py": "from .legacy import renew_legacy\n\ndef renew(name):\n return renew_legacy(name)\n",
44
+ "renew/legacy.py": "def renew_legacy(name):\n return {'provider':'legacy','name':name,'status':'renewed'}\n",
45
+ "consumer.py": "from renew import renew\n\ndef run():\n return renew('demo')\n"
46
+ },
47
+ "verifier": "from pathlib import Path\nimport sys\nsys.path.insert(0,str(Path.cwd()))\nimport renew\nfrom renew import service\nfrom renew.registry import ProviderRegistry\nassert renew.renew('a')=={'provider':'legacy','name':'a','status':'renewed'}\nassert renew.renew_for is service.renew_for and renew.register_provider is service.register_provider\nassert {'renew','renew_for','register_provider'} <= set(renew.__all__)\none=ProviderRegistry(); two=ProviderRegistry(); seen=[]\ndef handler(name): seen.append(name); return {'provider':'new','name':name}\none.register(' New ',handler)\nassert one.dispatch('new','b')=={'provider':'new','name':'b'} and seen==['b']\ntry: two.dispatch('new','x')\nexcept ValueError: pass\nelse: raise AssertionError('registry instances leaked state')\ntry: one.register('NEW',handler)\nexcept ValueError: pass\nelse: raise AssertionError('canonical duplicate accepted')\none.register('new',lambda name:{'provider':'replacement','name':name},replace=True)\nassert one.dispatch(' NEW ','c')['provider']=='replacement'\nfor bad in ('', ' ', None):\n try: one.register(bad,handler)\n except (ValueError,TypeError): pass\n else: raise AssertionError(('bad name accepted',bad))\ntry: one.register('bad',object())\nexcept TypeError: pass\nelse: raise AssertionError('non-callable accepted')\ntry: one.dispatch(' missing-provider ','x')\nexcept ValueError as e: assert 'missing-provider' in str(e)\nelse: raise AssertionError('unknown provider must fail')\nclass Boom(Exception): pass\none.register('boom',lambda name: (_ for _ in ()).throw(Boom(name)))\ntry: one.dispatch('boom','x')\nexcept Boom as e: assert str(e)=='x'\nelse: raise AssertionError('handler exception changed')\nservice.register_provider(' API ',handler)\nassert service.renew_for('api','z')=={'provider':'new','name':'z'}\nfrom consumer import run\nassert run()=={'provider':'legacy','name':'demo','status':'renewed'}\n"
48
+ },
49
+ {
50
+ "id": "complex-config-migration",
51
+ "class": "complex",
52
+ "description": "Idempotent, alias-safe configuration migration with per-key precedence and validation.",
53
+ "prompt": "Update settings.load_config to migrate the legacy flat endpoint/token keys into service.endpoint/service.token while accepting the new nested schema. New nested values win per key when the key is present; a missing nested key falls back to its legacy counterpart. Remove the consumed legacy keys, preserve unrelated top-level keys and unrelated service keys, and keep genuinely missing endpoint/token keys absent. The result must be a deep independent copy: never mutate or alias the input, including nested lists/dicts. Applying load_config to its own output must be idempotent. raw and an existing service value must be mappings; reject invalid shapes with TypeError before returning a partial result.",
54
+ "files": {
55
+ "settings.py": "def load_config(raw):\n # Current implementation only understands the legacy flat shape.\n result = dict(raw)\n service = {}\n if 'endpoint' in result:\n service['endpoint'] = result.pop('endpoint')\n if 'token' in result:\n service['token'] = result.pop('token')\n result['service'] = service\n return result\n"
56
+ },
57
+ "verifier": "from pathlib import Path\nfrom collections import UserDict\nimport importlib.util,copy\nspec=importlib.util.spec_from_file_location('settings',Path.cwd()/'settings.py')\nm=importlib.util.module_from_spec(spec); spec.loader.exec_module(m)\nlegacy={'endpoint':'e1','token':'t1','feature':{'flags':['a']}}; before=copy.deepcopy(legacy)\nout=m.load_config(legacy); assert legacy==before; assert out=={'service':{'endpoint':'e1','token':'t1'},'feature':{'flags':['a']}}\nout['feature']['flags'].append('b'); assert legacy==before\nnew={'service':{'endpoint':'e2','token':'t2','headers':{'x':['1']}},'feature':False}; before=copy.deepcopy(new)\nout=m.load_config(new); assert new==before and out==new and out is not new and out['service'] is not new['service']\nout['service']['headers']['x'].append('2'); assert new==before\nmixed={'endpoint':'old','token':'oldt','service':{'endpoint':None,'headers':{'a':1}},'x':1}\nout=m.load_config(mixed); assert out['service']=={'endpoint':None,'token':'oldt','headers':{'a':1}}; assert out['x']==1; assert 'endpoint' not in out and 'token' not in out\nempty=m.load_config({'x':1}); assert empty in ({'x':1},{'x':1,'service':{}}); assert 'endpoint' not in empty.get('service',{}) and 'token' not in empty.get('service',{})\nwrapped=UserDict({'endpoint':'e','service':{'token':'t'}}); assert m.load_config(wrapped)['service']=={'endpoint':'e','token':'t'}\nonce=m.load_config({'endpoint':'e','token':'t','nested':{'v':[1]}}); twice=m.load_config(once); assert twice==once and twice is not once and twice['nested'] is not once['nested']\nfor bad in (None,[],42):\n try: m.load_config(bad)\n except TypeError: pass\n else: raise AssertionError(('invalid raw accepted',bad))\nfor bad_service in (None,[],42,'x'):\n try: m.load_config({'service':bad_service,'endpoint':'e'})\n except TypeError: pass\n else: raise AssertionError(('invalid service accepted',bad_service))\n"
58
+ },
59
+ {
60
+ "id": "complex-dag-resolver",
61
+ "class": "complex",
62
+ "description": "Topological dependency graph resolver with cycle detection, tie-breaking, and batch scheduling.",
63
+ "prompt": "Implement a multi-stage dependency graph scheduler in the scheduler package. scheduler.graph.DependencyGraph allows adding nodes with dependencies via add_node(name, dependencies=None) where dependencies is an optional iterable of string node names; names and dependencies must be stripped non-empty strings. scheduler.resolver.resolve_order(graph) returns a deterministic topologically sorted list of node names; when multiple nodes are ready, tie-break by lexicographical order; raise scheduler.errors.CycleDetectedError (which stores the cycle path list in exc.cycle) if a dependency cycle exists, and raise scheduler.errors.MissingDependencyError (storing exc.missing_node) if a dependency is not registered in the graph. scheduler.resolver.resolve_batches(graph) returns list[list[str]] grouping independent nodes that can execute concurrently in each stage, with each stage's node list sorted lexicographically. Expose DependencyGraph, resolve_order, resolve_batches, CycleDetectedError, and MissingDependencyError in scheduler.__init__.",
64
+ "files": {
65
+ "scheduler/__init__.py": "from .graph import DependencyGraph\nfrom .resolver import resolve_order, resolve_batches\nfrom .errors import CycleDetectedError, MissingDependencyError\n\n__all__ = [\n 'DependencyGraph', 'resolve_order', 'resolve_batches',\n 'CycleDetectedError', 'MissingDependencyError',\n]\n",
66
+ "scheduler/errors.py": "class CycleDetectedError(ValueError):\n def __init__(self, cycle=None):\n super().__init__(f'Cycle detected: {cycle}')\n self.cycle = list(cycle) if cycle else []\n\nclass MissingDependencyError(KeyError):\n def __init__(self, missing_node=None):\n super().__init__(f'Missing dependency: {missing_node}')\n self.missing_node = missing_node\n",
67
+ "scheduler/graph.py": "class DependencyGraph:\n def __init__(self):\n self.nodes = {}\n\n def add_node(self, name, dependencies=None):\n # BUG: does not strip/validate names and stores dependencies as-is\n self.nodes[name] = list(dependencies or [])\n",
68
+ "scheduler/resolver.py": "# BUG: naive incomplete resolver without tie-breaking, cycle path detection, or batching\ndef resolve_order(graph):\n return list(graph.nodes.keys())\n\ndef resolve_batches(graph):\n return [[k] for k in graph.nodes.keys()]\n"
69
+ },
70
+ "verifier": "from pathlib import Path\nimport sys\nsys.path.insert(0, str(Path.cwd()))\nimport scheduler\nfrom scheduler import DependencyGraph, resolve_order, resolve_batches, CycleDetectedError, MissingDependencyError\nassert {'DependencyGraph', 'resolve_order', 'resolve_batches', 'CycleDetectedError', 'MissingDependencyError'} <= set(scheduler.__all__)\ng = DependencyGraph()\ng.add_node('c', ['a', 'b']); g.add_node('b'); g.add_node('a'); g.add_node('d', ['c'])\nassert resolve_order(g) == ['a', 'b', 'c', 'd']\nassert resolve_batches(g) == [['a', 'b'], ['c'], ['d']]\ng2 = DependencyGraph()\nfor name in ('delta', 'beta', 'alpha', 'gamma'): g2.add_node(name)\nassert resolve_order(g2) == ['alpha', 'beta', 'delta', 'gamma']\nassert resolve_batches(g2) == [['alpha', 'beta', 'delta', 'gamma']]\ng3 = DependencyGraph(); g3.add_node('task1', ['ghost'])\ntry:\n resolve_order(g3)\nexcept MissingDependencyError as exc:\n assert exc.missing_node == 'ghost'\nelse:\n raise AssertionError('missing dependency error not raised')\ng4 = DependencyGraph()\ng4.add_node('p1', ['p2']); g4.add_node('p2', ['p3']); g4.add_node('p3', ['p1'])\ntry:\n resolve_order(g4)\nexcept CycleDetectedError as exc:\n assert len(exc.cycle) >= 2\nelse:\n raise AssertionError('cycle detected error not raised')\nfor bad in ('', ' ', None):\n try: DependencyGraph().add_node(bad)\n except (ValueError, TypeError): pass\n else: raise AssertionError(f'bad node name accepted: {bad!r}')\n"
71
+ },
72
+ {
73
+ "id": "critical-resumable-migration",
74
+ "class": "critical",
75
+ "description": "Crash-resumable record migration with preflight validation and durable progress tracking.",
76
+ "prompt": "Implement migration.run_migration(store, journal_path). The store exposes list_ids(), read(record_id), and write_atomic(record_id, value). Preflight every record before writing anything: each record must be a mapping with schema 1 or 2; schema-1 records require a non-empty string name, while schema-2 records require profile.name. Convert schema 1 to schema 2 by moving name to profile.name, preserving enabled and unrelated fields, and never mutating objects returned by the store. Process record ids in deterministic sorted order. After each successful write, atomically persist a JSON journal {'completed':[...]} beside journal_path using a flushed/fsynced temporary file and os.replace. On restart, a valid journal may skip an id only when the current record is already schema 2; stale completed entries whose record is still schema 1 must be processed. A failed write must not be journaled. Remove the journal after all records are safely migrated. Malformed records or journals must raise ValueError before any write. The migration must be idempotent.",
77
+ "files": {
78
+ "migration.py": "def run_migration(store, journal_path):\n # BUG: neither validates nor journals progress and mutates source values.\n for record_id in store.list_ids():\n value = store.read(record_id)\n if value.get('schema') == 1:\n value['schema'] = 2\n value['profile'] = {'name': value.pop('name')}\n store.write_atomic(record_id, value)\n"
79
+ },
80
+ "verifier": "from pathlib import Path\nimport copy,importlib.util,json,tempfile\nspec=importlib.util.spec_from_file_location('migration',Path.cwd()/'migration.py')\nm=importlib.util.module_from_spec(spec); spec.loader.exec_module(m)\nclass Store:\n def __init__(self,data,fail=None): self.data=copy.deepcopy(data); self.fail=fail; self.writes=[]; self.read_values=[]\n def list_ids(self): return list(reversed(list(self.data)))\n def read(self,k):\n value=copy.deepcopy(self.data[k]); self.read_values.append(value); return value\n def write_atomic(self,k,v):\n if self.fail==k: raise OSError('crash')\n self.writes.append(k); self.data[k]=copy.deepcopy(v)\nwith tempfile.TemporaryDirectory() as td:\n journal=Path(td)/'progress.json'; original={'b':{'schema':1,'name':'Bee','enabled':False,'x':1},'a':{'schema':2,'profile':{'name':'Ay'},'enabled':True}}\n store=Store(original); m.run_migration(store,journal)\n assert store.writes==['b']; assert store.data['b']=={'schema':2,'profile':{'name':'Bee'},'enabled':False,'x':1}; assert not journal.exists(); assert original['b']['schema']==1\n before=copy.deepcopy(store.data); m.run_migration(store,journal); assert store.data==before and store.writes==['b']\nwith tempfile.TemporaryDirectory() as td:\n journal=Path(td)/'progress.json'; store=Store({'b':{'schema':1,'name':'B'},'a':{'schema':1,'name':'A'}},fail='b')\n try: m.run_migration(store,journal)\n except OSError: pass\n else: raise AssertionError('write failure must propagate')\n assert store.writes==['a']; assert json.loads(journal.read_text())=={'completed':['a']}\n store.fail=None; m.run_migration(store,journal); assert store.writes==['a','b']; assert all(v['schema']==2 for v in store.data.values()); assert not journal.exists()\nwith tempfile.TemporaryDirectory() as td:\n journal=Path(td)/'progress.json'; journal.write_text('{\"completed\":[\"a\"]}')\n store=Store({'a':{'schema':1,'name':'A'}}); m.run_migration(store,journal); assert store.writes==['a'] and not journal.exists()\nfor bad in ({'a':{'schema':1,'name':''}},{'a':{'schema':2,'profile':{}}},{'a':{'schema':9,'name':'x'}},{'a':[],'b':{'schema':1,'name':'B'}}):\n with tempfile.TemporaryDirectory() as td:\n store=Store(bad); journal=Path(td)/'j.json'\n try: m.run_migration(store,journal)\n except ValueError: pass\n else: raise AssertionError(('invalid data accepted',bad))\n assert store.writes==[]\nwith tempfile.TemporaryDirectory() as td:\n journal=Path(td)/'j.json'; journal.write_text('{bad'); store=Store({'a':{'schema':1,'name':'A'}})\n try: m.run_migration(store,journal)\n except ValueError: pass\n else: raise AssertionError('malformed journal accepted')\n assert store.writes==[]\n"
81
+ },
82
+ {
83
+ "id": "critical-atomic-state-write",
84
+ "class": "critical",
85
+ "description": "Crash-durable atomic replacement with permission preservation, directory fsync, and failure cleanup.",
86
+ "prompt": "Harden state.save_json for crash-durable replacement. Serialize JSON to a uniquely named temporary file in the destination directory, flush and fsync the file, then atomically replace the destination with os.replace and fsync the parent directory after replacement. Preserve the existing destination's permission bits; for a new destination use mode 0600. Serialization, write/fsync, or replace failures must clean up every temporary file. A failure before a successful os.replace must leave an existing destination byte-for-byte unchanged. Do not follow an existing destination symlink when choosing permissions. Keep load_json behavior and deterministic sort_keys output compatible.",
87
+ "files": {
88
+ "state.py": "import json\nfrom pathlib import Path\n\ndef load_json(path):\n with open(path,'r',encoding='utf-8') as f:\n return json.load(f)\n\ndef save_json(path, value):\n # BUG: direct write can truncate valid state on failure/crash.\n with open(path,'w',encoding='utf-8') as f:\n json.dump(value,f,sort_keys=True)\n"
89
+ },
90
+ "verifier": "from pathlib import Path\nimport importlib.util,tempfile,os,stat,unittest.mock as mock\nspec=importlib.util.spec_from_file_location('state',Path.cwd()/'state.py')\nm=importlib.util.module_from_spec(spec); spec.loader.exec_module(m)\nwith tempfile.TemporaryDirectory() as td:\n p=Path(td)/'state.json'; p.write_text('{\"old\":1}',encoding='utf-8'); os.chmod(p,0o640)\n real_replace=os.replace; real_fsync=os.fsync; events=[]\n def wrapped_replace(src,dst): events.append(('replace',Path(src),Path(dst))); return real_replace(src,dst)\n def wrapped_fsync(fd): events.append(('fsync',stat.S_ISDIR(os.fstat(fd).st_mode))); return real_fsync(fd)\n with mock.patch.object(os,'replace',wrapped_replace),mock.patch.object(os,'fsync',wrapped_fsync): m.save_json(p,{'z':2,'a':1})\n assert p.read_text(encoding='utf-8')=='{\"a\": 1, \"z\": 2}' and stat.S_IMODE(os.stat(p).st_mode)==0o640\n replace_index=next(i for i,e in enumerate(events) if e[0]=='replace'); assert any(e==('fsync',False) for e in events[:replace_index]); assert any(e==('fsync',True) for e in events[replace_index+1:])\n assert events[replace_index][1].parent==p.parent and events[replace_index][2]==p; assert not [x for x in p.parent.iterdir() if x!=p]\nwith tempfile.TemporaryDirectory() as td:\n p=Path(td)/'new.json'; m.save_json(p,{'x':1}); assert stat.S_IMODE(os.stat(p).st_mode)==0o600 and m.load_json(p)=={'x':1}\nwith tempfile.TemporaryDirectory() as td:\n p=Path(td)/'state.json'; p.write_text('{\"old\":1}',encoding='utf-8')\n def fail_replace(src,dst): raise OSError('replace failed')\n with mock.patch.object(os,'replace',fail_replace):\n try: m.save_json(p,{'new':2})\n except OSError: pass\n else: raise AssertionError('replace failure swallowed')\n assert p.read_text(encoding='utf-8')=='{\"old\":1}' and not [x for x in p.parent.iterdir() if x!=p]\nwith tempfile.TemporaryDirectory() as td:\n p=Path(td)/'state.json'; p.write_text('{\"old\":1}',encoding='utf-8')\n class Bad: pass\n try: m.save_json(p,{'bad':Bad()})\n except (TypeError,ValueError): pass\n else: raise AssertionError('serialization should fail')\n assert p.read_text(encoding='utf-8')=='{\"old\":1}' and not [x for x in p.parent.iterdir() if x!=p]\nwith tempfile.TemporaryDirectory() as td:\n target=Path(td)/'target'; target.write_text('secret'); os.chmod(target,0o777); link=Path(td)/'state.json'; link.symlink_to(target)\n m.save_json(link,{'safe':True}); assert not link.is_symlink(); assert target.read_text()=='secret'; assert stat.S_IMODE(os.stat(link).st_mode)==0o600\n"
91
+ },
92
+ {
93
+ "id": "critical-audit-event-wal",
94
+ "class": "critical",
95
+ "description": "Crash-durable Write-Ahead Log with record framing, CRC32 integrity, and torn-write recovery.",
96
+ "prompt": "Implement AuditWAL in audit_wal.py for crash-durable event logging. AuditWAL(path) creates or opens an append-only log file. append(event: dict) -> int validates event is a JSON mapping, serializes it to UTF-8, frames it with a 12-byte binary header: 4-byte magic b'WAL1', 4-byte big-endian uint32 payload length, and 4-byte big-endian uint32 CRC32 checksum of the payload bytes (using zlib.crc32). It writes the frame, flushes and fsyncs the file descriptor, and returns the 0-based entry sequence number. recover() -> list[dict] reads all valid events from the start. If the file ends with a torn/partial write (truncated header or incomplete payload at EOF), recover() must salvage all prior valid records, truncate the file back to the end of the last intact record using truncate() and fsync(), and return the valid records. If a corrupted header, bad magic bytes, or CRC32 mismatch occurs with valid data remaining after it (a non-terminal corruption), recover() must raise IntegrityError without truncating. Reopening an intact log preserves previously appended records and sequence numbers.",
97
+ "files": {
98
+ "audit_wal.py": "import json\n\nclass IntegrityError(Exception):\n pass\n\nclass AuditWAL:\n def __init__(self, path):\n self.path = path\n self._count = 0\n\n def append(self, event):\n # BUG: simple non-atomic JSON lines, no magic, no CRC32, no fsync\n with open(self.path, 'a') as f:\n f.write(json.dumps(event) + '\\n')\n seq = self._count\n self._count += 1\n return seq\n\n def recover(self):\n # BUG: fails to handle framing, CRC32 or torn writes\n records = []\n try:\n with open(self.path, 'r') as f:\n for line in f:\n records.append(json.loads(line))\n except FileNotFoundError:\n pass\n return records\n"
99
+ },
100
+ "verifier": "from pathlib import Path\nimport importlib.util, tempfile, os, struct, zlib\nspec = importlib.util.spec_from_file_location('audit_wal', Path.cwd() / 'audit_wal.py')\nm = importlib.util.module_from_spec(spec)\nspec.loader.exec_module(m)\nwith tempfile.TemporaryDirectory() as td:\n log_path = Path(td) / 'audit.wal'\n wal = m.AuditWAL(log_path)\n seq0 = wal.append({'action': 'login', 'user': 'alice'})\n seq1 = wal.append({'action': 'update', 'user': 'bob'})\n assert seq0 == 0 and seq1 == 1\n raw = log_path.read_bytes()\n assert raw.startswith(b'WAL1')\n magic, length, crc = struct.unpack('>4sII', raw[:12])\n assert magic == b'WAL1'\n payload = raw[12:12+length]\n assert (zlib.crc32(payload) & 0xffffffff) == crc\n recovered = wal.recover()\n assert len(recovered) == 2\n assert recovered[0]['user'] == 'alice' and recovered[1]['user'] == 'bob'\n wal.append({'action': 'delete', 'user': 'charlie'})\n before_len = log_path.stat().st_size\n log_path.write_bytes(log_path.read_bytes()[:-5])\n salvaged = wal.recover()\n assert len(salvaged) == 2\n assert log_path.stat().st_size < before_len\n seq2 = wal.append({'action': 'logout', 'user': 'alice'})\n assert seq2 == 2\n final_recs = wal.recover()\n assert len(final_recs) == 3 and final_recs[-1]['action'] == 'logout'\nwith tempfile.TemporaryDirectory() as td:\n log_path = Path(td) / 'audit2.wal'\n wal = m.AuditWAL(log_path)\n wal.append({'n': 1})\n wal.append({'n': 2})\n raw = bytearray(log_path.read_bytes())\n raw[15] ^= 0xFF\n log_path.write_bytes(bytes(raw))\n try:\n wal.recover()\n except m.IntegrityError:\n pass\n else:\n raise AssertionError('mid-log corruption must raise IntegrityError')\n"
101
+ }
102
+ ]
103
+ }
@@ -0,0 +1,41 @@
1
+ {
2
+ "schema_version": 2,
3
+ "repetitions": 3,
4
+ "timeout_seconds": 1800,
5
+ "max_repair_cycles": 2,
6
+ "matrix": [
7
+ {"id": "luna-direct", "strategy": "direct", "model": "gpt-5.6-luna", "reasoning_effort": "high"},
8
+ {"id": "terra-direct", "strategy": "direct", "model": "gpt-5.6-terra", "reasoning_effort": "high"},
9
+ {"id": "sol-direct", "strategy": "direct", "model": "gpt-5.6-sol", "reasoning_effort": "high"},
10
+ {
11
+ "id": "codex-flow-high",
12
+ "strategy": "flow",
13
+ "reasoning_policy": "fixed",
14
+ "parent": {"model": "gpt-5.6-sol", "reasoning_effort": "high"},
15
+ "worker": {"model": "gpt-5.6-luna", "reasoning_effort": "high"}
16
+ },
17
+ {
18
+ "id": "codex-flow-adaptive",
19
+ "strategy": "flow",
20
+ "reasoning_policy": "adaptive",
21
+ "parent": {
22
+ "model": "gpt-5.6-sol",
23
+ "reasoning_effort": {"routine": "high", "complex": "xhigh", "critical": "max"}
24
+ },
25
+ "worker": {
26
+ "model": "gpt-5.6-luna",
27
+ "reasoning_effort": {"routine": "high", "complex": "xhigh", "critical": "max"}
28
+ }
29
+ }
30
+ ],
31
+ "tasks": [
32
+ {
33
+ "id": "example-compatibility-refactor",
34
+ "class": "complex",
35
+ "source": "/absolute/path/to/frozen-benchmark-repo",
36
+ "base_ref": "0123456789abcdef0123456789abcdef01234567",
37
+ "prompt": "Implement the requested change without weakening acceptance criteria.",
38
+ "verify": ["python3", "/absolute/path/to/external-verifier.py"]
39
+ }
40
+ ]
41
+ }
@@ -0,0 +1,137 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "title": "codex-flow benchmark runner manifest",
4
+ "type": "object",
5
+ "required": ["schema_version", "tasks", "matrix"],
6
+ "properties": {
7
+ "schema_version": {"const": 2},
8
+ "repetitions": {"type": "integer", "minimum": 1, "default": 1},
9
+ "timeout_seconds": {"type": "integer", "minimum": 1, "default": 1800},
10
+ "max_repair_cycles": {"type": "integer", "minimum": 0, "default": 2},
11
+ "allow_floating_refs": {"type": "boolean", "default": false},
12
+ "matrix": {
13
+ "type": "array",
14
+ "minItems": 1,
15
+ "items": {
16
+ "oneOf": [
17
+ {
18
+ "type": "object",
19
+ "required": ["id", "strategy", "model", "reasoning_effort"],
20
+ "properties": {
21
+ "id": {"type": "string", "minLength": 1},
22
+ "strategy": {"const": "direct"},
23
+ "model": {"type": "string", "minLength": 1},
24
+ "reasoning_effort": {"enum": ["high", "xhigh", "max"]}
25
+ },
26
+ "additionalProperties": false
27
+ },
28
+ {
29
+ "type": "object",
30
+ "required": ["id", "strategy", "reasoning_policy", "parent", "worker"],
31
+ "properties": {
32
+ "id": {"type": "string", "minLength": 1},
33
+ "strategy": {"enum": ["flow", "runtime"]},
34
+ "reasoning_policy": {"enum": ["fixed", "adaptive"]},
35
+ "parent": {"$ref": "#/$defs/actor"},
36
+ "worker": {"$ref": "#/$defs/actor"},
37
+ "profile": {"type": "string"},
38
+ "routing_mode": {"type": "string"}
39
+ },
40
+ "allOf": [
41
+ {
42
+ "if": {"properties": {"reasoning_policy": {"const": "fixed"}}},
43
+ "then": {
44
+ "properties": {
45
+ "parent": {"$ref": "#/$defs/fixed_actor"},
46
+ "worker": {"$ref": "#/$defs/fixed_actor"}
47
+ }
48
+ }
49
+ },
50
+ {
51
+ "if": {"properties": {"reasoning_policy": {"const": "adaptive"}}},
52
+ "then": {
53
+ "properties": {
54
+ "parent": {"$ref": "#/$defs/adaptive_actor"},
55
+ "worker": {"$ref": "#/$defs/adaptive_actor"}
56
+ }
57
+ }
58
+ }
59
+ ],
60
+ "additionalProperties": false
61
+ }
62
+ ]
63
+ }
64
+ },
65
+ "tasks": {
66
+ "type": "array",
67
+ "minItems": 1,
68
+ "items": {
69
+ "type": "object",
70
+ "required": ["id", "class", "source", "base_ref", "prompt", "verify"],
71
+ "properties": {
72
+ "id": {"type": "string", "minLength": 1},
73
+ "class": {"enum": ["routine", "complex", "critical"]},
74
+ "source": {"type": "string", "minLength": 1},
75
+ "base_ref": {"type": "string", "minLength": 1},
76
+ "prompt": {"type": "string", "minLength": 1},
77
+ "verify": {"type": "array", "minItems": 1, "items": {"type": "string"}},
78
+ "max_repair_cycles": {"type": "integer", "minimum": 0}
79
+ },
80
+ "additionalProperties": false
81
+ }
82
+ }
83
+ },
84
+ "$defs": {
85
+ "actor": {
86
+ "type": "object",
87
+ "required": ["model", "reasoning_effort"],
88
+ "properties": {
89
+ "model": {"type": "string", "minLength": 1},
90
+ "reasoning_effort": {
91
+ "oneOf": [
92
+ {"enum": ["high", "xhigh", "max"]},
93
+ {
94
+ "type": "object",
95
+ "required": ["routine", "complex", "critical"],
96
+ "properties": {
97
+ "routine": {"enum": ["high", "xhigh", "max"]},
98
+ "complex": {"enum": ["high", "xhigh", "max"]},
99
+ "critical": {"enum": ["high", "xhigh", "max"]}
100
+ },
101
+ "additionalProperties": false
102
+ }
103
+ ]
104
+ }
105
+ },
106
+ "additionalProperties": false
107
+ },
108
+ "fixed_actor": {
109
+ "type": "object",
110
+ "required": ["model", "reasoning_effort"],
111
+ "properties": {
112
+ "model": {"type": "string", "minLength": 1},
113
+ "reasoning_effort": {"enum": ["high", "xhigh", "max"]}
114
+ },
115
+ "additionalProperties": false
116
+ },
117
+ "adaptive_actor": {
118
+ "type": "object",
119
+ "required": ["model", "reasoning_effort"],
120
+ "properties": {
121
+ "model": {"type": "string", "minLength": 1},
122
+ "reasoning_effort": {
123
+ "type": "object",
124
+ "required": ["routine", "complex", "critical"],
125
+ "properties": {
126
+ "routine": {"enum": ["high", "xhigh", "max"]},
127
+ "complex": {"enum": ["high", "xhigh", "max"]},
128
+ "critical": {"enum": ["high", "xhigh", "max"]}
129
+ },
130
+ "additionalProperties": false
131
+ }
132
+ },
133
+ "additionalProperties": false
134
+ }
135
+ },
136
+ "additionalProperties": false
137
+ }
@@ -0,0 +1,5 @@
1
+ {
2
+ "gpt-5.6-luna": {"input": 0.20, "cached_input": 0.02, "output": 1.20},
3
+ "gpt-5.6-terra": {"input": 2.00, "cached_input": 0.20, "output": 12.00},
4
+ "gpt-5.6-sol": {"input": 4.00, "cached_input": 0.40, "output": 20.00}
5
+ }
@@ -0,0 +1,90 @@
1
+ {
2
+ "schema_version": 2,
3
+ "profiles": {
4
+ "quick": {
5
+ "description": "One repetition across three high direct models, fixed-high flow, and adaptive-effort flow.",
6
+ "repetitions": 1,
7
+ "matrix": [
8
+ {"id": "luna-direct", "strategy": "direct", "model": "gpt-5.6-luna", "reasoning_effort": "high"},
9
+ {"id": "terra-direct", "strategy": "direct", "model": "gpt-5.6-terra", "reasoning_effort": "high"},
10
+ {"id": "sol-direct", "strategy": "direct", "model": "gpt-5.6-sol", "reasoning_effort": "high"},
11
+ {
12
+ "id": "codex-flow-high",
13
+ "strategy": "flow",
14
+ "reasoning_policy": "fixed",
15
+ "parent": {"model": "gpt-5.6-sol", "reasoning_effort": "high"},
16
+ "worker": {"model": "gpt-5.6-luna", "reasoning_effort": "high"}
17
+ },
18
+ {
19
+ "id": "codex-flow-adaptive",
20
+ "strategy": "flow",
21
+ "reasoning_policy": "adaptive",
22
+ "parent": {
23
+ "model": "gpt-5.6-sol",
24
+ "reasoning_effort": {"routine": "high", "complex": "xhigh", "critical": "max"}
25
+ },
26
+ "worker": {
27
+ "model": "gpt-5.6-luna",
28
+ "reasoning_effort": {"routine": "high", "complex": "xhigh", "critical": "max"}
29
+ }
30
+ }
31
+ ]
32
+ },
33
+ "full": {
34
+ "description": "Three repetitions across three high direct models, fixed-high flow, and adaptive-effort flow.",
35
+ "repetitions": 3,
36
+ "matrix": [
37
+ {"id": "luna-direct", "strategy": "direct", "model": "gpt-5.6-luna", "reasoning_effort": "high"},
38
+ {"id": "terra-direct", "strategy": "direct", "model": "gpt-5.6-terra", "reasoning_effort": "high"},
39
+ {"id": "sol-direct", "strategy": "direct", "model": "gpt-5.6-sol", "reasoning_effort": "high"},
40
+ {
41
+ "id": "codex-flow-high",
42
+ "strategy": "flow",
43
+ "reasoning_policy": "fixed",
44
+ "parent": {"model": "gpt-5.6-sol", "reasoning_effort": "high"},
45
+ "worker": {"model": "gpt-5.6-luna", "reasoning_effort": "high"}
46
+ },
47
+ {
48
+ "id": "codex-flow-adaptive",
49
+ "strategy": "flow",
50
+ "reasoning_policy": "adaptive",
51
+ "parent": {
52
+ "model": "gpt-5.6-sol",
53
+ "reasoning_effort": {"routine": "high", "complex": "xhigh", "critical": "max"}
54
+ },
55
+ "worker": {
56
+ "model": "gpt-5.6-luna",
57
+ "reasoning_effort": {"routine": "high", "complex": "xhigh", "critical": "max"}
58
+ }
59
+ }
60
+ ]
61
+ },
62
+ "agentic": {
63
+ "description": "One repetition comparing direct baselines with real FlowPilot runtime execution under efficient and balanced profiles.",
64
+ "repetitions": 1,
65
+ "matrix": [
66
+ {"id": "luna-direct", "strategy": "direct", "model": "gpt-5.6-luna", "reasoning_effort": "high"},
67
+ {"id": "terra-direct", "strategy": "direct", "model": "gpt-5.6-terra", "reasoning_effort": "high"},
68
+ {"id": "sol-direct", "strategy": "direct", "model": "gpt-5.6-sol", "reasoning_effort": "high"},
69
+ {
70
+ "id": "codex-flow-runtime-efficient",
71
+ "strategy": "runtime",
72
+ "reasoning_policy": "fixed",
73
+ "profile": "efficient",
74
+ "routing_mode": "delegate",
75
+ "parent": {"model": "gpt-5.6-sol", "reasoning_effort": "high"},
76
+ "worker": {"model": "gpt-5.6-luna", "reasoning_effort": "high"}
77
+ },
78
+ {
79
+ "id": "codex-flow-runtime-balanced",
80
+ "strategy": "runtime",
81
+ "reasoning_policy": "fixed",
82
+ "profile": "balanced",
83
+ "routing_mode": "delegate",
84
+ "parent": {"model": "gpt-5.6-sol", "reasoning_effort": "high"},
85
+ "worker": {"model": "gpt-5.6-luna", "reasoning_effort": "high"}
86
+ }
87
+ ]
88
+ }
89
+ }
90
+ }
@@ -0,0 +1,77 @@
1
+ {
2
+ "$schema": "https://json-schema.org/draft/2020-12/schema",
3
+ "title": "codex-flow benchmark strategy run",
4
+ "type": "object",
5
+ "required": [
6
+ "schema_version", "task_id", "task_class", "strategy_id", "strategy",
7
+ "reasoning_policy", "model", "reasoning_effort", "worker_model", "worker_reasoning_effort",
8
+ "passed", "first_passed", "input_tokens", "cached_input_tokens",
9
+ "output_tokens", "model_usage", "repair_cycles", "review_cycles"
10
+ ],
11
+ "properties": {
12
+ "schema_version": {"const": 2},
13
+ "task_id": {"type": "string", "minLength": 1},
14
+ "task_class": {"enum": ["routine", "complex", "critical"]},
15
+ "strategy_id": {"type": "string", "minLength": 1},
16
+ "strategy": {"enum": ["direct", "flow", "runtime"]},
17
+ "reasoning_policy": {"enum": ["fixed", "adaptive"]},
18
+ "model": {"type": "string", "minLength": 1},
19
+ "reasoning_effort": {"enum": ["high", "xhigh", "max"]},
20
+ "worker_model": {"type": ["string", "null"]},
21
+ "worker_reasoning_effort": {"type": ["string", "null"], "enum": ["high", "xhigh", "max", null]},
22
+ "passed": {"type": "boolean"},
23
+ "first_passed": {"type": "boolean"},
24
+ "input_tokens": {"type": "integer", "minimum": 0},
25
+ "cached_input_tokens": {"type": "integer", "minimum": 0},
26
+ "output_tokens": {"type": "integer", "minimum": 0},
27
+ "model_usage": {
28
+ "type": "array",
29
+ "minItems": 1,
30
+ "items": {
31
+ "type": "object",
32
+ "required": ["role", "model", "reasoning_effort", "calls", "input_tokens", "cached_input_tokens", "output_tokens"],
33
+ "properties": {
34
+ "role": {"enum": ["direct", "parent", "worker"]},
35
+ "model": {"type": "string", "minLength": 1},
36
+ "reasoning_effort": {"enum": ["high", "xhigh", "max"]},
37
+ "calls": {"type": "integer", "minimum": 1},
38
+ "input_tokens": {"type": "integer", "minimum": 0},
39
+ "cached_input_tokens": {"type": "integer", "minimum": 0},
40
+ "output_tokens": {"type": "integer", "minimum": 0}
41
+ },
42
+ "additionalProperties": false
43
+ }
44
+ },
45
+ "repair_cycles": {"type": "integer", "minimum": 0},
46
+ "review_cycles": {"type": "integer", "minimum": 0},
47
+ "wall_time_seconds": {"type": "number", "minimum": 0},
48
+ "source_commit": {"type": "string", "minLength": 1},
49
+ "repetition": {"type": "integer", "minimum": 1},
50
+ "codex_exit_code": {"type": "integer"},
51
+ "verification_excerpt": {"type": "string"},
52
+ "diagnostic_excerpt": {"type": "string"},
53
+ "notes": {"type": "string"}
54
+ },
55
+ "allOf": [
56
+ {
57
+ "if": {"properties": {"strategy": {"const": "direct"}}},
58
+ "then": {
59
+ "properties": {
60
+ "reasoning_policy": {"const": "fixed"},
61
+ "worker_model": {"type": "null"},
62
+ "worker_reasoning_effort": {"type": "null"}
63
+ }
64
+ }
65
+ },
66
+ {
67
+ "if": {"properties": {"strategy": {"enum": ["flow", "runtime"]}}},
68
+ "then": {
69
+ "properties": {
70
+ "worker_model": {"type": "string", "minLength": 1},
71
+ "worker_reasoning_effort": {"enum": ["high", "xhigh", "max"]}
72
+ }
73
+ }
74
+ }
75
+ ],
76
+ "additionalProperties": false
77
+ }
@@ -0,0 +1,50 @@
1
+ {
2
+ "schema_version": 2,
3
+ "tasks": [
4
+ {
5
+ "id": "routine-query-normalization",
6
+ "class": "routine",
7
+ "description": "Localized bug fix with edge-case normalization and no API change."
8
+ },
9
+ {
10
+ "id": "routine-env-precedence",
11
+ "class": "routine",
12
+ "description": "Configuration precedence bug requiring exact fallback semantics."
13
+ },
14
+ {
15
+ "id": "routine-header-sanitization",
16
+ "class": "routine",
17
+ "description": "HTTP header sanitization, case normalization, and injection protection."
18
+ },
19
+ {
20
+ "id": "complex-renew-provider-refactor",
21
+ "class": "complex",
22
+ "description": "Multi-file registry refactor with isolation, validation, normalization, and legacy compatibility."
23
+ },
24
+ {
25
+ "id": "complex-config-migration",
26
+ "class": "complex",
27
+ "description": "Idempotent, alias-safe configuration migration with per-key precedence and validation."
28
+ },
29
+ {
30
+ "id": "complex-dag-resolver",
31
+ "class": "complex",
32
+ "description": "Topological dependency graph resolver with cycle detection, tie-breaking, and batch scheduling."
33
+ },
34
+ {
35
+ "id": "critical-resumable-migration",
36
+ "class": "critical",
37
+ "description": "Crash-resumable record migration with preflight validation and durable progress tracking."
38
+ },
39
+ {
40
+ "id": "critical-atomic-state-write",
41
+ "class": "critical",
42
+ "description": "Crash-durable atomic replacement with permission preservation, directory fsync, and failure cleanup."
43
+ },
44
+ {
45
+ "id": "critical-audit-event-wal",
46
+ "class": "critical",
47
+ "description": "Crash-durable Write-Ahead Log with record framing, CRC32 integrity, and torn-write recovery."
48
+ }
49
+ ]
50
+ }
@@ -0,0 +1,34 @@
1
+ # bash completion for codex-flow
2
+ _codex_flow_completion() {
3
+ local cur="${COMP_WORDS[COMP_CWORD]}"
4
+ local commands="status strategy language update doctor overlay usage telemetry benchmark-local benchmark-corpus benchmark benchmark-analyze uninstall help"
5
+ if [[ "${COMP_CWORD}" -eq 1 ]]; then
6
+ COMPREPLY=( $(compgen -W "$commands" -- "$cur") )
7
+ elif [[ "${COMP_CWORD}" -eq 2 && "${COMP_WORDS[1]}" == "strategy" ]]; then
8
+ COMPREPLY=( $(compgen -W "show profiles enabled enable disable set routing plan" -- "$cur") )
9
+ elif [[ "${COMP_CWORD}" -eq 3 && "${COMP_WORDS[1]}" == "strategy" && "${COMP_WORDS[2]}" == "set" ]]; then
10
+ COMPREPLY=( $(compgen -W "efficient balanced quality speed" -- "$cur") )
11
+ elif [[ "${COMP_CWORD}" -eq 3 && "${COMP_WORDS[1]}" == "strategy" && "${COMP_WORDS[2]}" == "routing" ]]; then
12
+ COMPREPLY=( $(compgen -W "adaptive direct delegate" -- "$cur") )
13
+ elif [[ "${COMP_CWORD}" -ge 3 && "${COMP_WORDS[1]}" == "strategy" && "${COMP_WORDS[2]}" == "plan" ]]; then
14
+ COMPREPLY=( $(compgen -W "--profile --routing --review --fanout --complexity --uncertainty --risk --scope --parallelism --write-conflict --exploration-need --verification-cost --iteration-intensity --writable-workstreams --quality-intent --quota-pressure --max-threads --max-repairs" -- "$cur") )
15
+ elif [[ "${COMP_CWORD}" -eq 2 && "${COMP_WORDS[1]}" == "language" ]]; then
16
+ COMPREPLY=( $(compgen -W "auto zh en" -- "$cur") )
17
+ elif [[ "${COMP_CWORD}" -eq 2 && ( "${COMP_WORDS[1]}" == "benchmark-local" || "${COMP_WORDS[1]}" == "benchmark-corpus" ) ]]; then
18
+ COMPREPLY=( $(compgen -W "quick full" -- "$cur") )
19
+ elif [[ "${COMP_CWORD}" -eq 2 && ( "${COMP_WORDS[1]}" == "usage" || "${COMP_WORDS[1]}" == "telemetry" ) ]]; then
20
+ COMPREPLY=( $(compgen -W "last list show stats summary repair" -- "$cur") )
21
+ elif [[ "${COMP_WORDS[1]}" == "usage" || "${COMP_WORDS[1]}" == "telemetry" ]]; then
22
+ case "${COMP_WORDS[2]}" in
23
+ last) COMPREPLY=( $(compgen -W "--json" -- "$cur") ) ;;
24
+ list) COMPREPLY=( $(compgen -W "--json --today -n --limit -p --project" -- "$cur") ) ;;
25
+ show) COMPREPLY=( $(compgen -W "--json" -- "$cur") ) ;;
26
+ stats|summary) COMPREPLY=( $(compgen -W "--json -d --days -p --project" -- "$cur") ) ;;
27
+ repair) COMPREPLY=( $(compgen -W "--dry-run --json" -- "$cur") ) ;;
28
+ *) COMPREPLY=() ;;
29
+ esac
30
+ else
31
+ COMPREPLY=()
32
+ fi
33
+ }
34
+ complete -F _codex_flow_completion codex-flow