codex-flow 2.1.13__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codex_flow/__init__.py +28 -0
- codex_flow/__main__.py +9 -0
- codex_flow/cli.py +242 -0
- codex_flow/data/LICENSE +21 -0
- codex_flow/data/README.en.md +303 -0
- codex_flow/data/README.md +305 -0
- codex_flow/data/VERSION +1 -0
- codex_flow/data/apps/chatgpt-mcp/README.md +86 -0
- codex_flow/data/apps/chatgpt-mcp/__init__.py +1 -0
- codex_flow/data/apps/chatgpt-mcp/adapter.py +458 -0
- codex_flow/data/apps/chatgpt-mcp/server.py +358 -0
- codex_flow/data/apps/chatgpt-mcp/widget.html +927 -0
- codex_flow/data/apps/macos-overlay/README.en.md +121 -0
- codex_flow/data/apps/macos-overlay/README.md +123 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayRuntimeState.swift +126 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayScreenGeometry.swift +82 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayWindowController.swift +1052 -0
- codex_flow/data/apps/macos-overlay/Sources/Localization.swift +197 -0
- codex_flow/data/apps/macos-overlay/Sources/Models/TelemetryData.swift +1557 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/AccountSnapshotService.swift +1101 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/FlowPilotInstanceLock.swift +153 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/IPCServer.swift +298 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryQueryEngine.swift +800 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryWatcher.swift +135 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/UpdateService.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AccountView.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AnalyticsView.swift +566 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AutostartView.swift +293 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/BubbleView.swift +317 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HistoryView.swift +1124 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HoverRevealText.swift +165 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/InspectorSkillsToolsView.swift +121 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/LogoView.swift +182 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SleekSwitch.swift +117 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/StrategyModeView.swift +561 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SummaryView.swift +1273 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/UpdateView.swift +352 -0
- codex_flow/data/apps/macos-overlay/Sources/main.swift +340 -0
- codex_flow/data/apps/macos-overlay/Tests/OverlayScreenGeometryTests.swift +163 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryPhase1ContractTests.swift +357 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQueryEngineConcurrencyTests.swift +221 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQuotaSelectionTests.swift +158 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryWorkerTokenTests.swift +122 -0
- codex_flow/data/apps/macos-overlay/build.sh +75 -0
- codex_flow/data/benchmark/corpus.json +103 -0
- codex_flow/data/benchmark/manifest.example.json +41 -0
- codex_flow/data/benchmark/manifest.schema.json +137 -0
- codex_flow/data/benchmark/prices/gpt-5.6-2026-08-30.json +5 -0
- codex_flow/data/benchmark/profiles.json +90 -0
- codex_flow/data/benchmark/schema.json +77 -0
- codex_flow/data/benchmark/tasks.json +50 -0
- codex_flow/data/completions/codex-flow.bash +34 -0
- codex_flow/data/completions/codex-flow.zsh +52 -0
- codex_flow/data/glama.json +6 -0
- codex_flow/data/install-release.ps1 +126 -0
- codex_flow/data/install-release.sh +155 -0
- codex_flow/data/install.ps1 +349 -0
- codex_flow/data/install.sh +362 -0
- codex_flow/data/policy/benchmark.toml +49 -0
- codex_flow/data/policy/defaults.toml +70 -0
- codex_flow/data/scripts/analyze-benchmark.py +510 -0
- codex_flow/data/scripts/benchmark-local.py +171 -0
- codex_flow/data/scripts/check-recommendation.py +277 -0
- codex_flow/data/scripts/doctor.py +449 -0
- codex_flow/data/scripts/generate-release-manifest.py +74 -0
- codex_flow/data/scripts/localization.py +192 -0
- codex_flow/data/scripts/manage-hooks.py +448 -0
- codex_flow/data/scripts/manage-instructions.py +389 -0
- codex_flow/data/scripts/manage-shell.py +151 -0
- codex_flow/data/scripts/materialize-corpus.py +193 -0
- codex_flow/data/scripts/menu.py +646 -0
- codex_flow/data/scripts/migrations/0001_update_settings.py +80 -0
- codex_flow/data/scripts/package-release.py +132 -0
- codex_flow/data/scripts/render-benchmark-report.py +292 -0
- codex_flow/data/scripts/run-benchmark.py +829 -0
- codex_flow/data/scripts/strategies/__init__.py +28 -0
- codex_flow/data/scripts/strategies/balanced.py +115 -0
- codex_flow/data/scripts/strategies/base.py +363 -0
- codex_flow/data/scripts/strategies/efficient.py +158 -0
- codex_flow/data/scripts/strategies/lifecycle_runtime.py +590 -0
- codex_flow/data/scripts/strategies/quality.py +209 -0
- codex_flow/data/scripts/strategies/speed.py +108 -0
- codex_flow/data/scripts/strategies/task_budget_runtime.py +644 -0
- codex_flow/data/scripts/strategies/task_phase_runtime.py +341 -0
- codex_flow/data/scripts/strategies/work_unit_runtime.py +421 -0
- codex_flow/data/scripts/strategy_runtime.py +1091 -0
- codex_flow/data/scripts/telemetry.py +400 -0
- codex_flow/data/scripts/telemetry_core/__init__.py +192 -0
- codex_flow/data/scripts/telemetry_core/app_server.py +1192 -0
- codex_flow/data/scripts/telemetry_core/collector.py +1247 -0
- codex_flow/data/scripts/telemetry_core/common.py +421 -0
- codex_flow/data/scripts/telemetry_core/latency.py +593 -0
- codex_flow/data/scripts/telemetry_core/query.py +427 -0
- codex_flow/data/scripts/telemetry_core/quota_ledger.py +598 -0
- codex_flow/data/scripts/telemetry_core/render.py +460 -0
- codex_flow/data/scripts/telemetry_core/repair.py +223 -0
- codex_flow/data/scripts/ui.py +266 -0
- codex_flow/data/scripts/update-homebrew-formula.py +146 -0
- codex_flow/data/scripts/update_runtime_config.py +134 -0
- codex_flow/data/scripts/updater.py +1718 -0
- codex_flow/data/smithery.yaml +18 -0
- codex_flow/data/templates/agents/worker-explorer.toml +24 -0
- codex_flow/data/templates/agents/worker-implementer.toml +49 -0
- codex_flow/data/templates/agents/worker-reviewer.toml +25 -0
- codex_flow/data/templates/flow-pilot-instructions.md +35 -0
- codex_flow/data/templates/skills/flow-pilot/SKILL.md +577 -0
- codex_flow/mcp.py +35 -0
- codex_flow-2.1.13.dist-info/METADATA +342 -0
- codex_flow-2.1.13.dist-info/RECORD +113 -0
- codex_flow-2.1.13.dist-info/WHEEL +5 -0
- codex_flow-2.1.13.dist-info/entry_points.txt +3 -0
- codex_flow-2.1.13.dist-info/licenses/LICENSE +21 -0
- codex_flow-2.1.13.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
"""Built-in strategy registry for codex-flow."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from .balanced import STRATEGY as BALANCED
|
|
5
|
+
from .base import StrategySpec
|
|
6
|
+
from .efficient import STRATEGY as EFFICIENT
|
|
7
|
+
from .quality import STRATEGY as QUALITY
|
|
8
|
+
from .speed import STRATEGY as SPEED
|
|
9
|
+
|
|
10
|
+
_REGISTRY: dict[str, StrategySpec] = {
|
|
11
|
+
spec.name: spec
|
|
12
|
+
for spec in (EFFICIENT, BALANCED, QUALITY, SPEED)
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def names() -> tuple[str, ...]:
|
|
17
|
+
return tuple(_REGISTRY)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def get(name: str) -> StrategySpec:
|
|
21
|
+
try:
|
|
22
|
+
return _REGISTRY[name]
|
|
23
|
+
except KeyError as exc:
|
|
24
|
+
raise ValueError(f"invalid strategy: {name}") from exc
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def all_specs() -> tuple[StrategySpec, ...]:
|
|
28
|
+
return tuple(_REGISTRY.values())
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""Balanced quality/quota/latency strategy."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from .base import (
|
|
5
|
+
StagePolicy,
|
|
6
|
+
StrategySpec,
|
|
7
|
+
TaskBudgetPolicy,
|
|
8
|
+
WorkerBudget,
|
|
9
|
+
never,
|
|
10
|
+
small_low_risk_is_direct,
|
|
11
|
+
standard_effort,
|
|
12
|
+
)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def adaptive_route(task) -> str:
|
|
16
|
+
if small_low_risk_is_direct(task):
|
|
17
|
+
return "direct"
|
|
18
|
+
return "delegate" if task.complexity in {"routine", "complex", "critical"} and task.iteration_intensity != "one-shot" else "direct"
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def worker_budget(task) -> WorkerBudget:
|
|
22
|
+
if task.complexity in {"complex", "critical"} or task.scope == "repo-wide":
|
|
23
|
+
return WorkerBudget(3, 3, 1, 5, "medium")
|
|
24
|
+
if task.uncertainty == "high" or task.iteration_intensity == "heavy-loop":
|
|
25
|
+
return WorkerBudget(3, 2, 1, 4, "medium")
|
|
26
|
+
return WorkerBudget(2, 2, 1, 4, "medium")
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def implementation_soft_timeout(task) -> int:
|
|
30
|
+
if task.complexity == "critical" or task.risk == "critical":
|
|
31
|
+
return 1800
|
|
32
|
+
if task.complexity == "complex" or task.scope == "repo-wide" or task.iteration_intensity == "heavy-loop":
|
|
33
|
+
return 1500
|
|
34
|
+
return 1200
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def implementation_checkpoint_rearm_seconds(task) -> int:
|
|
38
|
+
if task.complexity == "critical" or task.risk == "critical":
|
|
39
|
+
return 360
|
|
40
|
+
if task.complexity == "complex" or task.scope in {"cross-module", "repo-wide"} or task.iteration_intensity == "heavy-loop":
|
|
41
|
+
return 300
|
|
42
|
+
return 240
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def implementation_repair_attempts(task) -> int:
|
|
46
|
+
if task.complexity in {"complex", "critical"} or task.risk == "critical" or task.scope == "repo-wide" or task.iteration_intensity == "heavy-loop":
|
|
47
|
+
return 2
|
|
48
|
+
return 1
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def implementation_maximum_work_units(task) -> int:
|
|
52
|
+
if task.scope == "repo-wide" or task.iteration_intensity == "heavy-loop":
|
|
53
|
+
semantic_maximum = 3
|
|
54
|
+
elif task.complexity in {"complex", "critical"} or task.scope == "cross-module" or task.risk == "critical":
|
|
55
|
+
semantic_maximum = 2
|
|
56
|
+
else:
|
|
57
|
+
semantic_maximum = 1
|
|
58
|
+
topology_floor = min(task.writable_workstreams, worker_budget(task).max_implementers)
|
|
59
|
+
return max(semantic_maximum, topology_floor)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def task_budget(task) -> TaskBudgetPolicy:
|
|
63
|
+
maximum_work_units = implementation_maximum_work_units(task)
|
|
64
|
+
if task.complexity == "critical" or task.risk == "critical" or task.scope == "repo-wide" or task.iteration_intensity == "heavy-loop":
|
|
65
|
+
soft_timeout, hard_timeout = 3000, 3600
|
|
66
|
+
elif task.complexity == "complex" or task.scope == "cross-module" or task.risk == "high":
|
|
67
|
+
soft_timeout, hard_timeout = 2700, 3300
|
|
68
|
+
else:
|
|
69
|
+
soft_timeout, hard_timeout = 2400, 3000
|
|
70
|
+
return TaskBudgetPolicy(
|
|
71
|
+
soft_timeout_seconds=soft_timeout,
|
|
72
|
+
hard_timeout_seconds=hard_timeout,
|
|
73
|
+
max_work_units=maximum_work_units,
|
|
74
|
+
max_implementation_attempts=maximum_work_units + 2,
|
|
75
|
+
max_replans=2,
|
|
76
|
+
max_replacements=2,
|
|
77
|
+
max_review_attempts=2,
|
|
78
|
+
parent_finalization_seconds=180,
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def lifecycle(task, stage: str) -> StagePolicy:
|
|
83
|
+
if stage == "exploration":
|
|
84
|
+
return StagePolicy("quorum", 1, 180, 1200, True, True, "parent_delta")
|
|
85
|
+
if stage == "implementation":
|
|
86
|
+
maximum_work_units = implementation_maximum_work_units(task)
|
|
87
|
+
bounded_mode = maximum_work_units > 1
|
|
88
|
+
return StagePolicy(
|
|
89
|
+
"required", 1, 240, 2400, False, False, "replan",
|
|
90
|
+
soft_timeout_seconds=implementation_soft_timeout(task),
|
|
91
|
+
checkpoint_rearm_seconds=implementation_checkpoint_rearm_seconds(task),
|
|
92
|
+
max_worker_repair_attempts=implementation_repair_attempts(task),
|
|
93
|
+
work_unit_mode="bounded" if bounded_mode else "single",
|
|
94
|
+
minimum_work_units=1,
|
|
95
|
+
join_between_work_units=bounded_mode,
|
|
96
|
+
maximum_work_units=maximum_work_units,
|
|
97
|
+
require_write_paths=bounded_mode,
|
|
98
|
+
)
|
|
99
|
+
if stage == "review":
|
|
100
|
+
return StagePolicy("required", 1, 180, 1800, True, False, "retry_review")
|
|
101
|
+
raise ValueError(f"invalid lifecycle stage: {stage}")
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
STRATEGY = StrategySpec(
|
|
105
|
+
name="balanced",
|
|
106
|
+
description="balance quality, quota consumption, and latency with moderate safe worker fan-out",
|
|
107
|
+
adaptive_route=adaptive_route,
|
|
108
|
+
effort=standard_effort,
|
|
109
|
+
worker_budget=worker_budget,
|
|
110
|
+
independent_review=never,
|
|
111
|
+
lifecycle=lifecycle,
|
|
112
|
+
allow_parallel_write=True,
|
|
113
|
+
quota_sensitive=True,
|
|
114
|
+
task_budget=task_budget,
|
|
115
|
+
)
|
|
@@ -0,0 +1,363 @@
|
|
|
1
|
+
"""Shared contracts for built-in codex-flow strategy modules."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
from typing import Any, Callable, Tuple
|
|
6
|
+
|
|
7
|
+
Task = Any
|
|
8
|
+
RouteFn = Callable[[Task], str]
|
|
9
|
+
EffortFn = Callable[[Task, str], str]
|
|
10
|
+
BudgetFn = Callable[[Task], "WorkerBudget"]
|
|
11
|
+
PredicateFn = Callable[[Task], bool]
|
|
12
|
+
CapabilityFn = Callable[[Task, str], str]
|
|
13
|
+
DemandFn = Callable[[Task], int]
|
|
14
|
+
NotesFn = Callable[[Task], Tuple[str, ...]]
|
|
15
|
+
LifecycleFn = Callable[[Task, str], "StagePolicy"]
|
|
16
|
+
TaskBudgetFn = Callable[[Task], "TaskBudgetPolicy | None"]
|
|
17
|
+
ReasoningRolloutFn = Callable[
|
|
18
|
+
[Task, str, "ReasoningRolloutPolicy", str, str], "ReasoningRolloutDecision"
|
|
19
|
+
]
|
|
20
|
+
|
|
21
|
+
STAGES = ("exploration", "implementation", "review")
|
|
22
|
+
JOIN_POLICIES = ("opportunistic", "quorum", "required")
|
|
23
|
+
FALLBACK_POLICIES = ("continue_partial", "parent_delta", "replan", "retry_review", "fail")
|
|
24
|
+
WORK_UNIT_MODES = ("single", "bounded")
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def worker_capability(_task: Task, _role: str) -> str:
|
|
28
|
+
"""Use the configured efficient-worker capability for this role."""
|
|
29
|
+
return "worker"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def zero_demand(_task: Task) -> int:
|
|
33
|
+
return 0
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def no_notes(_task: Task) -> tuple[str, ...]:
|
|
37
|
+
return ()
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
@dataclass(frozen=True)
|
|
41
|
+
class WorkerBudget:
|
|
42
|
+
"""Strategy preference envelope before Runtime safety/ceiling enforcement."""
|
|
43
|
+
|
|
44
|
+
max_explorers: int
|
|
45
|
+
max_implementers: int
|
|
46
|
+
max_reviewers: int
|
|
47
|
+
max_total_workers: int
|
|
48
|
+
speculation: str = "medium"
|
|
49
|
+
|
|
50
|
+
def validate(self) -> None:
|
|
51
|
+
for name, value in (
|
|
52
|
+
("max_explorers", self.max_explorers),
|
|
53
|
+
("max_implementers", self.max_implementers),
|
|
54
|
+
("max_reviewers", self.max_reviewers),
|
|
55
|
+
("max_total_workers", self.max_total_workers),
|
|
56
|
+
):
|
|
57
|
+
if value < 0:
|
|
58
|
+
raise ValueError(f"{name} cannot be negative")
|
|
59
|
+
if self.max_implementers < 1:
|
|
60
|
+
raise ValueError("max_implementers must be positive")
|
|
61
|
+
if self.max_total_workers < 1:
|
|
62
|
+
raise ValueError("max_total_workers must be positive")
|
|
63
|
+
if self.speculation not in {"low", "medium", "high"}:
|
|
64
|
+
raise ValueError(f"invalid speculation level: {self.speculation}")
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass(frozen=True)
|
|
68
|
+
class TaskBudgetPolicy:
|
|
69
|
+
"""Cumulative task budget carried across Worker attempts and replans.
|
|
70
|
+
|
|
71
|
+
Stage lifecycle limits describe one Worker stage. These task-level counters
|
|
72
|
+
are durable reservations across attempts/replans. `max_review_attempts`
|
|
73
|
+
counts read-only reviewer Worker starts independently from implementation
|
|
74
|
+
replans/replacements. `parent_finalization_seconds` is a deterministic tail
|
|
75
|
+
reserved after the review-stage hard window whenever reviewers are planned.
|
|
76
|
+
"""
|
|
77
|
+
|
|
78
|
+
soft_timeout_seconds: int
|
|
79
|
+
hard_timeout_seconds: int
|
|
80
|
+
max_work_units: int
|
|
81
|
+
max_implementation_attempts: int
|
|
82
|
+
max_replans: int
|
|
83
|
+
max_replacements: int
|
|
84
|
+
max_review_attempts: int = 0
|
|
85
|
+
parent_finalization_seconds: int = 120
|
|
86
|
+
|
|
87
|
+
@classmethod
|
|
88
|
+
def from_dict(cls, value: Any) -> "TaskBudgetPolicy":
|
|
89
|
+
if type(value) is not dict:
|
|
90
|
+
raise ValueError("task budget policy must be an object")
|
|
91
|
+
required = (
|
|
92
|
+
"soft_timeout_seconds",
|
|
93
|
+
"hard_timeout_seconds",
|
|
94
|
+
"max_work_units",
|
|
95
|
+
"max_implementation_attempts",
|
|
96
|
+
"max_replans",
|
|
97
|
+
"max_replacements",
|
|
98
|
+
"max_review_attempts",
|
|
99
|
+
"parent_finalization_seconds",
|
|
100
|
+
)
|
|
101
|
+
missing = [name for name in required if name not in value]
|
|
102
|
+
if missing:
|
|
103
|
+
raise ValueError(f"task budget policy missing fields: {missing}")
|
|
104
|
+
unknown = sorted(set(value).difference(required))
|
|
105
|
+
if unknown:
|
|
106
|
+
raise ValueError(f"task budget policy has unknown fields: {unknown}")
|
|
107
|
+
policy = cls(**{name: value[name] for name in required})
|
|
108
|
+
policy.validate()
|
|
109
|
+
return policy
|
|
110
|
+
|
|
111
|
+
def validate(self) -> None:
|
|
112
|
+
for name, value in (
|
|
113
|
+
("soft_timeout_seconds", self.soft_timeout_seconds),
|
|
114
|
+
("hard_timeout_seconds", self.hard_timeout_seconds),
|
|
115
|
+
("max_work_units", self.max_work_units),
|
|
116
|
+
("max_implementation_attempts", self.max_implementation_attempts),
|
|
117
|
+
("max_replans", self.max_replans),
|
|
118
|
+
("max_replacements", self.max_replacements),
|
|
119
|
+
("max_review_attempts", self.max_review_attempts),
|
|
120
|
+
("parent_finalization_seconds", self.parent_finalization_seconds),
|
|
121
|
+
):
|
|
122
|
+
if type(value) is not int:
|
|
123
|
+
raise ValueError(f"{name} must be an integer")
|
|
124
|
+
if value < 0:
|
|
125
|
+
raise ValueError(f"{name} cannot be negative")
|
|
126
|
+
if self.soft_timeout_seconds < 1:
|
|
127
|
+
raise ValueError("soft_timeout_seconds must be positive")
|
|
128
|
+
if self.hard_timeout_seconds <= self.soft_timeout_seconds:
|
|
129
|
+
raise ValueError("hard_timeout_seconds must be greater than soft_timeout_seconds")
|
|
130
|
+
if self.max_work_units < 1:
|
|
131
|
+
raise ValueError("max_work_units must be positive")
|
|
132
|
+
if self.max_implementation_attempts < 1:
|
|
133
|
+
raise ValueError("max_implementation_attempts must be positive")
|
|
134
|
+
if self.parent_finalization_seconds < 1:
|
|
135
|
+
raise ValueError("parent_finalization_seconds must be positive")
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
REASONING_ROLLOUT_MODES = ("legacy", "shadow", "adaptive")
|
|
139
|
+
REASONING_EFFORTS = ("high", "xhigh", "max")
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
@dataclass(frozen=True)
|
|
143
|
+
class ReasoningRolloutPolicy:
|
|
144
|
+
"""Optional efficient-worker reasoning rollout configuration."""
|
|
145
|
+
|
|
146
|
+
mode: str = "shadow"
|
|
147
|
+
minimum: str = "high"
|
|
148
|
+
routine: str = "high"
|
|
149
|
+
complex: str = "xhigh"
|
|
150
|
+
critical: str = "max"
|
|
151
|
+
|
|
152
|
+
@classmethod
|
|
153
|
+
def from_dict(cls, value: Any) -> "ReasoningRolloutPolicy":
|
|
154
|
+
if type(value) is not dict:
|
|
155
|
+
raise ValueError("reasoning rollout policy must be an object")
|
|
156
|
+
fields = ("mode", "minimum", "routine", "complex", "critical")
|
|
157
|
+
unknown = sorted(set(value).difference(fields))
|
|
158
|
+
if unknown:
|
|
159
|
+
raise ValueError(f"reasoning rollout policy has unknown fields: {unknown}")
|
|
160
|
+
policy = cls(**{name: value[name] for name in fields if name in value})
|
|
161
|
+
policy.validate()
|
|
162
|
+
return policy
|
|
163
|
+
|
|
164
|
+
def validate(self) -> None:
|
|
165
|
+
if type(self.mode) is not str or self.mode not in REASONING_ROLLOUT_MODES:
|
|
166
|
+
raise ValueError(f"invalid reasoning rollout mode: {self.mode}")
|
|
167
|
+
for name, value in (
|
|
168
|
+
("minimum", self.minimum),
|
|
169
|
+
("routine", self.routine),
|
|
170
|
+
("complex", self.complex),
|
|
171
|
+
("critical", self.critical),
|
|
172
|
+
):
|
|
173
|
+
if type(value) is not str or value not in REASONING_EFFORTS:
|
|
174
|
+
raise ValueError(f"invalid reasoning rollout {name}: {value}")
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
@dataclass(frozen=True)
|
|
178
|
+
class ReasoningRolloutDecision:
|
|
179
|
+
"""Planner output comparing legacy and rollout-selected Worker effort."""
|
|
180
|
+
|
|
181
|
+
mode: str
|
|
182
|
+
legacy_worker_reasoning: str
|
|
183
|
+
proposed_worker_reasoning: str
|
|
184
|
+
selected_worker_reasoning: str
|
|
185
|
+
applied: bool
|
|
186
|
+
|
|
187
|
+
def validate(self) -> None:
|
|
188
|
+
if type(self.mode) is not str or self.mode not in REASONING_ROLLOUT_MODES:
|
|
189
|
+
raise ValueError(f"invalid reasoning rollout mode: {self.mode}")
|
|
190
|
+
for name, value in (
|
|
191
|
+
("legacy_worker_reasoning", self.legacy_worker_reasoning),
|
|
192
|
+
("proposed_worker_reasoning", self.proposed_worker_reasoning),
|
|
193
|
+
("selected_worker_reasoning", self.selected_worker_reasoning),
|
|
194
|
+
):
|
|
195
|
+
if type(value) is not str or value not in REASONING_EFFORTS:
|
|
196
|
+
raise ValueError(f"invalid {name}: {value}")
|
|
197
|
+
if type(self.applied) is not bool:
|
|
198
|
+
raise ValueError("reasoning rollout applied must be boolean")
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
@dataclass(frozen=True)
|
|
202
|
+
class StagePolicy:
|
|
203
|
+
"""Strategy-owned lifecycle preference for one delegated execution stage.
|
|
204
|
+
|
|
205
|
+
Soft timeout is advisory convergence/checkpoint budget and never implies
|
|
206
|
+
cancellation by itself. `max_worker_repair_attempts` bounds Worker-local
|
|
207
|
+
validation/fix loops independently from Parent-level repair cycles.
|
|
208
|
+
|
|
209
|
+
`work_unit_mode=bounded` describes logical acceptance-bounded transactions,
|
|
210
|
+
not permission for overlapping writers. `retry_review` is a read-only review
|
|
211
|
+
fallback and must never reopen writable implementation scope.
|
|
212
|
+
"""
|
|
213
|
+
|
|
214
|
+
join_policy: str
|
|
215
|
+
min_successful_workers: int
|
|
216
|
+
idle_timeout_seconds: int
|
|
217
|
+
hard_timeout_seconds: int
|
|
218
|
+
cancel_if_superseded: bool = True
|
|
219
|
+
cancel_stragglers_after_quorum: bool = False
|
|
220
|
+
fallback_policy: str = "parent_delta"
|
|
221
|
+
soft_timeout_seconds: int | None = None
|
|
222
|
+
checkpoint_rearm_seconds: int | None = None
|
|
223
|
+
max_worker_repair_attempts: int | None = None
|
|
224
|
+
work_unit_mode: str = "single"
|
|
225
|
+
minimum_work_units: int = 1
|
|
226
|
+
join_between_work_units: bool = False
|
|
227
|
+
maximum_work_units: int | None = None
|
|
228
|
+
require_write_paths: bool = False
|
|
229
|
+
|
|
230
|
+
def validate(self) -> None:
|
|
231
|
+
for name, value in (
|
|
232
|
+
("min_successful_workers", self.min_successful_workers),
|
|
233
|
+
("idle_timeout_seconds", self.idle_timeout_seconds),
|
|
234
|
+
("hard_timeout_seconds", self.hard_timeout_seconds),
|
|
235
|
+
("minimum_work_units", self.minimum_work_units),
|
|
236
|
+
):
|
|
237
|
+
if type(value) is not int:
|
|
238
|
+
raise ValueError(f"{name} must be an integer")
|
|
239
|
+
if self.join_policy not in JOIN_POLICIES:
|
|
240
|
+
raise ValueError(f"invalid join policy: {self.join_policy}")
|
|
241
|
+
if self.min_successful_workers < 0:
|
|
242
|
+
raise ValueError("min_successful_workers cannot be negative")
|
|
243
|
+
if self.join_policy == "opportunistic" and self.min_successful_workers != 0:
|
|
244
|
+
raise ValueError("opportunistic stages must use min_successful_workers=0")
|
|
245
|
+
if self.join_policy in {"quorum", "required"} and self.min_successful_workers < 1:
|
|
246
|
+
raise ValueError(f"{self.join_policy} stages require at least one successful worker")
|
|
247
|
+
if self.idle_timeout_seconds < 1:
|
|
248
|
+
raise ValueError("idle_timeout_seconds must be positive")
|
|
249
|
+
if self.hard_timeout_seconds < self.idle_timeout_seconds:
|
|
250
|
+
raise ValueError("hard_timeout_seconds must be >= idle_timeout_seconds")
|
|
251
|
+
if self.soft_timeout_seconds is not None:
|
|
252
|
+
if type(self.soft_timeout_seconds) is not int:
|
|
253
|
+
raise ValueError("soft_timeout_seconds must be an integer when set")
|
|
254
|
+
if self.soft_timeout_seconds < 1:
|
|
255
|
+
raise ValueError("soft_timeout_seconds must be positive when set")
|
|
256
|
+
if self.soft_timeout_seconds >= self.hard_timeout_seconds:
|
|
257
|
+
raise ValueError("soft_timeout_seconds must be lower than hard_timeout_seconds")
|
|
258
|
+
if self.checkpoint_rearm_seconds is not None:
|
|
259
|
+
if type(self.checkpoint_rearm_seconds) is not int:
|
|
260
|
+
raise ValueError("checkpoint_rearm_seconds must be an integer when set")
|
|
261
|
+
if self.checkpoint_rearm_seconds < 1:
|
|
262
|
+
raise ValueError("checkpoint_rearm_seconds must be positive when set")
|
|
263
|
+
if self.checkpoint_rearm_seconds >= self.hard_timeout_seconds:
|
|
264
|
+
raise ValueError("checkpoint_rearm_seconds must be lower than hard_timeout_seconds")
|
|
265
|
+
if (
|
|
266
|
+
self.soft_timeout_seconds is not None
|
|
267
|
+
and self.soft_timeout_seconds + self.checkpoint_rearm_seconds >= self.hard_timeout_seconds
|
|
268
|
+
):
|
|
269
|
+
raise ValueError(
|
|
270
|
+
"checkpoint_rearm_seconds must leave time for a second checkpoint before hard_timeout_seconds"
|
|
271
|
+
)
|
|
272
|
+
if self.max_worker_repair_attempts is not None:
|
|
273
|
+
if type(self.max_worker_repair_attempts) is not int:
|
|
274
|
+
raise ValueError("max_worker_repair_attempts must be an integer when set")
|
|
275
|
+
if self.max_worker_repair_attempts < 0:
|
|
276
|
+
raise ValueError("max_worker_repair_attempts cannot be negative")
|
|
277
|
+
if self.work_unit_mode not in WORK_UNIT_MODES:
|
|
278
|
+
raise ValueError(f"invalid work_unit_mode: {self.work_unit_mode}")
|
|
279
|
+
if self.minimum_work_units < 1:
|
|
280
|
+
raise ValueError("minimum_work_units must be positive")
|
|
281
|
+
if type(self.join_between_work_units) is not bool:
|
|
282
|
+
raise ValueError("join_between_work_units must be boolean")
|
|
283
|
+
if self.maximum_work_units is not None:
|
|
284
|
+
if type(self.maximum_work_units) is not int:
|
|
285
|
+
raise ValueError("maximum_work_units must be an integer when set")
|
|
286
|
+
if self.maximum_work_units < 1:
|
|
287
|
+
raise ValueError("maximum_work_units must be positive when set")
|
|
288
|
+
if self.work_unit_mode == "single":
|
|
289
|
+
if self.minimum_work_units != 1:
|
|
290
|
+
raise ValueError("single work-unit mode requires minimum_work_units=1")
|
|
291
|
+
if self.join_between_work_units:
|
|
292
|
+
raise ValueError("single work-unit mode cannot join between work units")
|
|
293
|
+
if self.maximum_work_units is not None and self.maximum_work_units != 1:
|
|
294
|
+
raise ValueError("single work-unit mode requires maximum_work_units=1")
|
|
295
|
+
elif not self.join_between_work_units:
|
|
296
|
+
raise ValueError("bounded work-unit mode requires a Parent join between work units")
|
|
297
|
+
elif self.maximum_work_units is not None and self.maximum_work_units < self.minimum_work_units:
|
|
298
|
+
raise ValueError("bounded work-unit mode requires maximum_work_units >= minimum_work_units")
|
|
299
|
+
if type(self.require_write_paths) is not bool:
|
|
300
|
+
raise ValueError("require_write_paths must be boolean")
|
|
301
|
+
if self.fallback_policy not in FALLBACK_POLICIES:
|
|
302
|
+
raise ValueError(f"invalid fallback policy: {self.fallback_policy}")
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def standard_lifecycle(_task: Task, stage: str) -> StagePolicy:
|
|
306
|
+
"""V11-safe balanced lifecycle baseline for future strategies."""
|
|
307
|
+
if stage == "exploration":
|
|
308
|
+
return StagePolicy("quorum", 1, 180, 1200, True, True, "parent_delta")
|
|
309
|
+
if stage == "implementation":
|
|
310
|
+
return StagePolicy(
|
|
311
|
+
"required",
|
|
312
|
+
1,
|
|
313
|
+
240,
|
|
314
|
+
2400,
|
|
315
|
+
False,
|
|
316
|
+
False,
|
|
317
|
+
"replan",
|
|
318
|
+
soft_timeout_seconds=1200,
|
|
319
|
+
checkpoint_rearm_seconds=240,
|
|
320
|
+
max_worker_repair_attempts=1,
|
|
321
|
+
work_unit_mode="single",
|
|
322
|
+
minimum_work_units=1,
|
|
323
|
+
join_between_work_units=False,
|
|
324
|
+
maximum_work_units=1,
|
|
325
|
+
require_write_paths=False,
|
|
326
|
+
)
|
|
327
|
+
if stage == "review":
|
|
328
|
+
return StagePolicy("required", 1, 180, 1800, True, False, "retry_review")
|
|
329
|
+
raise ValueError(f"invalid lifecycle stage: {stage}")
|
|
330
|
+
|
|
331
|
+
|
|
332
|
+
@dataclass(frozen=True)
|
|
333
|
+
class StrategySpec:
|
|
334
|
+
name: str
|
|
335
|
+
description: str
|
|
336
|
+
adaptive_route: RouteFn
|
|
337
|
+
effort: EffortFn
|
|
338
|
+
worker_budget: BudgetFn
|
|
339
|
+
independent_review: PredicateFn
|
|
340
|
+
capability: CapabilityFn = worker_capability
|
|
341
|
+
exploration_bonus: DemandFn = zero_demand
|
|
342
|
+
reviewer_bonus: DemandFn = zero_demand
|
|
343
|
+
notes: NotesFn = no_notes
|
|
344
|
+
lifecycle: LifecycleFn = standard_lifecycle
|
|
345
|
+
allow_parallel_write: bool = False
|
|
346
|
+
quota_sensitive: bool = False
|
|
347
|
+
task_budget: TaskBudgetFn | None = None
|
|
348
|
+
reasoning_rollout: ReasoningRolloutFn | None = None
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def standard_effort(task: Task, role: str) -> str:
|
|
352
|
+
"""Cheap-parent / strong-worker baseline."""
|
|
353
|
+
if task.complexity == "critical" or task.risk == "critical":
|
|
354
|
+
return "xhigh" if role == "parent" else "max"
|
|
355
|
+
return "high" if role == "parent" else "xhigh"
|
|
356
|
+
|
|
357
|
+
|
|
358
|
+
def never(_task: Task) -> bool:
|
|
359
|
+
return False
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def small_low_risk_is_direct(task: Task) -> bool:
|
|
363
|
+
return task.complexity == "small" and task.risk in {"low", "medium"}
|
|
@@ -0,0 +1,158 @@
|
|
|
1
|
+
"""Quota-efficient strategy."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from .base import (
|
|
5
|
+
StagePolicy,
|
|
6
|
+
ReasoningRolloutDecision,
|
|
7
|
+
StrategySpec,
|
|
8
|
+
TaskBudgetPolicy,
|
|
9
|
+
WorkerBudget,
|
|
10
|
+
never,
|
|
11
|
+
small_low_risk_is_direct,
|
|
12
|
+
standard_effort,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
_EFFORT_RANK = {"high": 0, "xhigh": 1, "max": 2}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def _max_effort(*values: str) -> str:
|
|
20
|
+
return max(values, key=lambda value: _EFFORT_RANK[value])
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def adaptive_route(task) -> str:
|
|
24
|
+
if small_low_risk_is_direct(task):
|
|
25
|
+
return "direct"
|
|
26
|
+
return "delegate" if task.iteration_intensity != "one-shot" or task.scope in {"cross-module", "repo-wide"} else "direct"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def worker_budget(task) -> WorkerBudget:
|
|
30
|
+
if task.complexity == "critical" or task.scope == "repo-wide" or task.iteration_intensity == "heavy-loop":
|
|
31
|
+
return WorkerBudget(2, 2, 1, 5, "low")
|
|
32
|
+
if task.uncertainty == "high" or task.exploration_need == "high":
|
|
33
|
+
return WorkerBudget(2, 2, 1, 5, "low")
|
|
34
|
+
return WorkerBudget(1, 1, 1, 2, "low")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def implementation_soft_timeout(task) -> int:
|
|
38
|
+
if task.complexity == "critical" or task.risk == "critical":
|
|
39
|
+
return 1200
|
|
40
|
+
if task.complexity == "complex" or task.scope == "repo-wide" or task.iteration_intensity == "heavy-loop":
|
|
41
|
+
return 900
|
|
42
|
+
return 600
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def implementation_checkpoint_rearm_seconds(task) -> int:
|
|
46
|
+
if task.complexity == "critical" or task.risk == "critical":
|
|
47
|
+
return 300
|
|
48
|
+
if (
|
|
49
|
+
task.complexity == "complex"
|
|
50
|
+
or task.scope in {"cross-module", "repo-wide"}
|
|
51
|
+
or task.iteration_intensity == "heavy-loop"
|
|
52
|
+
):
|
|
53
|
+
return 240
|
|
54
|
+
return 180
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def implementation_repair_attempts(task) -> int:
|
|
58
|
+
if (
|
|
59
|
+
task.complexity in {"complex", "critical"}
|
|
60
|
+
or task.risk == "critical"
|
|
61
|
+
or task.scope == "repo-wide"
|
|
62
|
+
or task.iteration_intensity == "heavy-loop"
|
|
63
|
+
):
|
|
64
|
+
return 2
|
|
65
|
+
return 1
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def implementation_minimum_work_units(_task) -> int:
|
|
69
|
+
return 1
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def implementation_maximum_work_units(task) -> int:
|
|
73
|
+
if task.scope == "repo-wide" or task.iteration_intensity == "heavy-loop":
|
|
74
|
+
semantic_maximum = 3
|
|
75
|
+
elif task.complexity in {"complex", "critical"} or task.scope == "cross-module" or task.risk == "critical":
|
|
76
|
+
semantic_maximum = 2
|
|
77
|
+
else:
|
|
78
|
+
semantic_maximum = 1
|
|
79
|
+
topology_floor = min(task.writable_workstreams, worker_budget(task).max_implementers)
|
|
80
|
+
return max(semantic_maximum, topology_floor)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def task_budget(task) -> TaskBudgetPolicy:
|
|
84
|
+
max_work_units = implementation_maximum_work_units(task)
|
|
85
|
+
return TaskBudgetPolicy(
|
|
86
|
+
soft_timeout_seconds=1500,
|
|
87
|
+
hard_timeout_seconds=1800,
|
|
88
|
+
max_work_units=max_work_units,
|
|
89
|
+
max_implementation_attempts=max_work_units + 1,
|
|
90
|
+
max_replans=1,
|
|
91
|
+
max_replacements=1,
|
|
92
|
+
max_review_attempts=2,
|
|
93
|
+
parent_finalization_seconds=150,
|
|
94
|
+
)
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def reasoning_rollout(task, _role: str, policy, parent_reasoning: str, legacy_worker_reasoning: str) -> ReasoningRolloutDecision:
|
|
98
|
+
policy.validate()
|
|
99
|
+
if task.complexity == "critical" or task.risk == "critical":
|
|
100
|
+
class_target = policy.critical
|
|
101
|
+
elif task.complexity == "complex" or task.risk == "high":
|
|
102
|
+
class_target = policy.complex
|
|
103
|
+
else:
|
|
104
|
+
class_target = policy.routine
|
|
105
|
+
proposed = _max_effort(class_target, policy.minimum, parent_reasoning)
|
|
106
|
+
selected = proposed if policy.mode == "adaptive" else legacy_worker_reasoning
|
|
107
|
+
decision = ReasoningRolloutDecision(
|
|
108
|
+
mode=policy.mode,
|
|
109
|
+
legacy_worker_reasoning=legacy_worker_reasoning,
|
|
110
|
+
proposed_worker_reasoning=proposed,
|
|
111
|
+
selected_worker_reasoning=selected,
|
|
112
|
+
applied=policy.mode == "adaptive",
|
|
113
|
+
)
|
|
114
|
+
decision.validate()
|
|
115
|
+
return decision
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def lifecycle(task, stage: str) -> StagePolicy:
|
|
119
|
+
if stage == "exploration":
|
|
120
|
+
return StagePolicy("quorum", 1, 120, 900, True, True, "parent_delta")
|
|
121
|
+
if stage == "implementation":
|
|
122
|
+
minimum_work_units = implementation_minimum_work_units(task)
|
|
123
|
+
maximum_work_units = implementation_maximum_work_units(task)
|
|
124
|
+
bounded_mode = maximum_work_units > 1
|
|
125
|
+
return StagePolicy(
|
|
126
|
+
"required", 1, 180, 1800, False, False, "replan",
|
|
127
|
+
soft_timeout_seconds=implementation_soft_timeout(task),
|
|
128
|
+
checkpoint_rearm_seconds=implementation_checkpoint_rearm_seconds(task),
|
|
129
|
+
max_worker_repair_attempts=implementation_repair_attempts(task),
|
|
130
|
+
work_unit_mode="bounded" if bounded_mode else "single",
|
|
131
|
+
minimum_work_units=minimum_work_units,
|
|
132
|
+
join_between_work_units=bounded_mode,
|
|
133
|
+
maximum_work_units=maximum_work_units,
|
|
134
|
+
require_write_paths=bounded_mode,
|
|
135
|
+
)
|
|
136
|
+
if stage == "review":
|
|
137
|
+
return StagePolicy("quorum", 1, 150, 1200, True, True, "retry_review")
|
|
138
|
+
raise ValueError(f"invalid lifecycle stage: {stage}")
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
STRATEGY = StrategySpec(
|
|
142
|
+
name="efficient",
|
|
143
|
+
description="minimize expensive parent usage and total waste while offloading deep execution to efficient workers",
|
|
144
|
+
adaptive_route=adaptive_route,
|
|
145
|
+
effort=standard_effort,
|
|
146
|
+
worker_budget=worker_budget,
|
|
147
|
+
independent_review=never,
|
|
148
|
+
lifecycle=lifecycle,
|
|
149
|
+
# Efficient workers are deliberately cheap enough that proven parallel
|
|
150
|
+
# implementation should not require an aggressive fan-out override.
|
|
151
|
+
allow_parallel_write=True,
|
|
152
|
+
# Keep quota pressure focused on expensive Parent usage. The strategy's own
|
|
153
|
+
# low-speculation WorkerBudget remains the hard envelope (2 implementers,
|
|
154
|
+
# 5 total Workers at the largest task classes).
|
|
155
|
+
quota_sensitive=False,
|
|
156
|
+
task_budget=task_budget,
|
|
157
|
+
reasoning_rollout=reasoning_rollout,
|
|
158
|
+
)
|