codex-flow 2.1.13__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codex_flow/__init__.py +28 -0
- codex_flow/__main__.py +9 -0
- codex_flow/cli.py +242 -0
- codex_flow/data/LICENSE +21 -0
- codex_flow/data/README.en.md +303 -0
- codex_flow/data/README.md +305 -0
- codex_flow/data/VERSION +1 -0
- codex_flow/data/apps/chatgpt-mcp/README.md +86 -0
- codex_flow/data/apps/chatgpt-mcp/__init__.py +1 -0
- codex_flow/data/apps/chatgpt-mcp/adapter.py +458 -0
- codex_flow/data/apps/chatgpt-mcp/server.py +358 -0
- codex_flow/data/apps/chatgpt-mcp/widget.html +927 -0
- codex_flow/data/apps/macos-overlay/README.en.md +121 -0
- codex_flow/data/apps/macos-overlay/README.md +123 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayRuntimeState.swift +126 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayScreenGeometry.swift +82 -0
- codex_flow/data/apps/macos-overlay/Sources/Controllers/OverlayWindowController.swift +1052 -0
- codex_flow/data/apps/macos-overlay/Sources/Localization.swift +197 -0
- codex_flow/data/apps/macos-overlay/Sources/Models/TelemetryData.swift +1557 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/AccountSnapshotService.swift +1101 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/FlowPilotInstanceLock.swift +153 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/IPCServer.swift +298 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryQueryEngine.swift +800 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/TelemetryWatcher.swift +135 -0
- codex_flow/data/apps/macos-overlay/Sources/Services/UpdateService.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AccountView.swift +610 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AnalyticsView.swift +566 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/AutostartView.swift +293 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/BubbleView.swift +317 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HistoryView.swift +1124 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/HoverRevealText.swift +165 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/InspectorSkillsToolsView.swift +121 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/LogoView.swift +182 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SleekSwitch.swift +117 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/StrategyModeView.swift +561 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/SummaryView.swift +1273 -0
- codex_flow/data/apps/macos-overlay/Sources/Views/UpdateView.swift +352 -0
- codex_flow/data/apps/macos-overlay/Sources/main.swift +340 -0
- codex_flow/data/apps/macos-overlay/Tests/OverlayScreenGeometryTests.swift +163 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryPhase1ContractTests.swift +357 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQueryEngineConcurrencyTests.swift +221 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryQuotaSelectionTests.swift +158 -0
- codex_flow/data/apps/macos-overlay/Tests/TelemetryWorkerTokenTests.swift +122 -0
- codex_flow/data/apps/macos-overlay/build.sh +75 -0
- codex_flow/data/benchmark/corpus.json +103 -0
- codex_flow/data/benchmark/manifest.example.json +41 -0
- codex_flow/data/benchmark/manifest.schema.json +137 -0
- codex_flow/data/benchmark/prices/gpt-5.6-2026-08-30.json +5 -0
- codex_flow/data/benchmark/profiles.json +90 -0
- codex_flow/data/benchmark/schema.json +77 -0
- codex_flow/data/benchmark/tasks.json +50 -0
- codex_flow/data/completions/codex-flow.bash +34 -0
- codex_flow/data/completions/codex-flow.zsh +52 -0
- codex_flow/data/glama.json +6 -0
- codex_flow/data/install-release.ps1 +126 -0
- codex_flow/data/install-release.sh +155 -0
- codex_flow/data/install.ps1 +349 -0
- codex_flow/data/install.sh +362 -0
- codex_flow/data/policy/benchmark.toml +49 -0
- codex_flow/data/policy/defaults.toml +70 -0
- codex_flow/data/scripts/analyze-benchmark.py +510 -0
- codex_flow/data/scripts/benchmark-local.py +171 -0
- codex_flow/data/scripts/check-recommendation.py +277 -0
- codex_flow/data/scripts/doctor.py +449 -0
- codex_flow/data/scripts/generate-release-manifest.py +74 -0
- codex_flow/data/scripts/localization.py +192 -0
- codex_flow/data/scripts/manage-hooks.py +448 -0
- codex_flow/data/scripts/manage-instructions.py +389 -0
- codex_flow/data/scripts/manage-shell.py +151 -0
- codex_flow/data/scripts/materialize-corpus.py +193 -0
- codex_flow/data/scripts/menu.py +646 -0
- codex_flow/data/scripts/migrations/0001_update_settings.py +80 -0
- codex_flow/data/scripts/package-release.py +132 -0
- codex_flow/data/scripts/render-benchmark-report.py +292 -0
- codex_flow/data/scripts/run-benchmark.py +829 -0
- codex_flow/data/scripts/strategies/__init__.py +28 -0
- codex_flow/data/scripts/strategies/balanced.py +115 -0
- codex_flow/data/scripts/strategies/base.py +363 -0
- codex_flow/data/scripts/strategies/efficient.py +158 -0
- codex_flow/data/scripts/strategies/lifecycle_runtime.py +590 -0
- codex_flow/data/scripts/strategies/quality.py +209 -0
- codex_flow/data/scripts/strategies/speed.py +108 -0
- codex_flow/data/scripts/strategies/task_budget_runtime.py +644 -0
- codex_flow/data/scripts/strategies/task_phase_runtime.py +341 -0
- codex_flow/data/scripts/strategies/work_unit_runtime.py +421 -0
- codex_flow/data/scripts/strategy_runtime.py +1091 -0
- codex_flow/data/scripts/telemetry.py +400 -0
- codex_flow/data/scripts/telemetry_core/__init__.py +192 -0
- codex_flow/data/scripts/telemetry_core/app_server.py +1192 -0
- codex_flow/data/scripts/telemetry_core/collector.py +1247 -0
- codex_flow/data/scripts/telemetry_core/common.py +421 -0
- codex_flow/data/scripts/telemetry_core/latency.py +593 -0
- codex_flow/data/scripts/telemetry_core/query.py +427 -0
- codex_flow/data/scripts/telemetry_core/quota_ledger.py +598 -0
- codex_flow/data/scripts/telemetry_core/render.py +460 -0
- codex_flow/data/scripts/telemetry_core/repair.py +223 -0
- codex_flow/data/scripts/ui.py +266 -0
- codex_flow/data/scripts/update-homebrew-formula.py +146 -0
- codex_flow/data/scripts/update_runtime_config.py +134 -0
- codex_flow/data/scripts/updater.py +1718 -0
- codex_flow/data/smithery.yaml +18 -0
- codex_flow/data/templates/agents/worker-explorer.toml +24 -0
- codex_flow/data/templates/agents/worker-implementer.toml +49 -0
- codex_flow/data/templates/agents/worker-reviewer.toml +25 -0
- codex_flow/data/templates/flow-pilot-instructions.md +35 -0
- codex_flow/data/templates/skills/flow-pilot/SKILL.md +577 -0
- codex_flow/mcp.py +35 -0
- codex_flow-2.1.13.dist-info/METADATA +342 -0
- codex_flow-2.1.13.dist-info/RECORD +113 -0
- codex_flow-2.1.13.dist-info/WHEEL +5 -0
- codex_flow-2.1.13.dist-info/entry_points.txt +3 -0
- codex_flow-2.1.13.dist-info/licenses/LICENSE +21 -0
- codex_flow-2.1.13.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,593 @@
|
|
|
1
|
+
"""Redacted Worker latency events and deterministic rollout reports.
|
|
2
|
+
|
|
3
|
+
This module deliberately has no transcript or prompt integration. Callers
|
|
4
|
+
provide a small, already-decided lifecycle observation; the ledger stores only
|
|
5
|
+
validated enums/tokens, salted identifiers, and timing/counter facts.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import hashlib
|
|
11
|
+
import json
|
|
12
|
+
import math
|
|
13
|
+
import os
|
|
14
|
+
import re
|
|
15
|
+
import secrets
|
|
16
|
+
import time
|
|
17
|
+
from contextlib import contextmanager
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
from typing import Any, Iterator
|
|
20
|
+
|
|
21
|
+
from .common import STATE_ROOT
|
|
22
|
+
|
|
23
|
+
LATENCY_SCHEMA_VERSION = 1
|
|
24
|
+
LATENCY_FILE_NAME = "latency.jsonl"
|
|
25
|
+
LATENCY_SALT_FILE_NAME = ".latency-salt"
|
|
26
|
+
LATENCY_LOCK_FILE_NAME = ".latency.lock"
|
|
27
|
+
LOCK_TIMEOUT_SECONDS = 3.0
|
|
28
|
+
STALE_LOCK_SECONDS = 30.0
|
|
29
|
+
|
|
30
|
+
STRATEGIES = frozenset(("efficient", "balanced", "quality", "speed"))
|
|
31
|
+
TASK_CLASSES = frozenset(("small", "routine", "complex", "critical"))
|
|
32
|
+
STAGES = frozenset(("exploration", "implementation", "review"))
|
|
33
|
+
ROLES = frozenset(("explorer", "implementer", "reviewer"))
|
|
34
|
+
ROLLOUT_MODES = frozenset(("legacy", "shadow", "adaptive"))
|
|
35
|
+
EFFORTS = frozenset(("high", "xhigh", "max"))
|
|
36
|
+
OUTCOMES = frozenset(("completed", "failed", "cancelled", "timeout"))
|
|
37
|
+
BOUNDARIES = frozenset(("terminal", "checkpoint"))
|
|
38
|
+
MODEL_TOKEN = re.compile(r"^[A-Za-z0-9][A-Za-z0-9._:@+-]{0,127}$")
|
|
39
|
+
HASH_TOKEN = re.compile(r"^[0-9a-f]{64}$")
|
|
40
|
+
|
|
41
|
+
# Input aliases are intentionally finite. Unknown prompt/transcript/path
|
|
42
|
+
# fields fail closed instead of being silently carried into telemetry.
|
|
43
|
+
_INPUT_FIELDS = frozenset(
|
|
44
|
+
{
|
|
45
|
+
"event_id",
|
|
46
|
+
"task_id",
|
|
47
|
+
"worker_id",
|
|
48
|
+
"work_unit_id",
|
|
49
|
+
"strategy",
|
|
50
|
+
"task_class",
|
|
51
|
+
"stage",
|
|
52
|
+
"role",
|
|
53
|
+
"model",
|
|
54
|
+
"rollout_mode",
|
|
55
|
+
"legacy_effort",
|
|
56
|
+
"proposed_effort",
|
|
57
|
+
"selected_effort",
|
|
58
|
+
"observed_effort",
|
|
59
|
+
"legacy_worker_reasoning",
|
|
60
|
+
"proposed_worker_reasoning",
|
|
61
|
+
"selected_worker_reasoning",
|
|
62
|
+
"observed_worker_reasoning",
|
|
63
|
+
"outcome",
|
|
64
|
+
"boundary",
|
|
65
|
+
"event_type",
|
|
66
|
+
"started_at",
|
|
67
|
+
"finished_at",
|
|
68
|
+
"started",
|
|
69
|
+
"finished",
|
|
70
|
+
"duration_seconds",
|
|
71
|
+
"repair_count",
|
|
72
|
+
"repair_attempts",
|
|
73
|
+
"checkpoint_count",
|
|
74
|
+
"checkpoints",
|
|
75
|
+
}
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
class LatencyError(ValueError):
|
|
80
|
+
"""A concise, user-actionable latency telemetry validation error."""
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def latency_file(path: Path | str | None = None) -> Path:
|
|
84
|
+
if path is not None:
|
|
85
|
+
return Path(path)
|
|
86
|
+
configured = os.environ.get("CODEX_FLOW_LATENCY_FILE")
|
|
87
|
+
return Path(configured) if configured else STATE_ROOT / LATENCY_FILE_NAME
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _sidecar(path: Path, name: str) -> Path:
|
|
91
|
+
return path.parent / name
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _ensure_parent(path: Path) -> None:
|
|
95
|
+
if path.exists() and path.is_symlink():
|
|
96
|
+
raise LatencyError(f"refusing symlink telemetry target: {path.name}")
|
|
97
|
+
try:
|
|
98
|
+
path.mkdir(parents=True, exist_ok=True, mode=0o700)
|
|
99
|
+
try:
|
|
100
|
+
os.chmod(path, 0o700)
|
|
101
|
+
except OSError:
|
|
102
|
+
pass
|
|
103
|
+
except OSError as exc:
|
|
104
|
+
raise LatencyError(f"cannot prepare telemetry state: {exc}") from None
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def _reject_symlink(path: Path, label: str) -> None:
|
|
108
|
+
try:
|
|
109
|
+
if path.is_symlink():
|
|
110
|
+
raise LatencyError(f"refusing symlink {label}: {path.name}")
|
|
111
|
+
except OSError as exc:
|
|
112
|
+
raise LatencyError(f"cannot inspect {label}: {exc}") from None
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
@contextmanager
|
|
116
|
+
def _exclusive_lock(path: Path) -> Iterator[None]:
|
|
117
|
+
_ensure_parent(path.parent)
|
|
118
|
+
_reject_symlink(path, "latency lock")
|
|
119
|
+
deadline = time.monotonic() + LOCK_TIMEOUT_SECONDS
|
|
120
|
+
fd: int | None = None
|
|
121
|
+
while fd is None:
|
|
122
|
+
_reject_symlink(path, "latency lock")
|
|
123
|
+
try:
|
|
124
|
+
fd = os.open(str(path), os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
|
125
|
+
os.write(fd, f"{os.getpid()}\n".encode("ascii", "replace"))
|
|
126
|
+
os.fsync(fd)
|
|
127
|
+
os.close(fd)
|
|
128
|
+
fd = -1
|
|
129
|
+
try:
|
|
130
|
+
os.chmod(path, 0o600)
|
|
131
|
+
except OSError:
|
|
132
|
+
pass
|
|
133
|
+
except FileExistsError:
|
|
134
|
+
if time.monotonic() >= deadline:
|
|
135
|
+
raise LatencyError("latency telemetry lock is busy") from None
|
|
136
|
+
try:
|
|
137
|
+
age = max(0.0, time.time() - path.stat().st_mtime)
|
|
138
|
+
if age > STALE_LOCK_SECONDS:
|
|
139
|
+
path.unlink()
|
|
140
|
+
continue
|
|
141
|
+
except FileNotFoundError:
|
|
142
|
+
continue
|
|
143
|
+
except OSError:
|
|
144
|
+
pass
|
|
145
|
+
time.sleep(0.015)
|
|
146
|
+
except OSError as exc:
|
|
147
|
+
if fd is not None and fd >= 0:
|
|
148
|
+
try:
|
|
149
|
+
os.close(fd)
|
|
150
|
+
except OSError:
|
|
151
|
+
pass
|
|
152
|
+
raise LatencyError(f"cannot acquire latency telemetry lock: {exc}") from None
|
|
153
|
+
try:
|
|
154
|
+
yield
|
|
155
|
+
finally:
|
|
156
|
+
_reject_symlink(path, "latency lock")
|
|
157
|
+
try:
|
|
158
|
+
path.unlink()
|
|
159
|
+
except FileNotFoundError:
|
|
160
|
+
pass
|
|
161
|
+
except OSError:
|
|
162
|
+
# A lock leak is safer than deleting a replacement lock owned by a
|
|
163
|
+
# concurrent process after an unusual filesystem race.
|
|
164
|
+
pass
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def _string(value: Any, name: str, *, required: bool = True, limit: int = 512) -> str | None:
|
|
168
|
+
if value is None and not required:
|
|
169
|
+
return None
|
|
170
|
+
if type(value) is not str or not value.strip():
|
|
171
|
+
raise LatencyError(f"{name} must be a non-empty string")
|
|
172
|
+
value = value.strip()
|
|
173
|
+
if len(value) > limit or any(ord(ch) < 0x20 or ord(ch) == 0x7F for ch in value):
|
|
174
|
+
raise LatencyError(f"{name} contains invalid characters")
|
|
175
|
+
return value
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _enum(value: Any, name: str, allowed: frozenset[str], *, required: bool = True) -> str | None:
|
|
179
|
+
value = _string(value, name, required=required)
|
|
180
|
+
if value is None:
|
|
181
|
+
return None
|
|
182
|
+
if value not in allowed:
|
|
183
|
+
choices = ", ".join(sorted(allowed))
|
|
184
|
+
raise LatencyError(f"{name} must be one of: {choices}")
|
|
185
|
+
return value
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _timestamp(value: Any, name: str, *, required: bool = False) -> float | None:
|
|
189
|
+
if value is None and not required:
|
|
190
|
+
return None
|
|
191
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
192
|
+
raise LatencyError(f"{name} must be finite Unix seconds")
|
|
193
|
+
result = float(value)
|
|
194
|
+
if not math.isfinite(result) or result < 0:
|
|
195
|
+
raise LatencyError(f"{name} must be finite, non-negative Unix seconds")
|
|
196
|
+
if result > 100_000_000_000:
|
|
197
|
+
raise LatencyError(f"{name} looks like milliseconds; use Unix seconds")
|
|
198
|
+
return int(result) if result.is_integer() else result
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _count(value: Any, name: str, *, default: int = 0) -> int:
|
|
202
|
+
if value is None:
|
|
203
|
+
return default
|
|
204
|
+
if type(value) is not int or value < 0:
|
|
205
|
+
raise LatencyError(f"{name} must be a non-negative integer")
|
|
206
|
+
return value
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def _duration(value: Any) -> float | None:
|
|
210
|
+
if value is None:
|
|
211
|
+
return None
|
|
212
|
+
if isinstance(value, bool) or not isinstance(value, (int, float)):
|
|
213
|
+
raise LatencyError("duration_seconds must be finite, non-negative seconds")
|
|
214
|
+
result = float(value)
|
|
215
|
+
if not math.isfinite(result) or result < 0 or result > 100_000_000_000:
|
|
216
|
+
raise LatencyError("duration_seconds must be finite, non-negative seconds")
|
|
217
|
+
return int(result) if result.is_integer() else result
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def _first(event: dict[str, Any], *names: str) -> Any:
|
|
221
|
+
for name in names:
|
|
222
|
+
if name in event:
|
|
223
|
+
return event[name]
|
|
224
|
+
return None
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def _normalize(event: Any) -> dict[str, Any]:
|
|
228
|
+
if type(event) is not dict:
|
|
229
|
+
raise LatencyError("latency event must be a JSON object")
|
|
230
|
+
unknown = sorted(set(event).difference(_INPUT_FIELDS))
|
|
231
|
+
if unknown:
|
|
232
|
+
raise LatencyError(f"unknown latency event fields: {', '.join(unknown)}")
|
|
233
|
+
|
|
234
|
+
model = _string(event.get("model"), "model", limit=128)
|
|
235
|
+
if model is None or not MODEL_TOKEN.fullmatch(model):
|
|
236
|
+
raise LatencyError("model must be a safe model token, not a path or free-form text")
|
|
237
|
+
|
|
238
|
+
result: dict[str, Any] = {
|
|
239
|
+
"event_id": _string(event.get("event_id"), "event_id"),
|
|
240
|
+
"task_id": _string(event.get("task_id"), "task_id"),
|
|
241
|
+
"worker_id": _string(event.get("worker_id"), "worker_id"),
|
|
242
|
+
"work_unit_id": _string(event.get("work_unit_id"), "work_unit_id", required=False),
|
|
243
|
+
"strategy": _enum(event.get("strategy"), "strategy", STRATEGIES),
|
|
244
|
+
"task_class": _enum(event.get("task_class"), "task_class", TASK_CLASSES),
|
|
245
|
+
"stage": _enum(event.get("stage"), "stage", STAGES),
|
|
246
|
+
"role": _enum(event.get("role"), "role", ROLES),
|
|
247
|
+
"model": model,
|
|
248
|
+
"rollout_mode": _enum(event.get("rollout_mode"), "rollout_mode", ROLLOUT_MODES),
|
|
249
|
+
"legacy_effort": _enum(
|
|
250
|
+
_first(event, "legacy_effort", "legacy_worker_reasoning"),
|
|
251
|
+
"legacy_effort",
|
|
252
|
+
EFFORTS,
|
|
253
|
+
),
|
|
254
|
+
"proposed_effort": _enum(
|
|
255
|
+
_first(event, "proposed_effort", "proposed_worker_reasoning"),
|
|
256
|
+
"proposed_effort",
|
|
257
|
+
EFFORTS,
|
|
258
|
+
),
|
|
259
|
+
"selected_effort": _enum(
|
|
260
|
+
_first(event, "selected_effort", "selected_worker_reasoning"),
|
|
261
|
+
"selected_effort",
|
|
262
|
+
EFFORTS,
|
|
263
|
+
),
|
|
264
|
+
"observed_effort": _enum(
|
|
265
|
+
_first(event, "observed_effort", "observed_worker_reasoning"),
|
|
266
|
+
"observed_effort",
|
|
267
|
+
EFFORTS,
|
|
268
|
+
required=False,
|
|
269
|
+
),
|
|
270
|
+
"boundary": _enum(
|
|
271
|
+
_first(event, "boundary", "event_type") or "terminal",
|
|
272
|
+
"boundary",
|
|
273
|
+
BOUNDARIES,
|
|
274
|
+
),
|
|
275
|
+
"outcome": _enum(event.get("outcome"), "outcome", OUTCOMES, required=False),
|
|
276
|
+
"started_at": _timestamp(_first(event, "started_at", "started"), "started_at", required=True),
|
|
277
|
+
"finished_at": _timestamp(_first(event, "finished_at", "finished"), "finished_at"),
|
|
278
|
+
"duration_seconds": _duration(event.get("duration_seconds")),
|
|
279
|
+
"repair_count": _count(_first(event, "repair_count", "repair_attempts"), "repair_count"),
|
|
280
|
+
"checkpoint_count": _count(_first(event, "checkpoint_count", "checkpoints"), "checkpoint_count"),
|
|
281
|
+
}
|
|
282
|
+
if result["boundary"] == "terminal" and result["outcome"] is None:
|
|
283
|
+
raise LatencyError("terminal latency event requires outcome")
|
|
284
|
+
if result["boundary"] == "checkpoint" and result["outcome"] is not None:
|
|
285
|
+
raise LatencyError("checkpoint latency event must not carry a terminal outcome")
|
|
286
|
+
started = result["started_at"]
|
|
287
|
+
finished = result["finished_at"]
|
|
288
|
+
if finished is not None and finished < started:
|
|
289
|
+
raise LatencyError("finished_at cannot precede started_at")
|
|
290
|
+
if finished is not None:
|
|
291
|
+
derived = finished - started
|
|
292
|
+
supplied = result["duration_seconds"]
|
|
293
|
+
if supplied is not None and not math.isclose(float(supplied), float(derived), rel_tol=0.0, abs_tol=1e-6):
|
|
294
|
+
raise LatencyError("duration_seconds must match finished_at - started_at")
|
|
295
|
+
result["duration_seconds"] = int(derived) if float(derived).is_integer() else derived
|
|
296
|
+
elif result["duration_seconds"] is not None:
|
|
297
|
+
derived_finish = float(started) + float(result["duration_seconds"])
|
|
298
|
+
result["finished_at"] = int(derived_finish) if derived_finish.is_integer() else derived_finish
|
|
299
|
+
return result
|
|
300
|
+
|
|
301
|
+
|
|
302
|
+
def _salt(path: Path) -> str:
|
|
303
|
+
_reject_symlink(path, "latency salt")
|
|
304
|
+
if path.exists():
|
|
305
|
+
try:
|
|
306
|
+
value = path.read_text(encoding="ascii").strip()
|
|
307
|
+
except (OSError, UnicodeError) as exc:
|
|
308
|
+
raise LatencyError(f"cannot read latency salt: {exc}") from None
|
|
309
|
+
if not re.fullmatch(r"[0-9a-f]{64}", value):
|
|
310
|
+
raise LatencyError("latency salt is invalid")
|
|
311
|
+
return value
|
|
312
|
+
value = secrets.token_hex(32)
|
|
313
|
+
try:
|
|
314
|
+
fd = os.open(str(path), os.O_CREAT | os.O_EXCL | os.O_WRONLY, 0o600)
|
|
315
|
+
with os.fdopen(fd, "w", encoding="ascii") as handle:
|
|
316
|
+
handle.write(value + "\n")
|
|
317
|
+
handle.flush()
|
|
318
|
+
os.fsync(handle.fileno())
|
|
319
|
+
try:
|
|
320
|
+
os.chmod(path, 0o600)
|
|
321
|
+
except OSError:
|
|
322
|
+
pass
|
|
323
|
+
return value
|
|
324
|
+
except FileExistsError:
|
|
325
|
+
return _salt(path)
|
|
326
|
+
except OSError as exc:
|
|
327
|
+
raise LatencyError(f"cannot create latency salt: {exc}") from None
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _digest(salt: str, namespace: str, value: str) -> str:
|
|
331
|
+
return hashlib.sha256((namespace + "\0" + salt + "\0" + value).encode("utf-8")).hexdigest()
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _fingerprint(salt: str, event: dict[str, Any]) -> str:
|
|
335
|
+
encoded = json.dumps(event, ensure_ascii=False, sort_keys=True, separators=(",", ":")).encode("utf-8")
|
|
336
|
+
return hashlib.sha256(("event\0" + salt + "\0").encode("utf-8") + encoded).hexdigest()
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
def _redact(event: dict[str, Any], salt: str) -> dict[str, Any]:
|
|
340
|
+
redacted = {
|
|
341
|
+
"schema_version": LATENCY_SCHEMA_VERSION,
|
|
342
|
+
"event_id": _digest(salt, "event-id", event["event_id"]),
|
|
343
|
+
"task_id": _digest(salt, "task-id", event["task_id"]),
|
|
344
|
+
"worker_id": _digest(salt, "worker-id", event["worker_id"]),
|
|
345
|
+
"strategy": event["strategy"],
|
|
346
|
+
"task_class": event["task_class"],
|
|
347
|
+
"stage": event["stage"],
|
|
348
|
+
"role": event["role"],
|
|
349
|
+
"model": event["model"],
|
|
350
|
+
"rollout_mode": event["rollout_mode"],
|
|
351
|
+
"legacy_effort": event["legacy_effort"],
|
|
352
|
+
"proposed_effort": event["proposed_effort"],
|
|
353
|
+
"selected_effort": event["selected_effort"],
|
|
354
|
+
"observed_effort": event["observed_effort"],
|
|
355
|
+
"boundary": event["boundary"],
|
|
356
|
+
"outcome": event["outcome"],
|
|
357
|
+
"started_at": event["started_at"],
|
|
358
|
+
"finished_at": event["finished_at"],
|
|
359
|
+
"duration_seconds": event["duration_seconds"],
|
|
360
|
+
"repair_count": event["repair_count"],
|
|
361
|
+
"checkpoint_count": event["checkpoint_count"],
|
|
362
|
+
}
|
|
363
|
+
if event["work_unit_id"] is not None:
|
|
364
|
+
redacted["work_unit_id"] = _digest(salt, "work-unit-id", event["work_unit_id"])
|
|
365
|
+
redacted["event_fingerprint"] = _fingerprint(salt, event)
|
|
366
|
+
return redacted
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
def _read_events(path: Path) -> list[dict[str, Any]]:
|
|
370
|
+
_reject_symlink(path, "latency ledger")
|
|
371
|
+
if not path.exists():
|
|
372
|
+
return []
|
|
373
|
+
try:
|
|
374
|
+
with path.open("r", encoding="utf-8") as handle:
|
|
375
|
+
lines = handle.readlines()
|
|
376
|
+
except (OSError, UnicodeError) as exc:
|
|
377
|
+
raise LatencyError(f"cannot read latency ledger: {exc}") from None
|
|
378
|
+
events: list[dict[str, Any]] = []
|
|
379
|
+
for line_no, line in enumerate(lines, 1):
|
|
380
|
+
if not line.strip():
|
|
381
|
+
continue
|
|
382
|
+
try:
|
|
383
|
+
value = json.loads(line)
|
|
384
|
+
except json.JSONDecodeError:
|
|
385
|
+
raise LatencyError(f"invalid latency ledger JSON at line {line_no}") from None
|
|
386
|
+
if type(value) is not dict or value.get("schema_version") != LATENCY_SCHEMA_VERSION:
|
|
387
|
+
raise LatencyError(f"invalid latency ledger event at line {line_no}")
|
|
388
|
+
required = {
|
|
389
|
+
"schema_version", "event_id", "task_id", "worker_id", "strategy",
|
|
390
|
+
"task_class", "stage", "role", "model", "rollout_mode",
|
|
391
|
+
"legacy_effort", "proposed_effort", "selected_effort",
|
|
392
|
+
"observed_effort", "boundary", "outcome", "started_at",
|
|
393
|
+
"finished_at", "duration_seconds", "repair_count",
|
|
394
|
+
"checkpoint_count", "event_fingerprint",
|
|
395
|
+
}
|
|
396
|
+
if set(value) not in (required, required | {"work_unit_id"}):
|
|
397
|
+
raise LatencyError(f"invalid latency ledger fields at line {line_no}")
|
|
398
|
+
for hash_field in ("event_id", "task_id", "worker_id", "event_fingerprint"):
|
|
399
|
+
if type(value.get(hash_field)) is not str or not HASH_TOKEN.fullmatch(value[hash_field]):
|
|
400
|
+
raise LatencyError(f"invalid latency ledger {hash_field} at line {line_no}")
|
|
401
|
+
if "work_unit_id" in value and (
|
|
402
|
+
type(value["work_unit_id"]) is not str or not HASH_TOKEN.fullmatch(value["work_unit_id"])
|
|
403
|
+
):
|
|
404
|
+
raise LatencyError(f"invalid latency ledger work_unit_id at line {line_no}")
|
|
405
|
+
canonical = {
|
|
406
|
+
name: value.get(name)
|
|
407
|
+
for name in (
|
|
408
|
+
"event_id", "task_id", "worker_id", "work_unit_id", "strategy",
|
|
409
|
+
"task_class", "stage", "role", "model", "rollout_mode",
|
|
410
|
+
"legacy_effort", "proposed_effort", "selected_effort",
|
|
411
|
+
"observed_effort", "boundary", "outcome", "started_at",
|
|
412
|
+
"finished_at", "duration_seconds", "repair_count", "checkpoint_count",
|
|
413
|
+
)
|
|
414
|
+
}
|
|
415
|
+
try:
|
|
416
|
+
_normalize(canonical)
|
|
417
|
+
except LatencyError as exc:
|
|
418
|
+
raise LatencyError(f"invalid latency ledger event at line {line_no}: {exc}") from None
|
|
419
|
+
events.append(value)
|
|
420
|
+
return events
|
|
421
|
+
|
|
422
|
+
|
|
423
|
+
def record_latency_event(
|
|
424
|
+
event: Any,
|
|
425
|
+
*,
|
|
426
|
+
state_file: Path | str | None = None,
|
|
427
|
+
salt_file: Path | str | None = None,
|
|
428
|
+
lock_file: Path | str | None = None,
|
|
429
|
+
) -> dict[str, Any]:
|
|
430
|
+
"""Validate, redact, and append one event under a process lock."""
|
|
431
|
+
normalized = _normalize(event)
|
|
432
|
+
path = latency_file(state_file)
|
|
433
|
+
_ensure_parent(path.parent)
|
|
434
|
+
salt_path = Path(salt_file) if salt_file is not None else _sidecar(path, LATENCY_SALT_FILE_NAME)
|
|
435
|
+
lock_path = Path(lock_file) if lock_file is not None else _sidecar(path, LATENCY_LOCK_FILE_NAME)
|
|
436
|
+
with _exclusive_lock(lock_path):
|
|
437
|
+
salt = _salt(salt_path)
|
|
438
|
+
redacted = _redact(normalized, salt)
|
|
439
|
+
existing = _read_events(path)
|
|
440
|
+
for prior in existing:
|
|
441
|
+
if prior.get("event_id") != redacted["event_id"]:
|
|
442
|
+
continue
|
|
443
|
+
if prior.get("event_fingerprint") != redacted["event_fingerprint"]:
|
|
444
|
+
raise LatencyError("event_id was already recorded with a different payload")
|
|
445
|
+
return {
|
|
446
|
+
"recorded": False,
|
|
447
|
+
"deduplicated": True,
|
|
448
|
+
"event_id": redacted["event_id"],
|
|
449
|
+
}
|
|
450
|
+
_reject_symlink(path, "latency ledger")
|
|
451
|
+
try:
|
|
452
|
+
with path.open("a", encoding="utf-8", newline="\n") as handle:
|
|
453
|
+
json.dump(redacted, handle, ensure_ascii=False, sort_keys=True, separators=(",", ":"))
|
|
454
|
+
handle.write("\n")
|
|
455
|
+
handle.flush()
|
|
456
|
+
os.fsync(handle.fileno())
|
|
457
|
+
try:
|
|
458
|
+
os.chmod(path, 0o600)
|
|
459
|
+
except OSError:
|
|
460
|
+
pass
|
|
461
|
+
except OSError as exc:
|
|
462
|
+
raise LatencyError(f"cannot append latency event: {exc}") from None
|
|
463
|
+
return {"recorded": True, "deduplicated": False, "event_id": redacted["event_id"]}
|
|
464
|
+
|
|
465
|
+
|
|
466
|
+
def _nearest_rank(values: list[float], percentile: float) -> float | None:
|
|
467
|
+
if not values:
|
|
468
|
+
return None
|
|
469
|
+
ordered = sorted(values)
|
|
470
|
+
rank = max(1, math.ceil(percentile * len(ordered)))
|
|
471
|
+
value = ordered[rank - 1]
|
|
472
|
+
return int(value) if float(value).is_integer() else value
|
|
473
|
+
|
|
474
|
+
|
|
475
|
+
def _group_key(event: dict[str, Any]) -> tuple[Any, ...]:
|
|
476
|
+
return (
|
|
477
|
+
event.get("task_class"),
|
|
478
|
+
event.get("stage"),
|
|
479
|
+
event.get("model"),
|
|
480
|
+
event.get("selected_effort"),
|
|
481
|
+
event.get("observed_effort"),
|
|
482
|
+
event.get("rollout_mode"),
|
|
483
|
+
)
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
def _group_report(events: list[dict[str, Any]], key: tuple[Any, ...]) -> dict[str, Any]:
|
|
487
|
+
terminal = [e for e in events if e.get("boundary") == "terminal"]
|
|
488
|
+
observed = [e for e in terminal if e.get("duration_seconds") is not None]
|
|
489
|
+
uncensored = [e for e in observed if e.get("outcome") in {"completed", "failed"}]
|
|
490
|
+
durations = [float(e["duration_seconds"]) for e in uncensored]
|
|
491
|
+
success = sum(1 for e in terminal if e.get("outcome") == "completed")
|
|
492
|
+
failed = sum(1 for e in terminal if e.get("outcome") == "failed")
|
|
493
|
+
cancelled = sum(1 for e in terminal if e.get("outcome") == "cancelled")
|
|
494
|
+
timed_out = sum(1 for e in terminal if e.get("outcome") == "timeout")
|
|
495
|
+
missing = sum(1 for e in terminal if e.get("outcome") in {"completed", "failed"} and e.get("duration_seconds") is None)
|
|
496
|
+
censored = cancelled + timed_out
|
|
497
|
+
result: dict[str, Any] = {
|
|
498
|
+
"task_class": key[0],
|
|
499
|
+
"stage": key[1],
|
|
500
|
+
"model": key[2],
|
|
501
|
+
"selected_effort": key[3],
|
|
502
|
+
"observed_effort": key[4],
|
|
503
|
+
"rollout_mode": key[5],
|
|
504
|
+
"n": len(terminal),
|
|
505
|
+
"completed": len(uncensored),
|
|
506
|
+
"success": success,
|
|
507
|
+
"successes": success,
|
|
508
|
+
"failed": failed,
|
|
509
|
+
"cancelled": cancelled,
|
|
510
|
+
"timeout": timed_out,
|
|
511
|
+
"censored": censored,
|
|
512
|
+
"missing": missing,
|
|
513
|
+
"p50_seconds": _nearest_rank(durations, 0.50),
|
|
514
|
+
"p95_seconds": _nearest_rank(durations, 0.95),
|
|
515
|
+
"eligible_for_tuning": len(uncensored) >= 20 and key[4] is not None,
|
|
516
|
+
}
|
|
517
|
+
return result
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
def latency_report(*, state_file: Path | str | None = None) -> dict[str, Any]:
|
|
521
|
+
"""Return a stable nearest-rank report; this function never mutates policy."""
|
|
522
|
+
path = latency_file(state_file)
|
|
523
|
+
lock_path = _sidecar(path, LATENCY_LOCK_FILE_NAME)
|
|
524
|
+
with _exclusive_lock(lock_path):
|
|
525
|
+
events = _read_events(path)
|
|
526
|
+
grouped: dict[tuple[Any, ...], list[dict[str, Any]]] = {}
|
|
527
|
+
for event in events:
|
|
528
|
+
# Checkpoints are reported as a separate aggregate. They do not create
|
|
529
|
+
# zero-sized latency cohorts or dilute a terminal cohort's sample gate.
|
|
530
|
+
if event.get("boundary") != "terminal":
|
|
531
|
+
continue
|
|
532
|
+
grouped.setdefault(_group_key(event), []).append(event)
|
|
533
|
+
groups = [_group_report(grouped[key], key) for key in sorted(grouped, key=lambda item: tuple("" if v is None else str(v) for v in item))]
|
|
534
|
+
checkpoint_observations = sum(1 for e in events if e.get("boundary") == "checkpoint")
|
|
535
|
+
terminal = [e for e in events if e.get("boundary") == "terminal"]
|
|
536
|
+
observed = [e for e in terminal if e.get("duration_seconds") is not None]
|
|
537
|
+
uncensored = [e for e in observed if e.get("outcome") in {"completed", "failed"}]
|
|
538
|
+
total_durations = [float(e["duration_seconds"]) for e in uncensored]
|
|
539
|
+
success = sum(1 for e in terminal if e.get("outcome") == "completed")
|
|
540
|
+
failed = sum(1 for e in terminal if e.get("outcome") == "failed")
|
|
541
|
+
cancelled = sum(1 for e in terminal if e.get("outcome") == "cancelled")
|
|
542
|
+
timed_out = sum(1 for e in terminal if e.get("outcome") == "timeout")
|
|
543
|
+
missing = sum(1 for e in terminal if e.get("outcome") in {"completed", "failed"} and e.get("duration_seconds") is None)
|
|
544
|
+
censored = cancelled + timed_out
|
|
545
|
+
return {
|
|
546
|
+
"schema_version": LATENCY_SCHEMA_VERSION,
|
|
547
|
+
"n": len(terminal),
|
|
548
|
+
"completed": len(uncensored),
|
|
549
|
+
"success": success,
|
|
550
|
+
"successes": success,
|
|
551
|
+
"failed": failed,
|
|
552
|
+
"cancelled": cancelled,
|
|
553
|
+
"timeout": timed_out,
|
|
554
|
+
"censored": censored,
|
|
555
|
+
"missing": missing,
|
|
556
|
+
"checkpoint_observations": checkpoint_observations,
|
|
557
|
+
"p50_seconds": _nearest_rank(total_durations, 0.50),
|
|
558
|
+
"p95_seconds": _nearest_rank(total_durations, 0.95),
|
|
559
|
+
"eligible_for_tuning": any(group["eligible_for_tuning"] for group in groups),
|
|
560
|
+
"advisory": True,
|
|
561
|
+
"policy_mutation": False,
|
|
562
|
+
"groups": groups,
|
|
563
|
+
}
|
|
564
|
+
|
|
565
|
+
|
|
566
|
+
def format_latency_report(report: dict[str, Any]) -> str:
|
|
567
|
+
"""Compact deterministic text for humans; JSON remains the machine API."""
|
|
568
|
+
def seconds(value: Any) -> str:
|
|
569
|
+
return "missing" if value is None else f"{value}s"
|
|
570
|
+
|
|
571
|
+
lines = [
|
|
572
|
+
f"n={report['n']} completed={report['completed']} censored={report['censored']} missing={report['missing']} "
|
|
573
|
+
f"success={report['success']} p50={seconds(report['p50_seconds'])} p95={seconds(report['p95_seconds'])} "
|
|
574
|
+
f"eligible_for_tuning={'true' if report['eligible_for_tuning'] else 'false'}",
|
|
575
|
+
]
|
|
576
|
+
for group in report["groups"]:
|
|
577
|
+
lines.append(
|
|
578
|
+
f"{group['task_class']}/{group['stage']}/{group['model']}/selected={group['selected_effort']}/"
|
|
579
|
+
f"observed={group['observed_effort'] or 'missing'}/mode={group['rollout_mode']}: "
|
|
580
|
+
f"n={group['n']} p50={seconds(group['p50_seconds'])} p95={seconds(group['p95_seconds'])}"
|
|
581
|
+
)
|
|
582
|
+
return "\n".join(lines)
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
__all__ = [
|
|
586
|
+
"LATENCY_SCHEMA_VERSION",
|
|
587
|
+
"LATENCY_FILE_NAME",
|
|
588
|
+
"LatencyError",
|
|
589
|
+
"latency_file",
|
|
590
|
+
"record_latency_event",
|
|
591
|
+
"latency_report",
|
|
592
|
+
"format_latency_report",
|
|
593
|
+
]
|