froid-loop 0.11.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (116) hide show
  1. froid_loop/__init__.py +11 -0
  2. froid_loop/__main__.py +12 -0
  3. froid_loop/adapters/__init__.py +3 -0
  4. froid_loop/adapters/base.py +254 -0
  5. froid_loop/adapters/entrypoints.py +63 -0
  6. froid_loop/adapters/env_fault.py +290 -0
  7. froid_loop/adapters/generic.py +2013 -0
  8. froid_loop/adapters/mock.py +49 -0
  9. froid_loop/adapters/multiplexer.py +914 -0
  10. froid_loop/adapters/opencode_http.py +1687 -0
  11. froid_loop/adapters/profile.py +650 -0
  12. froid_loop/adapters/psmux_backend.py +1428 -0
  13. froid_loop/adapters/registry.py +322 -0
  14. froid_loop/adapters/tmux_backend.py +35 -0
  15. froid_loop/adapters/tmux_base.py +630 -0
  16. froid_loop/checks.py +187 -0
  17. froid_loop/cli.py +5041 -0
  18. froid_loop/data/__init__.py +0 -0
  19. froid_loop/data/froid_loop_hook.py +228 -0
  20. froid_loop/data/froid_loop_probe_hook.py +88 -0
  21. froid_loop/data/plugins/example/plugin.toml +21 -0
  22. froid_loop/data/plugins/tea/plugin.toml +184 -0
  23. froid_loop/data/plugins/tea/tea_plugin.py +258 -0
  24. froid_loop/data/plugins/unity/plugin.toml +140 -0
  25. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef +16 -0
  26. froid_loop/data/plugins/unity/unity_assets/FroidLoop.Unity.Editor.asmdef.meta +7 -0
  27. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs +221 -0
  28. froid_loop/data/plugins/unity/unity_assets/SceneAutoSaveGuard.cs.meta +11 -0
  29. froid_loop/data/plugins/unity/unity_assets/_folders/Editor.meta +8 -0
  30. froid_loop/data/plugins/unity/unity_assets/_folders/FroidLoop.meta +8 -0
  31. froid_loop/data/plugins/unity/unity_cleanup.py +125 -0
  32. froid_loop/data/plugins/unity/unity_dialog_probe.py +239 -0
  33. froid_loop/data/plugins/unity/unity_facts.md +17 -0
  34. froid_loop/data/plugins/unity/unity_plugin.py +415 -0
  35. froid_loop/data/plugins/unity/unity_quiesce.py +234 -0
  36. froid_loop/data/plugins/unity/unity_ready.py +230 -0
  37. froid_loop/data/plugins/unity/unity_seed_assets.py +298 -0
  38. froid_loop/data/plugins/unity/unity_setup.py +551 -0
  39. froid_loop/data/plugins/unity/unity_teardown.py +362 -0
  40. froid_loop/data/profiles/antigravity.toml +52 -0
  41. froid_loop/data/profiles/claude.toml +85 -0
  42. froid_loop/data/profiles/codex.toml +22 -0
  43. froid_loop/data/profiles/copilot.toml +52 -0
  44. froid_loop/data/profiles/gemini.toml +26 -0
  45. froid_loop/data/profiles/opencode.toml +54 -0
  46. froid_loop/data/settings/core.toml +458 -0
  47. froid_loop/data/skills/README.md +93 -0
  48. froid_loop/data/skills/froid-loop-resolve/SKILL.md +288 -0
  49. froid_loop/data/skills/froid-loop-setup/SKILL.md +161 -0
  50. froid_loop/data/skills/froid-loop-setup/assets/module-help.csv +3 -0
  51. froid_loop/data/skills/froid-loop-setup/assets/module.yaml +19 -0
  52. froid_loop/data/skills/froid-loop-sweep/SKILL.md +100 -0
  53. froid_loop/data/skills/froid-loop-sweep/automation-mode.md +127 -0
  54. froid_loop/data/skills/froid-loop-sweep/deferred-work-format.md +302 -0
  55. froid_loop/data/skills/froid-loop-sweep/migration-mode.md +86 -0
  56. froid_loop/decisions.py +202 -0
  57. froid_loop/deferredwork.py +2282 -0
  58. froid_loop/devcontract.py +892 -0
  59. froid_loop/diagnostics.py +1104 -0
  60. froid_loop/documents.py +532 -0
  61. froid_loop/engine.py +7732 -0
  62. froid_loop/envvars.py +111 -0
  63. froid_loop/escalation.py +225 -0
  64. froid_loop/events.py +266 -0
  65. froid_loop/fences.py +103 -0
  66. froid_loop/froidconfig.py +226 -0
  67. froid_loop/frontmatter.py +526 -0
  68. froid_loop/gates.py +133 -0
  69. froid_loop/install.py +2936 -0
  70. froid_loop/journal.py +178 -0
  71. froid_loop/machine.py +148 -0
  72. froid_loop/model.py +898 -0
  73. froid_loop/operatoractions.py +474 -0
  74. froid_loop/platform_util.py +1490 -0
  75. froid_loop/plugins/__init__.py +64 -0
  76. froid_loop/plugins/bus.py +259 -0
  77. froid_loop/plugins/context.py +319 -0
  78. froid_loop/plugins/loader.py +145 -0
  79. froid_loop/plugins/manifest.py +279 -0
  80. froid_loop/plugins/model.py +296 -0
  81. froid_loop/plugins/registry.py +245 -0
  82. froid_loop/plugins/trust.py +75 -0
  83. froid_loop/policy.py +1569 -0
  84. froid_loop/probe.py +1044 -0
  85. froid_loop/process_host.py +408 -0
  86. froid_loop/recovery_flow.py +1561 -0
  87. froid_loop/resolve.py +283 -0
  88. froid_loop/runs.py +4715 -0
  89. froid_loop/runsetup.py +1293 -0
  90. froid_loop/sanitize.py +593 -0
  91. froid_loop/settings_schema.py +276 -0
  92. froid_loop/signals.py +160 -0
  93. froid_loop/sprintstatus.py +609 -0
  94. froid_loop/statemachine.py +57 -0
  95. froid_loop/stories.py +615 -0
  96. froid_loop/stories_engine.py +796 -0
  97. froid_loop/sweep.py +1892 -0
  98. froid_loop/tokens.py +196 -0
  99. froid_loop/tui/__init__.py +11 -0
  100. froid_loop/tui/app.py +1584 -0
  101. froid_loop/tui/data.py +840 -0
  102. froid_loop/tui/launch.py +1003 -0
  103. froid_loop/tui/screens/__init__.py +1 -0
  104. froid_loop/tui/screens/dashboard.py +1071 -0
  105. froid_loop/tui/screens/modals.py +943 -0
  106. froid_loop/tui/screens/settings_screen.py +477 -0
  107. froid_loop/tui/settings.py +135 -0
  108. froid_loop/tui/widgets.py +981 -0
  109. froid_loop/verify.py +4545 -0
  110. froid_loop/workspace.py +320 -0
  111. froid_loop/worktree_flow.py +2301 -0
  112. froid_loop-0.11.1.dist-info/METADATA +728 -0
  113. froid_loop-0.11.1.dist-info/RECORD +116 -0
  114. froid_loop-0.11.1.dist-info/WHEEL +4 -0
  115. froid_loop-0.11.1.dist-info/entry_points.txt +2 -0
  116. froid_loop-0.11.1.dist-info/licenses/LICENSE +30 -0
@@ -0,0 +1,1687 @@
1
+ """OpenCode driver: sessions over the local HTTP server, no tmux, no hooks.
2
+
3
+ OpenCode (opencode.ai) is client/server: its TUI is just a client of a local
4
+ HTTP server (``opencode serve``) exposing an OpenAPI 3.1 API. This adapter
5
+ drives sessions entirely over that API (``injection="http"``,
6
+ ``observation="sse"``).
7
+
8
+ API contract — every fact below was pinned live against the real 1.18.2
9
+ binary (2026-07-16; the full probe record and the archived OpenAPI spec live
10
+ in git history, commit b85d8ca / PR #167). This list wins over memory:
11
+
12
+ - Readiness target is ``GET /global/health`` (``/health`` is the web-UI SPA
13
+ shell). With ``OPENCODE_SERVER_PASSWORD`` set *every* endpoint 401s, health
14
+ included — basic-auth username is literally ``opencode``; Bearer is
15
+ rejected.
16
+ - SSE ``GET /event``: flat ``data:``-only frames, first frame
17
+ ``server.connected``, ``server.heartbeat`` ≈ every 10 s, no server-side
18
+ filtering — clients filter by ``properties.sessionID``.
19
+ - ``session.idle`` fires even when aborting an idle session, so an idle event
20
+ alone is never proof a turn ran — it must pair with an assistant message
21
+ whose ``time.completed`` post-dates the last prompt sent.
22
+ - Poll fallback ``GET /session/status`` returns a ``{sessionID: status}`` map
23
+ in which idle sessions are simply **absent** — the rule is "absent ⇒ not
24
+ busy", never "wait for idle".
25
+ - Per-prompt ``model`` is the object form ``{providerID, modelID}``; the
26
+ ``"provider/model"`` string form belongs only to the config-file ``model``
27
+ key.
28
+ - ``OPENCODE_CONFIG_CONTENT`` outranks the project's ``opencode.json``
29
+ (applied as a final local-scope merge over all config files).
30
+ - Hermetic skills need ``OPENCODE_DISABLE_EXTERNAL_SKILLS=1`` **plus**
31
+ ``skills.paths=["<worktree>/.claude/skills"]`` inside the config content —
32
+ by default every server also sees the operator's personal
33
+ ``~/.claude/skills`` and ``~/.agents/skills``.
34
+ - ``opencode serve`` survives parent SIGKILL (reparents to init and keeps
35
+ serving); SIGTERM exits it cleanly.
36
+ - All ``time.*`` fields are epoch **milliseconds**; ``POST /session/:id/abort``
37
+ returns ``200 true`` even when nothing is running.
38
+
39
+ ``/event`` frame types the run-log renderer reads — pinned by a second live
40
+ probe of 1.18.2 (2026-07-25, 388 frames over 5 turns across two servers; the
41
+ full evidence table is posted on PR #279). Cross-checked against the server's
42
+ own OpenAPI at ``GET /doc``, whose ``Event`` union has 89 variants and is the
43
+ static contract — worth re-dumping on any version bump:
44
+
45
+ - ``message.updated`` — turn metadata only, no text. Carries ``info.id`` +
46
+ ``info.role``; **re-emits out of order** (a stale re-emit for the user
47
+ message lands mid-assistant-turn), so role must be keyed by message id, never
48
+ tracked as "last role seen".
49
+ - ``message.part.updated`` — **complete-once, NOT cumulative.** A part fires
50
+ exactly twice: once at creation with ``text: ""``, then once carrying the
51
+ full final text. Held at 4 / 711 / 5689 chars (2 fires each, 1 non-empty).
52
+ Rendering on this frame therefore yields one clean line per statement, with
53
+ no de-duplication needed. **Tool execution also surfaces here** — and only
54
+ here — as ``part.type == "tool"`` with ``part.tool`` and a ``state`` whose
55
+ ``status`` steps pending → running → terminal (``state.input`` throughout,
56
+ ``state.output`` on ``completed``, ``state.error`` on ``error``). Tool parts
57
+ carry no ``part.text``, so the tool branch must sit ABOVE the empty-text
58
+ bail-out in ``_inline_line`` to be reachable at all. ``ToolState`` is a
59
+ 4-variant union (pending / running / completed / error), so rendering on the
60
+ terminal pair is exhaustively once-per-tool-call; ``error`` is included from
61
+ the static contract (no tool failed during the live probe) so a failed call
62
+ cannot vanish from the transcript.
63
+ - ``message.part.delta`` — the separate per-token stream. At 5689 chars its 50
64
+ deltas concatenate **byte-exactly** to the single ``part.updated`` text, so
65
+ it is pure redundancy: never rendered inline, and excluded from the SSE trace
66
+ (``logs/`` is never trimmed by retention).
67
+ - ``command.executed`` — the **slash-command** surface, kept as that and nothing
68
+ more. It does **not** fire for agent tool use: 0 frames across live ``bash`` /
69
+ ``write`` / ``read`` / ``edit`` calls. Payload is ``name`` / ``arguments`` /
70
+ ``messageID`` / ``sessionID`` (all required, so it needs no allowlist).
71
+ - ``file.edited`` — payload is exactly ``{"file": "/abs/path"}``, schema-pinned
72
+ ``additionalProperties: false`` over that one key, so it carries **no
73
+ ``sessionID``** and the ``properties.sessionID`` filter would drop it before
74
+ either sink. It is reachable only through the explicit ``_SESSIONLESS_TYPES``
75
+ allowlist, sound here only because froid-loop runs one server per session.
76
+ ``file.watcher.updated`` is session-less too and deliberately NOT allowlisted
77
+ (see that constant).
78
+ - ``permission.asked`` — the ask frame. There is **no ``permission.updated``** in
79
+ 1.18.2. Payload is a ``permission`` **string** plus ``patterns`` /
80
+ ``metadata`` / ``always`` / ``tool`` / ``id`` / ``sessionID`` — not a nested
81
+ object with ``type`` / ``pattern``.
82
+ - ``permission.replied`` — carries **``reply``**, not ``response`` (plus
83
+ ``sessionID`` / ``requestID``).
84
+ - ``permission.v2.asked`` / ``permission.v2.replied`` exist in the union and are
85
+ **deliberately not consumed**: neither was seen live, and the v2 ask is a
86
+ different payload (``action`` / ``resources`` / ``save`` / ``metadata`` /
87
+ ``source``), so a branch for it would be written from the schema alone —
88
+ exactly the guess that made the three corrections above necessary. The v2
89
+ reply is shape-identical to the v1 one, but consuming it without the ask would
90
+ log a decision with no request. Add both together once a live sighting pins
91
+ what ``action`` / ``resources`` actually hold.
92
+ - There is **no ``tool.call`` / ``tool.response``** on this surface —
93
+ confirmed both statically in the 89-variant union and live.
94
+
95
+ Transport shape (the settled design drivers):
96
+
97
+ - **One ``opencode serve`` per session.** The API has no per-session env, and
98
+ the ``FROID_LOOP_*`` contract must reach tool subprocesses via the server
99
+ process env — so each session gets its own server spawned with
100
+ ``cwd=spec.cwd`` and the session env. Ports are OS-assigned free ports,
101
+ re-picked on a bind race.
102
+ - **Config injected via ``OPENCODE_CONFIG_CONTENT``** (outranks the project's
103
+ ``opencode.json``): a blanket permission allow (the bypass-flags
104
+ analogue), the hermetic-skills recipe above (project ``.claude/skills``
105
+ only — without it every session sees the operator's personal skills), and
106
+ the policy model when set. A per-session ``OPENCODE_SERVER_PASSWORD`` makes
107
+ the health poll self-discriminating against a foreign server on a reused
108
+ port and keeps other local processes from driving an allow-all server.
109
+ - **SSE ``session.idle`` ≙ the Stop hook**, filtered to this session's id —
110
+ child/subagent sessions share the stream and emit their own idles. SSE is
111
+ lossy upstream, so a silent or reconnecting stream degrades to an HTTP poll
112
+ (``GET /session/status`` + message-level proof-of-work): an idle event alone
113
+ is NOT proof a turn ran (abort emits one on an idle session),
114
+ so the fallback demands an assistant message completed *after* the last
115
+ prompt this adapter sent. OpenCode timestamps are epoch **milliseconds**.
116
+ - **Server death ≙ window death** (``crashed``, landed artifact honored);
117
+ stall verdicts under a live server pin ``accept_result=False`` — the
118
+ #48/#53 artifact-distrust invariant, unchanged.
119
+ - **Usage is read over HTTP before teardown** (server state is sqlite, not a
120
+ readable file tree): assistant-message token sums, stashed by session id,
121
+ raw messages dumped to ``tasks/<task_id>/messages.json`` as the transcript.
122
+ Child-session (subagent) tokens are not counted — the API scopes messages
123
+ per session.
124
+ - ``opencode serve`` survives parent death, so teardown is
125
+ authoritative: kill in ``run()``'s finally plus an atexit sweep. On Windows
126
+ the binary is an npm ``.cmd`` shim, so the kill goes straight to the
127
+ process-tree force-kill while the wrapper is still alive to enumerate.
128
+ """
129
+
130
+ from __future__ import annotations
131
+
132
+ import atexit
133
+ import json
134
+ import os
135
+ import queue
136
+ import secrets
137
+ import shutil
138
+ import socket
139
+ import subprocess
140
+ import sys
141
+ import threading
142
+ import time
143
+ from dataclasses import dataclass, field
144
+ from pathlib import Path
145
+ from typing import TYPE_CHECKING, Any
146
+
147
+ from .. import gates
148
+ from ..froidconfig import ProjectPaths
149
+ from ..journal import LOGS_DIR
150
+ from ..model import TokenUsage
151
+ from ..policy import Policy
152
+ from ..process_host import ProcessHostError, get_process_host
153
+ from .base import CodingCLIAdapter, SessionHandle, SessionResult, SessionSpec
154
+ from .env_fault import EnvFaultMixin
155
+ from .generic import (
156
+ BUDGET_NUDGE_TEXT,
157
+ HEARTBEAT_INTERVAL_S,
158
+ NUDGE_TEXT,
159
+ STALL_NUDGE_TEXT,
160
+ _DevSynthesisMixin,
161
+ _ResultFileMixin,
162
+ )
163
+ from .profile import CLIProfile
164
+
165
+ if TYPE_CHECKING:
166
+ from ..process_host import ProcessHost
167
+
168
+ # Spawn/readiness defaults; per-instance attributes so tests shrink them.
169
+ HEALTH_TIMEOUT_S = 30.0
170
+ HEALTH_POLL_S = 0.25
171
+ SSE_READ_TIMEOUT_S = 30.0 # server heartbeats ~10s; 3 misses = stream is gone
172
+ SILENCE_THRESHOLD_S = 30.0
173
+ RECONNECT_SLEEP_S = 1.0
174
+ KILL_WAIT_S = 5.0
175
+ # Straggler-reap poll cadence. Independent of generic's KILL_POLL_S: this teardown
176
+ # budgets waits at kill_wait_s scale (seconds), not the tmux grace scale, so a
177
+ # tight fixed poll keeps the reap responsive without a per-instance knob.
178
+ REAP_POLL_S = 0.1
179
+ SPAWN_ATTEMPTS = 3
180
+ POLL_TICK_S = 5.0 # max event-queue wait per loop tick (generic's cadence)
181
+ # Structured SSE trace (``logs/<task_id>.sse.jsonl``). Tier-1 knob deliberately:
182
+ # a module default plus the ``sse_trace`` instance attribute, NOT a policy field
183
+ # in core.toml. It changes nothing about how a run behaves — it only decides
184
+ # whether a debugging artifact is written — so it belongs with the timing knobs
185
+ # an operator flips in a REPL or a subclass, not in the run contract every
186
+ # settings file has to carry. Off means the sink is never opened.
187
+ SSE_TRACE = True
188
+
189
+ # ``/event`` frame types that carry NO ``properties.sessionID`` but are still
190
+ # attributed to this session, exempting them from the sessionID filter in
191
+ # `_dispatch_sse`. Sound ONLY because froid-loop spawns one `opencode serve` per
192
+ # session (see the transport notes in the module docstring): the server is
193
+ # single-tenant, so a server-global frame is unambiguously this session's. Any
194
+ # design that ever shares a server across sessions must empty this set.
195
+ #
196
+ # Deliberately an allowlist and not "allow every session-less frame": the live
197
+ # probe's session-less traffic was overwhelmingly noise that says nothing about
198
+ # the run — `plugin.added` alone was 90 of 388 frames, plus `server.heartbeat`,
199
+ # `catalog.updated`, `server.connected`, `reference.updated` and
200
+ # `integration.updated`. `file.watcher.updated` is excluded on purpose too: it
201
+ # is the editor-integration watcher firing for any change under the project
202
+ # (our own git operations, a build, an editor save), so it is not evidence the
203
+ # agent did anything, and for the changes the agent DID make it just
204
+ # double-reports what `file.edited` already carries.
205
+ _SESSIONLESS_TYPES = frozenset({"file.edited"})
206
+
207
+ # Fixed basic-auth username in OPENCODE_SERVER_PASSWORD mode (Bearer is rejected).
208
+ AUTH_USER = "opencode"
209
+
210
+
211
+ class OpencodeServerError(Exception):
212
+ """An ``opencode serve`` instance could not be spawned, readied or driven."""
213
+
214
+
215
+ def _require_httpx():
216
+ """Import httpx lazily — it ships as the ``opencode`` extra, so the
217
+ dep-free core never pays for it (the ``_psutil()`` pattern)."""
218
+ try:
219
+ import httpx # intentional lazy import — optional extra
220
+ except ImportError as exc:
221
+ raise OpencodeServerError(
222
+ "the opencode-http adapter needs httpx; "
223
+ "install it with `pip install 'froid-loop[opencode]'`"
224
+ ) from exc
225
+ return httpx
226
+
227
+
228
+ def _now_ms() -> int:
229
+ """Wall clock in epoch milliseconds — OpenCode's ``time.*`` unit.
230
+ Comparisons against ``SessionHandle.launched_ns`` must divide by 1e6 first;
231
+ a raw ns-vs-ms comparison is always False and silently disables the poll
232
+ fallback."""
233
+ return time.time_ns() // 1_000_000
234
+
235
+
236
+ def _free_port() -> int:
237
+ """An OS-assigned free localhost port. Racy by nature (the bind is
238
+ released before ``opencode serve`` re-binds); the spawn loop retries with a
239
+ fresh port when the server dies during the health poll."""
240
+ with socket.socket(socket.AF_INET, socket.SOCK_STREAM) as sock:
241
+ sock.bind(("127.0.0.1", 0))
242
+ return sock.getsockname()[1]
243
+
244
+
245
+ def _parse_sse_lines(lines) -> Any:
246
+ """Minimal SSE frame parser: accumulate ``data:`` lines until a blank line,
247
+ then yield the JSON-decoded payload. Tolerates comments, unknown fields and
248
+ undecodable payloads (skipped) — the stream is advisory, never trusted."""
249
+ data: list[str] = []
250
+ for line in lines:
251
+ if line == "":
252
+ if data:
253
+ try:
254
+ yield json.loads("\n".join(data))
255
+ except (json.JSONDecodeError, ValueError):
256
+ pass
257
+ data = []
258
+ continue
259
+ if line.startswith("data:"):
260
+ data.append(line[5:].lstrip())
261
+
262
+
263
+ def _sum_args(arguments: Any) -> str:
264
+ """Compact one-line summary of an arbitrary event payload value (a tool
265
+ part's ``state.input``, a ``permission.asked`` ``patterns``, a
266
+ ``command.executed`` ``arguments``, a ``session.error`` ``error``) for the
267
+ inline run log. Truncates long dicts/strings so a single line never explodes
268
+ the tail view, and accepts any shape — these payloads are server-controlled
269
+ and not all of them have a pinned schema."""
270
+ if not arguments:
271
+ return ""
272
+ try:
273
+ if isinstance(arguments, (dict, list)):
274
+ text = json.dumps(arguments, ensure_ascii=False)
275
+ else:
276
+ text = str(arguments)
277
+ except (TypeError, ValueError):
278
+ text = str(arguments)
279
+ if len(text) > 120:
280
+ text = text[:119] + "\u2026"
281
+ return text
282
+
283
+
284
+ # ANSI colors for inline-log marker lines, keyed by message role. The froid-loop
285
+ # TUI renders <task>.log through a pyte terminal emulator (tui/data.py LogView),
286
+ # which dispatches SGR to per-Char styles and maps pyte named colors via
287
+ # _rich_color straight through to Rich — so every [froid] marker shows colored in
288
+ # the LogView. A plain `cat`/editor shows the escape bytes; the structured trace
289
+ # stays uncoloured in <task>.sse.jsonl. Keys are opencode message.info.role
290
+ # values; an unseen role falls back to _ROLE_COLOR_DEFAULT so it still renders
291
+ # distinctly — add it to the map to pin its hue.
292
+ _ROLE_COLORS = {
293
+ "user": "\x1b[33m", # SGR 33 = yellow
294
+ "assistant": "\x1b[36m", # SGR 36 = cyan
295
+ }
296
+ _ROLE_COLOR_DEFAULT = "\x1b[35m" # SGR 35 = magenta — unseen roles
297
+ _TOOL_COLOR = "\x1b[32m" # SGR 32 = green — tool/cmd/file/permission (role-less)
298
+ _RESET = "\x1b[0m"
299
+
300
+ # Tool-part statuses that end a tool call. `ToolState` is a 4-variant union in
301
+ # the 1.18.2 OpenAPI (`pending`, `running`, `completed`, `error`), so these two
302
+ # are exhaustively the terminal ones — a tool call reaches exactly one of them,
303
+ # which is what makes "render on terminal" equal to "one line per tool". The
304
+ # live probe only ever saw pending → running → completed (nothing failed during
305
+ # it); `error` comes from the static contract, and including it is what keeps a
306
+ # FAILED tool call from vanishing from the transcript entirely.
307
+ _TOOL_TERMINAL_STATES = frozenset({"completed", "error"})
308
+
309
+
310
+ def _role_color(role: str) -> str:
311
+ """ANSI color for a message role; unseen roles fall back to magenta."""
312
+ return _ROLE_COLORS.get(role, _ROLE_COLOR_DEFAULT)
313
+
314
+
315
+ @dataclass
316
+ class _ServerSession:
317
+ """Everything the adapter tracks for one live ``opencode serve``."""
318
+
319
+ process: subprocess.Popen
320
+ port: int
321
+ base_url: str
322
+ password: str
323
+ log_fh: Any
324
+ # The spawned server's own stdout/stderr sink (``<task_id>.server.out``),
325
+ # kept separate from ``log_fh`` so the readable transcript stays clean. The
326
+ # server's INFO/diagnostic lines land here; ``log_fh`` carries only the
327
+ # curated ``[froid]`` lines written from the SSE reader thread.
328
+ server_fh: Any = None
329
+ # Structured SSE-trace JSONL sink (``<task_id>.sse.jsonl``); None when the
330
+ # ``sse_trace`` knob is off. Written from the SSE reader thread only.
331
+ event_fh: Any = None
332
+ # Monotonic line number stamped into the trace. Per SESSION, while the file
333
+ # is per TASK and opened in append mode — so a retried task restarts the seq
334
+ # at 1 partway down the file. Read a run of seq back to 1 as "new session",
335
+ # not as corruption; the ``ts`` field orders records across the whole file.
336
+ event_seq: int = 0
337
+ # Role keyed by opencode message id (``msg_*``), refreshed from
338
+ # ``message.updated.info.{id,role}``. The per-part / delta events carry a
339
+ # ``messageID`` but no role of their own, and ``message.updated`` frames
340
+ # arrive out of order (re-emits for an earlier message land mid-turn), so a
341
+ # "last role seen" would mislabel the assistant's reply as ``user:``. Keying
342
+ # by message id makes the lookup ordering-independent.
343
+ msg_roles: dict = field(default_factory=dict)
344
+ client: Any = None # control httpx.Client — main thread only
345
+ session_id: str = ""
346
+ events: queue.Queue = field(default_factory=queue.Queue)
347
+ sse_thread: threading.Thread | None = None
348
+ sse_stop: threading.Event = field(default_factory=threading.Event)
349
+ sse_connected: threading.Event = field(default_factory=threading.Event)
350
+ # Bumped by the SSE reader on any non-heartbeat frame (any session — the
351
+ # parent is silent while a child session streams, exactly like subagent
352
+ # output in a tmux pane). The wait loop snapshots it to re-arm the
353
+ # dev-stall grace window, mirroring generic._log_activity_key.
354
+ activity: int = 0
355
+ # Monotonic timestamp of the last SSE frame of any kind (heartbeats
356
+ # included) — a healthy-but-quiet stream keeps this fresh, so the wait
357
+ # loop only falls back to HTTP polling when BOTH its own dequeue clock and
358
+ # this are stale (a dead reader thread leaves it stale, preserving the
359
+ # degraded path).
360
+ last_frame_monotonic: float = 0.0
361
+ # Monotonic completion floor in epoch ms: the poll fallback only
362
+ # synthesizes an idle for an assistant message completed strictly after
363
+ # this. Starts at prompt-send, advances on every prompt this adapter sends
364
+ # and on every completion it consumes — otherwise one stale completed
365
+ # message re-synthesizes idle on every probe (each fake "Stop" refills the
366
+ # stall budget) and the session livelocks or burns its nudges.
367
+ floor_ms: int = 0
368
+
369
+
370
+ class OpencodeHttpAdapter(_ResultFileMixin, EnvFaultMixin, CodingCLIAdapter):
371
+ # Env-fault classification scans the SERVER's own stdout/stderr, not
372
+ # <task_id>.log — that file is the curated `[froid]` conversation transcript
373
+ # written by the SSE reader, so it carries the model's own words. Two reasons
374
+ # this must be .server.out: the provider's AI_APICallError logfmt lines only
375
+ # ever land there, and the profile's patterns are anchored on the assumption
376
+ # that a story quoting a provider error verbatim cannot reach the scanned
377
+ # bytes. Point this at the transcript and both properties break at once.
378
+ ENV_FAULT_LOG_SUFFIX = ".server.out"
379
+
380
+ injection = "http"
381
+ observation = "sse"
382
+ state = "remote"
383
+
384
+ def __init__(
385
+ self,
386
+ run_dir: Path,
387
+ policy: Policy,
388
+ profile: CLIProfile,
389
+ binary: str | None = None,
390
+ extra_args: tuple[str, ...] | None = None,
391
+ usage_grace_s: float | None = None,
392
+ stop_without_result_nudges: int | None = None,
393
+ events_dir: Path | None = None,
394
+ ):
395
+ # `events_dir` is accepted and unused: this family observes over SSE and
396
+ # fires no hooks, so it has no event channel to point at. It is part of
397
+ # the run description `runsetup.make_adapters` hands every family (#494),
398
+ # and refusing the kwarg here would make the bootstrap branch per family
399
+ # on a value that costs nothing to carry.
400
+ del events_dir
401
+ self._httpx = _require_httpx()
402
+ self.run_dir = run_dir
403
+ self.policy = policy
404
+ self.profile = profile
405
+ self.name = profile.name
406
+ self.binary = binary or profile.binary
407
+ # None = no extra serve args; unlike the tmux adapters there are no
408
+ # bypass flags to default to (permissions ride OPENCODE_CONFIG_CONTENT).
409
+ self.extra_args = extra_args
410
+ self._usage_grace_s = usage_grace_s if usage_grace_s is not None else profile.usage_grace_s
411
+ self._stop_nudges = (
412
+ stop_without_result_nudges
413
+ if stop_without_result_nudges is not None
414
+ else (
415
+ profile.stop_without_result_nudges
416
+ if profile.stop_without_result_nudges is not None
417
+ else policy.limits.stop_without_result_nudges
418
+ )
419
+ )
420
+ # Same base semantics as GenericAdapter: fail fast on a result-less
421
+ # Stop. The Phase 4 dev subclass raises these via _configure_dev_knobs.
422
+ self._stall_grace_s = 0.0
423
+ self._stall_nudges = 0
424
+ # Timing knobs — instance attributes so tests shrink them per-adapter.
425
+ self.health_timeout_s = HEALTH_TIMEOUT_S
426
+ self.health_poll_s = HEALTH_POLL_S
427
+ self.sse_read_timeout_s = SSE_READ_TIMEOUT_S
428
+ self.silence_threshold_s = SILENCE_THRESHOLD_S
429
+ self.reconnect_sleep_s = RECONNECT_SLEEP_S
430
+ self.kill_wait_s = KILL_WAIT_S
431
+ self.poll_tick_s = POLL_TICK_S
432
+ self.sse_trace = SSE_TRACE # False = never open the .sse.jsonl sink
433
+ self.result_grace_s: float | None = None # None = the mixin default
434
+ self.tasks_dir = run_dir / "tasks"
435
+ self.logs_dir = run_dir / LOGS_DIR
436
+ self.tasks_dir.mkdir(parents=True, exist_ok=True)
437
+ self.logs_dir.mkdir(parents=True, exist_ok=True)
438
+ self._sessions: dict[str, _ServerSession] = {}
439
+ self._usage: dict[str, TokenUsage] = {}
440
+ # opencode serve survives parent death: sweep whatever is
441
+ # still registered when the interpreter exits cooperatively. A hard
442
+ # SIGKILL of the engine still leaks — documented residual risk.
443
+ atexit.register(self._atexit_sweep)
444
+
445
+ # ------------------------------------------------------------- spawning
446
+
447
+ def _serve_argv(self, resolved_binary: str, port: int) -> list[str]:
448
+ """argv for one server. A seam: tests monkeypatch it to launch the
449
+ FakeOpencode sidecar wrapper-free."""
450
+ extra = self.extra_args or ()
451
+ return [
452
+ resolved_binary,
453
+ "serve",
454
+ "--port",
455
+ str(port),
456
+ "--hostname",
457
+ "127.0.0.1",
458
+ "--print-logs",
459
+ *extra,
460
+ ]
461
+
462
+ def _config_content(self, spec: SessionSpec) -> str:
463
+ """The OPENCODE_CONFIG_CONTENT JSON for this session:
464
+ blanket permission allow (the bypass-flags analogue), the hermetic
465
+ skills path (project skills only, paired with
466
+ OPENCODE_DISABLE_EXTERNAL_SKILLS=1 in the env), and the model when the
467
+ policy sets one (config-file model is the "provider/model" string
468
+ form)."""
469
+ config: dict[str, Any] = {
470
+ "permission": "allow",
471
+ "skills": {"paths": [str(Path(spec.cwd) / self.profile.skill_tree)]},
472
+ }
473
+ if spec.model:
474
+ config["model"] = spec.model
475
+ return json.dumps(config)
476
+
477
+ def _session_env(self, spec: SessionSpec, password: str) -> dict[str, str]:
478
+ return {
479
+ **os.environ,
480
+ **self.profile.env,
481
+ **spec.env,
482
+ "OPENCODE_DISABLE_EXTERNAL_SKILLS": "1",
483
+ "OPENCODE_SERVER_PASSWORD": password,
484
+ "OPENCODE_CONFIG_CONTENT": self._config_content(spec),
485
+ }
486
+
487
+ def _make_client(self, sess: _ServerSession):
488
+ return self._httpx.Client(
489
+ base_url=sess.base_url,
490
+ auth=(AUTH_USER, sess.password),
491
+ timeout=self._httpx.Timeout(10.0, connect=5.0),
492
+ )
493
+
494
+ def _spawn_server(self, spec: SessionSpec) -> _ServerSession:
495
+ """Spawn `opencode serve` and wait for readiness, retrying with a fresh
496
+ port when the process dies during the health poll (the free-port bind
497
+ is released before the server re-binds, so a collision is possible)."""
498
+ resolved = shutil.which(self.binary)
499
+ if resolved is None:
500
+ # shutil.which honors PATHEXT — on Windows the npm install is an
501
+ # `opencode.cmd` shim that a bare-name Popen (which appends only
502
+ # .exe) would never find.
503
+ raise OpencodeServerError(
504
+ f"opencode binary {self.binary!r} not found on PATH; " f"see `froid-loop validate`"
505
+ )
506
+ password = secrets.token_urlsafe(16)
507
+ env = self._session_env(spec, password)
508
+ log_path = self.logs_dir / f"{spec.task_id}.log"
509
+ log_fh = log_path.open("ab") # append: retries share one log
510
+ event_fh = None
511
+ server_fh = None
512
+ # The remaining sinks open under their own guard: each `open()` can fail
513
+ # on its own (ENOSPC, EMFILE, a permission race) with the earlier ones
514
+ # already open, and that happens ABOVE the retry loop's try/except — so
515
+ # without this the handles opened so far never reach `_close_spawn_sinks`
516
+ # and stay open for as long as the propagating traceback keeps this frame
517
+ # alive. One sink had nothing to leak here; three do.
518
+ try:
519
+ # Structured SSE trace, off entirely when the knob is down. Append
520
+ # like the other two sinks: the file is per TASK and every retry of
521
+ # that task writes into it, while `event_seq` is per SESSION and
522
+ # restarts at 1 for each one — so a seq dropping back to 1 mid-file
523
+ # marks a new session, not a truncation. `ts` (epoch ms) is the
524
+ # cross-session ordering key.
525
+ if self.sse_trace:
526
+ event_path = self.logs_dir / f"{spec.task_id}.sse.jsonl"
527
+ event_fh = event_path.open("a", encoding="utf-8") # one record per line
528
+ # The server's own stdout/stderr (INFO/diagnostic lines) is kept in a
529
+ # separate file so the readable transcript in <task_id>.log stays
530
+ # clean. This is also where a failed spawn's diagnostics land — the
531
+ # give-up error below names this path, not log_path.
532
+ server_path = self.logs_dir / f"{spec.task_id}.server.out"
533
+ server_fh = server_path.open("ab") # append: retries share one server log
534
+ except BaseException:
535
+ self._close_spawn_sinks(log_fh, server_fh, event_fh)
536
+ raise
537
+ last_error = "server did not become healthy"
538
+ try:
539
+ for _ in range(SPAWN_ATTEMPTS):
540
+ port = _free_port()
541
+ process = subprocess.Popen( # argv built from profile
542
+ self._serve_argv(resolved, port),
543
+ cwd=str(spec.cwd),
544
+ env=env,
545
+ stdout=server_fh,
546
+ stderr=subprocess.STDOUT,
547
+ stdin=subprocess.DEVNULL,
548
+ )
549
+ sess = _ServerSession(
550
+ process=process,
551
+ port=port,
552
+ base_url=f"http://127.0.0.1:{port}",
553
+ password=password,
554
+ log_fh=log_fh,
555
+ server_fh=server_fh,
556
+ event_fh=event_fh,
557
+ )
558
+ if self._await_healthy(sess):
559
+ sess.client = self._make_client(sess)
560
+ return sess
561
+ # Died or never readied: reap and try a fresh port. A live-but-
562
+ # unhealthy server is killed too — never leak what we spawned.
563
+ if process.poll() is None:
564
+ last_error = "server did not become healthy in time"
565
+ self._kill_process(sess)
566
+ else:
567
+ last_error = f"server exited rc={process.returncode} during startup"
568
+ except BaseException:
569
+ self._close_spawn_sinks(log_fh, server_fh, event_fh)
570
+ raise
571
+ self._close_spawn_sinks(log_fh, server_fh, event_fh)
572
+ raise OpencodeServerError(
573
+ f"could not start `{self.binary} serve` after {SPAWN_ATTEMPTS} attempts "
574
+ f"({last_error}); server log: {server_path}"
575
+ )
576
+
577
+ @staticmethod
578
+ def _close_spawn_sinks(log_fh: Any, server_fh: Any, event_fh: Any) -> None:
579
+ """Close whichever of the three per-task sinks are open, on the spawn
580
+ failure paths. Any of them can be None: `event_fh` when the sse_trace
581
+ knob is off, and either of the last two when the failure WAS the open
582
+ that would have created it.
583
+
584
+ Every close is guarded, matching `_teardown`. Both callers are already
585
+ reporting a failure — one is mid-`raise`, the other is about to raise
586
+ `OpencodeServerError` — so an OSError from a flush-on-close would
587
+ replace the failure actually worth reporting, and would strand the sinks
588
+ after it in this loop."""
589
+ for fh in (log_fh, server_fh, event_fh):
590
+ if fh is None:
591
+ continue
592
+ try:
593
+ fh.close()
594
+ except OSError:
595
+ pass
596
+
597
+ def _await_healthy(self, sess: _ServerSession) -> bool:
598
+ """Poll /global/health (authenticated) until it answers healthy. The
599
+ process liveness is re-checked after *every* probe: a foreign server
600
+ answering 200 on a stolen port must not mask our own corpse, and the
601
+ auth + shape check means a foreign 200 without our password never
602
+ reads as ready."""
603
+ deadline = time.monotonic() + self.health_timeout_s
604
+ with self._make_client(sess) as client:
605
+ while time.monotonic() < deadline:
606
+ healthy = False
607
+ try:
608
+ resp = client.get("/global/health")
609
+ healthy = resp.status_code == 200 and resp.json().get("healthy") is True
610
+ except Exception: # not up yet (conn refused, junk)
611
+ healthy = False
612
+ if sess.process.poll() is not None:
613
+ return False
614
+ if healthy:
615
+ return True
616
+ time.sleep(self.health_poll_s)
617
+ return False
618
+
619
+ # -------------------------------------------------------------- adapter
620
+
621
+ def start_session(self, spec: SessionSpec) -> SessionHandle:
622
+ task_dir = self.tasks_dir / spec.task_id
623
+ task_dir.mkdir(parents=True, exist_ok=True)
624
+ (task_dir / "prompt.txt").write_text(spec.prompt + "\n", encoding="utf-8")
625
+ # A re-armed/resumed run reuses task_ids; drop any prior cycle's result
626
+ # so a session that writes nothing can't be read as a stale completion.
627
+ (task_dir / "result.json").unlink(missing_ok=True)
628
+ # Same hazard, same reason, for the file the #194 tail scan reads (mirrors
629
+ # GenericAdapter.start_session, which unlinks its pane tee here). This one
630
+ # bites hardest on the path the classifier exists to serve: an env fault
631
+ # PAUSEs the run, the operator re-arms and resumes, and the next session
632
+ # reusing this task_id would scan the PREVIOUS cycle's provider error and
633
+ # pause again — however healthy the new session's own log. A pause loop
634
+ # that survives every re-arm, off one stale line.
635
+ #
636
+ # Unlinked HERE and not in _spawn_server, which deliberately opens the file
637
+ # "ab" so a spawn retry (free-port collision) keeps its predecessor's
638
+ # diagnostics inside the SAME session.
639
+ self._env_fault_log_path(spec.task_id).unlink(missing_ok=True)
640
+
641
+ launched_ns = time.time_ns()
642
+ sess = self._spawn_server(spec)
643
+ # Registered before the API handshake so the atexit sweep (and kill())
644
+ # covers a crash mid-setup; run()'s finally-kill only exists once
645
+ # start_session has returned a handle.
646
+ self._sessions[spec.task_id] = sess
647
+ try:
648
+ resp = sess.client.post("/session", json={"title": spec.task_id})
649
+ if resp.status_code != 200:
650
+ raise OpencodeServerError(
651
+ f"POST /session failed: {resp.status_code} {resp.text[:200]}"
652
+ )
653
+ sess.session_id = resp.json()["id"]
654
+
655
+ self._start_sse_reader(sess)
656
+ # Wait for the stream to actually attach before prompting: a fast
657
+ # turn can emit session.idle before the subscription exists, and a
658
+ # lost idle degrades every completion to the (slow) poll fallback.
659
+ if not sess.sse_connected.wait(timeout=self.health_timeout_s):
660
+ raise OpencodeServerError("event stream did not connect")
661
+
662
+ self._prompt(sess, self.profile.render_prompt(spec.prompt))
663
+ except Exception:
664
+ self._sessions.pop(spec.task_id, None)
665
+ self._teardown(sess)
666
+ raise
667
+ return SessionHandle(
668
+ task_id=spec.task_id, native_id=sess.session_id, launched_ns=launched_ns
669
+ )
670
+
671
+ def _prompt(self, sess: _ServerSession, text: str) -> None:
672
+ """prompt_async — the injection primitive for both the initial prompt
673
+ and the nudges. Advances the completion floor: anything completed
674
+ before this send is a previous turn's evidence. The floor moves only
675
+ once the server ACCEPTS the prompt (204) — a rejected/failed send
676
+ starts no new turn, and consuming the floor for it would discard
677
+ still-valid completion evidence of the previous turn."""
678
+ sent_ms = _now_ms() # sampled before the POST: it precedes the new turn
679
+ resp = sess.client.post(
680
+ f"/session/{sess.session_id}/prompt_async",
681
+ json={"parts": [{"type": "text", "text": text}]},
682
+ )
683
+ if resp.status_code != 204:
684
+ raise OpencodeServerError(f"prompt_async failed: {resp.status_code} {resp.text[:200]}")
685
+ sess.floor_ms = max(sess.floor_ms, sent_ms)
686
+
687
+ def send_text(self, handle: SessionHandle, text: str) -> None:
688
+ """Nudge the running session. Best-effort: a server that died between
689
+ the liveness probe and the nudge is caught as `crashed` on the next
690
+ tick, not by blowing up the completion loop."""
691
+ sess = self._sessions.get(handle.task_id)
692
+ if sess is None:
693
+ return
694
+ try:
695
+ self._prompt(sess, text)
696
+ except Exception: # nosec B110 - next tick's poll() settles liveness
697
+ pass
698
+
699
+ def _start_sse_reader(self, sess: _ServerSession) -> None:
700
+ thread = threading.Thread(
701
+ target=self._sse_loop,
702
+ args=(sess,),
703
+ name=f"opencode-sse-{sess.port}",
704
+ daemon=True,
705
+ )
706
+ sess.sse_thread = thread
707
+ thread.start()
708
+
709
+ def _sse_loop(self, sess: _ServerSession) -> None:
710
+ """SSE reader: owns its own client (created and closed here — kill()
711
+ never touches it; killing the server is what unblocks the read, with
712
+ the read timeout as backstop). Filters idle/error to this session's id
713
+ (child sessions share the stream), counts every other non-heartbeat
714
+ frame as activity, and turns any disconnect into a single `gap`
715
+ sentinel so the wait loop probes over HTTP for what the stream may
716
+ have dropped."""
717
+ httpx = self._httpx
718
+ while not sess.sse_stop.is_set():
719
+ try:
720
+ with httpx.Client(
721
+ base_url=sess.base_url,
722
+ auth=(AUTH_USER, sess.password),
723
+ timeout=httpx.Timeout(5.0, read=self.sse_read_timeout_s),
724
+ ) as client:
725
+ with client.stream("GET", "/event") as resp:
726
+ if resp.status_code != 200:
727
+ raise OpencodeServerError(f"/event -> {resp.status_code}")
728
+ # The server registers the subscriber once the response
729
+ # starts; events published after this are delivered.
730
+ sess.sse_connected.set()
731
+ for event in _parse_sse_lines(resp.iter_lines()):
732
+ if sess.sse_stop.is_set():
733
+ return
734
+ self._dispatch_sse(sess, event)
735
+ except Exception: # nosec B110 - reader must never die silently
736
+ pass
737
+ if sess.sse_stop.is_set():
738
+ return
739
+ sess.events.put("gap")
740
+ sess.sse_stop.wait(self.reconnect_sleep_s)
741
+
742
+ def _dispatch_sse(self, sess: _ServerSession, event: Any) -> None:
743
+ if not isinstance(event, dict):
744
+ return
745
+ sess.last_frame_monotonic = time.monotonic()
746
+ etype = event.get("type")
747
+ if etype in ("server.heartbeat", "server.connected"):
748
+ return
749
+ # Any substantive frame — including child-session traffic — proves the
750
+ # session tree is working (the parent is silent while a subagent
751
+ # streams, exactly like subagent output in a tmux pane log).
752
+ sess.activity += 1
753
+ props = event.get("properties") or {}
754
+ # Session-scoped frames must name this session (child/subagent sessions
755
+ # share the stream). A short allowlist of SESSION-LESS types passes
756
+ # anyway: some frames worth logging carry no sessionID at all, so there
757
+ # is no id to match and this filter would drop them before either sink.
758
+ # See `_SESSIONLESS_TYPES` for which ones, why it is an allowlist, and
759
+ # what deliberately stays out.
760
+ #
761
+ # Both sinks sit below this gate, so an allowlisted frame is traced as
762
+ # well as rendered. That is intended: the trace's contract is one record
763
+ # per frame the adapter acted on, and it is the file you read to explain
764
+ # a line in the transcript — a rendered line with no matching record
765
+ # would make the two sinks disagree. It costs nothing here because the
766
+ # allowlist admits one low-rate type, not the noisy session-less bulk.
767
+ if props.get("sessionID") != sess.session_id and not (
768
+ # `in` on a frozenset raises TypeError for an unhashable value, and
769
+ # `type` is server-controlled, so narrow before the membership test
770
+ # (the isinstance check below this filter is deliberately kept there
771
+ # — moving it up would change the activity counter above).
772
+ isinstance(etype, str)
773
+ and etype in _SESSIONLESS_TYPES
774
+ ):
775
+ return
776
+ if not isinstance(etype, str):
777
+ # Every branch below compares `type` against a string literal, so a
778
+ # frame carrying a non-string (or absent) type can match none of
779
+ # them. Dropping it here narrows the value for the typed helpers and
780
+ # keeps a malformed frame out of the trace — it is not a frame the
781
+ # adapter acted on. Liveness/activity above already counted it.
782
+ return
783
+ # Before the control queue: emit a structured trace record and render
784
+ # one human-readable line into the run log. Both helpers no-op on
785
+ # unknown types, so unhandled frames leave both sinks byte-identical to
786
+ # today, and idle/error still queue below — control flow is unchanged.
787
+ #
788
+ # Guarded as a block because these two sinks sit UPSTREAM of the
789
+ # idle/error queue puts on a frame-shaped payload we do not control:
790
+ # both helpers walk nested `.get` chains (props["part"], ["info"],
791
+ # ["permission"]), so a server that ships a string where a dict is
792
+ # expected raises AttributeError here. Unguarded that unwinds all the
793
+ # way to the reader's connection-level catch in `_sse_loop`, which tears
794
+ # the stream down, reconnects, and drops whatever the server had already
795
+ # buffered — a logging bug silently degrading completion signaling. Same
796
+ # never-die doctrine as the reader itself: logging is advisory, the
797
+ # queue puts below are not. `_emit_event` runs first (it is the simpler
798
+ # of the two) so a render bug still leaves the offending frame recorded
799
+ # in the trace that would be used to diagnose it.
800
+ try:
801
+ self._emit_event(sess, etype, props)
802
+ self._render_inline(sess, etype, props)
803
+ except Exception: # nosec B110 - logging must never disturb the queue puts below
804
+ pass
805
+ if etype == "session.idle":
806
+ sess.events.put("idle")
807
+ elif etype == "session.error":
808
+ sess.events.put("error")
809
+
810
+ def _render_inline(self, sess: _ServerSession, etype: str, props: dict) -> None:
811
+ """Append one human-readable progress line to the run log for the event
812
+ types worth watching live. No-ops on anything else, so an unhandled
813
+ frame writes nothing.
814
+
815
+ ``log_fh`` is the readable transcript sink (``<task_id>.log``): it
816
+ carries ONLY these curated ``[froid]`` lines. The server's own INFO /
817
+ diagnostic stdout is kept apart in ``<task_id>.server.out`` (see
818
+ ``server_fh``) so the transcript stays a clean, line-wrapped
819
+ conversation rather than a jumble of server logging. Only this SSE
820
+ reader thread writes ``log_fh``, and each line is a single ``write()``
821
+ plus ``flush()`` — so it always lands as a whole line at EOF.
822
+
823
+ Every event-type string below, and the payload shape each branch reads,
824
+ is pinned in the module docstring's ``/event`` section against a live
825
+ 1.18.2 probe — read that before adding or changing a branch here. That
826
+ is deliberately the only copy of those facts: three of this renderer's
827
+ original branches named frames the server does not send, and a second
828
+ copy is a second thing to get wrong on the next version bump.
829
+
830
+ ``message.updated`` carries only turn metadata (notably ``info.role``
831
+ keyed by ``info.id``) and renders no inline line of its own — it just
832
+ records the role under the message id so the following
833
+ ``message.part.*`` lines resolve the right speaker. Keying by message
834
+ id (not "last role seen") is required because ``message.updated``
835
+ frames arrive out of order: a re-emit for the user message lands mid-
836
+ assistant-turn and would otherwise flip the prefix to ``user:`` on the
837
+ assistant's own reply text. That also avoids a flood of role-only
838
+ announce lines like ``[froid] assistant: assistant``."""
839
+ if etype == "message.updated":
840
+ info = props.get("info") or {}
841
+ mid = info.get("id")
842
+ role = info.get("role")
843
+ if mid and role:
844
+ sess.msg_roles[mid] = role
845
+ return
846
+ role = "assistant"
847
+ if etype == "message.part.updated":
848
+ mid = (props.get("part") or {}).get("messageID")
849
+ if mid and mid in sess.msg_roles:
850
+ role = sess.msg_roles[mid]
851
+ line = self._inline_line(etype, props, role)
852
+ if line is None:
853
+ return
854
+ if sess.log_fh is None: # bare-sess unit tests / disabled inline log
855
+ return
856
+ try:
857
+ # CRLF line endings: the froid-loop TUI renders this log through a
858
+ # pyte terminal emulator, where a bare LF moves the cursor down
859
+ # without returning to column 0 (a real PTY's ONLCR does that, but
860
+ # pyte is a pure VT100 emulator). LF-only lines staircase right,
861
+ # so emit CRLF like real terminal output (the claude/tmux captures
862
+ # carry it via cursor-addressing; our plain text must carry it raw).
863
+ sess.log_fh.write(line.replace("\n", "\r\n").encode("utf-8"))
864
+ sess.log_fh.flush()
865
+ except OSError:
866
+ pass
867
+
868
+ def _inline_line(self, etype: str, props: dict, role: str) -> str | None:
869
+ # ``message.updated`` is consumed in ``_render_inline`` (role refresh
870
+ # only); it renders no line here.
871
+ # assistant / user assembled text. Per-token ``message.part.delta``
872
+ # frames are NOT rendered inline: they concatenate to exactly the text
873
+ # ``message.part.updated`` already carries complete, so emitting both
874
+ # duplicates every statement as a one-word-per-line flood. For the same
875
+ # reason they are excluded from the SSE trace too (see ``_emit_event``);
876
+ # nothing consumes deltas anywhere in this adapter.
877
+ if etype == "message.part.updated":
878
+ part = props.get("part") or {}
879
+ # Tool execution rides this same frame as a `tool`-typed part —
880
+ # there is no tool.call/tool.response event on this surface, and
881
+ # `command.executed` never fires for it (both pinned live). Tool
882
+ # parts carry no `text`, so this branch MUST sit above the
883
+ # empty-text `return None` below or it can never be reached: that
884
+ # exact ordering is the difference between rendering the agent's
885
+ # actions and rendering nothing but its prose.
886
+ if part.get("type") == "tool":
887
+ return self._tool_line(part)
888
+ text = part.get("text") or ""
889
+ body = text.strip() # drop surrounding blanks; keep internal structure
890
+ if not body: # empty/non-text part — no value live
891
+ return None
892
+ role = role or "assistant"
893
+ # A blank line separates each turn from the previous one (the
894
+ # separator sits ABOVE the header, body follows directly beneath),
895
+ # and the role-colored marker line anchors the turn boundary.
896
+ # Internal newlines in the body are preserved so multi-paragraph
897
+ # reasoning reads as prose under the one header.
898
+ return f"\n{_role_color(role)}[froid] {role}:{_RESET}\n{body}\n"
899
+ # The SLASH-COMMAND surface — NOT the tool surface. Agent tool use never
900
+ # reaches here (0 `command.executed` frames across live bash/write/read/
901
+ # edit calls); it renders from the `tool`-typed part above. Kept because
902
+ # a slash command invoked through this server is still worth a line and
903
+ # its payload is pinned (`name`/`arguments`/`messageID`, all required),
904
+ # but do not "fix" tool rendering back into this branch.
905
+ if etype == "command.executed":
906
+ args = _sum_args(props.get("arguments"))
907
+ return f"{_TOOL_COLOR}[froid] cmd: {props.get('name') or '?'}{(' ' + args) if args else ''}{_RESET}\n"
908
+ if etype == "file.edited":
909
+ # Session-less: reachable only via `_SESSIONLESS_TYPES`.
910
+ return f"{_TOOL_COLOR}[froid] file: {props.get('file') or '?'}{_RESET}\n"
911
+ # Permission surface. Field names are the live 1.18.2 ones: the ask frame
912
+ # is `permission.asked` (there is no `permission.updated`) carrying a
913
+ # `permission` STRING plus a `patterns` list — not a nested object with
914
+ # `type`/`pattern` — and the reply carries `reply`, not `response`.
915
+ # `metadata` (the concrete command) and `always` are left to the trace;
916
+ # `patterns` is what the permission actually matched on.
917
+ if etype == "permission.asked":
918
+ pats = _sum_args(props.get("patterns"))
919
+ return (
920
+ f"{_TOOL_COLOR}[froid] perm ask: {props.get('permission') or '?'}"
921
+ f"{(' ' + pats) if pats else ''}{_RESET}\n"
922
+ )
923
+ if etype == "permission.replied":
924
+ return f"{_TOOL_COLOR}[froid] perm reply: {props.get('reply') or '?'}{_RESET}\n"
925
+ # A failing turn is the one thing a reader most wants in the transcript:
926
+ # without this the log just stops mid-turn while the queue put below
927
+ # ends the session. Same green/tool hue as the other role-less markers —
928
+ # the color family is "not a speaker", not "severity". `error` has no
929
+ # pinned shape on this surface, so summarize whatever it carries rather
930
+ # than reaching into fields that may not exist.
931
+ if etype == "session.error":
932
+ detail = _sum_args(props.get("error"))
933
+ return f"{_TOOL_COLOR}[froid] error:{(' ' + detail) if detail else ''}{_RESET}\n"
934
+ return None
935
+
936
+ @staticmethod
937
+ def _tool_line(part: dict) -> str | None:
938
+ """One line for a finished tool call, or None while it is still running.
939
+
940
+ A `tool`-typed ``message.part.updated`` part fires once per state
941
+ transition — ``pending`` → ``running`` → terminal — carrying the SAME
942
+ part id each time. Rendering every fire would print each tool call three
943
+ times, so the line is emitted only on a terminal status
944
+ (``_TOOL_TERMINAL_STATES``): a tool call reaches exactly one terminal
945
+ state, which makes that gate exactly "once per tool call". The gate is
946
+ the whole correctness argument here — a test that only asserts a tool
947
+ renders at all passes with it removed.
948
+
949
+ Same green/tool hue as the other role-less markers: a tool call is an
950
+ action, not a speaker. ``state.input`` is summarized inline (that is the
951
+ "what did the agent just do" the reader wants); ``state.output`` is
952
+ deliberately NOT — it is a full command's stdout or a whole file's text
953
+ on this surface, and the complete payload is in the SSE trace for anyone
954
+ who needs it. A failure appends the error, because a tool that failed
955
+ silently is exactly what a reader would go looking for."""
956
+ state = part.get("state") or {}
957
+ status = state.get("status")
958
+ if status not in _TOOL_TERMINAL_STATES:
959
+ return None
960
+ args = _sum_args(state.get("input"))
961
+ suffix = ""
962
+ if status == "error":
963
+ detail = _sum_args(state.get("error"))
964
+ suffix = f" -> error{(': ' + detail) if detail else ''}"
965
+ return (
966
+ f"{_TOOL_COLOR}[froid] tool: {part.get('tool') or '?'}"
967
+ f"{(' ' + args) if args else ''}{suffix}{_RESET}\n"
968
+ )
969
+
970
+ def _emit_event(self, sess: _ServerSession, etype: str, props: dict) -> None:
971
+ """Append one structured JSON object to the session's
972
+ ``<task_id>.sse.jsonl`` sink — a record for every acted-on frame except
973
+ the per-token deltas, for post-hoc replay/debugging. No-ops when the
974
+ sink is off (``event_fh is None``, i.e. the ``sse_trace`` knob is down).
975
+
976
+ ``message.part.delta`` is the one exclusion. Its text concatenates
977
+ byte-exactly to the ``message.part.updated`` record already in the file
978
+ (pinned live — see the module docstring), so every delta is bytes spent
979
+ re-storing text the trace already holds, and they dominate the frame
980
+ count on any real turn. ``logs/`` is never trimmed by retention, so that
981
+ cost is permanent. Drop the branch to get them back."""
982
+ if sess.event_fh is None:
983
+ return
984
+ if etype == "message.part.delta":
985
+ return
986
+ sess.event_seq += 1
987
+ record = {
988
+ "seq": sess.event_seq,
989
+ "type": etype,
990
+ "properties": props,
991
+ "ts": _now_ms(), # epoch ms — the unit every opencode time.* uses
992
+ }
993
+ try:
994
+ sess.event_fh.write(json.dumps(record, ensure_ascii=False) + "\n")
995
+ sess.event_fh.flush()
996
+ except (OSError, TypeError, ValueError):
997
+ pass
998
+
999
+ # ------------------------------------------------------ completion loop
1000
+
1001
+ def _result_json(self, handle: SessionHandle, spec: SessionSpec, *, wait: bool) -> dict | None:
1002
+ if not wait:
1003
+ return self._read_result(handle.task_id)
1004
+ # Pass the grace explicitly: the mixin's grace_s default is bound at
1005
+ # def time, so an instance override is the only reachable knob.
1006
+ if self.result_grace_s is not None:
1007
+ return self._await_result(handle.task_id, grace_s=self.result_grace_s)
1008
+ return self._await_result(handle.task_id)
1009
+
1010
+ def wait_for_completion(self, handle: SessionHandle, spec: SessionSpec) -> SessionResult:
1011
+ sess = self._sessions.get(handle.task_id)
1012
+ if sess is None:
1013
+ raise OpencodeServerError(f"no live opencode server for task {handle.task_id!r}")
1014
+ deadline = time.monotonic() + spec.timeout_s
1015
+ # Wall-clock co-bound (#157): a host suspend freezes time.monotonic(),
1016
+ # silently extending the monotonic deadline by the nap's length. The
1017
+ # wall clock keeps counting through a suspend, so it may EXPIRE the
1018
+ # deadline — never extend it; all sub-waits below stay monotonic (a
1019
+ # wall clock stepped backward must not stretch the session).
1020
+ wall_deadline = time.time() + spec.timeout_s
1021
+ session_id = sess.session_id
1022
+ nudges_left = self._stop_nudges
1023
+ # Mirrors generic.wait_for_completion: stall grace starts at launch,
1024
+ # re-arms on activity or a result-less Stop (idle), and is spent via
1025
+ # wake-nudges bounded by the monotonic spec.stall_nudges_cap (#149).
1026
+ stall_deadline = time.monotonic() + self._stall_grace_s if self._stall_grace_s > 0 else None
1027
+ last_activity = sess.activity
1028
+ stall_nudges_left = self._stall_nudges
1029
+ stall_nudges_sent = 0
1030
+ # Loop-owned silence clock: updated on every dequeue, so a dead reader
1031
+ # thread degrades to the poll fallback instead of disabling it.
1032
+ last_seen = time.monotonic()
1033
+ # monotonic ts of the last heartbeat.json overwrite; None = not yet
1034
+ # written, so the first tick always stamps one.
1035
+ last_heartbeat: float | None = None
1036
+ # Session-budget guard (#158), mirroring generic.wait_for_completion:
1037
+ # latched on the first cap crossing; budget_deadline is the enforce-mode
1038
+ # monotonic grace expiry, checked every tick (sampling itself rides the
1039
+ # heartbeat cadence).
1040
+ budget_tripped = False
1041
+ budget_weighted: int | None = None
1042
+ budget_deadline: float | None = None
1043
+ # wall-clock co-bound for the grace (#157 pattern): a host suspend
1044
+ # freezes time.monotonic(), silently stretching the "bounded" wrap-up
1045
+ # window; the wall clock may EXPIRE the grace — never extend it.
1046
+ budget_wall_deadline: float | None = None
1047
+
1048
+ while True:
1049
+ remaining = deadline - time.monotonic()
1050
+ wall_expired = time.time() >= wall_deadline
1051
+ if remaining <= 0 or wall_expired:
1052
+ if remaining <= 0 and wall_expired:
1053
+ expired = "both"
1054
+ elif remaining <= 0:
1055
+ expired = "monotonic"
1056
+ else:
1057
+ # wall-only expiry with monotonic time to spare: the
1058
+ # monotonic clock stood still — the suspend signature.
1059
+ expired = "wall"
1060
+ self._note_lifecycle(
1061
+ handle.task_id,
1062
+ "timeout-fired",
1063
+ expired_clock=expired,
1064
+ timeout_s=spec.timeout_s,
1065
+ mono_remaining_s=round(remaining, 3),
1066
+ )
1067
+ self._abort(sess)
1068
+ transcript = self._capture_usage(handle, sess)
1069
+ return SessionResult(
1070
+ status="timeout",
1071
+ session_id=session_id,
1072
+ transcript_path=transcript,
1073
+ timeout_fired_at=time.time(),
1074
+ timeout_expired_clock=expired,
1075
+ budget_weighted=budget_weighted,
1076
+ )
1077
+ # Hard-stop poll (#319), per-iteration and deliberately NOT inside the
1078
+ # heartbeat throttle below: the loop blocks up to `POLL_TICK_S` (5s) per
1079
+ # tick, so *detection* normally lands well inside `stop_run`'s 10s
1080
+ # grace window — the common case, not a bound: the dispatch legs below
1081
+ # the wait are bounded only by the client's own timeouts, and the
1082
+ # generic adapter is no better off (its `_await_result` waits
1083
+ # RESULT_GRACE_S on a healthy box). Beyond detection, this arm then
1084
+ # makes two
1085
+ # HTTP round-trips against a server that may itself be wedged, and the
1086
+ # client's 10s per-phase timeout applies to each. So the arm is NOT
1087
+ # bounded by the grace window, by design: it gives the engine its best
1088
+ # chance to tear itself down cleanly, and when the server will not answer
1089
+ # it degrades to `stop_run`'s force-kill backstop — the same outcome
1090
+ # every native-Windows stop had before #319, never a worse one. Don't
1091
+ # "fix" this by trimming the timeouts: the same two calls serve the
1092
+ # timeout arm, where the transcript is the whole diagnostic payload.
1093
+ #
1094
+ # Mirror the timeout arm exactly — without `_abort` the in-flight HTTP
1095
+ # turn keeps running until teardown. Return the verdict; never raise
1096
+ # `RunStopped` here, and never unlink the request file: the engine
1097
+ # consumes it and attributes the stop.
1098
+ if self._hard_stop_requested():
1099
+ self._note_lifecycle(handle.task_id, "stop-abort-fired")
1100
+ self._abort(sess)
1101
+ transcript = self._capture_usage(handle, sess)
1102
+ return SessionResult(
1103
+ status="aborted",
1104
+ session_id=session_id,
1105
+ transcript_path=transcript,
1106
+ budget_weighted=budget_weighted,
1107
+ )
1108
+ now = time.monotonic()
1109
+ if last_heartbeat is None or now - last_heartbeat >= HEARTBEAT_INTERVAL_S:
1110
+ last_heartbeat = now
1111
+ self._write_heartbeat(
1112
+ handle.task_id,
1113
+ {
1114
+ "ts": time.time(),
1115
+ "remaining_s": round(remaining, 3),
1116
+ "stall_armed": stall_deadline is not None,
1117
+ "stall_nudges_sent": stall_nudges_sent,
1118
+ },
1119
+ )
1120
+ # Mid-session spec-status transition sampling (#276 M2) rides the
1121
+ # same heartbeat cadence — a no-op on the plain HTTP adapter; the
1122
+ # OpencodeDevAdapter shares _DevSynthesisMixin, so the hook records
1123
+ # transitions there exactly as on the generic dev adapter.
1124
+ self._observe_tick(handle, spec)
1125
+ # Budget sampling rides the heartbeat cadence — no extra knob.
1126
+ # Usage comes over HTTP (server state is sqlite, not a file).
1127
+ if (
1128
+ not budget_tripped
1129
+ and spec.token_budget is not None
1130
+ and spec.token_budget_mode in ("warn", "enforce")
1131
+ ):
1132
+ weighted = self._sample_weighted_usage(sess, spec)
1133
+ if weighted is not None and weighted > spec.token_budget:
1134
+ budget_tripped = True
1135
+ budget_weighted = weighted
1136
+ self._note_lifecycle(
1137
+ handle.task_id,
1138
+ "budget-tripped",
1139
+ weighted=weighted,
1140
+ budget=spec.token_budget,
1141
+ mode=spec.token_budget_mode,
1142
+ )
1143
+ try:
1144
+ gates.notify(
1145
+ self.policy,
1146
+ self.run_dir,
1147
+ "froid-loop session over token budget",
1148
+ f"{handle.task_id}: weighted spend {weighted} crossed the "
1149
+ f"{spec.token_budget} per-session cap "
1150
+ f"(mode={spec.token_budget_mode})",
1151
+ )
1152
+ except OSError:
1153
+ # observe-degrade: an unwritable ATTENTION file is
1154
+ # observability, never a reason to break the loop
1155
+ # (the _write_heartbeat doctrine).
1156
+ pass
1157
+ # nosec below: bandit B105 pattern-matches the "token"
1158
+ # in token_budget_mode as a hardcoded-password compare;
1159
+ # it is a mode enum, not a credential.
1160
+ if spec.token_budget_mode == "enforce": # nosec B105
1161
+ if spec.token_budget_grace_s <= 0:
1162
+ # zero grace = terminate at trip, no nudge — but
1163
+ # server death still wins (artifact honored via
1164
+ # the crash path), exactly like grace expiry.
1165
+ if sess.process.poll() is not None:
1166
+ transcript = self._capture_usage(handle, sess)
1167
+ return self._final(
1168
+ handle,
1169
+ spec,
1170
+ "crashed",
1171
+ session_id,
1172
+ transcript,
1173
+ budget_weighted=weighted,
1174
+ )
1175
+ self._note_lifecycle(
1176
+ handle.task_id,
1177
+ "over-budget-fired",
1178
+ weighted=weighted,
1179
+ budget=spec.token_budget,
1180
+ grace_s=spec.token_budget_grace_s,
1181
+ zero_grace=True,
1182
+ )
1183
+ self._abort(sess)
1184
+ transcript = self._capture_usage(handle, sess)
1185
+ return SessionResult(
1186
+ status="over_budget",
1187
+ session_id=session_id,
1188
+ transcript_path=transcript,
1189
+ budget_weighted=weighted,
1190
+ )
1191
+ try:
1192
+ self.send_text(handle, BUDGET_NUDGE_TEXT)
1193
+ except Exception: # nosec B110 - best-effort nudge
1194
+ # a dead/hung server can't take the nudge; the
1195
+ # grace still arms — the next tick's process
1196
+ # poll scores a dead server crashed.
1197
+ pass
1198
+ budget_deadline = time.monotonic() + spec.token_budget_grace_s
1199
+ budget_wall_deadline = time.time() + spec.token_budget_grace_s
1200
+ if budget_deadline is not None and (
1201
+ time.monotonic() >= budget_deadline
1202
+ or (budget_wall_deadline is not None and time.time() >= budget_wall_deadline)
1203
+ ):
1204
+ # Grace expired with no completion (wall co-bound included: a
1205
+ # suspend-frozen monotonic clock must not stretch the window,
1206
+ # #157). Server death ≙ window death (the crash path honors a
1207
+ # landed artifact); a live server ends over_budget WITHOUT
1208
+ # reading the result file — an artifact under a live session is
1209
+ # never trusted (#48/#53).
1210
+ if sess.process.poll() is not None:
1211
+ transcript = self._capture_usage(handle, sess)
1212
+ return self._final(
1213
+ handle,
1214
+ spec,
1215
+ "crashed",
1216
+ session_id,
1217
+ transcript,
1218
+ budget_weighted=budget_weighted,
1219
+ )
1220
+ self._note_lifecycle(
1221
+ handle.task_id,
1222
+ "over-budget-fired",
1223
+ weighted=budget_weighted,
1224
+ budget=spec.token_budget,
1225
+ grace_s=spec.token_budget_grace_s,
1226
+ zero_grace=False,
1227
+ )
1228
+ self._abort(sess)
1229
+ transcript = self._capture_usage(handle, sess)
1230
+ return SessionResult(
1231
+ status="over_budget",
1232
+ session_id=session_id,
1233
+ transcript_path=transcript,
1234
+ budget_weighted=budget_weighted,
1235
+ )
1236
+ try:
1237
+ event: str | None = sess.events.get(timeout=min(remaining, self.poll_tick_s))
1238
+ except queue.Empty:
1239
+ event = None
1240
+ if event is not None:
1241
+ last_seen = time.monotonic()
1242
+
1243
+ # Second poll (#319) — see the arm at the top of the loop. What follows
1244
+ # here is the dispatch: `_probe_completion`'s two GETs, which are NOT
1245
+ # throttled (once a turn goes quiet past SILENCE_THRESHOLD_S they run on
1246
+ # every tick), a `_session_status` GET, or `_result_json(wait=True)`'s
1247
+ # RESULT_GRACE_S wait. Each is bounded only by the client's own timeouts,
1248
+ # so a single iteration can outlast `stop_run`'s 10s grace. Polling here
1249
+ # keeps at most one leg between two checks. It cannot bound an in-flight
1250
+ # socket read, so when one does outlast the window the stop degrades to
1251
+ # the force-kill backstop exactly as it did before #319.
1252
+ if self._hard_stop_requested():
1253
+ self._note_lifecycle(handle.task_id, "stop-abort-fired")
1254
+ self._abort(sess)
1255
+ transcript = self._capture_usage(handle, sess)
1256
+ return SessionResult(
1257
+ status="aborted",
1258
+ session_id=session_id,
1259
+ transcript_path=transcript,
1260
+ budget_weighted=budget_weighted,
1261
+ )
1262
+ if event == "error":
1263
+ # session.error may precede a retry, not a turn-end (status
1264
+ # "retry" exists); only a PROVABLY settled session reads as a
1265
+ # result-less Stop — an errored turn may never get
1266
+ # time.completed, so waiting on proof-of-work alone would burn
1267
+ # the whole timeout. A dead server takes the crash path; an
1268
+ # unknowable status (probe failure) keeps waiting rather than
1269
+ # mis-nudging a session that may still be retrying — timeout_s
1270
+ # bounds a persistently unreadable one.
1271
+ if sess.process.poll() is not None:
1272
+ transcript = self._capture_usage(handle, sess)
1273
+ return self._final(
1274
+ handle,
1275
+ spec,
1276
+ "crashed",
1277
+ session_id,
1278
+ transcript,
1279
+ budget_weighted=budget_weighted,
1280
+ )
1281
+ if self._session_status(sess) is not False:
1282
+ continue
1283
+ event = "idle"
1284
+
1285
+ if event in (None, "gap"):
1286
+ if sess.process.poll() is not None:
1287
+ # Server death ≙ window death: the crash path vouches for a
1288
+ # landed artifact (accept_result=True), same as generic.
1289
+ transcript = self._capture_usage(handle, sess)
1290
+ return self._final(
1291
+ handle,
1292
+ spec,
1293
+ "crashed",
1294
+ session_id,
1295
+ transcript,
1296
+ budget_weighted=budget_weighted,
1297
+ )
1298
+ silent = (
1299
+ time.monotonic() - max(last_seen, sess.last_frame_monotonic)
1300
+ > self.silence_threshold_s
1301
+ )
1302
+ if (event == "gap" or silent) and self._probe_completion(sess):
1303
+ event = "idle" # fall through to the Stop path below
1304
+ last_seen = time.monotonic()
1305
+ else:
1306
+ if stall_deadline is not None:
1307
+ # Re-arm on activity (any SSE traffic since arming) —
1308
+ # a session streaming subagent work is working, not
1309
+ # stalled; only genuine silence trips the stall below.
1310
+ key = sess.activity
1311
+ if last_activity is None or key != last_activity:
1312
+ last_activity = key
1313
+ stall_deadline = time.monotonic() + self._stall_grace_s
1314
+ continue
1315
+ if time.monotonic() >= stall_deadline:
1316
+ if self._session_status(sess):
1317
+ # Provably mid-turn (a busy child/parent the
1318
+ # SSE missed): re-arm rather than injecting a
1319
+ # prompt into a working session or declaring
1320
+ # it stalled after the nudge budget is spent.
1321
+ stall_deadline = time.monotonic() + self._stall_grace_s
1322
+ continue
1323
+ if stall_nudges_left > 0 and (
1324
+ spec.stall_nudges_cap is None
1325
+ or stall_nudges_sent < spec.stall_nudges_cap
1326
+ ):
1327
+ # Unknown status (None) proceeds to the nudge —
1328
+ # a transport too broken to answer the probe
1329
+ # would fail the nudge too, and burning the
1330
+ # bounded budget converges to an honest stall.
1331
+ stall_nudges_left -= 1
1332
+ stall_nudges_sent += 1
1333
+ self.send_text(handle, STALL_NUDGE_TEXT)
1334
+ stall_deadline = time.monotonic() + self._stall_grace_s
1335
+ last_activity = sess.activity
1336
+ continue
1337
+ # Re-probe liveness before finalizing: a hard death
1338
+ # in the gap since the top-of-tick check flows
1339
+ # through the crash path (artifact honored) instead
1340
+ # of a stall that discards a just-flushed result.
1341
+ if sess.process.poll() is not None:
1342
+ transcript = self._capture_usage(handle, sess)
1343
+ return self._final(
1344
+ handle,
1345
+ spec,
1346
+ "crashed",
1347
+ session_id,
1348
+ transcript,
1349
+ budget_weighted=budget_weighted,
1350
+ )
1351
+ transcript = self._capture_usage(handle, sess)
1352
+ return self._final(
1353
+ handle,
1354
+ spec,
1355
+ "stalled",
1356
+ session_id,
1357
+ transcript,
1358
+ accept_result=False,
1359
+ budget_weighted=budget_weighted,
1360
+ )
1361
+ continue
1362
+
1363
+ if event == "idle":
1364
+ result_json = self._result_json(handle, spec, wait=True)
1365
+ if result_json is not None:
1366
+ transcript = self._capture_usage(handle, sess)
1367
+ return SessionResult(
1368
+ status="completed",
1369
+ result_json=result_json,
1370
+ session_id=session_id,
1371
+ transcript_path=transcript,
1372
+ budget_weighted=budget_weighted,
1373
+ )
1374
+ if nudges_left > 0:
1375
+ nudges_left -= 1
1376
+ self.send_text(handle, NUDGE_TEXT)
1377
+ continue
1378
+ if self._stall_grace_s <= 0:
1379
+ transcript = self._capture_usage(handle, sess)
1380
+ return self._final(
1381
+ handle,
1382
+ spec,
1383
+ "stalled",
1384
+ session_id,
1385
+ transcript,
1386
+ budget_weighted=budget_weighted,
1387
+ )
1388
+ # A result-less Stop, but the session may have ended its turn
1389
+ # awaiting a background process: open/re-arm the idle-grace
1390
+ # window; a fresh Stop lands here again and resets it.
1391
+ stall_deadline = time.monotonic() + self._stall_grace_s
1392
+ last_activity = sess.activity
1393
+ # a real turn-end proves the session responsive: restore the
1394
+ # wake-nudge budget (the monotonic cap still bounds the total).
1395
+ stall_nudges_left = self._stall_nudges
1396
+ continue
1397
+
1398
+ def _session_status(self, sess: _ServerSession) -> bool | None:
1399
+ """Tri-state /session/status probe: True = busy/retrying, False =
1400
+ provably settled (absent from the map or explicit idle),
1401
+ None = unknowable (unreachable server, non-200). Unknown is not
1402
+ settled: callers must not treat a failed probe as proof a turn ended
1403
+ (the liveness-parity doctrine); the process poll settles real death."""
1404
+ try:
1405
+ resp = sess.client.get("/session/status")
1406
+ if resp.status_code != 200:
1407
+ return None
1408
+ status = resp.json().get(sess.session_id) or {}
1409
+ return status.get("type") in ("busy", "retry")
1410
+ except Exception: # probe is advisory
1411
+ return None
1412
+
1413
+ def _probe_completion(self, sess: _ServerSession) -> bool:
1414
+ """HTTP fallback for a lossy stream: the turn is finished
1415
+ only when the session is PROVABLY settled (an unknowable status is not
1416
+ settled) AND an assistant message completed strictly after the
1417
+ completion floor exists — an idle status alone is not proof a turn ran
1418
+ (abort emits idle even on an idle session). Consuming the evidence advances the floor so one
1419
+ completion can never be consumed twice."""
1420
+ if self._session_status(sess) is not False:
1421
+ return False
1422
+ try:
1423
+ resp = sess.client.get(f"/session/{sess.session_id}/message")
1424
+ if resp.status_code != 200:
1425
+ return False
1426
+ completed = 0
1427
+ for msg in resp.json():
1428
+ info = msg.get("info") or {}
1429
+ if info.get("role") != "assistant":
1430
+ continue
1431
+ done_ms = (info.get("time") or {}).get("completed") or 0
1432
+ completed = max(completed, int(done_ms))
1433
+ except Exception: # probe is advisory
1434
+ return False
1435
+ if completed > sess.floor_ms:
1436
+ sess.floor_ms = completed
1437
+ return True
1438
+ return False
1439
+
1440
+ def _abort(self, sess: _ServerSession) -> None:
1441
+ if sess.client is None or not sess.session_id or sess.process.poll() is not None:
1442
+ return
1443
+ try:
1444
+ sess.client.post(f"/session/{sess.session_id}/abort")
1445
+ except Exception: # nosec B110 - abort is best-effort
1446
+ pass
1447
+
1448
+ # ----------------------------------------------------------------- usage
1449
+
1450
+ def _sample_weighted_usage(self, sess: _ServerSession, spec: SessionSpec) -> int | None:
1451
+ """Mid-session cumulative weighted spend over HTTP, or None when the
1452
+ guard must stay inert this tick (no live session yet, non-200, a
1453
+ transport error). Never raises — sampling must not break the wait
1454
+ loop."""
1455
+ if sess.client is None or not sess.session_id:
1456
+ return None
1457
+ try:
1458
+ resp = sess.client.get(f"/session/{sess.session_id}/message")
1459
+ if resp.status_code != 200:
1460
+ return None
1461
+ usage = _sum_usage(resp.json())
1462
+ except Exception: # sampling is advisory
1463
+ return None
1464
+ return usage.weighted_total(spec.cache_read_weight)
1465
+
1466
+ def _capture_usage(self, handle: SessionHandle, sess: _ServerSession) -> str | None:
1467
+ """Read usage over HTTP before teardown (state is server-side sqlite):
1468
+ dump the raw messages as the transcript and stash the token sum by
1469
+ session id for read_usage(). Best-effort in full — the crashed path
1470
+ runs this against a dead server and the verdict must not change."""
1471
+ if sess.client is None or not sess.session_id:
1472
+ return None
1473
+ try:
1474
+ resp = sess.client.get(f"/session/{sess.session_id}/message")
1475
+ if resp.status_code != 200:
1476
+ return None
1477
+ messages = resp.json()
1478
+ path = self.tasks_dir / handle.task_id / "messages.json"
1479
+ path.write_text(json.dumps(messages, ensure_ascii=False, indent=2), encoding="utf-8")
1480
+ self._usage[sess.session_id] = _sum_usage(messages)
1481
+ return str(path)
1482
+ except Exception: # usage is metadata, never a gate
1483
+ return None
1484
+
1485
+ def read_usage(self, result: SessionResult) -> TokenUsage | None:
1486
+ if not result.session_id:
1487
+ return None
1488
+ return self._usage.get(result.session_id)
1489
+
1490
+ # -------------------------------------------------------------- teardown
1491
+
1492
+ def kill(self, handle: SessionHandle) -> None:
1493
+ sess = self._sessions.pop(handle.task_id, None)
1494
+ if sess is None:
1495
+ return
1496
+ self._abort(sess)
1497
+ self._teardown(sess)
1498
+
1499
+ def _teardown(self, sess: _ServerSession) -> None:
1500
+ sess.sse_stop.set()
1501
+ self._kill_process(sess)
1502
+ if sess.sse_thread is not None:
1503
+ # Bounded: killing the server closed the stream socket, which
1504
+ # unblocks the reader; the join is a courtesy, never a gate.
1505
+ sess.sse_thread.join(timeout=2.0)
1506
+ if sess.client is not None:
1507
+ try:
1508
+ sess.client.close()
1509
+ except Exception: # nosec B110 - closing is best-effort
1510
+ pass
1511
+ try:
1512
+ sess.log_fh.close()
1513
+ except OSError:
1514
+ pass
1515
+ if sess.server_fh is not None:
1516
+ try:
1517
+ sess.server_fh.close()
1518
+ except OSError:
1519
+ pass
1520
+ if sess.event_fh is not None:
1521
+ try:
1522
+ sess.event_fh.close()
1523
+ except OSError:
1524
+ pass
1525
+
1526
+ def _kill_process(self, sess: _ServerSession) -> None:
1527
+ process = sess.process
1528
+ if process.poll() is not None:
1529
+ return
1530
+ try:
1531
+ host = get_process_host()
1532
+ except ProcessHostError:
1533
+ # An explicit-but-bogus FROID_LOOP_PROCESS_HOST override raises loudly
1534
+ # (deliberate doctrine — never silently mis-signal). But the lookup
1535
+ # precedes the first signal, so the server must not be left alive
1536
+ # behind the raise: one legacy Popen root strike (no host → no tree
1537
+ # kill; the win32 kill() reaps only the .cmd wrapper — an accepted
1538
+ # degrade on a loud config error), then re-raise. Mirrors the tmux
1539
+ # adapter's kill().
1540
+ try:
1541
+ if sys.platform == "win32":
1542
+ process.kill()
1543
+ else:
1544
+ process.terminate()
1545
+ except OSError:
1546
+ pass
1547
+ raise
1548
+ # Harvest the server's descendant tree BEFORE the first signal, while the
1549
+ # tree is intact (#183): a tool subprocess the server detached (setsid, a
1550
+ # double-fork) outlives the root's SIGTERM and would keep writing into the
1551
+ # worktree the engine is about to merge/remove. Post-kill it reparents to
1552
+ # init and is unreachable, so snapshot it now; each member's pid-reuse
1553
+ # identity rides along from the enumeration itself, and the reap below is
1554
+ # identity-guarded via alive_and_ours, never a bare (reusable) pid. {} =
1555
+ # no descendants / psutil absent → the root ladder alone, as before.
1556
+ tree = host.descendants(process.pid)
1557
+ # The live Popen handle pins the pid (win32 handle / unreaped POSIX
1558
+ # child), so signalling it cannot hit a reused pid — the identity
1559
+ # confirmation force_kill's contract asks for.
1560
+ if sys.platform == "win32":
1561
+ # Both the npm install and the test launcher are `.cmd` wrappers:
1562
+ # Popen.pid is cmd.exe. Go straight to the tree force-kill while
1563
+ # the tree is still intact — a polite taskkill can reap cmd.exe
1564
+ # alone (orphaning the server with the port bound), and once the
1565
+ # wrapper is gone `/T` can never enumerate the child again.
1566
+ try:
1567
+ host.force_kill(process.pid)
1568
+ except Exception: # nosec B110 - already-gone races are fine
1569
+ pass
1570
+ else:
1571
+ try:
1572
+ host.terminate(process.pid) # SIGTERM exits opencode cleanly
1573
+ except OSError:
1574
+ pass
1575
+ try:
1576
+ process.wait(timeout=self.kill_wait_s)
1577
+ except subprocess.TimeoutExpired:
1578
+ try:
1579
+ host.force_kill(process.pid)
1580
+ except Exception: # nosec B110 - already-gone races are fine
1581
+ pass
1582
+ try:
1583
+ process.wait(timeout=self.kill_wait_s)
1584
+ except subprocess.TimeoutExpired:
1585
+ pass
1586
+ self._reap_descendants(host, tree)
1587
+
1588
+ def _reap_descendants(self, host: ProcessHost, tree: dict[int, float | None]) -> None:
1589
+ """Reap harvested straggler descendants the root signal missed — a detached
1590
+ tool subprocess that escaped the server's process group. Terminate → bounded
1591
+ wait ≤ ``kill_wait_s`` → force-kill, identity-guarded: a None identity is
1592
+ unconfirmable (a possible pid reuse), so it is never signalled or polled at
1593
+ all — even a SIGTERM to a recycled pid kills an innocent process (the tmux
1594
+ adapter's straggler doctrine, mirrored). Same already-gone swallow as the
1595
+ root ladder; best-effort, never a teardown gate."""
1596
+
1597
+ def _survivors() -> list[int]:
1598
+ return [
1599
+ pid
1600
+ for pid, identity in tree.items()
1601
+ if identity is not None and host.alive_and_ours(pid, identity)
1602
+ ]
1603
+
1604
+ survivors = _survivors()
1605
+ if not survivors:
1606
+ return
1607
+ for pid in survivors:
1608
+ try:
1609
+ host.terminate(pid)
1610
+ except OSError:
1611
+ pass
1612
+ deadline = time.monotonic() + self.kill_wait_s
1613
+ while True:
1614
+ survivors = _survivors()
1615
+ if not survivors or time.monotonic() >= deadline:
1616
+ break
1617
+ time.sleep(REAP_POLL_S)
1618
+ for pid in survivors:
1619
+ try:
1620
+ host.force_kill(pid)
1621
+ except Exception: # nosec B110 - already-gone races are fine
1622
+ pass
1623
+
1624
+ def _atexit_sweep(self) -> None:
1625
+ for task_id in list(self._sessions):
1626
+ sess = self._sessions.pop(task_id, None)
1627
+ if sess is not None:
1628
+ self._teardown(sess)
1629
+
1630
+
1631
+ class OpencodeDevAdapter(_DevSynthesisMixin, OpencodeHttpAdapter):
1632
+ """Dev/review adapter for the generic ``froid-build-auto`` skill over HTTP.
1633
+
1634
+ That skill writes NO ``result.json`` — its outcome lives in the terminal
1635
+ spec it leaves on disk, which :class:`_DevSynthesisMixin` locates and
1636
+ synthesizes into the legacy result dict via :mod:`devcontract` (the same
1637
+ machinery as GenericDevAdapter, mixin-shared rather than duplicated).
1638
+ Selected by ``policy.dev.skill == "froid-dev-auto"`` (see
1639
+ ``cli._make_adapters``).
1640
+ """
1641
+
1642
+ def __init__(self, *args, paths: ProjectPaths, **kwargs):
1643
+ super().__init__(*args, **kwargs)
1644
+ self.paths = paths
1645
+ self._configure_dev_knobs()
1646
+ # task_id -> server process, kept past kill(): kill() pops the
1647
+ # _ServerSession registry, but _post_kill_reconcile still needs to
1648
+ # settle liveness after the teardown.
1649
+ self._server_procs: dict[str, subprocess.Popen] = {}
1650
+
1651
+ def start_session(self, spec: SessionSpec) -> SessionHandle:
1652
+ handle = super().start_session(spec)
1653
+ self._server_procs[spec.task_id] = self._sessions[spec.task_id].process
1654
+ return handle
1655
+
1656
+ def _probe_alive(self, handle: SessionHandle) -> bool | None:
1657
+ proc = self._server_procs.get(handle.task_id)
1658
+ if proc is None:
1659
+ return False # never spawned: nothing we own is alive
1660
+ # Never None: the live Popen handle pins the pid, so poll() is always
1661
+ # answerable — unlike a tmux transport there is no probe that can hang.
1662
+ return proc.poll() is None
1663
+
1664
+
1665
+ def _sum_usage(messages: Any) -> TokenUsage:
1666
+ """Sum assistant-message token counts. Reasoning tokens are
1667
+ billed as output; OpenCode's cache read/write map onto the claude-style
1668
+ cache_read/cache_creation fields. Child-session (subagent) tokens are not
1669
+ visible here — the messages endpoint is scoped per session."""
1670
+ usage = TokenUsage()
1671
+ if not isinstance(messages, list):
1672
+ return usage
1673
+ for msg in messages:
1674
+ info = (msg or {}).get("info") or {}
1675
+ if info.get("role") != "assistant":
1676
+ continue
1677
+ tokens = info.get("tokens") or {}
1678
+ cache = tokens.get("cache") or {}
1679
+ usage.add(
1680
+ TokenUsage(
1681
+ input_tokens=int(tokens.get("input") or 0),
1682
+ output_tokens=int(tokens.get("output") or 0) + int(tokens.get("reasoning") or 0),
1683
+ cache_read_tokens=int(cache.get("read") or 0),
1684
+ cache_creation_tokens=int(cache.get("write") or 0),
1685
+ )
1686
+ )
1687
+ return usage