okstra 0.164.0 → 0.165.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (84) hide show
  1. package/README.md +1 -1
  2. package/docs/architecture.md +12 -8
  3. package/docs/cli.md +7 -3
  4. package/docs/for-ai/README.md +2 -2
  5. package/docs/for-ai/skills/okstra-inspect.md +2 -2
  6. package/docs/for-ai/skills/okstra-user-response.md +2 -2
  7. package/docs/project-structure-overview.md +15 -9
  8. package/package.json +1 -1
  9. package/runtime/BUILD.json +2 -2
  10. package/runtime/agents/workers/antigravity-worker.md +9 -7
  11. package/runtime/agents/workers/codex-worker.md +9 -7
  12. package/runtime/agents/workers/grok-worker.md +6 -4
  13. package/runtime/agents/workers/kimi-worker.md +6 -4
  14. package/runtime/bin/okstra-antigravity-exec.sh +1 -340
  15. package/runtime/bin/okstra-claude-exec.sh +1 -178
  16. package/runtime/bin/okstra-codex-exec.sh +1 -467
  17. package/runtime/bin/okstra-provider-exec.py +165 -190
  18. package/runtime/bin/okstra-trace-cleanup.sh +14 -7
  19. package/runtime/bin/okstra-wrapper-status.py +26 -19
  20. package/runtime/prompts/lead/adapters/cmux.md +1 -1
  21. package/runtime/prompts/lead/convergence.md +36 -8
  22. package/runtime/prompts/lead/okstra-lead-contract.md +23 -1
  23. package/runtime/prompts/lead/plan-body-verification.md +9 -1
  24. package/runtime/prompts/lead/report-writer.md +1 -0
  25. package/runtime/prompts/lead/team-contract.md +3 -3
  26. package/runtime/prompts/profiles/_common-contract.md +9 -1
  27. package/runtime/prompts/profiles/_coverage-critic.md +1 -1
  28. package/runtime/prompts/profiles/_implementation-diff-review.md +3 -1
  29. package/runtime/prompts/profiles/_implementation-self-check.md +1 -1
  30. package/runtime/prompts/profiles/_implementation-verifier.md +3 -1
  31. package/runtime/prompts/profiles/implementation-planning.md +5 -3
  32. package/runtime/python/okstra_ctl/adapters/hosts/claude-code/relay.md +1 -1
  33. package/runtime/python/okstra_ctl/adapters/hosts/external/relay.md +1 -1
  34. package/runtime/python/okstra_ctl/adapters/providers/antigravity/adapter.py +148 -0
  35. package/runtime/python/okstra_ctl/adapters/providers/claude/adapter.py +55 -0
  36. package/runtime/python/okstra_ctl/adapters/providers/codex/adapter.py +41 -0
  37. package/runtime/python/okstra_ctl/adapters/providers/grok/adapter.py +44 -0
  38. package/runtime/python/okstra_ctl/adapters/providers/kimi/adapter.py +42 -0
  39. package/runtime/python/okstra_ctl/dispatch_core.py +5 -1
  40. package/runtime/python/okstra_ctl/dispatch_state.py +10 -0
  41. package/runtime/python/okstra_ctl/domain/provider.py +5 -1
  42. package/runtime/python/okstra_ctl/domain/worker_exec.py +102 -0
  43. package/runtime/python/okstra_ctl/domain/worker_role.py +34 -0
  44. package/runtime/python/okstra_ctl/domain/worker_stream.py +261 -0
  45. package/runtime/python/okstra_ctl/incremental_scope.py +16 -4
  46. package/runtime/python/okstra_ctl/report_html/common.py +71 -25
  47. package/runtime/python/okstra_ctl/report_html/models.py +5 -0
  48. package/runtime/python/okstra_ctl/report_html/render.py +1 -1
  49. package/runtime/python/okstra_ctl/report_html/run_usage.py +19 -0
  50. package/runtime/python/okstra_ctl/report_html/view_models/implementation_planning.py +14 -0
  51. package/runtime/python/okstra_ctl/report_views.py +44 -16
  52. package/runtime/python/okstra_ctl/stage_citations.py +52 -15
  53. package/runtime/python/okstra_ctl/user_response.py +45 -29
  54. package/runtime/python/okstra_ctl/wizard.py +13 -9
  55. package/runtime/python/okstra_ctl/worker_prompt_policy.py +10 -3
  56. package/runtime/python/okstra_ctl/worker_request.py +140 -0
  57. package/runtime/python/okstra_ctl/worker_runner.py +622 -0
  58. package/runtime/python/okstra_token_usage/collect.py +8 -1
  59. package/runtime/python/okstra_token_usage/report.py +42 -0
  60. package/runtime/python/okstra_token_usage/task_totals.py +88 -0
  61. package/runtime/schemas/final-report-v1.0.schema.json +70 -0
  62. package/runtime/schemas/final-report-v2.0.schema.json +90 -0
  63. package/runtime/skills/okstra-inspect/SKILL.md +1 -2
  64. package/runtime/skills/okstra-inspect/facets/logs.md +5 -5
  65. package/runtime/skills/okstra-inspect/facets/run-audit.md +3 -3
  66. package/runtime/skills/okstra-run/SKILL.md +1 -1
  67. package/runtime/skills/okstra-user-response/SKILL.md +15 -5
  68. package/runtime/templates/report-writer-prompt-preamble.md +1 -0
  69. package/runtime/templates/reports/html/assets/base.css +8 -4
  70. package/runtime/templates/reports/html/base.template.html +12 -6
  71. package/runtime/templates/reports/html/i18n/en.json +29 -6
  72. package/runtime/templates/reports/html/i18n/ko.json +29 -6
  73. package/runtime/templates/reports/html/macros/forms.html +9 -3
  74. package/runtime/templates/reports/html/tasks/implementation-planning.template.html +14 -19
  75. package/runtime/templates/reports/report.js +59 -26
  76. package/runtime/templates/reports/user-response.template.md +12 -8
  77. package/runtime/validators/validate-run.py +88 -7
  78. package/runtime/validators/validate_session_conformance.py +62 -1
  79. package/src/cli-registry.mjs +0 -7
  80. package/runtime/bin/okstra-wrapper-agy-stream.py +0 -61
  81. package/runtime/python/okstra_ctl/error_issue.py +0 -640
  82. package/runtime/python/okstra_ctl/issue_signals.py +0 -186
  83. package/runtime/skills/okstra-inspect/facets/error-issue.md +0 -77
  84. package/src/commands/inspect/error-issue.mjs +0 -27
@@ -0,0 +1,622 @@
1
+ """Run one worker CLI and record what happened.
2
+
3
+ Shared by every provider: the entrypoint scripts parse arguments and hand over
4
+ here. What differs per provider is the command (an ``ExecutionStrategy``); what
5
+ differs per surface is the presentation.
6
+
7
+ Idle is measured from stream arrival, never from the log file's mtime. The
8
+ screen deliberately drops thinking events, so an mtime-based watchdog would
9
+ SIGTERM a healthy worker in the middle of a long reasoning stretch.
10
+ """
11
+ from __future__ import annotations
12
+
13
+ import json
14
+ import os
15
+ import selectors
16
+ import signal
17
+ import subprocess
18
+ import sys
19
+ import time
20
+ from functools import partial
21
+ from pathlib import Path
22
+ from typing import Any, Callable, Mapping
23
+
24
+ from .domain.worker_exec import (
25
+ STREAM_JSON,
26
+ ExecCommand,
27
+ ExecutionStrategy,
28
+ WorkerExecRequest,
29
+ )
30
+ from .domain.worker_stream import Normalise, final_text, format_live, format_log
31
+
32
+ LIVE = "live"
33
+ QUIET = "quiet"
34
+
35
+ _SELECT_TIMEOUT_SECONDS = 0.25
36
+ _TERM_GRACE_SECONDS = 5
37
+ # How long the streams stay open after the worker process itself is gone. A pipe
38
+ # holds at most its capacity of already-written output when its writer exits, so
39
+ # this only has to cover a drain, never a producer.
40
+ _DRAIN_AFTER_EXIT_SECONDS = 2
41
+ _TIMEOUT_EXIT_CODE = 124
42
+ _READ_SIZE = 8192
43
+ _WRITE_SIZE = 8192
44
+ _NO_STATUS_EXTRA: Mapping[str, Any] = {}
45
+
46
+ # How many progress lines reach the log copy before it starts eliding. Progress
47
+ # is where a text CLI's bulk is — a dispatch that only reads one file already
48
+ # puts kilobytes of tool echo on stderr, and observed sidecars reach 8MB and
49
+ # dominate a project's `.okstra/` bytes. The cap is run-wide rather than
50
+ # per-block because block boundaries are a provider's own vocabulary and this
51
+ # runner has none; the cost is that a very long run keeps its opening rather
52
+ # than a sample throughout, which the elision notices make visible.
53
+ _LOG_PROGRESS_LINE_CAP = 5000
54
+ _ELISION_NOTICE_EVERY = 500
55
+
56
+ # The signals that end this process without raising anything Python can catch on
57
+ # the way out. SIGINT is absent on purpose: it arrives as KeyboardInterrupt and
58
+ # the exception path already closes the sidecar.
59
+ _ABNORMAL_SIGNALS = (signal.SIGTERM, signal.SIGHUP)
60
+
61
+
62
+ def run_worker(
63
+ strategy: ExecutionStrategy,
64
+ request: WorkerExecRequest,
65
+ *,
66
+ presentation: str,
67
+ log_path: Path,
68
+ status_path: Path,
69
+ status_extra: Mapping[str, Any] = _NO_STATUS_EXTRA,
70
+ ) -> int:
71
+ command = strategy.build_command(request)
72
+ started_monotonic = time.monotonic()
73
+ status = _started_status(status_extra, log_path)
74
+ _write_status(status_path, status)
75
+ guard = _AbnormalExit(status_path, status, started_monotonic)
76
+
77
+ try:
78
+ with guard:
79
+ exit_code, timed_out, idle_seconds = _launch(
80
+ command,
81
+ log_path,
82
+ presentation=presentation,
83
+ idle_timeout_seconds=request.idle_timeout_seconds,
84
+ on_spawn=guard.watch,
85
+ )
86
+ except BaseException as exc:
87
+ # Whatever ended this run — an OS error, a Ctrl-C, a bug in this file —
88
+ # the sidecar has to stop saying `started`. Nothing downstream rewrites
89
+ # it, so `worker_liveness` would read the worker as still working
90
+ # forever. No exit code is invented — its absence is already how every
91
+ # reader spells a run that failed. The exits that never raise at all are
92
+ # the guard's to close.
93
+ guard.close(f"{type(exc).__name__}: {exc}")
94
+ raise
95
+
96
+ status = _closed(status, started_monotonic)
97
+ status["exit_code"] = exit_code
98
+ if timed_out:
99
+ status.update(
100
+ timeout=True,
101
+ idle_at_ts=status["ended_ts"],
102
+ idle_seconds=idle_seconds,
103
+ terminated_by="idle-watchdog",
104
+ )
105
+ _write_status(status_path, status)
106
+ return exit_code
107
+
108
+
109
+ def _launch(
110
+ command: ExecCommand,
111
+ log_path: Path,
112
+ *,
113
+ presentation: str,
114
+ idle_timeout_seconds: int,
115
+ on_spawn: Callable[[subprocess.Popen[bytes]], None],
116
+ ) -> tuple[int, bool, int]:
117
+ log_path.parent.mkdir(parents=True, exist_ok=True)
118
+ with log_path.open("w", encoding="utf-8") as log_file:
119
+ # The strategy decided where this provider runs — some CLIs work in the
120
+ # stage tree, others in the project root and reach the tree by flag.
121
+ process = subprocess.Popen(
122
+ list(command.argv),
123
+ cwd=str(command.cwd),
124
+ stdin=subprocess.PIPE if command.stdin_text is not None else None,
125
+ stdout=subprocess.PIPE,
126
+ stderr=_stderr_target(command.stream_format),
127
+ start_new_session=True,
128
+ env=_child_env(),
129
+ )
130
+ on_spawn(process)
131
+ return _pump(
132
+ process,
133
+ log_file,
134
+ stream_format=command.stream_format,
135
+ normalise=command.normalise,
136
+ presentation=presentation,
137
+ idle_timeout_seconds=idle_timeout_seconds,
138
+ stdin_text=command.stdin_text,
139
+ )
140
+
141
+
142
+ class _AbnormalExit:
143
+ """Close the status sidecar for the exits that raise nothing at all.
144
+
145
+ ``except BaseException`` around the run covers an exception, including the
146
+ ``KeyboardInterrupt`` a SIGINT raises. It does not cover SIGTERM or SIGHUP:
147
+ their default disposition ends the process outright, and nothing in this
148
+ file runs (measured — a bash ``trap … EXIT`` does fire on SIGTERM, which is
149
+ why the shell wrappers needed no equivalent of this class). Those two are
150
+ the common abnormal exits: a pane kill, ``okstra-trace-cleanup.sh``, session
151
+ teardown. Without this the sidecar stays at ``started`` and
152
+ ``worker_liveness`` reads a dead worker as a working one.
153
+
154
+ SIGKILL and a host crash remain uncovered because nothing can cover them. A
155
+ sidecar still reading ``started`` is the residue they leave, and no reader
156
+ should take it as proof the worker is alive.
157
+ """
158
+
159
+ def __init__(
160
+ self, status_path: Path, status: Mapping[str, Any], started_monotonic: float
161
+ ) -> None:
162
+ self._status_path = status_path
163
+ self._status = status
164
+ self._started_monotonic = started_monotonic
165
+ self._process: subprocess.Popen[bytes] | None = None
166
+ self._restore: dict[int, Any] = {}
167
+
168
+ def watch(self, process: subprocess.Popen[bytes]) -> None:
169
+ """Adopt the child, so a signal tears down its group rather than orphan it."""
170
+ self._process = process
171
+
172
+ def close(self, failure: str) -> None:
173
+ _write_status(
174
+ self._status_path,
175
+ {**_closed(self._status, self._started_monotonic), "failure": failure},
176
+ )
177
+
178
+ def __enter__(self) -> _AbnormalExit:
179
+ for number in _ABNORMAL_SIGNALS:
180
+ try:
181
+ self._restore[number] = signal.signal(number, self._handle)
182
+ except ValueError:
183
+ # Handlers install from the main thread only. A caller running
184
+ # the runner off-thread keeps the exception path and nothing
185
+ # more, which is what it had before this class existed.
186
+ break
187
+ return self
188
+
189
+ def __exit__(self, *_exception: Any) -> bool:
190
+ for number, previous in self._restore.items():
191
+ signal.signal(number, previous)
192
+ self._restore.clear()
193
+ return False
194
+
195
+ def _handle(self, number: int, _frame: Any) -> None:
196
+ if self._process is not None:
197
+ _terminate(self._process)
198
+ self.close(f"signal {signal.Signals(number).name}")
199
+ # Then die the way the sender asked, so the exit code still names the
200
+ # signal instead of reporting a clean stop this run did not make.
201
+ signal.signal(number, signal.SIG_DFL)
202
+ os.kill(os.getpid(), number)
203
+
204
+
205
+ def _closed(status: Mapping[str, Any], started_monotonic: float) -> dict[str, Any]:
206
+ """The sidecar's terminal shape, whatever it was that ended the run.
207
+
208
+ ``stage`` is the field every reader keys on — ``wrapper_status.is_terminal``,
209
+ the pane reclaim and the dispatch record all ask whether it reads ``exited``.
210
+ """
211
+ return {
212
+ **status,
213
+ "stage": "exited",
214
+ "ended_ts": int(time.time()),
215
+ "duration_ms": int((time.monotonic() - started_monotonic) * 1000),
216
+ }
217
+
218
+
219
+ def _started_status(status_extra: Mapping[str, Any], log_path: Path) -> dict[str, Any]:
220
+ """The sidecar as it looks before the CLI has produced anything.
221
+
222
+ ``status_extra`` carries what only the entrypoint knows — which wrapper the
223
+ caller invoked and which role it dispatched — and is placed where the
224
+ wrapper-written sidecars already carry those keys.
225
+ """
226
+ return {
227
+ "schemaVersion": 1,
228
+ **dict(status_extra),
229
+ "pid": os.getpid(),
230
+ "started_ts": int(time.time()),
231
+ "log_path": str(log_path),
232
+ "stage": "started",
233
+ }
234
+
235
+
236
+ def _stderr_target(stream_format: str) -> int:
237
+ """Whether the child's two streams stay apart.
238
+
239
+ A JSON-stream CLI puts events on stdout and only its own error text on
240
+ stderr, so folding the two loses nothing and leaves one reader. A text CLI
241
+ splits meaning across them — the result on stdout, progress on stderr — and
242
+ merging destroys the only way to tell the answer from the noise.
243
+ """
244
+ return subprocess.STDOUT if stream_format == STREAM_JSON else subprocess.PIPE
245
+
246
+
247
+ def _pump(
248
+ process: subprocess.Popen[bytes],
249
+ log_file,
250
+ *,
251
+ stream_format: str,
252
+ normalise: Normalise,
253
+ presentation: str,
254
+ idle_timeout_seconds: int,
255
+ stdin_text: str | None = None,
256
+ ) -> tuple[int, bool, int]:
257
+ selector = selectors.DefaultSelector()
258
+ readers, finalize_log = _register_output(
259
+ selector,
260
+ process,
261
+ log_file,
262
+ stream_format=stream_format,
263
+ normalise=normalise,
264
+ presentation=presentation,
265
+ )
266
+ outgoing = _register_prompt(selector, process, stdin_text)
267
+ closing_text: str | None = None
268
+ last_output = time.monotonic()
269
+ timed_out = False
270
+ idle_seconds = 0
271
+ drain_deadline: float | None = None
272
+
273
+ while selector.get_map():
274
+ for key, events in selector.select(timeout=_SELECT_TIMEOUT_SECONDS):
275
+ if events & selectors.EVENT_WRITE:
276
+ outgoing = _push_prompt(selector, key, outgoing)
277
+ continue
278
+ chunk = os.read(key.fd, _READ_SIZE)
279
+ if not chunk:
280
+ selector.unregister(key.fileobj)
281
+ continue
282
+ # Either stream is proof of life: a text CLI reports progress for
283
+ # minutes before a result exists to send.
284
+ last_output = time.monotonic()
285
+ closing_text = readers[key.fd].feed(chunk) or closing_text
286
+ idle_seconds = int(time.monotonic() - last_output)
287
+ if (
288
+ idle_timeout_seconds
289
+ and idle_seconds >= idle_timeout_seconds
290
+ and process.poll() is None
291
+ ):
292
+ timed_out = True
293
+ _terminate(process)
294
+ drain_deadline = _drain_deadline(process, drain_deadline)
295
+ if drain_deadline is not None and time.monotonic() >= drain_deadline:
296
+ _stop_reading(selector)
297
+
298
+ for reader in readers.values():
299
+ closing_text = reader.flush() or closing_text
300
+ finalize_log()
301
+
302
+ exit_code = process.wait()
303
+ if closing_text is None:
304
+ # A JSON stream is supposed to end in a result event. Ending without one
305
+ # means the CLI stopped without saying what it concluded, and the caller
306
+ # would otherwise read an empty stdout under `quiet` as a run that
307
+ # simply had nothing to report. A text CLI has no result event to miss.
308
+ if stream_format == STREAM_JSON:
309
+ print(
310
+ f"okstra worker: no result event in the CLI's output — "
311
+ f"see {log_file.name}",
312
+ file=sys.stderr,
313
+ flush=True,
314
+ )
315
+ elif presentation == QUIET:
316
+ print(closing_text, flush=True)
317
+ return (_TIMEOUT_EXIT_CODE if timed_out else exit_code), timed_out, idle_seconds
318
+
319
+
320
+ def _drain_deadline(
321
+ process: subprocess.Popen[bytes], deadline: float | None
322
+ ) -> float | None:
323
+ """When to stop reading, once the worker process itself is gone.
324
+
325
+ Without this the loop ends only at EOF on every stream, and a worker that
326
+ leaves a background process behind — a dev server, a file watcher, the
327
+ watcher the container-build skill starts on purpose — leaves that process
328
+ holding the write end of the pipe open for as long as it lives. The shell
329
+ wrappers returned as soon as `wait <pid>` did, so blocking on a grandchild is
330
+ a regression rather than a policy, and the exit code the caller gets stays
331
+ the worker's own.
332
+
333
+ The deadline is set once and never pushed back: output arriving after the
334
+ worker exited is the grandchild's, and letting it extend the wait would
335
+ restore exactly the hang this bounds.
336
+ """
337
+ if deadline is not None or process.poll() is None:
338
+ return deadline
339
+ return time.monotonic() + _DRAIN_AFTER_EXIT_SECONDS
340
+
341
+
342
+ def _stop_reading(selector: selectors.BaseSelector) -> None:
343
+ """Drop every remaining stream without signalling whoever still holds it.
344
+
345
+ A process the worker deliberately left running is not this runner's to kill.
346
+ The idle watchdog is what ends a run that went wrong; this only ends the
347
+ reading of a run that already finished.
348
+ """
349
+ for key in list(selector.get_map().values()):
350
+ selector.unregister(key.fileobj)
351
+
352
+
353
+ def _register_output(
354
+ selector: selectors.BaseSelector,
355
+ process: subprocess.Popen[bytes],
356
+ log_file,
357
+ *,
358
+ stream_format: str,
359
+ normalise: Normalise,
360
+ presentation: str,
361
+ ) -> tuple[dict[int, _LineReader], Callable[[], None]]:
362
+ """Register every stream this child speaks, each with its own destination.
363
+
364
+ Returns the readers plus the finalizer a capped sink needs to report what it
365
+ dropped once the streams are done. A JSON stream caps nothing.
366
+ """
367
+ assert process.stdout is not None
368
+ if stream_format == STREAM_JSON:
369
+ sinks = [
370
+ (process.stdout, partial(_emit_event, log_file, presentation, normalise))
371
+ ]
372
+ finalize_log: Callable[[], None] = _nothing_to_finalize
373
+ else:
374
+ assert process.stderr is not None
375
+ progress = _ProgressSink(log_file, presentation)
376
+ sinks = [
377
+ (process.stdout, partial(_emit_result, log_file)),
378
+ (process.stderr, progress),
379
+ ]
380
+ finalize_log = progress.report_elided
381
+ readers: dict[int, _LineReader] = {}
382
+ for stream, emit in sinks:
383
+ selector.register(stream, selectors.EVENT_READ)
384
+ readers[stream.fileno()] = _LineReader(emit)
385
+ return readers, finalize_log
386
+
387
+
388
+ def _nothing_to_finalize() -> None:
389
+ """A JSON stream elides nothing, so it has nothing to report at the end."""
390
+
391
+
392
+ class _LineReader:
393
+ """Whole lines out of one byte stream, handed to that stream's sink.
394
+
395
+ One per stream rather than one shared buffer: the two streams of a text CLI
396
+ arrive interleaved, and a shared buffer would splice half a progress line
397
+ onto the front of the result.
398
+ """
399
+
400
+ def __init__(self, emit: Callable[[str], str | None]) -> None:
401
+ self._emit = emit
402
+ self._pending = b""
403
+
404
+ def feed(self, chunk: bytes) -> str | None:
405
+ self._pending += chunk
406
+ *complete, self._pending = self._pending.split(b"\n")
407
+ return self._drain(complete)
408
+
409
+ def flush(self) -> str | None:
410
+ """Whatever the stream ended on without a closing newline."""
411
+ trailing, self._pending = self._pending, b""
412
+ return self._drain([trailing] if trailing else [])
413
+
414
+ def _drain(self, raw_lines: list[bytes]) -> str | None:
415
+ closing: str | None = None
416
+ for raw in raw_lines:
417
+ closing = self._emit(raw.decode("utf-8", "replace")) or closing
418
+ return closing
419
+
420
+
421
+ def _register_prompt(
422
+ selector: selectors.BaseSelector,
423
+ process: subprocess.Popen[bytes],
424
+ stdin_text: str | None,
425
+ ) -> bytes:
426
+ """Queue the prompt for a CLI that reads it from stdin.
427
+
428
+ Fed inside the pump rather than written whole before it. A prompt larger
429
+ than the pipe buffer — worker prompts are tens of kilobytes — would
430
+ otherwise block this process while the CLI blocks writing the stdout that
431
+ nobody is draining yet, and the idle watchdog could not even reach that
432
+ stall because it only runs once the pump is looping.
433
+ """
434
+ if stdin_text is None or process.stdin is None:
435
+ return b""
436
+ os.set_blocking(process.stdin.fileno(), False)
437
+ selector.register(process.stdin, selectors.EVENT_WRITE)
438
+ return stdin_text.encode("utf-8")
439
+
440
+
441
+ def _push_prompt(
442
+ selector: selectors.BaseSelector, key: selectors.SelectorKey, payload: bytes
443
+ ) -> bytes:
444
+ """Hand over as much of the prompt as the pipe will take right now."""
445
+ try:
446
+ written = os.write(key.fd, payload[:_WRITE_SIZE]) if payload else 0
447
+ except BlockingIOError:
448
+ return payload
449
+ except OSError:
450
+ # The CLI exited before reading its prompt. Its exit code is the story;
451
+ # this half-delivered write is not.
452
+ _close_prompt(selector, key)
453
+ return b""
454
+ payload = payload[written:]
455
+ if not payload:
456
+ # EOF is what tells the CLI its prompt is complete.
457
+ _close_prompt(selector, key)
458
+ return payload
459
+
460
+
461
+ def _close_prompt(selector: selectors.BaseSelector, key: selectors.SelectorKey) -> None:
462
+ selector.unregister(key.fileobj)
463
+ key.fileobj.close()
464
+
465
+
466
+ def _emit_result(log_file, line: str) -> None:
467
+ """A text CLI's stdout is its answer, so the caller gets it in either mode.
468
+
469
+ ``quiet`` withholds progress, not the result — and unlike a JSON stream
470
+ there is no result event to hold back and print at the end, so the answer
471
+ passes through as it arrives.
472
+ """
473
+ _write_log(log_file, [line])
474
+ print(line, flush=True)
475
+
476
+
477
+ class _ProgressSink:
478
+ """A text CLI's stderr: shown live, archived up to a cap.
479
+
480
+ Progress goes to this process's stderr rather than its stdout so the
481
+ caller's stdout stays the result alone. A pane shows both, and a pane is
482
+ exactly where ``live`` lands.
483
+
484
+ Only the log copy is capped, never the screen and never the result stream.
485
+ That asymmetry is the point: a truncated tool echo costs detail, while a
486
+ truncated answer costs the whole post-mortem — and the caller's streams are
487
+ read by a person or a subagent that was promised the CLI's output verbatim.
488
+ """
489
+
490
+ def __init__(self, log_file, presentation: str) -> None:
491
+ self._log_file = log_file
492
+ self._presentation = presentation
493
+ self._archived = 0
494
+ self._elided = 0
495
+
496
+ def __call__(self, line: str) -> None:
497
+ if self._presentation == LIVE:
498
+ print(line, file=sys.stderr, flush=True)
499
+ if self._archived < _LOG_PROGRESS_LINE_CAP:
500
+ self._archived += 1
501
+ _write_log(self._log_file, [line])
502
+ return
503
+ self._elided += 1
504
+ # Periodic rather than only at the end: a reader tailing the log has to
505
+ # see that the run is still producing progress, not a file that stopped.
506
+ if self._elided % _ELISION_NOTICE_EVERY == 0:
507
+ self._note_elision()
508
+
509
+ def report_elided(self) -> None:
510
+ """Close the archive with the exact total, if the last notice missed some."""
511
+ if self._elided % _ELISION_NOTICE_EVERY:
512
+ self._note_elision()
513
+
514
+ def _note_elision(self) -> None:
515
+ _write_log(
516
+ self._log_file,
517
+ [f" [okstra log-cap] {self._elided} progress line(s) elided"],
518
+ )
519
+
520
+
521
+ def _emit_event(
522
+ log_file, presentation: str, normalise: Normalise, line: str
523
+ ) -> str | None:
524
+ """Write one line of a JSON stream out; return closing text if this is it.
525
+
526
+ The provider's own wire shape gets no further than ``normalise`` — what
527
+ reaches the projections is the normalised vocabulary, which is the only
528
+ thing this runner and the formatter are allowed to know.
529
+ """
530
+ stripped = line.strip()
531
+ if not stripped:
532
+ return None
533
+ try:
534
+ event = json.loads(stripped)
535
+ except ValueError:
536
+ # Not an event. The CLI's stderr is folded into this stream, so this is
537
+ # where its own error text arrives — the caller has to see it, and the
538
+ # archive alone does not show it to anyone. stderr rather than stdout so
539
+ # it reaches the caller without posing as progress, in either mode.
540
+ _write_log(log_file, [stripped])
541
+ print(stripped, file=sys.stderr, flush=True)
542
+ return None
543
+ if not isinstance(event, dict):
544
+ return None
545
+
546
+ closing: str | None = None
547
+ for entry in normalise(event):
548
+ _write_log(log_file, format_log(entry))
549
+ if presentation == LIVE:
550
+ for row in format_live(entry):
551
+ print(row, flush=True)
552
+ text = final_text(entry)
553
+ if text is not None:
554
+ closing = text
555
+ if closing is not None:
556
+ # Whatever the presentation, the archive ends with the worker's own
557
+ # conclusion — otherwise it stops at the last tool call and the reader
558
+ # never learns what the run decided.
559
+ _write_log(log_file, closing.splitlines())
560
+ return closing
561
+
562
+
563
+ def _write_log(log_file, lines) -> None:
564
+ for line in lines:
565
+ log_file.write(line + "\n")
566
+ log_file.flush()
567
+
568
+
569
+ def _terminate(process: subprocess.Popen[bytes]) -> None:
570
+ """SIGTERM the worker's process group, then SIGKILL whatever survives.
571
+
572
+ The group, not the process: a CLI spawns shells and build tools, and
573
+ signalling only the direct child leaves those running with the pipe open.
574
+ ``start_new_session=True`` at spawn is what makes the group addressable by
575
+ the child's own pid.
576
+ """
577
+ try:
578
+ os.killpg(process.pid, signal.SIGTERM)
579
+ except ProcessLookupError:
580
+ return
581
+ try:
582
+ process.wait(timeout=_TERM_GRACE_SECONDS)
583
+ except subprocess.TimeoutExpired:
584
+ try:
585
+ os.killpg(process.pid, signal.SIGKILL)
586
+ except ProcessLookupError:
587
+ pass
588
+ process.wait()
589
+
590
+
591
+ def _child_env() -> dict[str, str]:
592
+ """The worker's environment with git's fsmonitor disabled.
593
+
594
+ The main worktree's fsmonitor daemon leaks its IPC socket into task
595
+ worktrees, so git status/commit there intermittently fail with
596
+ `fsmonitor_ipc__send_query`. GIT_CONFIG_* scopes the override to this
597
+ process tree without touching the user's repo config, appending after any
598
+ GIT_CONFIG_* the caller already set.
599
+ """
600
+ env = dict(os.environ)
601
+ index = int(env.get("GIT_CONFIG_COUNT", "0") or "0")
602
+ env[f"GIT_CONFIG_KEY_{index}"] = "core.fsmonitor"
603
+ env[f"GIT_CONFIG_VALUE_{index}"] = "false"
604
+ env["GIT_CONFIG_COUNT"] = str(index + 1)
605
+ return env
606
+
607
+
608
+ def _write_status(path: Path, status: Mapping[str, Any]) -> None:
609
+ """Best-effort status write; a sidecar failure must not break the run.
610
+
611
+ Replaced whole rather than rewritten in place: readers poll this file while
612
+ the worker is still running and must never see a half-written document.
613
+ """
614
+ temporary = Path(f"{path}.tmp")
615
+ try:
616
+ path.parent.mkdir(parents=True, exist_ok=True)
617
+ temporary.write_text(
618
+ json.dumps(status, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
619
+ )
620
+ os.replace(temporary, path)
621
+ except OSError:
622
+ return
@@ -704,14 +704,17 @@ def _populate_usage_summary(
704
704
  workers = state.get("workers", [])
705
705
  lead = state.get("leadUsage") or {}
706
706
  lead_total = lead.get("totalTokens", 0) or 0
707
+ lead_cache_read = lead.get("cacheReadTokens", 0) or 0
707
708
  lead_billable = lead.get("billableEquivalentTokens", 0) or 0
708
709
  lead_cost = lead.get("estimatedCostUsd", 0) or 0
709
710
  worker_total = sum((w.get("usage") or {}).get("totalTokens", 0) or 0 for w in workers)
711
+ worker_cache_read = sum((w.get("usage") or {}).get("cacheReadTokens", 0) or 0 for w in workers)
710
712
  worker_billable = sum((w.get("usage") or {}).get("billableEquivalentTokens", 0) or 0 for w in workers)
711
713
  worker_cost = sum((w.get("usage") or {}).get("estimatedCostUsd", 0) or 0 for w in workers)
712
714
  cli_cost = sum((w.get("usage") or {}).get("cliEstimatedCostUsd", 0) or 0 for w in workers)
713
715
  if unattributed_usage is not None:
714
716
  worker_total += unattributed_usage.get("totalTokens", 0) or 0
717
+ worker_cache_read += unattributed_usage.get("cacheReadTokens", 0) or 0
715
718
  worker_billable += unattributed_usage.get("billableEquivalentTokens", 0) or 0
716
719
  worker_cost += unattributed_usage.get("estimatedCostUsd", 0) or 0
717
720
 
@@ -733,6 +736,9 @@ def _populate_usage_summary(
733
736
  "leadTotalTokens": lead_total,
734
737
  "workerTotalTokens": worker_total,
735
738
  "grandTotalTokens": lead_total + worker_total,
739
+ "leadCacheReadTokens": lead_cache_read,
740
+ "workerCacheReadTokens": worker_cache_read,
741
+ "grandCacheReadTokens": lead_cache_read + worker_cache_read,
736
742
  "leadBillableEquivalentTokens": lead_billable,
737
743
  "workerBillableEquivalentTokens": worker_billable,
738
744
  "grandBillableEquivalentTokens": lead_billable + worker_billable,
@@ -750,7 +756,8 @@ def _populate_usage_summary(
750
756
  "unattributedTeamSessions": unattributed_sessions or [],
751
757
  "unattributedWorkerUsage": unattributed_usage,
752
758
  "definitions": {
753
- "totalTokens": "Sum of input + output + cache_creation + cache_read tokens (raw processed volume; matches Anthropic API breakdown). Cache reads are 95%+ in long sessions.",
759
+ "totalTokens": "Sum of input + output + cache_creation tokens — the volume the session put through the model once. cache_read is excluded and reported separately as cacheReadTokens: a session re-reads its whole context from cache every turn, so folding it in here would count the same tokens once per turn.",
760
+ "cacheReadTokens": "Context re-read from cache. Billed at 0.1x base input, so it lands in billableEquivalentTokens and in the cost even though it is not part of totalTokens.",
754
761
  "billableEquivalentTokens": "Tokens normalized to base-input-price units (cache_creation_5m x1.25, cache_creation_1h x2.0, cache_read x0.1, output x5). 5m vs 1h is split from usage.cache_creation when the API breakdown is present; otherwise all cache_creation falls into 5m.",
755
762
  "estimatedCostUsd": "USD cost using public list pricing for the model recorded in the session. cliWorkers covers attributable registered-provider CLI calls.",
756
763
  },