alissa-tools-github-revloop 0.16.8__tar.gz → 0.16.9__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {alissa_tools_github_revloop-0.16.8/src/main/alissa_tools_github_revloop.egg-info → alissa_tools_github_revloop-0.16.9}/PKG-INFO +1 -1
  2. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/__main__.py +25 -1
  3. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/loop.py +324 -5
  4. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/state.py +232 -2
  5. alissa_tools_github_revloop-0.16.9/src/main/alissa/tools/github/revloop/version +1 -0
  6. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9/src/main/alissa_tools_github_revloop.egg-info}/PKG-INFO +1 -1
  7. alissa_tools_github_revloop-0.16.8/src/main/alissa/tools/github/revloop/version +0 -1
  8. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/LICENSE +0 -0
  9. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/MANIFEST.in +0 -0
  10. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/NOTICE +0 -0
  11. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/README.md +0 -0
  12. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/requirements.txt +0 -0
  13. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/setup.cfg +0 -0
  14. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/setup.py +0 -0
  15. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/__init__.py +0 -0
  16. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/alissa.py +0 -0
  17. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/config.py +0 -0
  18. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/ghclient.py +0 -0
  19. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/proc.py +0 -0
  20. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/prreview.py +0 -0
  21. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/version.py +0 -0
  22. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/webui/__init__.py +0 -0
  23. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/webui/__main__.py +0 -0
  24. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/webui/auth.py +0 -0
  25. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/webui/page.py +0 -0
  26. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/webui/server.py +0 -0
  27. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/webui/sources.py +0 -0
  28. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa/tools/github/revloop/webui/sysinfo.py +0 -0
  29. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa_tools_github_revloop.egg-info/SOURCES.txt +0 -0
  30. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa_tools_github_revloop.egg-info/dependency_links.txt +0 -0
  31. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa_tools_github_revloop.egg-info/entry_points.txt +0 -0
  32. {alissa_tools_github_revloop-0.16.8 → alissa_tools_github_revloop-0.16.9}/src/main/alissa_tools_github_revloop.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: alissa-tools-github-revloop
3
- Version: 0.16.8
3
+ Version: 0.16.9
4
4
  Summary: ALISSA-TOOLS-GITHUB-REVLOOP
5
5
  Home-page: https://alissa.app
6
6
  Author: Fahera
@@ -19,7 +19,7 @@ from .config import (
19
19
  resolve_config_path,
20
20
  )
21
21
  from .ghclient import IdentityMismatch
22
- from .loop import ReviewWatcher
22
+ from .loop import LedgerUnwritable, ReviewWatcher
23
23
  from .proc import CommandError
24
24
 
25
25
  log = logging.getLogger(__name__)
@@ -189,6 +189,16 @@ def main(argv: list[str] | None = None) -> int:
189
189
  datefmt="%H:%M:%S",
190
190
  )
191
191
 
192
+ # STARTUP and STEADY STATE are separated on purpose (issue #62). The
193
+ # handlers below label a FileNotFoundError / ValueError "config error" and
194
+ # exit 2, which is right for the startup phase -- a missing config file or
195
+ # an unparseable value cannot be fixed by trying again. It was fatally
196
+ # wrong for the poll loop, where the same classes mean a transient
197
+ # subprocess ENOENT or a bad response: `run_forever` now firewalls those
198
+ # per iteration and never lets them reach here at all. The one-shot modes
199
+ # (`--pr`, `--once`) still surface a failure to their caller, as a
200
+ # one-shot must -- including a refused pass, which leaves through the
201
+ # LedgerUnwritable handler below with exit 1 rather than looking clean.
192
202
  try:
193
203
  config = resolve_config(args)
194
204
  log.info("workspace: %s", config.workspace_root)
@@ -207,6 +217,20 @@ def main(argv: list[str] | None = None) -> int:
207
217
  watcher.poll_once()
208
218
  else:
209
219
  watcher.run_forever()
220
+ except LedgerUnwritable as exc:
221
+ # Only reachable from `--once`: run_forever handles its own refusals and
222
+ # keeps polling. A one-shot REPORTS rather than retries, so this must
223
+ # not look like a clean pass to `... --once && echo ok` or to a health
224
+ # probe. Exit 1 ("the environment failed"), not the 2 reserved for
225
+ # "your config is wrong" -- a config error tells you to edit a file, an
226
+ # unwritable ledger tells you to look at the volume mount.
227
+ print(
228
+ f"ledger error: {exc} is not writable — no decisions were taken. "
229
+ "The daemon refuses to spawn, escalate, grant or post what it "
230
+ "cannot record; fix the volume mount or its ownership.",
231
+ file=sys.stderr,
232
+ )
233
+ return 1
210
234
  except IdentityMismatch as exc:
211
235
  print(f"identity error: {exc}", file=sys.stderr)
212
236
  return 2
@@ -16,7 +16,7 @@ import secrets
16
16
  import sqlite3
17
17
  import time
18
18
  from collections import Counter
19
- from dataclasses import dataclass
19
+ from dataclasses import dataclass, field
20
20
  from enum import Enum
21
21
  from pathlib import Path
22
22
 
@@ -99,6 +99,152 @@ MAX_VERDICT_POST_ATTEMPTS = 5
99
99
  # day, per stuck PR, with the operator paged exactly once.
100
100
  MAX_VERDICT_POST_BACKOFF_SECONDS = 60 * 60
101
101
 
102
+ # -- the poll-failure firewall (issue #62) ------------------------------------
103
+ #
104
+ # 2026-07-29 killed the Railway daemon three times in one day: a subprocess
105
+ # ENOENT on the `alissa` CLI (an image-layer file that vanished mid-run) and a
106
+ # readonly-sqlite snapshot write both escaped `poll_once`, and __main__'s
107
+ # startup-shaped handlers turned each into `exit 2`. One bad poll must never end
108
+ # the daemon: the steady state's contract is degraded-but-alive with a loud
109
+ # signal, never a silent exit. Startup config errors keep the fast exit -- those
110
+ # a restart genuinely cannot fix.
111
+
112
+ # Ceiling on the doubling backoff a failing poll applies. Matches the
113
+ # rate-limit branch's cap: a daemon that has been failing for a quarter of an
114
+ # hour gains nothing from polling more often than every 15 minutes, and a
115
+ # recovered substrate is picked up within one window.
116
+ POLL_BACKOFF_CAP_SECONDS = 900
117
+
118
+ # How long one exception CLASS must fire on every consecutive poll before the
119
+ # firewall stops calling it transient and escalates to a page-worthy ERROR.
120
+ # Half an hour is several backoff windows -- long past anything a container
121
+ # blip explains -- and still inside the window an operator can act on the same
122
+ # day. A different class arriving resets the streak: that is a different
123
+ # condition, not a continuation of this one.
124
+ POLL_ESCALATE_SECONDS = 30 * 60
125
+
126
+ # Streak limiting for the firewall's log line: the first few failures of a
127
+ # streak are logged in full, then one in every POLL_FAILURE_LOG_EVERY, so a
128
+ # substrate outage costs a handful of lines an hour instead of one per poll.
129
+ # The escalation crossing and the recovery line are logged unconditionally --
130
+ # both are state changes, and suppressing either would hide the very transition
131
+ # the log exists to show.
132
+ POLL_FAILURE_LOG_HEAD = 3
133
+ POLL_FAILURE_LOG_EVERY = 10
134
+
135
+ # Poll failures whose message is the whole diagnosis: CommandError already
136
+ # carries the command and its stderr, so a traceback adds noise. Anything else
137
+ # reaching the firewall is by definition unanticipated -- log where it came
138
+ # from.
139
+ EXPECTED_POLL_FAILURES = (CommandError,)
140
+
141
+
142
+ class LedgerUnwritable(RuntimeError):
143
+ """Raised by `poll_once` when the ledger gate refuses the pass.
144
+
145
+ A dedicated signal rather than an empty result, because the two callers
146
+ have to tell "refused" apart from "polled, nothing to do" and a bare `[]`
147
+ cannot (PR #63 round-2 major and one of its minors, both closed by this):
148
+
149
+ * `run_forever` must leave the firewall's failure streak COMPLETELY alone.
150
+ A refused pass is not evidence that anything cleared -- it is evidence
151
+ the daemon did not look -- so counting it as a success printed a false
152
+ recovery line and re-armed the escalation clock for a fault that was
153
+ still failing. That is the same power the RateLimited branch was stripped
154
+ of in round 2, for the same reason.
155
+ * `--once` must exit non-zero. A one-shot reports rather than retries, and
156
+ a health probe or `... --once && echo ok` reading a refused pass as a
157
+ clean one is the one failure mode it cannot survive.
158
+ """
159
+
160
+
161
+ @dataclass
162
+ class Streak:
163
+ """A run of consecutive identical outcomes, with the log policy attached:
164
+ how many, how long, whether it has been escalated, and whether THIS one is
165
+ worth a line.
166
+
167
+ Extracted so the two callers that need "streak, limit, escalate, recover"
168
+ -- the poll firewall and the ledger gate -- cannot disagree about it (PR #63
169
+ round-2 nit; the gate's hand-rolled copy dropped the crossing bypass, which
170
+ made the parameters table's "the escalation crossing is never suppressed"
171
+ false at any poll interval other than 60s).
172
+
173
+ Deliberately a value object taking `now` from its caller rather than reading
174
+ the clock: the escalation rule is a statement about elapsed time, and it has
175
+ to be testable without sleeping through it.
176
+ """
177
+
178
+ count: int = 0
179
+ first_at: float = 0.0
180
+ escalated: bool = False
181
+
182
+ def record(self, now: float) -> tuple[bool, bool]:
183
+ """Fold one occurrence in. Returns (log_this_one, escalated_just_now).
184
+
185
+ The crossing BYPASSES the streak limit: it is a state change, and
186
+ suppressing it would hide the one transition the log exists to show.
187
+ """
188
+ if self.count == 0:
189
+ self.first_at, self.escalated = now, False
190
+ self.count += 1
191
+ crossing = not self.escalated and now - self.first_at >= POLL_ESCALATE_SECONDS
192
+ if crossing:
193
+ self.escalated = True
194
+ should_log = (
195
+ crossing
196
+ or self.count <= POLL_FAILURE_LOG_HEAD
197
+ or self.count % POLL_FAILURE_LOG_EVERY == 0
198
+ )
199
+ return should_log, crossing
200
+
201
+ def held(self, now: float) -> float:
202
+ """Seconds since the streak began. Zero when there is no streak."""
203
+ return 0.0 if self.count == 0 else now - self.first_at
204
+
205
+ def clear(self) -> None:
206
+ self.count, self.first_at, self.escalated = 0, 0.0, False
207
+
208
+ def resolve(self, now: float) -> tuple[int, float] | None:
209
+ """End the streak. Returns (occurrences, seconds), or None if there was
210
+ no streak -- the caller logs the recovery, which is the only evidence in
211
+ the log that a degraded daemon came back on its own."""
212
+ if self.count == 0:
213
+ return None
214
+ ended = (self.count, now - self.first_at)
215
+ self.clear()
216
+ return ended
217
+
218
+
219
+ @dataclass
220
+ class PollFailures:
221
+ """The firewall's memory of the current run of consecutive poll failures.
222
+
223
+ All the counting, limiting, escalation and recovery lives in `Streak`; what
224
+ is genuinely this class's own is the KEY. A streak is identified by the
225
+ exception CLASS, and a different class arriving mid-outage starts a new one
226
+ (re-arming escalation) because it is a different fault -- reporting "ENOENT
227
+ has been failing for 40 minutes" when the last 30 were sqlite errors would
228
+ be a lie the operator acts on.
229
+ """
230
+
231
+ kind: str | None = None
232
+ streak: Streak = field(default_factory=Streak)
233
+
234
+ def record(self, exc: BaseException, now: float) -> tuple[bool, bool]:
235
+ """Fold one failure in. Returns (log_this_one, escalated_just_now)."""
236
+ kind = type(exc).__name__
237
+ if kind != self.kind:
238
+ self.kind = kind
239
+ self.streak.clear()
240
+ return self.streak.record(now)
241
+
242
+ def resolve(self, now: float) -> tuple[int, float] | None:
243
+ """Clear the streak on a successful poll."""
244
+ self.kind = None
245
+ return self.streak.resolve(now)
246
+
247
+
102
248
  # The GitHub review states the CI gate can produce for a round it refused to
103
249
  # approve: a red head lands as CHANGES_REQUESTED, a rollup that never concluded
104
250
  # as COMMENTED. Read by _convergence_reason, which must not converge on an
@@ -809,6 +955,14 @@ class ReviewWatcher:
809
955
  # rollup (two API calls) on every poll, forever, for every PR with an
810
956
  # owed approve. In-memory for the same reason _dry_run_drift is.
811
957
  self._dry_run_rollups: dict[tuple[str, int, int, str], str] = {}
958
+ # Consecutive passes refused by the ledger gate in poll_once, and when
959
+ # the refusal began -- the same streak-limit-then-escalate shape the
960
+ # poll firewall uses, for the same reason: a read-only volume refuses
961
+ # every pass, and one line per poll would bury the condition it is
962
+ # reporting. In memory because it describes THIS process's degraded
963
+ # state, and because the only durable place to put it is the ledger
964
+ # that cannot be written.
965
+ self._ledger_streak = Streak()
812
966
 
813
967
  # -- per-PR decision ---------------------------------------------------
814
968
 
@@ -2769,6 +2923,42 @@ class ReviewWatcher:
2769
2923
  # -- polling -----------------------------------------------------------
2770
2924
 
2771
2925
  def poll_once(self) -> list[tuple[str, Decision]]:
2926
+ # THE LEDGER GATE (issue #62, PR #63 round-1 blocker). Nothing below
2927
+ # may run when the ledger cannot record what it does.
2928
+ #
2929
+ # Keeping the correctness writes strict aborts the pass that fails, but
2930
+ # the firewall in run_forever hands the loop straight back here -- and
2931
+ # by then the side effect is already taken. Concretely, over a
2932
+ # read-only volume: _spawn enqueues a reviewer session, record_spawn
2933
+ # raises, the pass dies, and the next pass finds no spawn row (the
2934
+ # in-flight check is a READ of the row that never landed), so it
2935
+ # enqueues another one. Every poll. Each a live agent that submits a
2936
+ # real review and burns the round budget. Before the firewall existed
2937
+ # the daemon died after one such duplicate -- bad, but bounded.
2938
+ #
2939
+ # So the gate is above everything, including the reap sweep (a kill is
2940
+ # a side effect and record_reap is a correctness write). The pass takes
2941
+ # no decisions and returns empty; the loop stays alive and keeps
2942
+ # probing. One race remains and is deliberate: a volume that flips
2943
+ # read-only BETWEEN this probe and record_spawn costs one duplicate,
2944
+ # which is the pre-firewall blast radius, and every pass after it is
2945
+ # gated. Closing it would mean recording before enqueuing, which trades
2946
+ # this for an orphan row that wedges the round for a full stale window
2947
+ # on any enqueue failure.
2948
+ # DRY-RUN IS EXEMPT, and vacuously so: it already suppresses every side
2949
+ # effect AND every correctness write (`_spawn` skips record_spawn, the
2950
+ # reaper logs instead of killing, the drift/cap-out/deferral paths
2951
+ # return before both their comment and their record). Its only ledger
2952
+ # write is the snapshot, which this module classifies as best-effort
2953
+ # telemetry and which _write_snapshot writes in dry-run deliberately.
2954
+ # So the gate would protect nothing there and cost the operator the one
2955
+ # tool that answers "what would you do right now" -- asked, precisely,
2956
+ # during the substrate incident this whole change is about.
2957
+ if not self.config.dry_run and not self.state.writable():
2958
+ self._note_ledger_unwritable()
2959
+ raise LedgerUnwritable(str(self.config.state_db))
2960
+ self._note_ledger_writable()
2961
+
2772
2962
  # Sweep BEFORE evaluating: a full worker is exactly when a fresh spawn
2773
2963
  # needs the slot a finished session is squatting on. Deliberately not
2774
2964
  # inside the per-request loop below — the sweep must reach sessions
@@ -2806,6 +2996,48 @@ class ReviewWatcher:
2806
2996
  )
2807
2997
  return results
2808
2998
 
2999
+ def _note_ledger_unwritable(self) -> None:
3000
+ """Report a pass refused because the ledger cannot record it.
3001
+
3002
+ Shares `Streak` with the poll firewall, so the streak limit, the
3003
+ escalation window and -- the part the hand-rolled copy got wrong -- the
3004
+ crossing's bypass of that limit are one implementation. A daemon that
3005
+ is up, polling, and deciding NOTHING is the most misleading state it
3006
+ can be in, so once the condition outlasts POLL_ESCALATE_SECONDS every
3007
+ logged line says so at page-worthy level.
3008
+ """
3009
+ now = time.monotonic()
3010
+ should_log, crossing = self._ledger_streak.record(now)
3011
+ if not should_log:
3012
+ return
3013
+ log.log(
3014
+ logging.ERROR if self._ledger_streak.escalated else logging.WARNING,
3015
+ "ledger at %s cannot be written — skipping this pass entirely "
3016
+ "(%d consecutive, %.0f min)%s: the daemon will not spawn, escalate, "
3017
+ "grant or post what it cannot record. It is alive and re-probing "
3018
+ "every poll; no review will be queued until the volume is writable.",
3019
+ self.config.state_db,
3020
+ self._ledger_streak.count,
3021
+ self._ledger_streak.held(now) / 60,
3022
+ " — this is no longer transient" if crossing else "",
3023
+ )
3024
+
3025
+ def _note_ledger_writable(self) -> None:
3026
+ """Announce that the gate has re-opened. Unconditional, like the
3027
+ firewall's recovery line: the operator's last word on a degraded
3028
+ daemon must not be the degradation."""
3029
+ ended = self._ledger_streak.resolve(time.monotonic())
3030
+ if ended is None:
3031
+ return
3032
+ skipped, seconds = ended
3033
+ log.info(
3034
+ "ledger at %s is writable again after %d skipped pass(es) over "
3035
+ "%.0fs — resuming normal decisions",
3036
+ self.config.state_db,
3037
+ skipped,
3038
+ seconds,
3039
+ )
3040
+
2809
3041
  def _stage_record(self, slug: str, decision: Decision) -> dict:
2810
3042
  """One per-item entry of a poll snapshot's compact JSON: the PR
2811
3043
  reference (the slug and the number parsed from it), the current stage
@@ -2884,7 +3116,16 @@ class ReviewWatcher:
2884
3116
  def run_forever(self) -> None:
2885
3117
  # preflight() is the caller's responsibility -- the CLI runs it once for
2886
3118
  # every mode, so calling it here too would double every check.
3119
+ #
3120
+ # Every exception a poll can raise is caught HERE (issue #62). The
3121
+ # daemon's steady state has no fatal errors: a transient subprocess
3122
+ # ENOENT, a parse error, a readonly ledger -- any of them is one bad
3123
+ # poll, and poll N+1 may well succeed. Only KeyboardInterrupt and
3124
+ # SystemExit pass through (neither is an `Exception`), and only startup
3125
+ # -- resolve_config, before this loop is ever entered -- still exits
3126
+ # fast.
2887
3127
  backoff = self.config.poll_interval
3128
+ failures = PollFailures()
2888
3129
  while True:
2889
3130
  # The sleep lives INSIDE the KeyboardInterrupt guard: with a 60s
2890
3131
  # poll interval (up to 900s backing off) the loop spends nearly
@@ -2894,13 +3135,91 @@ class ReviewWatcher:
2894
3135
  try:
2895
3136
  self.poll_once()
2896
3137
  backoff = self.config.poll_interval
3138
+ self._note_poll_recovered(failures)
3139
+ except LedgerUnwritable:
3140
+ # Already logged, streak-limited and escalating, by the
3141
+ # gate itself. What matters HERE is what is NOT done: the
3142
+ # firewall's `failures` is left completely untouched, so a
3143
+ # fault that is still failing keeps counting toward its own
3144
+ # page instead of having the clock re-armed by a pass the
3145
+ # daemon never took. The backoff DOES reset to the poll
3146
+ # interval, deliberately: probing at cadence is the point
3147
+ # of the gate, and inheriting a failing streak's 15-minute
3148
+ # backoff would leave a healed volume unnoticed that long.
3149
+ backoff = self.config.poll_interval
2897
3150
  except RateLimited as exc:
2898
- backoff = min(backoff * 2, 900)
3151
+ # Not a failure of the daemon: GitHub is telling it to slow
3152
+ # down, and it does. It does NOT count toward the firewall's
3153
+ # streak -- and, just as deliberately, it does not END one
3154
+ # either (PR #63 round-1 major). A rate limit is not
3155
+ # evidence that a substrate fault cleared: `review_requests`
3156
+ # is the first GitHub call in the pass, so it can pre-empt
3157
+ # the failing call site entirely, and resolving the streak
3158
+ # here let a busy hour re-arm the escalation clock forever
3159
+ # and cancel a page the DoD requires. The streak is left
3160
+ # exactly as it was; only a genuinely successful poll ends
3161
+ # one, and that path logs the recovery.
3162
+ backoff = min(backoff * 2, POLL_BACKOFF_CAP_SECONDS)
2899
3163
  log.warning("rate limited (%s) — backing off %ds", exc, backoff)
2900
- except CommandError as exc:
2901
- backoff = min(backoff * 2, 900)
2902
- log.error("poll failed: %s — retrying in %ds", exc, backoff)
3164
+ except Exception as exc:
3165
+ backoff = min(backoff * 2, POLL_BACKOFF_CAP_SECONDS)
3166
+ self._note_poll_failure(failures, exc, backoff)
2903
3167
  time.sleep(backoff)
2904
3168
  except KeyboardInterrupt:
2905
3169
  log.info("stopping")
2906
3170
  return
3171
+
3172
+ def _note_poll_failure(
3173
+ self, failures: PollFailures, exc: Exception, backoff: int
3174
+ ) -> None:
3175
+ """Log one firewalled poll failure, streak-limited and escalating.
3176
+
3177
+ WARNING while the fault still looks transient, ERROR once the same
3178
+ class has fired on every poll for POLL_ESCALATE_SECONDS -- the daemon
3179
+ is alive either way, so the log level is the only thing that can tell
3180
+ an operator "this one is not healing".
3181
+ """
3182
+ should_log, crossing = failures.record(exc, time.monotonic())
3183
+ if not should_log:
3184
+ return
3185
+ # A class the firewall did not anticipate gets its traceback; the ones
3186
+ # that carry their own diagnosis do not (see EXPECTED_POLL_FAILURES).
3187
+ traced = not isinstance(exc, EXPECTED_POLL_FAILURES)
3188
+ if crossing:
3189
+ log.error(
3190
+ "poll has failed with %s on every attempt for %.0f min "
3191
+ "(%d consecutive failures, latest: %s) — the daemon is alive and "
3192
+ "still retrying every %ds, but this is no longer transient",
3193
+ failures.kind,
3194
+ failures.streak.held(time.monotonic()) / 60,
3195
+ failures.streak.count,
3196
+ exc,
3197
+ backoff,
3198
+ exc_info=traced,
3199
+ )
3200
+ return
3201
+ log.log(
3202
+ logging.ERROR if failures.streak.escalated else logging.WARNING,
3203
+ "poll failed (%s: %s) — failure %d of this streak; retrying in %ds",
3204
+ failures.kind,
3205
+ exc,
3206
+ failures.streak.count,
3207
+ backoff,
3208
+ exc_info=traced,
3209
+ )
3210
+
3211
+ @staticmethod
3212
+ def _note_poll_recovered(failures: PollFailures) -> None:
3213
+ """Announce that a failing streak ended. Unconditional: the recovery is
3214
+ the counterpart of the escalation, and a streak that healed silently
3215
+ leaves an operator reading the last ERROR as the current state."""
3216
+ ended = failures.resolve(time.monotonic())
3217
+ if ended is None:
3218
+ return
3219
+ count, seconds = ended
3220
+ log.info(
3221
+ "poll recovered after %d consecutive failure(s) over %.0fs — "
3222
+ "resuming normal polling",
3223
+ count,
3224
+ seconds,
3225
+ )
@@ -26,12 +26,15 @@ it, alongside the untouched legacy ledgers.
26
26
  from __future__ import annotations
27
27
 
28
28
  import json
29
+ import logging
29
30
  import sqlite3
30
31
  import time
31
- from collections.abc import Iterable
32
+ from collections.abc import Callable, Iterable
32
33
  from dataclasses import dataclass
33
34
  from pathlib import Path
34
35
 
36
+ log = logging.getLogger(__name__)
37
+
35
38
  # Poll-snapshot retention: the newest N rows are kept, older ones pruned on
36
39
  # every write. Fixed, not a config key -- `poll_snapshots` is an observation
37
40
  # buffer for a future console sidecar, and a bounded ring is all it needs (it
@@ -39,6 +42,13 @@ from pathlib import Path
39
42
  # observable buffer size, so it is pinned by a test.
40
43
  SNAPSHOT_RETENTION = 1000
41
44
 
45
+ # Streak limiting for the best-effort telemetry writer's WARN (issue #62), on
46
+ # the same rule the poll firewall uses: the first few failures in full, then
47
+ # one in ten. A ledger that has gone read-only fails on every single poll, and
48
+ # the warning is worth nothing if it drowns the decisions around it.
49
+ TELEMETRY_LOG_HEAD = 3
50
+ TELEMETRY_LOG_EVERY = 10
51
+
42
52
  # Shared between SCHEMA and the migration so the two can never drift.
43
53
  _SPAWNS_TABLE = """
44
54
  CREATE TABLE IF NOT EXISTS spawns (
@@ -231,6 +241,16 @@ class State:
231
241
  for any path an operator can type.
232
242
  """
233
243
  path = Path(path).expanduser()
244
+ # Kept so a best-effort telemetry write can RECONNECT after a failure
245
+ # (see _reconnect): the daemon's ledger lives on a platform volume, and
246
+ # a remount leaves the open handle pointing at a file descriptor that
247
+ # is gone while the path is perfectly good again.
248
+ self._path = path
249
+ self._read_only = read_only
250
+ # Consecutive failures of the best-effort writer, for streak-limited
251
+ # logging and for firing the one reconnect attempt on the FIRST failure
252
+ # of a streak rather than on every write.
253
+ self._telemetry_failures = 0
234
254
  if read_only:
235
255
  uri = Path(path).absolute().as_uri() + "?mode=ro"
236
256
  self._db = sqlite3.connect(uri, uri=True)
@@ -304,6 +324,167 @@ class State:
304
324
  return False # fresh database, nothing to migrate
305
325
  return [r["name"] for r in info if r["pk"]] != ["session"]
306
326
 
327
+ # -- best-effort writes (issue #62) ------------------------------------
328
+ #
329
+ # CLASSIFICATION. Every write in this class is one of two kinds, and only
330
+ # one of them may ever be swallowed:
331
+ #
332
+ # * TELEMETRY -- `record_snapshot`, and nothing else. `poll_snapshots` is an
333
+ # observation buffer: the daemon never reads it back to make a decision
334
+ # (only `read_snapshots`, for the console, does), so a row lost to a
335
+ # read-only volume costs one missing datapoint on a dashboard. On
336
+ # 2026-07-29 it cost the whole daemon instead -- the sqlite exception
337
+ # escaped `poll_once` and killed the process mid-poll.
338
+ #
339
+ # * CORRECTNESS -- every other write here (`record_spawn`, `record_reap`,
340
+ # `record_ping`, `record_escalation`, `record_grant`, the `verdict_posts`
341
+ # writes, `age_out_spawn`). Each is a dedupe key or an in-flight marker
342
+ # for an action the daemon TAKES: swallowing one does not lose a
343
+ # datapoint, it re-spawns a reviewer round, re-pages an operator, or
344
+ # re-grants a cap. Those stay strict and raise.
345
+ #
346
+ # STRICTNESS IS NOT, BY ITSELF, THE PROTECTION -- and the first draft of
347
+ # this change claimed it was (PR #63 round-1 blocker). Raising aborts the
348
+ # pass that failed; it says nothing about the next one. The poll firewall
349
+ # then hands the loop straight back to the same code path, and the side
350
+ # effect the write was meant to dedupe has ALREADY been taken -- so a
351
+ # read-only volume turned "enqueue a reviewer, fail to record it" into a
352
+ # fresh reviewer session every poll, indefinitely, where before it merely
353
+ # killed the daemon after one. What actually protects the side effect is
354
+ # `writable()` above, checked by `loop.poll_once` before the pass takes any
355
+ # decision at all: the daemon does not take an action it cannot record.
356
+ # Strictness is what makes an unrecordable action VISIBLE; the gate is what
357
+ # makes it not repeat.
358
+
359
+ @staticmethod
360
+ def _write_probe(db: sqlite3.Connection) -> bool:
361
+ """Can this connection actually write? A no-op header write, which
362
+ exercises exactly the path a real write needs, changes nothing, and
363
+ costs one page. Read-only-ness is the thing being detected, so it
364
+ cannot be answered by inspecting the file's mode: sqlite decides it at
365
+ open time and a handle can be read-only over a writable file (and, for
366
+ one recoverable moment, the reverse)."""
367
+ try:
368
+ version = db.execute("PRAGMA user_version").fetchone()[0]
369
+ db.execute(f"PRAGMA user_version = {int(version)}")
370
+ db.commit()
371
+ except sqlite3.DatabaseError:
372
+ return False
373
+ return True
374
+
375
+ def writable(self) -> bool:
376
+ """Whether the ledger can accept a write RIGHT NOW.
377
+
378
+ The daemon asks this before it takes any action it would have to
379
+ record (issue #62, round-1 blocker). Keeping a correctness write strict
380
+ aborts the pass that fails, but the poll firewall hands the loop
381
+ straight back to the same code path -- so without this gate a read-only
382
+ volume turns "enqueue a reviewer, then fail to record it" into a fresh
383
+ reviewer session every poll, forever, each one a live agent. The
384
+ invariant the gate buys is simple: the daemon does not take an action
385
+ it cannot record.
386
+
387
+ A failing probe retries through `_reconnect`, whose candidate is
388
+ write-probed before adoption -- so a stale handle over a live file
389
+ heals here too, and only a genuinely unwritable ledger answers False.
390
+ A read-only `State` (the console's) is never writable by construction.
391
+ """
392
+ if self._read_only:
393
+ return False
394
+ return self._write_probe(self._db) or self._reconnect()
395
+
396
+ def _reconnect(self) -> bool:
397
+ """Swap in a fresh connection, but ONLY if the fresh one is better.
398
+ True when the swap happened -- which, because the candidate is
399
+ write-probed, is also proof that the ledger is writable.
400
+
401
+ Deliberately raw: it re-establishes the connection and NOTHING else --
402
+ no schema script, no migration. The reconnect exists for the
403
+ stale-handle-after-remount case, where the database on disk is the one
404
+ this process already migrated; re-running DDL through a path that only
405
+ a failed telemetry write reaches would be a far larger act than the
406
+ failure justifies.
407
+
408
+ The candidate is WRITE-PROBED before it is adopted, and the old
409
+ connection is kept when the probe fails, because a blind reconnect
410
+ makes the read-only case permanently worse rather than better: sqlite
411
+ decides read-only-ness when it OPENS the file, so a handle opened while
412
+ the volume was read-only stays read-only for the rest of its life even
413
+ after the volume comes back -- while the handle opened before the fault
414
+ heals by itself the moment writes are possible again. Replacing the
415
+ healable handle with a poisoned one would trade a transient outage for
416
+ a permanent one.
417
+
418
+ WRITE MODE ONLY. A read-only `State` is the console's, it must never
419
+ write, and there is nothing a reconnect could improve for it -- so it
420
+ returns False rather than swapping one equivalent handle for another.
421
+ That also keeps the contract absolute: a True from here always means a
422
+ candidate passed the write probe, which is what `writable()` relies on.
423
+ """
424
+ if self._read_only:
425
+ return False
426
+ candidate: sqlite3.Connection | None = None
427
+ try:
428
+ candidate = sqlite3.connect(str(self._path))
429
+ candidate.row_factory = sqlite3.Row
430
+ except sqlite3.Error as exc:
431
+ log.debug("state: reconnect to %s declined: %s", self._path, exc)
432
+ return False
433
+ if not self._write_probe(candidate):
434
+ log.debug("state: reconnect to %s declined (candidate cannot write)", self._path)
435
+ try:
436
+ candidate.close()
437
+ except sqlite3.Error:
438
+ pass
439
+ return False
440
+ try:
441
+ self._db.close()
442
+ except sqlite3.Error:
443
+ pass # already broken; the point was to replace it
444
+ self._db = candidate
445
+ return True
446
+
447
+ def _write_telemetry(self, write: "Callable[[], None]", what: str) -> bool:
448
+ """Run a TELEMETRY write, absorbing any database error. True on success.
449
+
450
+ One reconnect attempt on the FIRST failure of a streak (not on every
451
+ one: a database that is read-only stays read-only, and reconnecting per
452
+ poll would add a file open to every pass for nothing), then a
453
+ streak-limited WARN and back to polling.
454
+ """
455
+ try:
456
+ write()
457
+ except sqlite3.DatabaseError as exc:
458
+ first = self._telemetry_failures == 0
459
+ if first and self._reconnect():
460
+ try:
461
+ write()
462
+ except sqlite3.DatabaseError as retry_exc:
463
+ exc = retry_exc
464
+ else:
465
+ log.info(
466
+ "state: %s succeeded after reconnecting to %s",
467
+ what, self._path,
468
+ )
469
+ return True
470
+ self._telemetry_failures += 1
471
+ n = self._telemetry_failures
472
+ if n <= TELEMETRY_LOG_HEAD or n % TELEMETRY_LOG_EVERY == 0:
473
+ log.warning(
474
+ "state: %s failed (%s: %s) — failure %d of this streak; "
475
+ "telemetry is best-effort, the loop keeps polling",
476
+ what, type(exc).__name__, exc, n,
477
+ )
478
+ return False
479
+ if self._telemetry_failures:
480
+ log.info(
481
+ "state: %s succeeded after %d failed attempt(s) — telemetry "
482
+ "is persisting again",
483
+ what, self._telemetry_failures,
484
+ )
485
+ self._telemetry_failures = 0
486
+ return True
487
+
307
488
  def close(self) -> None:
308
489
  self._db.close()
309
490
 
@@ -689,7 +870,7 @@ class State:
689
870
  awaiting_post: int = 0,
690
871
  abandoned: int = 0,
691
872
  stages: list[dict],
692
- ) -> None:
873
+ ) -> bool:
693
874
  """Append one poll-pass observation, then prune to the newest
694
875
  SNAPSHOT_RETENTION rows. `ts` is stamped here (wall-clock seconds,
695
876
  like every other row in this ledger); `stages` is the compact
@@ -698,7 +879,56 @@ class State:
698
879
  dry-run included -- and pruned on write, so the table is
699
880
  self-bounding. The count kwargs default to 0 so a caller need only
700
881
  pass the ones a given pass produced.
882
+
883
+ BEST-EFFORT, and the only write in this class that is (issue #62): a
884
+ snapshot observes the pass, it is not something the daemon has to
885
+ remember, so a database error here is absorbed, reported once per
886
+ streak-limited window, and the loop keeps polling. Returns whether the
887
+ row landed, for a caller that wants to say so; nothing in the daemon
888
+ depends on it.
701
889
  """
890
+ return self._write_telemetry(
891
+ lambda: self._insert_snapshot(
892
+ duration_ms=duration_ms,
893
+ candidates=candidates,
894
+ spawned=spawned,
895
+ stale_reenqueued=stale_reenqueued,
896
+ in_flight=in_flight,
897
+ deferred=deferred,
898
+ converged=converged,
899
+ capped=capped,
900
+ escalated=escalated,
901
+ skipped=skipped,
902
+ reaped=reaped,
903
+ posted=posted,
904
+ awaiting_post=awaiting_post,
905
+ abandoned=abandoned,
906
+ stages=stages,
907
+ ),
908
+ "poll snapshot",
909
+ )
910
+
911
+ def _insert_snapshot(
912
+ self,
913
+ *,
914
+ duration_ms: int,
915
+ candidates: int,
916
+ spawned: int,
917
+ stale_reenqueued: int,
918
+ in_flight: int,
919
+ deferred: int,
920
+ converged: int,
921
+ capped: int,
922
+ escalated: int,
923
+ skipped: int,
924
+ reaped: int,
925
+ posted: int,
926
+ awaiting_post: int,
927
+ abandoned: int,
928
+ stages: list[dict],
929
+ ) -> None:
930
+ """The snapshot INSERT + prune itself, strict. Split out so the
931
+ best-effort wrapper can RETRY it verbatim after a reconnect."""
702
932
  self._db.execute(
703
933
  "INSERT INTO poll_snapshots "
704
934
  "(ts, duration_ms, candidates, spawned, stale_reenqueued, "
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: alissa-tools-github-revloop
3
- Version: 0.16.8
3
+ Version: 0.16.9
4
4
  Summary: ALISSA-TOOLS-GITHUB-REVLOOP
5
5
  Home-page: https://alissa.app
6
6
  Author: Fahera