alissa-tools-github-revloop 0.16.12__tar.gz → 0.16.14__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. {alissa_tools_github_revloop-0.16.12/src/main/alissa_tools_github_revloop.egg-info → alissa_tools_github_revloop-0.16.14}/PKG-INFO +1 -1
  2. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/__main__.py +8 -0
  3. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/config.py +48 -0
  4. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/loop.py +404 -2
  5. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/state.py +13 -2
  6. alissa_tools_github_revloop-0.16.14/src/main/alissa/tools/github/revloop/version +1 -0
  7. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/webui/page.py +1 -1
  8. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14/src/main/alissa_tools_github_revloop.egg-info}/PKG-INFO +1 -1
  9. alissa_tools_github_revloop-0.16.12/src/main/alissa/tools/github/revloop/version +0 -1
  10. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/LICENSE +0 -0
  11. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/MANIFEST.in +0 -0
  12. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/NOTICE +0 -0
  13. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/README.md +0 -0
  14. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/requirements.txt +0 -0
  15. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/setup.cfg +0 -0
  16. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/setup.py +0 -0
  17. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/__init__.py +0 -0
  18. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/alissa.py +0 -0
  19. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/ghclient.py +0 -0
  20. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/proc.py +0 -0
  21. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/prreview.py +0 -0
  22. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/version.py +0 -0
  23. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/webui/__init__.py +0 -0
  24. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/webui/__main__.py +0 -0
  25. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/webui/auth.py +0 -0
  26. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/webui/server.py +0 -0
  27. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/webui/sources.py +0 -0
  28. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa/tools/github/revloop/webui/sysinfo.py +0 -0
  29. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa_tools_github_revloop.egg-info/SOURCES.txt +0 -0
  30. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa_tools_github_revloop.egg-info/dependency_links.txt +0 -0
  31. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa_tools_github_revloop.egg-info/entry_points.txt +0 -0
  32. {alissa_tools_github_revloop-0.16.12 → alissa_tools_github_revloop-0.16.14}/src/main/alissa_tools_github_revloop.egg-info/top_level.txt +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: alissa-tools-github-revloop
3
- Version: 0.16.12
3
+ Version: 0.16.14
4
4
  Summary: ALISSA-TOOLS-GITHUB-REVLOOP
5
5
  Home-page: https://alissa.app
6
6
  Author: Fahera
@@ -122,6 +122,13 @@ def build_parser() -> argparse.ArgumentParser:
122
122
  help="page-worthy threshold: more live reviewer sessions than this "
123
123
  "after a sweep and the daemon logs loudly",
124
124
  )
125
+ over.add_argument(
126
+ "--max-concurrent-sessions",
127
+ type=int,
128
+ metavar="N",
129
+ help="spawn gate: at this many live reviewer sessions an owed round "
130
+ "waits for a slot instead of spawning (must be <= --reap-session-cap)",
131
+ )
125
132
  over.add_argument(
126
133
  "--checks-wait-seconds",
127
134
  type=int,
@@ -165,6 +172,7 @@ def overrides_from(args: argparse.Namespace) -> dict:
165
172
  "on_missing_hub": args.on_missing_hub,
166
173
  "reap_grace_seconds": args.reap_grace_seconds,
167
174
  "reap_session_cap": args.reap_session_cap,
175
+ "max_concurrent_sessions": args.max_concurrent_sessions,
168
176
  "checks_wait_seconds": args.checks_wait_seconds,
169
177
  "dry_run": args.dry_run,
170
178
  }
@@ -129,6 +129,7 @@ CONFIG_KEYS = (
129
129
  "on_missing_hub",
130
130
  "reap_grace_seconds",
131
131
  "reap_session_cap",
132
+ "max_concurrent_sessions",
132
133
  "checks_wait_seconds",
133
134
  "dry_run",
134
135
  )
@@ -170,6 +171,25 @@ DEFAULT_REAP_GRACE_SECONDS = 30 * 60
170
171
  # healthy deployment reaches, not a capacity limit.
171
172
  DEFAULT_REAP_SESSION_CAP = 6
172
173
 
174
+ # The spawn gate: how many reviewer sessions of THIS daemon's own grammar may be
175
+ # live before an owed round waits for a slot instead of spawning (issue #70).
176
+ #
177
+ # Distinct from `reap_session_cap` above in kind, not just in number: that one is
178
+ # an ALARM on a condition the loop cannot fix (sessions the sweep could not
179
+ # reap), this one is a LIMIT the loop enforces on itself before it acts. Nothing
180
+ # bounded concurrency before it -- `round_cap` bounds rounds per PR, and the
181
+ # alarm only logs -- so a merge wave spawned one interactive claude session per
182
+ # PR, all at once, against a fixed container budget. On 2026-07-29 the 18:45-19:00Z
183
+ # burst pegged the deployment's 2 vCPU ceiling with 4+ concurrent reviewers plus
184
+ # the poll loop; throttled sessions review slower, hold their round slots longer,
185
+ # and widen the very burst that is starving them.
186
+ #
187
+ # 4 is the deployed shape's honest ceiling: two vCPUs, and a reviewer session is
188
+ # a full interactive agent. It is deliberately BELOW the reap alarm (6) so the
189
+ # steady state never pages -- and `Config.build` refuses a config where the alarm
190
+ # sits under the limit, which would page on healthy load.
191
+ DEFAULT_MAX_CONCURRENT_SESSIONS = 4
192
+
173
193
  # How long a round holds its APPROVE while the head's CI rollup is still
174
194
  # running (or unreadable) before it gives up and records the verdict as a
175
195
  # COMMENT instead. An approve from the reviewer identity is the operator's cue
@@ -233,6 +253,11 @@ class Config:
233
253
  reap_grace_seconds: int = DEFAULT_REAP_GRACE_SECONDS
234
254
  reap_session_cap: int = DEFAULT_REAP_SESSION_CAP
235
255
 
256
+ # The spawn gate's limit -- see DEFAULT_MAX_CONCURRENT_SESSIONS. At or above
257
+ # it an owed round defers to a later poll instead of spawning; it burns no
258
+ # round number and no attempt while it waits.
259
+ max_concurrent_sessions: int = DEFAULT_MAX_CONCURRENT_SESSIONS
260
+
236
261
  # The bound on holding a round's approve for a rollup that has not settled;
237
262
  # see DEFAULT_CHECKS_WAIT_SECONDS. 0 is legal and means "never hold": a
238
263
  # rollup that is not already green degrades the verdict to a comment on the
@@ -357,6 +382,28 @@ class Config:
357
382
  # state of a working loop -- an alarm that always fires is noise.
358
383
  raise ValueError(f"reap_session_cap must be >= 1, got {session_cap}")
359
384
 
385
+ max_sessions = int(
386
+ raw.get("max_concurrent_sessions", cls.max_concurrent_sessions)
387
+ )
388
+ if max_sessions < 1:
389
+ # 0 would defer every round forever: no session may spawn, so no
390
+ # slot ever frees. "Review nothing" is not a tuning value.
391
+ raise ValueError(
392
+ f"max_concurrent_sessions must be >= 1, got {max_sessions}"
393
+ )
394
+ if session_cap < max_sessions:
395
+ # The alarm would then fire on load the gate considers healthy --
396
+ # every poll of a fully-loaded, correctly-behaving daemon pages the
397
+ # operator, and a page that fires in the steady state trains people
398
+ # to ignore the one that matters. Refused at load rather than
399
+ # discovered at 3am.
400
+ raise ValueError(
401
+ f"reap_session_cap ({session_cap}) must be >= "
402
+ f"max_concurrent_sessions ({max_sessions}): the cap is the "
403
+ f"page-worthy alarm and the gate is the spawn limit, so an "
404
+ f"alarm below the limit pages on healthy load"
405
+ )
406
+
360
407
  checks_wait = int(raw.get("checks_wait_seconds", cls.checks_wait_seconds))
361
408
  if checks_wait < 0:
362
409
  raise ValueError(f"checks_wait_seconds must be >= 0, got {checks_wait}")
@@ -404,6 +451,7 @@ class Config:
404
451
  on_missing_hub=hub_mode,
405
452
  reap_grace_seconds=grace,
406
453
  reap_session_cap=session_cap,
454
+ max_concurrent_sessions=max_sessions,
407
455
  checks_wait_seconds=checks_wait,
408
456
  dry_run=bool(raw.get("dry_run", False)),
409
457
  )
@@ -141,6 +141,89 @@ POLL_FAILURE_LOG_EVERY = 10
141
141
  EXPECTED_POLL_FAILURES = (CommandError,)
142
142
 
143
143
 
144
+ # -- the spawn gate (issue #70) -----------------------------------------------
145
+ #
146
+ # The one INFO line a poll pass emits about deferrals, summarizing them all.
147
+ # Per PASS, never per round: a wave of eight PRs behind a gate of four is one
148
+ # fact about the container, not eight facts about PRs, and the deployed daemon
149
+ # polls every 30s. Streak-limited on the same rule as the poll firewall (the
150
+ # first few in full, then one in ten) so a long queue costs a handful of lines
151
+ # an hour.
152
+ DEFERRAL_SUMMARY = (
153
+ "spawn gate: %d round(s) deferred — %d/%d reviewer sessions live. "
154
+ "%s Nothing is lost: a deferred round burns no round number and no "
155
+ "attempt, and the oldest waiter takes the next free slot."
156
+ )
157
+
158
+ # What the summary becomes once the gate has been shut, with NOTHING spawning,
159
+ # for longer than POLL_ESCALATE_SECONDS (PR #71 round-1 [major]). Deferral
160
+ # itself is never page-worthy -- a container at its limit that keeps handing
161
+ # out freed slots is the gate working -- but a gate that has spawned nothing
162
+ # for half an hour is not that. Three session classes can hold it shut with the
163
+ # reap alarm silent: a BUSY session is never reaped whatever its PR's state, a
164
+ # hand-spawned `review-pr-<n>` on an OPEN PR is out of the reaper's scope by
165
+ # design (issue #46), and an undecidable session is spared every poll. Four of
166
+ # those against the shipped defaults (limit 4, alarm 6) is a fleet-wide review
167
+ # outage that no other channel reports: the gated rounds write no ledger row,
168
+ # so the stale-round probe cannot see them either.
169
+ #
170
+ # It states the OBSERVATION and hands the diagnosis to the operator (PR #71
171
+ # round-2 [nit]): the predicate is "deferred and started nothing for half an
172
+ # hour", which four legitimately slow reviews satisfy exactly as well as a
173
+ # wedged session. Both are worth an operator's eyes at that duration, but only
174
+ # the survivor list tells them which they have -- so the line points there
175
+ # instead of naming a cause it cannot know.
176
+ DEFERRAL_STALLED = (
177
+ "spawn gate: %d round(s) deferred — %d/%d reviewer sessions live, and "
178
+ "NOTHING has spawned for %.0f min. %s Nothing has started for long enough "
179
+ "that this may no longer be back-pressure doing its job: the reap sweep "
180
+ "never frees a busy session, and never frees a hand-spawned "
181
+ "review-pr-<n> on an open PR, so a wedged session holds its slot "
182
+ "indefinitely — check the sweep's survivors above."
183
+ )
184
+
185
+ # The recovery line for an ESCALATED stall: the gate started something again
186
+ # while rounds are still waiting. Logged unconditionally, outside the streak
187
+ # limit, on the rule _note_ledger_writable states -- the operator's last word
188
+ # on a degraded daemon must not be the degradation.
189
+ GATE_STALL_CLEARED = (
190
+ "spawn gate: spawning again after %d pass(es) over %.0f min with nothing "
191
+ "started — %d round(s) still waiting, and the queue is moving"
192
+ )
193
+
194
+ # The recovery line, logged once when a deferral streak ends -- the operator's
195
+ # only evidence in the log that a queue drained on its own.
196
+ DEFERRAL_CLEARED = (
197
+ "spawn gate: clear after %d pass(es) over %.0f min — every owed round "
198
+ "spawned"
199
+ )
200
+
201
+
202
+ @dataclass(frozen=True)
203
+ class Waiting:
204
+ """One round's place in the spawn queue, from its FIRST deferral.
205
+
206
+ `seq` is the ordering key and it is a COUNTER, not a clock: two rounds
207
+ deferred in the same pass are microseconds apart, and a wall-clock tie
208
+ would be broken by whatever `sorted` felt like -- which is the search's own
209
+ order, i.e. the starvation this exists to prevent. `since` is monotonic and
210
+ only ever reported, never compared.
211
+ """
212
+
213
+ seq: int
214
+ since: float
215
+
216
+
217
+ def _slug_key(slug: str) -> tuple[str, int]:
218
+ """`acme/widgets#7` -> the `_waiting` key `("acme/widgets", 7)`.
219
+
220
+ The poll walk carries decisions keyed by slug and the gate keys by
221
+ (repo, number); this is the one place the two spellings meet.
222
+ """
223
+ repo, _, number = slug.partition("#")
224
+ return repo, int(number)
225
+
226
+
144
227
  class LedgerUnwritable(RuntimeError):
145
228
  """Raised by `poll_once` when the ledger gate refuses the pass.
146
229
 
@@ -851,6 +934,17 @@ class Action(str, Enum):
851
934
  # this one has given up, and the poll snapshot and console aggregate the
852
935
  # action rather than the reason.
853
936
  ABANDONED = "abandoned"
937
+ # The round is owed and nothing about the PR blocks it -- the SPAWN GATE
938
+ # does: `max_concurrent_sessions` reviewer sessions are already live, so it
939
+ # waits for a slot and is retried on a later poll (issue #70).
940
+ #
941
+ # NOT folded into the liveness deferral's IN_FLIGHT/`deferred` pair, which
942
+ # it superficially resembles: that one names a live session still working
943
+ # its round, and the console counts it as an ACTIVE SESSION
944
+ # (webui.sources: in_flight + deferred). A gated round has no session at
945
+ # all -- counting it as one would inflate exactly the number this gate
946
+ # exists to hold down.
947
+ QUEUED = "queued"
854
948
 
855
949
 
856
950
  @dataclass(frozen=True)
@@ -1017,6 +1111,38 @@ class ReviewWatcher:
1017
1111
  # de-duplication, and a corpus that outlived its pass would answer the
1018
1112
  # next one from titles and statuses that have since moved.
1019
1113
  self._pass_tasks: list[Task] | None = None
1114
+ # -- the spawn gate's per-pass and cross-pass state (issue #70) ------
1115
+ # How many own-grammar reviewer sessions are live THIS pass, and
1116
+ # whether anyone has looked. The sweep seeds both (it lists the
1117
+ # sessions anyway); a count of None with `_census_probed` True means
1118
+ # the list could not be read, which the gate treats as "unknown" and
1119
+ # fails OPEN -- see _live_session_count.
1120
+ self._session_census: int | None = None
1121
+ self._census_probed = False
1122
+ self._census_warned = False
1123
+ # (repo full name, PR number) -> where that round sits in the spawn
1124
+ # queue, from the first pass that deferred it. Cross-pass and
1125
+ # in-memory: it is a FAIRNESS ORDER, not a decision the daemon must
1126
+ # remember -- a restart costs the queue its order for one pass (every
1127
+ # waiter is re-stamped on its next deferral) and nothing else, which
1128
+ # does not justify a ledger table on the poll path. Pruned every pass
1129
+ # against the live candidate set, so a merged or converged PR cannot
1130
+ # hold a place forever.
1131
+ self._waiting: dict[tuple[str, int], Waiting] = {}
1132
+ self._wait_seq = 0
1133
+ # Consecutive passes that deferred at least one round -- the summary
1134
+ # line's streak limiter, and ONLY that. It never escalates: a queue
1135
+ # that keeps draining is the gate working, and a wave permanently
1136
+ # larger than the limit would otherwise page forever on a daemon that
1137
+ # is reviewing everything, just serially.
1138
+ self._gate_streak = Streak()
1139
+ # Consecutive passes that deferred a round and spawned NOTHING -- the
1140
+ # escalating one (PR #71 round-1 [major]). Cleared by any spawn,
1141
+ # because a spawn is the proof the queue is moving; once it outlasts
1142
+ # POLL_ESCALATE_SECONDS the summary switches to DEFERRAL_STALLED at
1143
+ # WARNING. Separate from the limiter above so the predicate that pages
1144
+ # is "the gate is stuck", not "the gate is busy".
1145
+ self._gate_stall = Streak()
1020
1146
  # Consecutive passes refused by the ledger gate in poll_once, and when
1021
1147
  # the refusal began -- the same streak-limit-then-escalate shape the
1022
1148
  # poll firewall uses, for the same reason: a read-only volume refuses
@@ -1240,6 +1366,22 @@ class ReviewWatcher:
1240
1366
  deferred = self._defer_stale_round(pr, round_, age, cap)
1241
1367
  if deferred is not None:
1242
1368
  return deferred
1369
+
1370
+ # THE SPAWN GATE, and it sits here -- past every branch that decides
1371
+ # WHETHER a round is owed, immediately before the one that acts.
1372
+ # Upstream of it the loop is only reading; downstream it starts an
1373
+ # interactive agent. So a gated round has passed convergence, the cap
1374
+ # and CR9 re-entry identically to one that spawns: re-entry-granted
1375
+ # rounds queue through the gate like any other, delayed and never
1376
+ # denied. A stale-round respawn is gated too -- it is a spawn, and the
1377
+ # dead session it replaces is exactly as absent next poll.
1378
+ held = self._gate_spawn(pr, round_)
1379
+ if held is not None:
1380
+ return held
1381
+
1382
+ if age is not None:
1383
+ # Logged only once the gate has let the respawn through, so the
1384
+ # line cannot claim a re-enqueue that back-pressure then deferred.
1243
1385
  log.warning(
1244
1386
  "%s round %d has been in flight %.0f min with no submitted review "
1245
1387
  "and its session is gone or finished — re-enqueuing (reviewer "
@@ -1251,6 +1393,206 @@ class ReviewWatcher:
1251
1393
 
1252
1394
  return self._spawn(pr, round_, task, cap, reenqueued=age is not None)
1253
1395
 
1396
+ # -- the spawn gate ----------------------------------------------------
1397
+
1398
+ def _gate_spawn(self, pr: PullRequest, round_: int) -> Decision | None:
1399
+ """Hold an owed round back when the container is already full.
1400
+
1401
+ Returns a QUEUED Decision when the round must wait, or None to spawn.
1402
+
1403
+ DEFERRAL IS NOT FAILURE, and every part of that is load-bearing:
1404
+
1405
+ * it writes NO spawn-ledger row, so the round number is untouched (the
1406
+ next poll computes the same `completed + 1`) and no attempt is spent;
1407
+ * with no row there is no `spawn_age`, so the stale-round probe and its
1408
+ respawn branch cannot see the round at all -- a round can sit gated
1409
+ for hours without ever being "in flight 90 minutes";
1410
+ * it posts nothing: no PR comment, no operator page, no escalation.
1411
+ The only trace is one summary line per pass (see _note_deferrals).
1412
+
1413
+ The count is of sessions matching THIS package's grammar and no other
1414
+ (`Alissa.list_review_sessions` filters on `parse_session_name`), which
1415
+ is the same ownership boundary the reaper draws: other lanes share the
1416
+ container and are invisible here in both directions. Hand-spawned
1417
+ `review-pr-<n>` sessions are ours by that grammar and DO count -- they
1418
+ consume the same CPU, and a gate that ignored them would let a
1419
+ hand-driven round and a daemon round each think it had the last slot.
1420
+
1421
+ Sessions this pass has already enqueued count too (`_spawn` increments
1422
+ the census): the census is read once per pass, so without that a single
1423
+ pass would hand the same free slot to every PR in the wave.
1424
+ """
1425
+ limit = self.config.max_concurrent_sessions
1426
+ live = self._live_session_count()
1427
+ key = (pr.full_name, pr.number)
1428
+ if live is None or live < limit:
1429
+ # Spawning: this round is no longer waiting on anything. Dropped
1430
+ # here rather than in `_spawn`, which can still bail on a missing
1431
+ # hub or review task -- a round that never reaches the enqueue is
1432
+ # not holding a queue place either.
1433
+ self._waiting.pop(key, None)
1434
+ return None
1435
+
1436
+ wait = self._waiting.get(key)
1437
+ if wait is None:
1438
+ self._wait_seq += 1
1439
+ wait = Waiting(seq=self._wait_seq, since=time.monotonic())
1440
+ self._waiting[key] = wait
1441
+ return Decision(
1442
+ Action.QUEUED,
1443
+ f"round {round_} deferred — {live}/{limit} reviewer sessions live "
1444
+ f"(waiting {int(time.monotonic() - wait.since)}s)",
1445
+ round_,
1446
+ )
1447
+
1448
+ def _live_session_count(self) -> int | None:
1449
+ """Own-grammar reviewer sessions live this pass, or None if unknown.
1450
+
1451
+ Seeded by the reap sweep, which lists them anyway, so the gate costs no
1452
+ extra `alissa tmux ls` in the daemon. A caller that reaches the gate
1453
+ without a sweep (a direct `evaluate`, or a pass whose sweep bailed
1454
+ early) probes once and memoizes for the pass.
1455
+
1456
+ None -- the list could not be read -- FAILS OPEN: the spawn proceeds.
1457
+ The alternative was tried on paper and rejected: `alissa tmux ls` is
1458
+ also the reaper's only input, so a CLI that cannot answer it means
1459
+ nothing is being reaped either, and refusing every spawn would turn one
1460
+ broken subprocess into a fleet-wide review outage while the container
1461
+ sat idle. Failing open restores exactly the pre-gate behaviour for as
1462
+ long as the outage lasts, and it is loud (once per pass) about doing
1463
+ so.
1464
+ """
1465
+ if not self._census_probed:
1466
+ self._census_probed = True
1467
+ try:
1468
+ self._session_census = len(self.alissa.list_review_sessions())
1469
+ except CommandError as exc:
1470
+ self._session_census = None
1471
+ log.warning(
1472
+ "spawn gate: could not count live reviewer sessions (%s) — "
1473
+ "allowing spawns this pass rather than stalling the loop on "
1474
+ "a session list the reaper cannot read either",
1475
+ exc,
1476
+ )
1477
+ self._census_warned = True
1478
+ if self._session_census is None and not self._census_warned:
1479
+ self._census_warned = True
1480
+ log.warning(
1481
+ "spawn gate: the live reviewer-session count is unknown this "
1482
+ "pass — allowing spawns (the gate is back-pressure, not a "
1483
+ "safety interlock)"
1484
+ )
1485
+ return self._session_census
1486
+
1487
+ def _gate_order(
1488
+ self, requests: list[tuple[str, str, int]]
1489
+ ) -> list[tuple[str, str, int]]:
1490
+ """This pass's candidates, oldest waiter first.
1491
+
1492
+ FIFO fairness, and it needs its own order because the walk order it
1493
+ replaces is NOT fair: `review_requests` is a `search/issues` query, and
1494
+ that API sorts by best match by default -- a relevance ranking with no
1495
+ relation to how long a round has been waiting, and not even stable
1496
+ between calls. Under a gate that hands out one slot per freed session,
1497
+ an unstable order starves whichever PR keeps losing the coin toss.
1498
+
1499
+ So: everything that has already been deferred goes first, in the order
1500
+ it was FIRST deferred (`Waiting.seq`), then everything else in the
1501
+ search's own order. A round that has waited two passes therefore takes
1502
+ the next free slot ahead of one that has waited none -- which is the
1503
+ whole anti-starvation guarantee -- and a pass with no deferrals is
1504
+ byte-for-byte the old walk.
1505
+ """
1506
+ if not self._waiting:
1507
+ return requests
1508
+ return sorted(
1509
+ requests,
1510
+ key=lambda item: (
1511
+ self._waiting[(f"{item[0]}/{item[1]}", item[2])].seq
1512
+ if (f"{item[0]}/{item[1]}", item[2]) in self._waiting
1513
+ else self._wait_seq + 1
1514
+ ),
1515
+ )
1516
+
1517
+ def _note_deferrals(self, results: list[tuple[str, Decision]]) -> None:
1518
+ """One streak-limited line per pass summarizing the gate.
1519
+
1520
+ INFO while the queue MOVES -- a full container that keeps handing out
1521
+ freed slots is the gate working, and paging on it would make normal
1522
+ load indistinguishable from a leak. It escalates to WARNING on a
1523
+ different predicate: passes that deferred and spawned NOTHING,
1524
+ consecutively, for longer than POLL_ESCALATE_SECONDS. That is not
1525
+ back-pressure, it is a review outage the reap alarm can miss entirely
1526
+ (see DEFERRAL_STALLED), and this module's own rule for it is stated at
1527
+ _note_ledger_unwritable: a daemon that is up, polling, and deciding
1528
+ nothing is the most misleading state it can be in.
1529
+
1530
+ The two are separate streaks on purpose. One spawn proves the queue is
1531
+ moving and clears the stall, but it does NOT end the deferral episode
1532
+ the log limiter is rationing -- collapsing them would either page on a
1533
+ draining wave or reset the limiter every pass and log one line per
1534
+ poll, which is the spam the summary exists to avoid.
1535
+ """
1536
+ now = time.monotonic()
1537
+ held = [(slug, d) for slug, d in results if d.action is Action.QUEUED]
1538
+ if not held:
1539
+ self._gate_stall.clear()
1540
+ ended = self._gate_streak.resolve(now)
1541
+ if ended is not None:
1542
+ passes, seconds = ended
1543
+ log.info(DEFERRAL_CLEARED, passes, seconds / 60)
1544
+ return
1545
+
1546
+ should_log, _ = self._gate_streak.record(now)
1547
+ if any(d.action is Action.SPAWNED for _, d in results):
1548
+ # The queue is draining: rounds waited, but a slot freed and the
1549
+ # oldest waiter took it. Whatever the backlog, this is not a stall.
1550
+ # A stall that had ESCALATED says so out loud on its way out, past
1551
+ # the streak limit -- otherwise the last word an operator has on
1552
+ # the gate is a WARNING the daemon has already recovered from.
1553
+ escalated = self._gate_stall.escalated
1554
+ ended = self._gate_stall.resolve(now)
1555
+ if escalated and ended is not None:
1556
+ passes, seconds = ended
1557
+ log.info(GATE_STALL_CLEARED, passes, seconds / 60, len(held))
1558
+ else:
1559
+ _, crossing = self._gate_stall.record(now)
1560
+ # The crossing BYPASSES the streak limit, on the same reasoning as
1561
+ # the ledger gate's: it is a state change, and suppressing it would
1562
+ # hide the one transition the line exists to show.
1563
+ should_log = should_log or crossing
1564
+ if not should_log:
1565
+ return
1566
+ # A round is only ever deferred against a census the gate could read,
1567
+ # so this is never the fallback -- an unreadable count fails open and
1568
+ # produces no deferrals at all.
1569
+ live = self._session_census or 0
1570
+ waiting = "Waiting, oldest first: " + ", ".join(
1571
+ slug for slug, _ in sorted(
1572
+ held,
1573
+ key=lambda item: self._waiting.get(
1574
+ _slug_key(item[0]), Waiting(seq=0, since=0.0)
1575
+ ).seq,
1576
+ )
1577
+ ) + "."
1578
+ if self._gate_stall.escalated:
1579
+ log.warning(
1580
+ DEFERRAL_STALLED,
1581
+ len(held),
1582
+ live,
1583
+ self.config.max_concurrent_sessions,
1584
+ self._gate_stall.held(now) / 60,
1585
+ waiting,
1586
+ )
1587
+ return
1588
+ log.info(
1589
+ DEFERRAL_SUMMARY,
1590
+ len(held),
1591
+ live,
1592
+ self.config.max_concurrent_sessions,
1593
+ waiting,
1594
+ )
1595
+
1254
1596
  # -- native verdict post -----------------------------------------------
1255
1597
 
1256
1598
  def _close_round_natively(
@@ -2402,6 +2744,10 @@ class ReviewWatcher:
2402
2744
  sessions = self.alissa.list_review_sessions()
2403
2745
  except CommandError as exc:
2404
2746
  log.warning("reap sweep skipped: could not list sessions: %s", exc)
2747
+ # Probed and unanswerable. The spawn gate reads the same list, so
2748
+ # it is told the count is unknown rather than left to pay for a
2749
+ # second failing subprocess per candidate PR.
2750
+ self._census_probed, self._session_census = True, None
2405
2751
  return 0
2406
2752
 
2407
2753
  # Drop cached resolutions for sessions that are gone. Names are unique
@@ -2422,6 +2768,13 @@ class ReviewWatcher:
2422
2768
  completed_cache: dict[tuple[str, int, str | None], float | None] = {}
2423
2769
  holdouts: dict[str, str] = {}
2424
2770
  reaped: list[str] = []
2771
+ # Dry-run only: what a production sweep would have killed here. The
2772
+ # spawn gate's census has to be seeded with production's slot count or
2773
+ # `--dry-run` reports rounds QUEUED that production would spawn -- the
2774
+ # opposite error to the one _spawn's self-increment fixes, in the one
2775
+ # tool an operator reaches for during exactly this incident class (PR
2776
+ # #71 round-1 [minor]). Empty in production, where `reaped` carries it.
2777
+ would_reap: list[str] = []
2425
2778
 
2426
2779
  for ses in sessions:
2427
2780
  idle_for = time.time() - ses.last_activity
@@ -2469,6 +2822,7 @@ class ReviewWatcher:
2469
2822
  if self.config.dry_run:
2470
2823
  log.info("[dry-run] would reap reviewer session %s (%s)", ses.name, evidence)
2471
2824
  holdouts[ses.name] = f"dry-run: would have reaped ({evidence})"
2825
+ would_reap.append(ses.name)
2472
2826
  continue
2473
2827
  try:
2474
2828
  self.alissa.kill_session(ses.name)
@@ -2483,7 +2837,7 @@ class ReviewWatcher:
2483
2837
  reaped.append(ses.name)
2484
2838
  log.info("reaped reviewer session %s (%s)", ses.name, evidence)
2485
2839
 
2486
- self._check_session_cap(sessions, reaped, holdouts)
2840
+ self._check_session_cap(sessions, reaped, holdouts, would_reap)
2487
2841
  return len(reaped)
2488
2842
 
2489
2843
  def _hold(self, holdouts: dict[str, str], ses: ManagedSession, why: str) -> None:
@@ -2646,6 +3000,7 @@ class ReviewWatcher:
2646
3000
  sessions: list[ManagedSession],
2647
3001
  reaped: list[str],
2648
3002
  holdouts: dict[str, str],
3003
+ would_reap: list[str] | None = None,
2649
3004
  ) -> None:
2650
3005
  """Page-worthy log when the sweep is not keeping up.
2651
3006
 
@@ -2673,6 +3028,23 @@ class ReviewWatcher:
2673
3028
  """
2674
3029
  killed = set(reaped)
2675
3030
  remaining = sorted(s.name for s in sessions if s.name not in killed)
3031
+ # The spawn gate's census for this pass, taken from the same post-sweep
3032
+ # list the alarm counts: what the sweep could not free is exactly what
3033
+ # is still holding CPU when the gate decides. Seeded here so the gate
3034
+ # never runs its own `alissa tmux ls`.
3035
+ #
3036
+ # `would_reap` is the dry-run correction and nothing else: a dry-run
3037
+ # sweep kills nothing, so without it the census counts slots production
3038
+ # would have freed and the diagnostic reports rounds gated that
3039
+ # production spawns. Deliberately NOT subtracted from `remaining`
3040
+ # below -- the alarm's own count is documented as "sessions this pass
3041
+ # could not free", and a dry-run pass genuinely freed none of them
3042
+ # (each carries a `dry-run: would have reaped` holdout reason that says
3043
+ # so). That reading predates the gate; the census is the part the gate
3044
+ # owns.
3045
+ pretend = set(would_reap or ())
3046
+ self._census_probed = True
3047
+ self._session_census = sum(1 for name in remaining if name not in pretend)
2676
3048
  if len(remaining) <= self.config.reap_session_cap:
2677
3049
  self._paged_cap = None
2678
3050
  return
@@ -2804,6 +3176,16 @@ class ReviewWatcher:
2804
3176
  dry_run=self.config.dry_run,
2805
3177
  )
2806
3178
 
3179
+ # This pass's own spawns count against the gate immediately (issue
3180
+ # #70). The census is read once per pass and a just-enqueued session
3181
+ # takes a moment to appear in `alissa tmux ls`, so without this a
3182
+ # merge wave would hand one free slot to every PR in it -- the
3183
+ # unbounded burst the gate exists to prevent, reintroduced by a
3184
+ # caching detail. Incremented in dry-run too: the diagnostic's job is
3185
+ # to report the decisions production would take.
3186
+ if self._session_census is not None:
3187
+ self._session_census += 1
3188
+
2807
3189
  if not self.config.dry_run:
2808
3190
  self.state.record_spawn(
2809
3191
  repo=pr.full_name,
@@ -3104,6 +3486,14 @@ class ReviewWatcher:
3104
3486
  # to this one's title searches as if it were current, and it is the one
3105
3487
  # piece of per-pass state whose staleness would be invisible.
3106
3488
  self._pass_tasks = None
3489
+ # ...and so does the spawn gate's session census: a count carried over
3490
+ # from the previous pass would gate this one on sessions that have
3491
+ # since been reaped (or miss ones that have since spawned). Cleared
3492
+ # with `_pass_tasks` and for the same reason. The `_waiting` queue is
3493
+ # deliberately NOT cleared -- it is what remembers who has waited
3494
+ # longest ACROSS passes.
3495
+ self._session_census = None
3496
+ self._census_probed = self._census_warned = False
3107
3497
 
3108
3498
  # THE LEDGER GATE (issue #62, PR #63 round-1 blocker). Nothing below
3109
3499
  # may run when the ledger cannot record what it does.
@@ -3162,8 +3552,17 @@ class ReviewWatcher:
3162
3552
  requests = self.github.review_requests(self.config.repos)
3163
3553
  log.info("%d PR(s) with a review pending from %s", len(requests), self.github.login)
3164
3554
 
3555
+ # Forget queue places belonging to PRs this pass cannot act on at all
3556
+ # (merged, converged, withdrawn): they are gone from the search, so
3557
+ # nothing would ever clear them and the oldest-first order would be
3558
+ # anchored to a PR that is never coming back.
3559
+ live_keys = {(f"{owner}/{repo}", number) for owner, repo, number in requests}
3560
+ self._waiting = {
3561
+ key: wait for key, wait in self._waiting.items() if key in live_keys
3562
+ }
3563
+
3165
3564
  results = []
3166
- for owner, repo, number in requests:
3565
+ for owner, repo, number in self._gate_order(requests):
3167
3566
  slug = f"{owner}/{repo}#{number}"
3168
3567
  if not self.config.watches(f"{owner}/{repo}"):
3169
3568
  continue
@@ -3179,6 +3578,8 @@ class ReviewWatcher:
3179
3578
  log.log(level, "%s → %s (%s)", slug, decision.action.value, decision.reason)
3180
3579
  results.append((slug, decision))
3181
3580
 
3581
+ self._note_deferrals(results)
3582
+
3182
3583
  # Persist one poll_snapshots row per pass, built entirely from the
3183
3584
  # Decision list already in hand plus the reap count -- no new GitHub
3184
3585
  # calls. Written in dry-run too: a snapshot OBSERVES the pass, it is
@@ -3289,6 +3690,7 @@ class ReviewWatcher:
3289
3690
  if d.action is Action.IN_FLIGHT and d.deferred
3290
3691
  )
3291
3692
  self.state.record_snapshot(
3693
+ queued=counts[Action.QUEUED],
3292
3694
  duration_ms=duration_ms,
3293
3695
  candidates=len(results),
3294
3696
  spawned=spawned,
@@ -188,6 +188,12 @@ CREATE TABLE IF NOT EXISTS poll_snapshots (
188
188
  posted INTEGER NOT NULL DEFAULT 0,
189
189
  awaiting_post INTEGER NOT NULL DEFAULT 0,
190
190
  abandoned INTEGER NOT NULL DEFAULT 0,
191
+ -- Rounds the spawn gate held back this pass because the container was
192
+ -- already at max_concurrent_sessions (issue #70). Its own column rather
193
+ -- than a share of `deferred`: that one counts rounds whose LIVE session is
194
+ -- still working, and the console reads the two together as active
195
+ -- sessions -- a gated round has no session at all.
196
+ queued INTEGER NOT NULL DEFAULT 0,
191
197
  stages_json TEXT NOT NULL
192
198
  );
193
199
  """
@@ -204,6 +210,7 @@ _ADDED_COLUMNS = {
204
210
  ("posted", "INTEGER NOT NULL DEFAULT 0"),
205
211
  ("awaiting_post", "INTEGER NOT NULL DEFAULT 0"),
206
212
  ("abandoned", "INTEGER NOT NULL DEFAULT 0"),
213
+ ("queued", "INTEGER NOT NULL DEFAULT 0"),
207
214
  ),
208
215
  "verdict_posts": (
209
216
  ("checks_held_at", "INTEGER"),
@@ -970,6 +977,7 @@ class State:
970
977
  posted: int = 0,
971
978
  awaiting_post: int = 0,
972
979
  abandoned: int = 0,
980
+ queued: int = 0,
973
981
  stages: list[dict],
974
982
  ) -> bool:
975
983
  """Append one poll-pass observation, then prune to the newest
@@ -1004,6 +1012,7 @@ class State:
1004
1012
  posted=posted,
1005
1013
  awaiting_post=awaiting_post,
1006
1014
  abandoned=abandoned,
1015
+ queued=queued,
1007
1016
  stages=stages,
1008
1017
  ),
1009
1018
  "poll snapshot",
@@ -1026,6 +1035,7 @@ class State:
1026
1035
  posted: int,
1027
1036
  awaiting_post: int,
1028
1037
  abandoned: int,
1038
+ queued: int,
1029
1039
  stages: list[dict],
1030
1040
  ) -> None:
1031
1041
  """The snapshot INSERT + prune itself, strict. Split out so the
@@ -1034,8 +1044,8 @@ class State:
1034
1044
  "INSERT INTO poll_snapshots "
1035
1045
  "(ts, duration_ms, candidates, spawned, stale_reenqueued, "
1036
1046
  "in_flight, deferred, converged, capped, escalated, skipped, "
1037
- "reaped, posted, awaiting_post, abandoned, stages_json) "
1038
- "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
1047
+ "reaped, posted, awaiting_post, abandoned, queued, stages_json) "
1048
+ "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
1039
1049
  (
1040
1050
  int(time.time()),
1041
1051
  duration_ms,
@@ -1052,6 +1062,7 @@ class State:
1052
1062
  posted,
1053
1063
  awaiting_post,
1054
1064
  abandoned,
1065
+ queued,
1055
1066
  json.dumps(stages, separators=(",", ":")),
1056
1067
  ),
1057
1068
  )
@@ -204,7 +204,7 @@ section.panel {
204
204
  border-color: var(--status-in-progress); background: var(--status-in-progress-bg); }
205
205
  .pill.in-flight { color: var(--status-committed);
206
206
  border-color: var(--status-committed); background: var(--status-committed-bg); }
207
- .pill.deferred { color: var(--status-pending);
207
+ .pill.deferred, .pill.queued { color: var(--status-pending);
208
208
  border-color: var(--status-pending); background: var(--status-pending-bg); }
209
209
  .pill.converged, .pill.posted { color: var(--status-in-progress);
210
210
  border-color: var(--status-in-progress); background: var(--status-in-progress-bg); }
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: alissa-tools-github-revloop
3
- Version: 0.16.12
3
+ Version: 0.16.14
4
4
  Summary: ALISSA-TOOLS-GITHUB-REVLOOP
5
5
  Home-page: https://alissa.app
6
6
  Author: Fahera