autonomous-sdlc-harness 0.4.0 → 0.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -101,20 +101,25 @@
101
101
  # newest events across every live run and applies it to ALL of them, through the
102
102
  # ordinary pause protocol and nothing else:
103
103
  #
104
- # * IT NEVER KILLS A RUN, and it never invents a mechanism. It drops
105
- # `<state_dir>/PAUSE` into each running working copy, tags the record
106
- # `paused_by=usage` and records `usage_resume_at`; the engine yields at its
104
+ # * IT NEVER KILLS A RUN, and it never invents a mechanism. It tags each running
105
+ # record `paused_by=usage` with its `usage_resume_at`, then drops
106
+ # `<state_dir>/PAUSE` into its working copy; the engine yields at its
107
107
  # next clean checkpoint, writes PAUSE_ACK, and classify_run_exit marks it
108
108
  # `paused` — the same path a hand-dropped PAUSE takes. Once the window has
109
109
  # reset the gate drops `<state_dir>/RESUME`, and the pause-resume pass above
110
110
  # re-launches the run with no further involvement from here.
111
+ # * A TAGGED PAUSE WHOSE RESET TIME IS MISSING IS GIVEN THE FALLBACK — the
112
+ # same hour the pause side assumes when no reset was reported — written,
113
+ # logged and reported once, and it then resumes on the wall clock like any
114
+ # other.
111
115
  # * A RUN PAUSED BY HAND IS NEVER AUTO-RESUMED. The resume side acts on the
112
116
  # `paused_by=usage` tag alone, and a hand pause carries no tag.
113
117
  # * WHILE A PAUSE IS IN EFFECT THE HOLD MARKER IS UP, which is what defers a
114
118
  # fresh inbox drop (it stays in the inbox) and skips the watchdog above:
115
119
  # launching into a full window spends a run on an immediate refusal. A
116
120
  # REMOTE drop is dispatched through the hold — the job gates itself (see
117
- # REMOTE DISPATCH).
121
+ # REMOTE DISPATCH). The hold is bounded by each pause's recorded or repaired
122
+ # `usage_resume_at`: it comes down on the pass that drops RESUME.
118
123
  # * THE WINDOW TYPES ARE ASSESSED INDEPENDENTLY — the 5-hour one and the
119
124
  # rolling weekly one — so a 5-hour window that has just reset cannot mask a
120
125
  # weekly window sitting at its cap. The worst state across every window of
@@ -349,11 +354,17 @@
349
354
  # advances it to the epoch taken just before its query; a failed poll
350
355
  # advances nothing and pauses nothing.
351
356
  # * THE DECISION, when the run leaves `running`: `paused` for `budget` ->
352
- # `continue`; for `user` -> `stop`; by the usage gate (`usage`) -> WAIT IN
353
- # THE JOB, the gate's own auto-resume and resume_paused_runs relaunching it,
354
- # when the reset falls before HARNESS_JOB_DEADLINE_EPOCH and the runner is
355
- # self-hosted or the wait is at most REMOTE_WAIT_MAX_SECS, else
356
- # `wait-poller`; with no pause requested — the run's own API-overload
357
+ # `continue`; for `user` -> `stop`; by the usage gate (`usage`) -> a lost
358
+ # `usage_resume_at` is first given the gate's fallback and reported as
359
+ # assumed, then WAIT IN THE JOB, the gate's own auto-resume and
360
+ # resume_paused_runs relaunching it, when the reset falls before
361
+ # HARNESS_JOB_DEADLINE_EPOCH and the runner is self-hosted or the wait is at
362
+ # most REMOTE_WAIT_MAX_SECS, else `wait-poller`. A wait still paused
363
+ # REMOTE_WAIT_MAX_SECS past the gate's last chance to resume it — the later
364
+ # of that reset and its own start, plus USAGE_CHECK_INTERVAL_SECS and
365
+ # POLL_INTERVAL_SECS, the most the throttled gate can lag the reset — ends
366
+ # in `wait-poller` with one notification, on every runner; with no pause
367
+ # requested — the run's own API-overload
357
368
  # self-pause (`overload`) -> auto-resume, else `stop`. `failed` ->
358
369
  # auto-resume unless the stall watchdog gave up, else `stop`. Every other
359
370
  # status -> `stop`. An empty HARNESS_JOB_DEADLINE_EPOCH bounds nothing.
@@ -1024,6 +1035,11 @@ USAGE_WARNING_DEBOUNCE="${USAGE_WARNING_DEBOUNCE:-2}"
1024
1035
  # than at it: the reported instant is the account's, not this machine's, and a
1025
1036
  # resume that lands a moment early is refused and costs the run its session.
1026
1037
  USAGE_RESUME_MARGIN_SECS="${USAGE_RESUME_MARGIN_SECS:-120}"
1038
+ # The resume time the gate assumes, from now, for a pause with no reported reset
1039
+ # and for a usage-paused record whose `usage_resume_at` was lost. Assigned
1040
+ # plainly, with no environment override: it is a guess at a missing value, not a
1041
+ # policy knob.
1042
+ USAGE_FALLBACK_RESUME_SECS=3600
1027
1043
  # The weekly window's own trigger threshold, as a fraction of its reported
1028
1044
  # utilization. Its `allowed_warning` fires from about half the weekly budget
1029
1045
  # onward — informational, not a signal that anything is about to be refused — so
@@ -1219,8 +1235,9 @@ notify() {
1219
1235
  # gate, and empty otherwise — which is the whole of how a
1220
1236
  # gate pause is told apart from a hand-dropped one. A hand
1221
1237
  # pause is never auto-resumed precisely because it has no
1222
- # value here. Written the moment the PAUSE is REQUESTED,
1223
- # while the record is still `running`, and cleared by a
1238
+ # value here. Written together with `usage_resume_at` in
1239
+ # one write, BEFORE the PAUSE is dropped, while the record
1240
+ # is still `running`, and cleared by a
1224
1241
  # real resume, by the gate's stale-tag sweep, and by
1225
1242
  # launch_run on a reused branch key — see the gate for why
1226
1243
  # clearing it any earlier than those strands the run.
@@ -1228,7 +1245,10 @@ notify() {
1228
1245
  # BINDING worst-state window reset (the overage window's
1229
1246
  # while `isUsingOverage`) plus USAGE_RESUME_MARGIN_SECS.
1230
1247
  # The ONLY state the wall-clock resume reads, and written
1231
- # and cleared together with `paused_by`.
1248
+ # and cleared together with `paused_by`, in one write. A
1249
+ # tagged `paused` record found without a usable value
1250
+ # has it written alone, by usage_resume_at_var's repair:
1251
+ # now plus USAGE_FALLBACK_RESUME_SECS.
1232
1252
  # remote_stopped_at the epoch second `remote-run.sh stop` sent the branch's
1233
1253
  # stop marker and asked GitHub to cancel its runs. Written
1234
1254
  # by `remote-run.sh stop` alone, only on an existing
@@ -1244,7 +1264,7 @@ notify() {
1244
1264
  # `user` job mode did, on a `harness pause <branch>` run;
1245
1265
  # `overload` nobody requested it — the run's own
1246
1266
  # API-overload self-pause. `user` and `budget` are written
1247
- # when the PAUSE is dropped, while still `running`;
1267
+ # before the PAUSE is dropped, while still `running`;
1248
1268
  # classify_run_exit settles the reason as the run pauses,
1249
1269
  # `user` first, then `usage`, then `budget`. Cleared when
1250
1270
  # job mode relaunches the run — plus `killed`, a registry-only
@@ -1290,7 +1310,8 @@ notify() {
1290
1310
  # -----------------------------------------------------------------------------
1291
1311
  # The bodies are lib/harness-run-lib.sh's THE RUN REGISTRY, shared with every
1292
1312
  # script that reads or writes this file; these wrappers bind them to $REGISTRY.
1293
- # registry_set <branch> <key> <value>; registry_get <branch> <key>.
1313
+ # registry_set <branch> <key> <value> [<key> <value> …] — every pair in one
1314
+ # write; registry_get <branch> <key>.
1294
1315
  registry_init() { hr_registry_init "$REGISTRY"; }
1295
1316
  registry_set() { hr_registry_set "$REGISTRY" "$@"; }
1296
1317
  registry_get() { hr_registry_get "$REGISTRY" "$@"; }
@@ -1697,9 +1718,14 @@ lane_release_if_idle() {
1697
1718
  # --effort "<agentEffort>" \
1698
1719
  # --output-format stream-json --verbose \
1699
1720
  # --add-dir <worktree> \
1700
- # --add-dir <MAIN_REPO>/<state_dir>
1721
+ # --add-dir <MAIN_REPO>/<state_dir> \
1722
+ # [--add-dir <dir> ...]
1701
1723
  #
1702
- # and NEVER a permission-bypass flag: the profile's deny floor is what keeps an
1724
+ # where the bracketed tail is JOB MODE ONLY: one --add-dir per
1725
+ # permissions.additionalDirectories entry of the profile, in file order, minus
1726
+ # empty entries and the two directories above; outside job mode the line ends at
1727
+ # the state directory. See spawn_engine for why the profile's list is repeated.
1728
+ # And NEVER a permission-bypass flag: the profile's deny floor is what keeps an
1703
1729
  # unattended run in its lane, and bypassing it makes every refusal decorative.
1704
1730
  # Both run-setting flags are CONDITIONAL: an unset key leaves its flag off the
1705
1731
  # line entirely rather than passing an empty argument.
@@ -1857,6 +1883,32 @@ ${GLOBAL_STOP}. End at 'branch ready for review' — never merge, never push to
1857
1883
  effort_args=(--effort "$AGENT_EFFORT")
1858
1884
  fi
1859
1885
 
1886
+ # JOB MODE ONLY: every permissions.additionalDirectories entry of the profile
1887
+ # (the plugin roots `init --plugin-root-entries` wrote) also goes on the line as
1888
+ # an --add-dir, because a job session was refused reads under a root the profile
1889
+ # file already granted. `init` stays the one producer of the list. File order;
1890
+ # an empty entry and one equal to the two directories already passed are
1891
+ # skipped. An absent, unparseable or keyless profile adds nothing and never
1892
+ # blocks the launch — `doctor --remote-job` refuses an unusable profile earlier.
1893
+ local extra_dir_args extra_dirs_logged
1894
+ extra_dir_args=()
1895
+ extra_dirs_logged=""
1896
+ if [ "$JOB_MODE" = "1" ]; then
1897
+ local profile_dirs profile_dir
1898
+ profile_dirs="$(jq -r '.permissions.additionalDirectories[]? // empty' "$SETTINGS_PROFILE" 2>/dev/null)" || profile_dirs=""
1899
+ while IFS= read -r profile_dir; do
1900
+ [ -n "$profile_dir" ] || continue
1901
+ case "${profile_dir%/}" in
1902
+ "${worktree%/}" | "${main_state%/}") continue ;;
1903
+ esac
1904
+ extra_dir_args+=(--add-dir "$profile_dir")
1905
+ extra_dirs_logged="${extra_dirs_logged} '${profile_dir}'"
1906
+ done <<EOF
1907
+ $profile_dirs
1908
+ EOF
1909
+ [ -z "$extra_dirs_logged" ] || log "job mode: '$branch' also gets --add-dir from the profile's additionalDirectories:${extra_dirs_logged}"
1910
+ fi
1911
+
1860
1912
  # The formatter is the tail of the pipeline; a passthrough keeps the raw events
1861
1913
  # in the log rather than breaking the pipe when it is not runnable.
1862
1914
  local formatter="$FORMAT_STREAM"
@@ -1889,7 +1941,10 @@ ${GLOBAL_STOP}. End at 'branch ready for review' — never merge, never push to
1889
1941
  cd "$worktree" || exit 97
1890
1942
  # `--add-dir "$worktree"` is NOT redundant with the profile: that file grants
1891
1943
  # the sibling-worktree glob through Edit/Write/Read rules, not through
1892
- # additionalDirectories.
1944
+ # additionalDirectories. The job-mode extras after the two fixed --add-dir
1945
+ # flags duplicate the profile's additionalDirectories on purpose: a job
1946
+ # session was refused reads under a directory that file granted, and a
1947
+ # launch flag does not depend on how the settings file is merged.
1893
1948
  #
1894
1949
  # The run is streamed as JSON events through the formatter so the per-run log
1895
1950
  # shows the orchestrator heartbeat and the sub-agent dispatches LIVE and
@@ -1908,7 +1963,8 @@ ${GLOBAL_STOP}. End at 'branch ready for review' — never merge, never push to
1908
1963
  ${effort_args[@]+"${effort_args[@]}"} \
1909
1964
  --output-format stream-json --verbose \
1910
1965
  --add-dir "$worktree" \
1911
- --add-dir "$main_state" 2>>"$log_path" |
1966
+ --add-dir "$main_state" \
1967
+ ${extra_dir_args[@]+"${extra_dir_args[@]}"} 2>>"$log_path" |
1912
1968
  tee -a "${log_path%.log}.stream.jsonl" |
1913
1969
  "$formatter" >>"$log_path"
1914
1970
  rc=${PIPESTATUS[0]}
@@ -1998,8 +2054,7 @@ launch_run() {
1998
2054
  registry_set "$branch" stall_restarts 0
1999
2055
  registry_set "$branch" stall_warned ""
2000
2056
  registry_set "$branch" stall_killing ""
2001
- registry_set "$branch" paused_by ""
2002
- registry_set "$branch" usage_resume_at ""
2057
+ registry_set "$branch" paused_by "" usage_resume_at ""
2003
2058
  registry_set "$branch" resume_kind ""
2004
2059
  registry_set "$branch" park_loop_cycles 0
2005
2060
 
@@ -2029,8 +2084,7 @@ launch_remote_run() {
2029
2084
  registry_set "$branch" stall_restarts 0
2030
2085
  registry_set "$branch" stall_warned ""
2031
2086
  registry_set "$branch" stall_killing ""
2032
- registry_set "$branch" paused_by ""
2033
- registry_set "$branch" usage_resume_at ""
2087
+ registry_set "$branch" paused_by "" usage_resume_at ""
2034
2088
  registry_set "$branch" resume_kind ""
2035
2089
  registry_set "$branch" park_loop_cycles 0
2036
2090
 
@@ -2171,8 +2225,7 @@ classify_run_exit() {
2171
2225
  reason=overload
2172
2226
  fi
2173
2227
  fi
2174
- registry_set "$branch" pause_reason "$reason"
2175
- registry_set "$branch" status paused
2228
+ registry_set "$branch" pause_reason "$reason" status paused
2176
2229
  log "run '$branch' paused (PAUSE honored, reason $reason) — rc=$rc"
2177
2230
  if [ "$reason" = "user" ]; then
2178
2231
  notify paused "$branch" "$log_path" "paused as you asked — run /autonomous-sdlc-harness:branch-resume $branch to continue"
@@ -3633,6 +3686,34 @@ usage_paused_count() {
3633
3686
  jq -r '[.runs | to_entries[] | select(.value.status=="paused" and .value.paused_by=="usage")] | length' "$REGISTRY" 2>/dev/null
3634
3687
  }
3635
3688
 
3689
+ # usage_resume_at_var <branch> — sets USAGE_RESUME_AT to the record's usable
3690
+ # `usage_resume_at` epoch and USAGE_RESUME_REPAIRED=0, returning 0. A record that
3691
+ # is `paused` with `paused_by=usage` and no usable value has LOST it: it gets
3692
+ # now + USAGE_FALLBACK_RESUME_SECS, written and logged, with
3693
+ # USAGE_RESUME_REPAIRED=1. Anything else — an untagged record above all, so a hand
3694
+ # pause is never given a time — is left alone, with USAGE_RESUME_AT empty and a
3695
+ # return of 1. Call it UNSUBSTITUTED: `log` writes to stdout.
3696
+ usage_resume_at_var() {
3697
+ local b="$1" raw
3698
+ USAGE_RESUME_AT=""
3699
+ USAGE_RESUME_REPAIRED=0
3700
+ raw="$(registry_get "$b" usage_resume_at)"
3701
+ case "$raw" in
3702
+ '' | *[!0-9]*) ;;
3703
+ *)
3704
+ USAGE_RESUME_AT="$((10#$raw))"
3705
+ return 0
3706
+ ;;
3707
+ esac
3708
+ [ "$(registry_get "$b" status)" = "paused" ] || return 1
3709
+ [ "$(registry_get "$b" paused_by)" = "usage" ] || return 1
3710
+ USAGE_RESUME_AT=$(($(date +%s) + USAGE_FALLBACK_RESUME_SECS))
3711
+ USAGE_RESUME_REPAIRED=1
3712
+ registry_set "$b" usage_resume_at "$USAGE_RESUME_AT"
3713
+ log "usage: '$b' is usage-paused with no usable usage_resume_at ('$raw') — assuming the gate's fallback, resume ~$(stall_human_time "$USAGE_RESUME_AT")"
3714
+ return 0
3715
+ }
3716
+
3636
3717
  # The gate itself, throttled to USAGE_CHECK_INTERVAL_SECS and run in three parts,
3637
3718
  # in this order: the resume side and the stale-tag cleanup, then the assessment
3638
3719
  # and the pauses it justifies, then the hold marker. Resume-before-pause is what
@@ -3649,7 +3730,7 @@ usage_gate() {
3649
3730
  # Every one of these is initialized, not merely declared: `set -u` makes a
3650
3731
  # DECLARED-BUT-UNSET name an error on first read, and several of the branches
3651
3732
  # below are reached without every name having been assigned in that iteration.
3652
- local b status pb ra wt state_rel="" state_abs=""
3733
+ local b status pb ra wt lp state_rel="" state_abs=""
3653
3734
  while IFS= read -r b; do
3654
3735
  [ -n "$b" ] || continue
3655
3736
  # The auto-resume side skips a remote run: its job gates itself, and
@@ -3689,17 +3770,31 @@ usage_gate() {
3689
3770
  # one case where the question cannot be ANSWERED, and there the tag stays.
3690
3771
  if [ -z "$wt" ] ||
3691
3772
  { [ -n "$state_abs" ] && [ ! -f "$state_abs/PAUSE" ] && [ ! -f "$state_abs/PAUSE_ACK" ]; }; then
3692
- registry_set "$b" paused_by ""
3693
- registry_set "$b" usage_resume_at ""
3773
+ registry_set "$b" paused_by "" usage_resume_at ""
3694
3774
  fi
3695
3775
  continue
3696
3776
  fi
3697
3777
 
3698
3778
  [ "$status" = "paused" ] || continue
3699
- ra="$(registry_get "$b" usage_resume_at)"
3700
- # No usable resume time is not a reason to resume: leave it paused and let an
3701
- # operator's own RESUME be the trigger, exactly as for a hand pause.
3702
- case "$ra" in '' | *[!0-9]*) continue ;; esac
3779
+ # A tagged pause with no usable resume time has LOST it; left alone it would
3780
+ # never resume and would hold the launch hold up for good. The helper gives it
3781
+ # the fallback time and this pass reports that once — the repaired value is
3782
+ # usable and in the future, so no later pass notifies again. In job mode a
3783
+ # lost value is left to run_job's `usage)` arm, which repairs it and owns the
3784
+ # job's notifications; repairing here as well would send two for one pause.
3785
+ if [ "$JOB_MODE" = "1" ]; then
3786
+ ra="$(registry_get "$b" usage_resume_at)"
3787
+ case "$ra" in '' | *[!0-9]*) continue ;; esac
3788
+ else
3789
+ usage_resume_at_var "$b" || continue
3790
+ ra="$USAGE_RESUME_AT"
3791
+ if [ "$USAGE_RESUME_REPAIRED" = "1" ]; then
3792
+ lp="$(registry_get "$b" log_path)"
3793
+ [ -n "$lp" ] || lp="$LOGS_DIR/$b.log"
3794
+ notify paused "$b" "$lp" "usage pause lost its recorded reset time — assuming ~$(stall_human_time "$ra") and resuming then; drop ${state_rel:-<state_dir>}/RESUME in ${wt:-its working copy} to resume sooner"
3795
+ continue
3796
+ fi
3797
+ fi
3703
3798
  [ "$now" -ge "$ra" ] || continue
3704
3799
  if [ -z "$wt" ] || [ ! -d "$wt" ] || [ -z "$state_abs" ]; then
3705
3800
  # The trigger cannot be placed where the run would read it. Leave BOTH tags
@@ -3714,8 +3809,7 @@ usage_gate() {
3714
3809
  # Cleared TOGETHER with the trigger: the pause-resume pass owns the relaunch
3715
3810
  # from here, and a tag left behind would make the next hand pause look like
3716
3811
  # this gate's.
3717
- registry_set "$b" paused_by ""
3718
- registry_set "$b" usage_resume_at ""
3812
+ registry_set "$b" paused_by "" usage_resume_at ""
3719
3813
  done <<EOF
3720
3814
  $(registry_branches)
3721
3815
  EOF
@@ -3773,7 +3867,7 @@ EOF
3773
3867
  # pauses would never be resumed by the wall clock. An hour is the fallback:
3774
3868
  # long enough not to thrash, short enough that a wrong guess costs one hour.
3775
3869
  case "$resume_at" in '' | *[!0-9]*) resume_at=0 ;; esac
3776
- [ "$resume_at" -le 0 ] && resume_at=$((now + 3600))
3870
+ [ "$resume_at" -le 0 ] && resume_at=$((now + USAGE_FALLBACK_RESUME_SECS))
3777
3871
  while IFS= read -r b; do
3778
3872
  [ -n "$b" ] || continue
3779
3873
  # The pause side skips a remote run: its job gates itself.
@@ -3793,11 +3887,14 @@ EOF
3793
3887
  log "usage auto-pause: the state directory in '$wt' is unresolvable — cannot pause '$b'"
3794
3888
  continue
3795
3889
  fi
3890
+ # Tagged BEFORE the engine acknowledges, on purpose — see invariant 2 — and
3891
+ # so before PAUSE exists: an engine acknowledging a PAUSE whose tag is not
3892
+ # yet recorded is classified `overload`. Both keys land in one write. A tag
3893
+ # whose `touch` then fails is a `running` record with no PAUSE, which the
3894
+ # stale-tag sweep in (1) clears on the next pass.
3895
+ registry_set "$b" paused_by usage usage_resume_at "$resume_at"
3796
3896
  mkdir -p "$wt/$state_rel" 2>/dev/null || true
3797
3897
  touch "$wt/$state_rel/PAUSE"
3798
- # Tagged BEFORE the engine acknowledges, on purpose — see invariant 2.
3799
- registry_set "$b" paused_by usage
3800
- registry_set "$b" usage_resume_at "$resume_at"
3801
3898
  log "usage auto-pause (state=$state, trigger=$USAGE_PAUSE_TRIGGER): dropped $state_rel/PAUSE in $wt (auto-resume ~$(stall_human_time "$resume_at"))"
3802
3899
  done <<EOF
3803
3900
  $(registry_branches)
@@ -3808,7 +3905,9 @@ EOF
3808
3905
  # --- (3) The launch hold: up while a usage pause is in effect OR being
3809
3906
  # initiated, down otherwise. Derived from the registry every pass rather than
3810
3907
  # toggled, so a marker left behind by a watcher that died mid-pause is cleared
3811
- # by the next one instead of holding the inbox forever.
3908
+ # by the next one instead of holding the inbox forever. A tagged pause stops
3909
+ # counting on the pass whose part (1) drops RESUME and clears both tags, so
3910
+ # the hold is bounded by the recorded or repaired `usage_resume_at`.
3812
3911
  local held
3813
3912
  held="$(usage_paused_count)"
3814
3913
  case "$held" in '' | *[!0-9]*) held=0 ;; esac
@@ -4006,8 +4105,8 @@ job_control_poll() {
4006
4105
  registry_set "$branch" control_polled_at "$before"
4007
4106
  if [ "$rc" = "0" ]; then
4008
4107
  JOB_USER_PAUSE_DROPPED=1
4009
- touch "$state_abs/PAUSE"
4010
4108
  registry_set "$branch" pause_reason user
4109
+ touch "$state_abs/PAUSE"
4011
4110
  log "job: a 'harness pause $branch' run was created at or after $since — dropped PAUSE (reason user)"
4012
4111
  fi
4013
4112
  job_write_status "$branch" "$remote_status" continue "job started"
@@ -4021,16 +4120,22 @@ job_budget_pass() {
4021
4120
  after="$(job_int "${REMOTE_SELF_PAUSE_AFTER_SECS:-}")" || return 0
4022
4121
  [ $(($(date +%s) - JOB_START_EPOCH)) -ge "$after" ] || return 0
4023
4122
  JOB_BUDGET_PAUSE_DROPPED=1
4024
- touch "$state_abs/PAUSE"
4025
4123
  [ "$(registry_get "$branch" pause_reason)" = "user" ] || registry_set "$branch" pause_reason budget
4124
+ touch "$state_abs/PAUSE"
4026
4125
  log "job: ${after}s of the hosted time budget have passed — dropped PAUSE (reason budget)"
4027
4126
  }
4028
4127
 
4029
- # job_usage_wait_ok <branch> — 0 when a usage pause is waited out in the job.
4030
- # An empty usage_resume_at means the gate has already dropped RESUME.
4128
+ # job_usage_wait_ok <branch> — 0 when a usage pause is waited out in the job;
4129
+ # leaves usage_resume_at_var's USAGE_RESUME_AT and USAGE_RESUME_REPAIRED set.
4130
+ # With no usable usage_resume_at: `paused_by=usage` still set means the value was
4131
+ # LOST, and the decision is made on the helper's repaired epoch — an hour out,
4132
+ # so `wait-poller` on a hosted runner at the default REMOTE_WAIT_MAX_SECS;
4133
+ # `paused_by` empty means the gate already dropped
4134
+ # RESUME, so 0, and the next resume_paused_runs relaunches the run.
4031
4135
  job_usage_wait_ok() {
4032
4136
  local ra deadline
4033
- ra="$(job_int "$(registry_get "$1" usage_resume_at)")" || return 0
4137
+ usage_resume_at_var "$1" || return 0
4138
+ ra="$USAGE_RESUME_AT"
4034
4139
  deadline="$(job_int "${HARNESS_JOB_DEADLINE_EPOCH:-}")" || deadline=""
4035
4140
  if [ -n "$deadline" ] && [ "$ra" -ge "$deadline" ]; then
4036
4141
  return 1
@@ -4068,8 +4173,8 @@ job_auto_resume() {
4068
4173
  # is one-shot: put it back, so the relaunched session still yields at its next
4069
4174
  # clean checkpoint instead of running on until the step timeout kills it.
4070
4175
  if [ "$JOB_BUDGET_PAUSE_DROPPED" = "1" ]; then
4071
- touch "$state_abs/PAUSE"
4072
4176
  registry_set "$branch" pause_reason budget
4177
+ touch "$state_abs/PAUSE"
4073
4178
  log "job: the hosted time budget's PAUSE was pending at the resume of '$branch' after $why — re-dropped it"
4074
4179
  fi
4075
4180
  notify resumed "$branch" "$log_path" "automatic resume $count/$REMOTE_AUTO_RESUME_MAX after $why ($(job_label))"
@@ -4099,8 +4204,7 @@ run_job() {
4099
4204
  registry_set "$branch" stall_restarts 0
4100
4205
  registry_set "$branch" stall_warned ""
4101
4206
  registry_set "$branch" stall_killing ""
4102
- registry_set "$branch" paused_by ""
4103
- registry_set "$branch" usage_resume_at ""
4207
+ registry_set "$branch" paused_by "" usage_resume_at ""
4104
4208
  registry_set "$branch" resume_kind ""
4105
4209
  registry_set "$branch" park_loop_cycles 0
4106
4210
  registry_set "$branch" auto_resumes 0
@@ -4168,6 +4272,7 @@ run_job() {
4168
4272
 
4169
4273
  # The supervision loop: the header's JOB MODE block states each decision.
4170
4274
  local status reason ra when decision="stop" detail="" usage_waiting=0 restarts
4275
+ local wait_ok=1 usage_wait_start=0 usage_wait_ra=0
4171
4276
  while :; do
4172
4277
  status="$(registry_get "$branch" status)"
4173
4278
  case "$status" in
@@ -4194,25 +4299,49 @@ run_job() {
4194
4299
  break
4195
4300
  ;;
4196
4301
  usage)
4197
- ra="$(registry_get "$branch" usage_resume_at)"
4198
- when="the reset"
4199
- [ -n "$ra" ] && when="the reset at ~$(stall_human_time "$ra")"
4200
4302
  if [ "$usage_waiting" = "0" ]; then
4201
- if ! job_usage_wait_ok "$branch"; then
4303
+ # Decided BEFORE `ra` is read, so a lost value is repaired first and
4304
+ # both notifications name the repaired time.
4305
+ wait_ok=1
4306
+ job_usage_wait_ok "$branch" || wait_ok=0
4307
+ ra="$USAGE_RESUME_AT"
4308
+ when="the reset"
4309
+ [ -n "$ra" ] && when="the reset at ~$(stall_human_time "$ra")"
4310
+ [ "$USAGE_RESUME_REPAIRED" = "1" ] &&
4311
+ when="the assumed reset at ~$(stall_human_time "$ra") (the recorded reset time was lost)"
4312
+ if [ "$wait_ok" = "0" ]; then
4202
4313
  decision=wait-poller
4203
4314
  detail="usage limit reached; the resume poller resumes it after $when"
4204
4315
  notify paused "$branch" "$log_path" "usage limit reached — resumes automatically after $when"
4205
4316
  break
4206
4317
  fi
4207
4318
  usage_waiting=1
4319
+ # Captured once: the gate's own resume clears usage_resume_at later
4320
+ # in this same wait, and the bound below must not move with it.
4321
+ usage_wait_start="$(date +%s)"
4322
+ usage_wait_ra="$(job_int "$ra")" || usage_wait_ra=0
4323
+ [ "$usage_wait_ra" -gt "$usage_wait_start" ] || usage_wait_ra="$usage_wait_start"
4208
4324
  log "job: '$branch' is usage-paused — waiting in the job for $when"
4209
4325
  notify paused "$branch" "$log_path" "usage limit reached — waiting in the job for $when"
4210
4326
  fi
4211
4327
  sleep "$POLL_INTERVAL_SECS"
4212
4328
  usage_gate
4213
4329
  resume_paused_runs
4214
- if [ "$(registry_get "$branch" status)" = "running" ]; then
4330
+ status="$(registry_get "$branch" status)"
4331
+ if [ "$status" = "running" ]; then
4215
4332
  registry_set "$branch" pause_reason ""
4333
+ elif [ "$status" = "paused" ] &&
4334
+ [ "$(date +%s)" -gt $((usage_wait_ra + USAGE_CHECK_INTERVAL_SECS + POLL_INTERVAL_SECS + REMOTE_WAIT_MAX_SECS)) ]; then
4335
+ # The bound on the wait itself, self-hosted included. It is measured
4336
+ # from the gate's last chance to resume, not from the reset: the gate
4337
+ # is throttled to USAGE_CHECK_INTERVAL_SECS and runs after a
4338
+ # POLL_INTERVAL_SECS sleep, so its resume can lag the reset by both.
4339
+ # REMOTE_WAIT_MAX_SECS past that with no resume, the job hands the
4340
+ # run over rather than waiting on nothing.
4341
+ decision=wait-poller
4342
+ detail="usage limit reached; the in-job usage wait passed its bound (REMOTE_WAIT_MAX_SECS=${REMOTE_WAIT_MAX_SECS}s past the reset) without a resume"
4343
+ notify paused "$branch" "$log_path" "usage limit: the in-job wait passed ${REMOTE_WAIT_MAX_SECS}s after the reset without a resume — run /autonomous-sdlc-harness:branch-resume $branch to continue"
4344
+ break
4216
4345
  fi
4217
4346
  ;;
4218
4347
  *)