autonomous-sdlc-harness 0.4.0 → 0.4.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -38,6 +38,22 @@
38
38
  # runtime sits in `retrieval/`
39
39
  # cli/src/retrieval/runtime.ts RETRIEVAL_CACHE_DIRNAME, that `retrieval/`
40
40
  #
41
+ # ACTION PINS.
42
+ # actions/checkout@v5
43
+ # actions/setup-node@v5
44
+ # actions/cache/restore@v5
45
+ # actions/cache@v5
46
+ # actions/upload-artifact@v6
47
+ # Each is the lowest major whose own action.yml declares `runs.using: node24`,
48
+ # per a maintainer's lookups of each action's action.yml and release notes on
49
+ # 2026-09-29T08:21Z (UTC). Each of those majors' release notes requires Actions
50
+ # Runner 2.327.1 or newer; a GitHub-hosted runner already has it, and a
51
+ # self-hosted one must run it (docs/remote-execution.md, section 8). A major
52
+ # tag, not a commit sha: every action here is GitHub's own `actions/`
53
+ # organisation, the owner of the runner and GITHUB_TOKEN this job already
54
+ # trusts, and a major tag takes the action's own patch and security releases
55
+ # where a sha would freeze your copy. Pin shas in your copy if you want them.
56
+ #
41
57
  # WHAT IT READS.
42
58
  # Secrets: CLAUDE_CODE_OAUTH_TOKEN and/or ANTHROPIC_API_KEY (one is required;
43
59
  # billing follows the API key when both are set), HARNESS_GIT_TOKEN (optional:
@@ -72,12 +88,12 @@
72
88
  # the step timeout. The job-level timeout-minutes is the self-hosted limit; a
73
89
  # hosted job is stopped at its own limit whatever that says.
74
90
  #
75
- # WHY THE TIMEOUT IS COMPUTED INTO GITHUB_ENV. Whether a step's timeout-minutes
76
- # accepts an expression over `runner.environment` could not be checked against
77
- # GitHub's workflow-syntax documentation when this file was written (no network
78
- # access), so the first step computes every budget value from
79
- # `runner.environment` into GITHUB_ENV, and the harness step reads a single
80
- # `env` value. Gate 12 records the real behaviour.
91
+ # WHY THE TIMEOUT IS COMPUTED INTO GITHUB_ENV. The first step computes every
92
+ # budget value from `runner.environment` into GITHUB_ENV, so the harness step's
93
+ # timeout-minutes reads a single `env` value rather than an expression over the
94
+ # runner kind. Gate 12 round 2, on 2026-09-29, found every harness-run.yml run
95
+ # accepted and its `Run the harness` step run: the expression-valued
96
+ # timeout-minutes is accepted.
81
97
  #
82
98
  # THE PLUGIN PIN. `claude plugin marketplace add --help` and
83
99
  # `claude plugin install --help` (Claude Code 2.1.282) offer no ref or version:
@@ -88,17 +104,19 @@
88
104
  # rendered CLI version.
89
105
  #
90
106
  # WHY retention-days IS SET. The `harness-state` bundle is the only remote copy
91
- # of a run's clarifications and carried counts; a run parked or paused longer
92
- # than its retention cannot be answered and loses those counts. 400 is the
107
+ # of a run's clarifications, uncommitted planning drafts and carried counts; a
108
+ # run parked or paused longer than its retention cannot be answered and loses
109
+ # those drafts and counts. 400 is the
93
110
  # largest retention any repository can configure, and `actions/upload-artifact`
94
111
  # caps a larger `retention-days` at the repository's own maximum, so the value
95
112
  # means "as long as this repository allows" — the same bound as leaving it
96
113
  # unset. It is explicit so that lowering it is a visible edit. The real bound is
97
114
  # the repository's Artifact and log retention setting, which
98
- # `autonomous-sdlc-harness doctor --check-github` reads. The cap (rather than a
99
- # rejection) is carried from the action's `@actions/artifact` retention code
100
- # and could not be re-checked against its v4 README when this file was written
101
- # (no network access); Gate 12 records the real behaviour.
115
+ # `autonomous-sdlc-harness doctor --check-github` reads. Gate 12 round 2, on
116
+ # 2026-09-29, observed the cap rather than a rejection on `actions/upload-artifact@v4`:
117
+ # the artifact's `expires_at` fell 90 days after its creation, with the
118
+ # repository's artifact-and-log-retention at {"days":90,"maximum_allowed_days":400}.
119
+ # That was v4; the v6 pinned above has not yet been observed by a gate round.
102
120
 
103
121
  name: harness-run
104
122
  run-name: harness ${{ inputs.action }} ${{ inputs.branch }}
@@ -248,7 +266,7 @@ jobs:
248
266
 
249
267
  - name: Check out the run's branch
250
268
  if: env.HARNESS_STOPPED != '1'
251
- uses: actions/checkout@v4
269
+ uses: actions/checkout@v5
252
270
  with:
253
271
  ref: ${{ inputs.branch }}
254
272
  fetch-depth: 0
@@ -280,9 +298,11 @@ jobs:
280
298
 
281
299
  - name: Set up Node
282
300
  if: env.HARNESS_STOPPED != '1'
283
- uses: actions/setup-node@v4
301
+ uses: actions/setup-node@v5
284
302
  with:
285
303
  node-version: '22'
304
+ # v5 caches by package.json's packageManager; the job installs nothing through npm's cache, and a lockfile-less repository must not fail this step.
305
+ package-manager-cache: false
286
306
 
287
307
  - name: Install the claude CLI when absent
288
308
  if: env.HARNESS_STOPPED != '1'
@@ -327,7 +347,7 @@ jobs:
327
347
 
328
348
  - name: Restore the docs-retrieval cache
329
349
  if: env.HARNESS_STOPPED != '1' && steps.config.outputs.retrieval == 'true'
330
- uses: actions/cache/restore@v4
350
+ uses: actions/cache/restore@v5
331
351
  with:
332
352
  path: ${{ env.HARNESS_RETRIEVAL_CACHE }}
333
353
  key: harness-retrieval-${{ runner.os }}-${{ env.HARNESS_CLI_VERSION }}
@@ -345,7 +365,7 @@ jobs:
345
365
 
346
366
  - name: Preflight with doctor
347
367
  if: env.HARNESS_STOPPED != '1'
348
- run: npx --yes "autonomous-sdlc-harness@$HARNESS_CLI_VERSION" doctor
368
+ run: npx --yes "autonomous-sdlc-harness@$HARNESS_CLI_VERSION" doctor --remote-job
349
369
 
350
370
  - name: Bootstrap the checkout
351
371
  if: env.HARNESS_STOPPED != '1'
@@ -377,7 +397,7 @@ jobs:
377
397
 
378
398
  - name: Upload the state bundle
379
399
  if: always() && env.SCRIPTS_DIR != ''
380
- uses: actions/upload-artifact@v4
400
+ uses: actions/upload-artifact@v6
381
401
  with:
382
402
  name: harness-state
383
403
  path: ${{ runner.temp }}/harness-state
@@ -411,7 +431,7 @@ jobs:
411
431
 
412
432
  - name: Check out the dispatched ref
413
433
  if: env.HARNESS_STOPPED != '1'
414
- uses: actions/checkout@v4
434
+ uses: actions/checkout@v5
415
435
 
416
436
  - name: Read the configuration
417
437
  id: config
@@ -430,13 +450,15 @@ jobs:
430
450
 
431
451
  - name: Set up Node
432
452
  if: env.HARNESS_STOPPED != '1' && steps.config.outputs.retrieval == 'true'
433
- uses: actions/setup-node@v4
453
+ uses: actions/setup-node@v5
434
454
  with:
435
455
  node-version: '22'
456
+ # As in the run job: no npm cache is used, and a lockfile-less repository must not fail this step.
457
+ package-manager-cache: false
436
458
 
437
459
  - name: Restore and save the docs-retrieval cache
438
460
  if: env.HARNESS_STOPPED != '1' && steps.config.outputs.retrieval == 'true'
439
- uses: actions/cache@v4
461
+ uses: actions/cache@v5
440
462
  with:
441
463
  path: ${{ env.HARNESS_RETRIEVAL_CACHE }}
442
464
  key: harness-retrieval-${{ runner.os }}-${{ env.HARNESS_CLI_VERSION }}
@@ -9,8 +9,13 @@
9
9
  {{pushEnvPath}}
10
10
  {{qaCredentialsPath}}
11
11
  {{clientEnvPath}}
12
- # The per-machine settings file the agent runner writes beside the committed profiles.
12
+ # The per-machine settings file the agent runner writes.
13
13
  .claude/settings.local.json
14
+ # The unattended run's permission profile. It names this checkout's absolute paths, so a copy is
15
+ # wrong on every other clone; the watcher and `doctor` read it from the main checkout only - a linked
16
+ # worktree carries none - and a remote job's own `init` generates its own. A copy an earlier release
17
+ # committed stays tracked despite this rule until it is untracked by hand - `doctor` reports that state.
18
+ {{permissionProfilePath}}
14
19
  # The presence-only marker one user's "don't ask again" answer writes to silence the change-request
15
20
  # offer. It lives at the main worktree's root and is checked there from every worktree of this
16
21
  # checkout, so one answer covers them all - its existence is the whole signal, so it has no
@@ -101,20 +101,25 @@
101
101
  # newest events across every live run and applies it to ALL of them, through the
102
102
  # ordinary pause protocol and nothing else:
103
103
  #
104
- # * IT NEVER KILLS A RUN, and it never invents a mechanism. It drops
105
- # `<state_dir>/PAUSE` into each running working copy, tags the record
106
- # `paused_by=usage` and records `usage_resume_at`; the engine yields at its
104
+ # * IT NEVER KILLS A RUN, and it never invents a mechanism. It tags each running
105
+ # record `paused_by=usage` with its `usage_resume_at`, then drops
106
+ # `<state_dir>/PAUSE` into its working copy; the engine yields at its
107
107
  # next clean checkpoint, writes PAUSE_ACK, and classify_run_exit marks it
108
108
  # `paused` — the same path a hand-dropped PAUSE takes. Once the window has
109
109
  # reset the gate drops `<state_dir>/RESUME`, and the pause-resume pass above
110
110
  # re-launches the run with no further involvement from here.
111
+ # * A TAGGED PAUSE WHOSE RESET TIME IS MISSING IS GIVEN THE FALLBACK — the
112
+ # same hour the pause side assumes when no reset was reported — written,
113
+ # logged and reported once, and it then resumes on the wall clock like any
114
+ # other.
111
115
  # * A RUN PAUSED BY HAND IS NEVER AUTO-RESUMED. The resume side acts on the
112
116
  # `paused_by=usage` tag alone, and a hand pause carries no tag.
113
117
  # * WHILE A PAUSE IS IN EFFECT THE HOLD MARKER IS UP, which is what defers a
114
118
  # fresh inbox drop (it stays in the inbox) and skips the watchdog above:
115
119
  # launching into a full window spends a run on an immediate refusal. A
116
120
  # REMOTE drop is dispatched through the hold — the job gates itself (see
117
- # REMOTE DISPATCH).
121
+ # REMOTE DISPATCH). The hold is bounded by each pause's recorded or repaired
122
+ # `usage_resume_at`: it comes down on the pass that drops RESUME.
118
123
  # * THE WINDOW TYPES ARE ASSESSED INDEPENDENTLY — the 5-hour one and the
119
124
  # rolling weekly one — so a 5-hour window that has just reset cannot mask a
120
125
  # weekly window sitting at its cap. The worst state across every window of
@@ -349,11 +354,17 @@
349
354
  # advances it to the epoch taken just before its query; a failed poll
350
355
  # advances nothing and pauses nothing.
351
356
  # * THE DECISION, when the run leaves `running`: `paused` for `budget` ->
352
- # `continue`; for `user` -> `stop`; by the usage gate (`usage`) -> WAIT IN
353
- # THE JOB, the gate's own auto-resume and resume_paused_runs relaunching it,
354
- # when the reset falls before HARNESS_JOB_DEADLINE_EPOCH and the runner is
355
- # self-hosted or the wait is at most REMOTE_WAIT_MAX_SECS, else
356
- # `wait-poller`; with no pause requested — the run's own API-overload
357
+ # `continue`; for `user` -> `stop`; by the usage gate (`usage`) -> a lost
358
+ # `usage_resume_at` is first given the gate's fallback and reported as
359
+ # assumed, then WAIT IN THE JOB, the gate's own auto-resume and
360
+ # resume_paused_runs relaunching it, when the reset falls before
361
+ # HARNESS_JOB_DEADLINE_EPOCH and the runner is self-hosted or the wait is at
362
+ # most REMOTE_WAIT_MAX_SECS, else `wait-poller`. A wait still paused
363
+ # REMOTE_WAIT_MAX_SECS past the gate's last chance to resume it — the later
364
+ # of that reset and its own start, plus USAGE_CHECK_INTERVAL_SECS and
365
+ # POLL_INTERVAL_SECS, the most the throttled gate can lag the reset — ends
366
+ # in `wait-poller` with one notification, on every runner; with no pause
367
+ # requested — the run's own API-overload
357
368
  # self-pause (`overload`) -> auto-resume, else `stop`. `failed` ->
358
369
  # auto-resume unless the stall watchdog gave up, else `stop`. Every other
359
370
  # status -> `stop`. An empty HARNESS_JOB_DEADLINE_EPOCH bounds nothing.
@@ -1024,6 +1035,11 @@ USAGE_WARNING_DEBOUNCE="${USAGE_WARNING_DEBOUNCE:-2}"
1024
1035
  # than at it: the reported instant is the account's, not this machine's, and a
1025
1036
  # resume that lands a moment early is refused and costs the run its session.
1026
1037
  USAGE_RESUME_MARGIN_SECS="${USAGE_RESUME_MARGIN_SECS:-120}"
1038
+ # The resume time the gate assumes, from now, for a pause with no reported reset
1039
+ # and for a usage-paused record whose `usage_resume_at` was lost. Assigned
1040
+ # plainly, with no environment override: it is a guess at a missing value, not a
1041
+ # policy knob.
1042
+ USAGE_FALLBACK_RESUME_SECS=3600
1027
1043
  # The weekly window's own trigger threshold, as a fraction of its reported
1028
1044
  # utilization. Its `allowed_warning` fires from about half the weekly budget
1029
1045
  # onward — informational, not a signal that anything is about to be refused — so
@@ -1219,8 +1235,9 @@ notify() {
1219
1235
  # gate, and empty otherwise — which is the whole of how a
1220
1236
  # gate pause is told apart from a hand-dropped one. A hand
1221
1237
  # pause is never auto-resumed precisely because it has no
1222
- # value here. Written the moment the PAUSE is REQUESTED,
1223
- # while the record is still `running`, and cleared by a
1238
+ # value here. Written together with `usage_resume_at` in
1239
+ # one write, BEFORE the PAUSE is dropped, while the record
1240
+ # is still `running`, and cleared by a
1224
1241
  # real resume, by the gate's stale-tag sweep, and by
1225
1242
  # launch_run on a reused branch key — see the gate for why
1226
1243
  # clearing it any earlier than those strands the run.
@@ -1228,7 +1245,10 @@ notify() {
1228
1245
  # BINDING worst-state window reset (the overage window's
1229
1246
  # while `isUsingOverage`) plus USAGE_RESUME_MARGIN_SECS.
1230
1247
  # The ONLY state the wall-clock resume reads, and written
1231
- # and cleared together with `paused_by`.
1248
+ # and cleared together with `paused_by`, in one write. A
1249
+ # tagged `paused` record found without a usable value
1250
+ # has it written alone, by usage_resume_at_var's repair:
1251
+ # now plus USAGE_FALLBACK_RESUME_SECS.
1232
1252
  # remote_stopped_at the epoch second `remote-run.sh stop` sent the branch's
1233
1253
  # stop marker and asked GitHub to cancel its runs. Written
1234
1254
  # by `remote-run.sh stop` alone, only on an existing
@@ -1244,7 +1264,7 @@ notify() {
1244
1264
  # `user` job mode did, on a `harness pause <branch>` run;
1245
1265
  # `overload` nobody requested it — the run's own
1246
1266
  # API-overload self-pause. `user` and `budget` are written
1247
- # when the PAUSE is dropped, while still `running`;
1267
+ # before the PAUSE is dropped, while still `running`;
1248
1268
  # classify_run_exit settles the reason as the run pauses,
1249
1269
  # `user` first, then `usage`, then `budget`. Cleared when
1250
1270
  # job mode relaunches the run — plus `killed`, a registry-only
@@ -1252,7 +1272,9 @@ notify() {
1252
1272
  # bundle still says `running`, or a finished run left no
1253
1273
  # bundle, and `expired`, a registry-only value it derives
1254
1274
  # when a finished run's bundle has expired (the job can no
1255
- # longer take an answer, and the carried counts are lost);
1275
+ # longer take an answer, and the carried counts and any
1276
+ # planning drafts not yet committed that it carried are
1277
+ # lost);
1256
1278
  # `status.json` never carries either. Both map to
1257
1279
  # `paused` rather than `failed` because a `failed` record
1258
1280
  # has no resume path, while the ledger on the branch is
@@ -1290,7 +1312,8 @@ notify() {
1290
1312
  # -----------------------------------------------------------------------------
1291
1313
  # The bodies are lib/harness-run-lib.sh's THE RUN REGISTRY, shared with every
1292
1314
  # script that reads or writes this file; these wrappers bind them to $REGISTRY.
1293
- # registry_set <branch> <key> <value>; registry_get <branch> <key>.
1315
+ # registry_set <branch> <key> <value> [<key> <value> …] — every pair in one
1316
+ # write; registry_get <branch> <key>.
1294
1317
  registry_init() { hr_registry_init "$REGISTRY"; }
1295
1318
  registry_set() { hr_registry_set "$REGISTRY" "$@"; }
1296
1319
  registry_get() { hr_registry_get "$REGISTRY" "$@"; }
@@ -1697,9 +1720,14 @@ lane_release_if_idle() {
1697
1720
  # --effort "<agentEffort>" \
1698
1721
  # --output-format stream-json --verbose \
1699
1722
  # --add-dir <worktree> \
1700
- # --add-dir <MAIN_REPO>/<state_dir>
1723
+ # --add-dir <MAIN_REPO>/<state_dir> \
1724
+ # [--add-dir <dir> ...]
1701
1725
  #
1702
- # and NEVER a permission-bypass flag: the profile's deny floor is what keeps an
1726
+ # where the bracketed tail is JOB MODE ONLY: one --add-dir per
1727
+ # permissions.additionalDirectories entry of the profile, in file order, minus
1728
+ # empty entries and the two directories above; outside job mode the line ends at
1729
+ # the state directory. See spawn_engine for why the profile's list is repeated.
1730
+ # And NEVER a permission-bypass flag: the profile's deny floor is what keeps an
1703
1731
  # unattended run in its lane, and bypassing it makes every refusal decorative.
1704
1732
  # Both run-setting flags are CONDITIONAL: an unset key leaves its flag off the
1705
1733
  # line entirely rather than passing an empty argument.
@@ -1857,6 +1885,32 @@ ${GLOBAL_STOP}. End at 'branch ready for review' — never merge, never push to
1857
1885
  effort_args=(--effort "$AGENT_EFFORT")
1858
1886
  fi
1859
1887
 
1888
+ # JOB MODE ONLY: every permissions.additionalDirectories entry of the profile
1889
+ # (the plugin roots `init --plugin-root-entries` wrote) also goes on the line as
1890
+ # an --add-dir, because a job session was refused reads under a root the profile
1891
+ # file already granted. `init` stays the one producer of the list. File order;
1892
+ # an empty entry and one equal to the two directories already passed are
1893
+ # skipped. An absent, unparseable or keyless profile adds nothing and never
1894
+ # blocks the launch — `doctor --remote-job` refuses an unusable profile earlier.
1895
+ local extra_dir_args extra_dirs_logged
1896
+ extra_dir_args=()
1897
+ extra_dirs_logged=""
1898
+ if [ "$JOB_MODE" = "1" ]; then
1899
+ local profile_dirs profile_dir
1900
+ profile_dirs="$(jq -r '.permissions.additionalDirectories[]? // empty' "$SETTINGS_PROFILE" 2>/dev/null)" || profile_dirs=""
1901
+ while IFS= read -r profile_dir; do
1902
+ [ -n "$profile_dir" ] || continue
1903
+ case "${profile_dir%/}" in
1904
+ "${worktree%/}" | "${main_state%/}") continue ;;
1905
+ esac
1906
+ extra_dir_args+=(--add-dir "$profile_dir")
1907
+ extra_dirs_logged="${extra_dirs_logged} '${profile_dir}'"
1908
+ done <<EOF
1909
+ $profile_dirs
1910
+ EOF
1911
+ [ -z "$extra_dirs_logged" ] || log "job mode: '$branch' also gets --add-dir from the profile's additionalDirectories:${extra_dirs_logged}"
1912
+ fi
1913
+
1860
1914
  # The formatter is the tail of the pipeline; a passthrough keeps the raw events
1861
1915
  # in the log rather than breaking the pipe when it is not runnable.
1862
1916
  local formatter="$FORMAT_STREAM"
@@ -1889,7 +1943,10 @@ ${GLOBAL_STOP}. End at 'branch ready for review' — never merge, never push to
1889
1943
  cd "$worktree" || exit 97
1890
1944
  # `--add-dir "$worktree"` is NOT redundant with the profile: that file grants
1891
1945
  # the sibling-worktree glob through Edit/Write/Read rules, not through
1892
- # additionalDirectories.
1946
+ # additionalDirectories. The job-mode extras after the two fixed --add-dir
1947
+ # flags duplicate the profile's additionalDirectories on purpose: a job
1948
+ # session was refused reads under a directory that file granted, and a
1949
+ # launch flag does not depend on how the settings file is merged.
1893
1950
  #
1894
1951
  # The run is streamed as JSON events through the formatter so the per-run log
1895
1952
  # shows the orchestrator heartbeat and the sub-agent dispatches LIVE and
@@ -1908,7 +1965,8 @@ ${GLOBAL_STOP}. End at 'branch ready for review' — never merge, never push to
1908
1965
  ${effort_args[@]+"${effort_args[@]}"} \
1909
1966
  --output-format stream-json --verbose \
1910
1967
  --add-dir "$worktree" \
1911
- --add-dir "$main_state" 2>>"$log_path" |
1968
+ --add-dir "$main_state" \
1969
+ ${extra_dir_args[@]+"${extra_dir_args[@]}"} 2>>"$log_path" |
1912
1970
  tee -a "${log_path%.log}.stream.jsonl" |
1913
1971
  "$formatter" >>"$log_path"
1914
1972
  rc=${PIPESTATUS[0]}
@@ -1998,8 +2056,7 @@ launch_run() {
1998
2056
  registry_set "$branch" stall_restarts 0
1999
2057
  registry_set "$branch" stall_warned ""
2000
2058
  registry_set "$branch" stall_killing ""
2001
- registry_set "$branch" paused_by ""
2002
- registry_set "$branch" usage_resume_at ""
2059
+ registry_set "$branch" paused_by "" usage_resume_at ""
2003
2060
  registry_set "$branch" resume_kind ""
2004
2061
  registry_set "$branch" park_loop_cycles 0
2005
2062
 
@@ -2029,8 +2086,7 @@ launch_remote_run() {
2029
2086
  registry_set "$branch" stall_restarts 0
2030
2087
  registry_set "$branch" stall_warned ""
2031
2088
  registry_set "$branch" stall_killing ""
2032
- registry_set "$branch" paused_by ""
2033
- registry_set "$branch" usage_resume_at ""
2089
+ registry_set "$branch" paused_by "" usage_resume_at ""
2034
2090
  registry_set "$branch" resume_kind ""
2035
2091
  registry_set "$branch" park_loop_cycles 0
2036
2092
 
@@ -2171,8 +2227,7 @@ classify_run_exit() {
2171
2227
  reason=overload
2172
2228
  fi
2173
2229
  fi
2174
- registry_set "$branch" pause_reason "$reason"
2175
- registry_set "$branch" status paused
2230
+ registry_set "$branch" pause_reason "$reason" status paused
2176
2231
  log "run '$branch' paused (PAUSE honored, reason $reason) — rc=$rc"
2177
2232
  if [ "$reason" = "user" ]; then
2178
2233
  notify paused "$branch" "$log_path" "paused as you asked — run /autonomous-sdlc-harness:branch-resume $branch to continue"
@@ -3633,6 +3688,34 @@ usage_paused_count() {
3633
3688
  jq -r '[.runs | to_entries[] | select(.value.status=="paused" and .value.paused_by=="usage")] | length' "$REGISTRY" 2>/dev/null
3634
3689
  }
3635
3690
 
3691
+ # usage_resume_at_var <branch> — sets USAGE_RESUME_AT to the record's usable
3692
+ # `usage_resume_at` epoch and USAGE_RESUME_REPAIRED=0, returning 0. A record that
3693
+ # is `paused` with `paused_by=usage` and no usable value has LOST it: it gets
3694
+ # now + USAGE_FALLBACK_RESUME_SECS, written and logged, with
3695
+ # USAGE_RESUME_REPAIRED=1. Anything else — an untagged record above all, so a hand
3696
+ # pause is never given a time — is left alone, with USAGE_RESUME_AT empty and a
3697
+ # return of 1. Call it UNSUBSTITUTED: `log` writes to stdout.
3698
+ usage_resume_at_var() {
3699
+ local b="$1" raw
3700
+ USAGE_RESUME_AT=""
3701
+ USAGE_RESUME_REPAIRED=0
3702
+ raw="$(registry_get "$b" usage_resume_at)"
3703
+ case "$raw" in
3704
+ '' | *[!0-9]*) ;;
3705
+ *)
3706
+ USAGE_RESUME_AT="$((10#$raw))"
3707
+ return 0
3708
+ ;;
3709
+ esac
3710
+ [ "$(registry_get "$b" status)" = "paused" ] || return 1
3711
+ [ "$(registry_get "$b" paused_by)" = "usage" ] || return 1
3712
+ USAGE_RESUME_AT=$(($(date +%s) + USAGE_FALLBACK_RESUME_SECS))
3713
+ USAGE_RESUME_REPAIRED=1
3714
+ registry_set "$b" usage_resume_at "$USAGE_RESUME_AT"
3715
+ log "usage: '$b' is usage-paused with no usable usage_resume_at ('$raw') — assuming the gate's fallback, resume ~$(stall_human_time "$USAGE_RESUME_AT")"
3716
+ return 0
3717
+ }
3718
+
3636
3719
  # The gate itself, throttled to USAGE_CHECK_INTERVAL_SECS and run in three parts,
3637
3720
  # in this order: the resume side and the stale-tag cleanup, then the assessment
3638
3721
  # and the pauses it justifies, then the hold marker. Resume-before-pause is what
@@ -3649,7 +3732,7 @@ usage_gate() {
3649
3732
  # Every one of these is initialized, not merely declared: `set -u` makes a
3650
3733
  # DECLARED-BUT-UNSET name an error on first read, and several of the branches
3651
3734
  # below are reached without every name having been assigned in that iteration.
3652
- local b status pb ra wt state_rel="" state_abs=""
3735
+ local b status pb ra wt lp state_rel="" state_abs=""
3653
3736
  while IFS= read -r b; do
3654
3737
  [ -n "$b" ] || continue
3655
3738
  # The auto-resume side skips a remote run: its job gates itself, and
@@ -3689,17 +3772,31 @@ usage_gate() {
3689
3772
  # one case where the question cannot be ANSWERED, and there the tag stays.
3690
3773
  if [ -z "$wt" ] ||
3691
3774
  { [ -n "$state_abs" ] && [ ! -f "$state_abs/PAUSE" ] && [ ! -f "$state_abs/PAUSE_ACK" ]; }; then
3692
- registry_set "$b" paused_by ""
3693
- registry_set "$b" usage_resume_at ""
3775
+ registry_set "$b" paused_by "" usage_resume_at ""
3694
3776
  fi
3695
3777
  continue
3696
3778
  fi
3697
3779
 
3698
3780
  [ "$status" = "paused" ] || continue
3699
- ra="$(registry_get "$b" usage_resume_at)"
3700
- # No usable resume time is not a reason to resume: leave it paused and let an
3701
- # operator's own RESUME be the trigger, exactly as for a hand pause.
3702
- case "$ra" in '' | *[!0-9]*) continue ;; esac
3781
+ # A tagged pause with no usable resume time has LOST it; left alone it would
3782
+ # never resume and would hold the launch hold up for good. The helper gives it
3783
+ # the fallback time and this pass reports that once — the repaired value is
3784
+ # usable and in the future, so no later pass notifies again. In job mode a
3785
+ # lost value is left to run_job's `usage)` arm, which repairs it and owns the
3786
+ # job's notifications; repairing here as well would send two for one pause.
3787
+ if [ "$JOB_MODE" = "1" ]; then
3788
+ ra="$(registry_get "$b" usage_resume_at)"
3789
+ case "$ra" in '' | *[!0-9]*) continue ;; esac
3790
+ else
3791
+ usage_resume_at_var "$b" || continue
3792
+ ra="$USAGE_RESUME_AT"
3793
+ if [ "$USAGE_RESUME_REPAIRED" = "1" ]; then
3794
+ lp="$(registry_get "$b" log_path)"
3795
+ [ -n "$lp" ] || lp="$LOGS_DIR/$b.log"
3796
+ notify paused "$b" "$lp" "usage pause lost its recorded reset time — assuming ~$(stall_human_time "$ra") and resuming then; drop ${state_rel:-<state_dir>}/RESUME in ${wt:-its working copy} to resume sooner"
3797
+ continue
3798
+ fi
3799
+ fi
3703
3800
  [ "$now" -ge "$ra" ] || continue
3704
3801
  if [ -z "$wt" ] || [ ! -d "$wt" ] || [ -z "$state_abs" ]; then
3705
3802
  # The trigger cannot be placed where the run would read it. Leave BOTH tags
@@ -3714,8 +3811,7 @@ usage_gate() {
3714
3811
  # Cleared TOGETHER with the trigger: the pause-resume pass owns the relaunch
3715
3812
  # from here, and a tag left behind would make the next hand pause look like
3716
3813
  # this gate's.
3717
- registry_set "$b" paused_by ""
3718
- registry_set "$b" usage_resume_at ""
3814
+ registry_set "$b" paused_by "" usage_resume_at ""
3719
3815
  done <<EOF
3720
3816
  $(registry_branches)
3721
3817
  EOF
@@ -3773,7 +3869,7 @@ EOF
3773
3869
  # pauses would never be resumed by the wall clock. An hour is the fallback:
3774
3870
  # long enough not to thrash, short enough that a wrong guess costs one hour.
3775
3871
  case "$resume_at" in '' | *[!0-9]*) resume_at=0 ;; esac
3776
- [ "$resume_at" -le 0 ] && resume_at=$((now + 3600))
3872
+ [ "$resume_at" -le 0 ] && resume_at=$((now + USAGE_FALLBACK_RESUME_SECS))
3777
3873
  while IFS= read -r b; do
3778
3874
  [ -n "$b" ] || continue
3779
3875
  # The pause side skips a remote run: its job gates itself.
@@ -3793,11 +3889,14 @@ EOF
3793
3889
  log "usage auto-pause: the state directory in '$wt' is unresolvable — cannot pause '$b'"
3794
3890
  continue
3795
3891
  fi
3892
+ # Tagged BEFORE the engine acknowledges, on purpose — see invariant 2 — and
3893
+ # so before PAUSE exists: an engine acknowledging a PAUSE whose tag is not
3894
+ # yet recorded is classified `overload`. Both keys land in one write. A tag
3895
+ # whose `touch` then fails is a `running` record with no PAUSE, which the
3896
+ # stale-tag sweep in (1) clears on the next pass.
3897
+ registry_set "$b" paused_by usage usage_resume_at "$resume_at"
3796
3898
  mkdir -p "$wt/$state_rel" 2>/dev/null || true
3797
3899
  touch "$wt/$state_rel/PAUSE"
3798
- # Tagged BEFORE the engine acknowledges, on purpose — see invariant 2.
3799
- registry_set "$b" paused_by usage
3800
- registry_set "$b" usage_resume_at "$resume_at"
3801
3900
  log "usage auto-pause (state=$state, trigger=$USAGE_PAUSE_TRIGGER): dropped $state_rel/PAUSE in $wt (auto-resume ~$(stall_human_time "$resume_at"))"
3802
3901
  done <<EOF
3803
3902
  $(registry_branches)
@@ -3808,7 +3907,9 @@ EOF
3808
3907
  # --- (3) The launch hold: up while a usage pause is in effect OR being
3809
3908
  # initiated, down otherwise. Derived from the registry every pass rather than
3810
3909
  # toggled, so a marker left behind by a watcher that died mid-pause is cleared
3811
- # by the next one instead of holding the inbox forever.
3910
+ # by the next one instead of holding the inbox forever. A tagged pause stops
3911
+ # counting on the pass whose part (1) drops RESUME and clears both tags, so
3912
+ # the hold is bounded by the recorded or repaired `usage_resume_at`.
3812
3913
  local held
3813
3914
  held="$(usage_paused_count)"
3814
3915
  case "$held" in '' | *[!0-9]*) held=0 ;; esac
@@ -4006,8 +4107,8 @@ job_control_poll() {
4006
4107
  registry_set "$branch" control_polled_at "$before"
4007
4108
  if [ "$rc" = "0" ]; then
4008
4109
  JOB_USER_PAUSE_DROPPED=1
4009
- touch "$state_abs/PAUSE"
4010
4110
  registry_set "$branch" pause_reason user
4111
+ touch "$state_abs/PAUSE"
4011
4112
  log "job: a 'harness pause $branch' run was created at or after $since — dropped PAUSE (reason user)"
4012
4113
  fi
4013
4114
  job_write_status "$branch" "$remote_status" continue "job started"
@@ -4021,16 +4122,22 @@ job_budget_pass() {
4021
4122
  after="$(job_int "${REMOTE_SELF_PAUSE_AFTER_SECS:-}")" || return 0
4022
4123
  [ $(($(date +%s) - JOB_START_EPOCH)) -ge "$after" ] || return 0
4023
4124
  JOB_BUDGET_PAUSE_DROPPED=1
4024
- touch "$state_abs/PAUSE"
4025
4125
  [ "$(registry_get "$branch" pause_reason)" = "user" ] || registry_set "$branch" pause_reason budget
4126
+ touch "$state_abs/PAUSE"
4026
4127
  log "job: ${after}s of the hosted time budget have passed — dropped PAUSE (reason budget)"
4027
4128
  }
4028
4129
 
4029
- # job_usage_wait_ok <branch> — 0 when a usage pause is waited out in the job.
4030
- # An empty usage_resume_at means the gate has already dropped RESUME.
4130
+ # job_usage_wait_ok <branch> — 0 when a usage pause is waited out in the job;
4131
+ # leaves usage_resume_at_var's USAGE_RESUME_AT and USAGE_RESUME_REPAIRED set.
4132
+ # With no usable usage_resume_at: `paused_by=usage` still set means the value was
4133
+ # LOST, and the decision is made on the helper's repaired epoch — an hour out,
4134
+ # so `wait-poller` on a hosted runner at the default REMOTE_WAIT_MAX_SECS;
4135
+ # `paused_by` empty means the gate already dropped
4136
+ # RESUME, so 0, and the next resume_paused_runs relaunches the run.
4031
4137
  job_usage_wait_ok() {
4032
4138
  local ra deadline
4033
- ra="$(job_int "$(registry_get "$1" usage_resume_at)")" || return 0
4139
+ usage_resume_at_var "$1" || return 0
4140
+ ra="$USAGE_RESUME_AT"
4034
4141
  deadline="$(job_int "${HARNESS_JOB_DEADLINE_EPOCH:-}")" || deadline=""
4035
4142
  if [ -n "$deadline" ] && [ "$ra" -ge "$deadline" ]; then
4036
4143
  return 1
@@ -4068,8 +4175,8 @@ job_auto_resume() {
4068
4175
  # is one-shot: put it back, so the relaunched session still yields at its next
4069
4176
  # clean checkpoint instead of running on until the step timeout kills it.
4070
4177
  if [ "$JOB_BUDGET_PAUSE_DROPPED" = "1" ]; then
4071
- touch "$state_abs/PAUSE"
4072
4178
  registry_set "$branch" pause_reason budget
4179
+ touch "$state_abs/PAUSE"
4073
4180
  log "job: the hosted time budget's PAUSE was pending at the resume of '$branch' after $why — re-dropped it"
4074
4181
  fi
4075
4182
  notify resumed "$branch" "$log_path" "automatic resume $count/$REMOTE_AUTO_RESUME_MAX after $why ($(job_label))"
@@ -4099,8 +4206,7 @@ run_job() {
4099
4206
  registry_set "$branch" stall_restarts 0
4100
4207
  registry_set "$branch" stall_warned ""
4101
4208
  registry_set "$branch" stall_killing ""
4102
- registry_set "$branch" paused_by ""
4103
- registry_set "$branch" usage_resume_at ""
4209
+ registry_set "$branch" paused_by "" usage_resume_at ""
4104
4210
  registry_set "$branch" resume_kind ""
4105
4211
  registry_set "$branch" park_loop_cycles 0
4106
4212
  registry_set "$branch" auto_resumes 0
@@ -4168,6 +4274,7 @@ run_job() {
4168
4274
 
4169
4275
  # The supervision loop: the header's JOB MODE block states each decision.
4170
4276
  local status reason ra when decision="stop" detail="" usage_waiting=0 restarts
4277
+ local wait_ok=1 usage_wait_start=0 usage_wait_ra=0
4171
4278
  while :; do
4172
4279
  status="$(registry_get "$branch" status)"
4173
4280
  case "$status" in
@@ -4194,25 +4301,49 @@ run_job() {
4194
4301
  break
4195
4302
  ;;
4196
4303
  usage)
4197
- ra="$(registry_get "$branch" usage_resume_at)"
4198
- when="the reset"
4199
- [ -n "$ra" ] && when="the reset at ~$(stall_human_time "$ra")"
4200
4304
  if [ "$usage_waiting" = "0" ]; then
4201
- if ! job_usage_wait_ok "$branch"; then
4305
+ # Decided BEFORE `ra` is read, so a lost value is repaired first and
4306
+ # both notifications name the repaired time.
4307
+ wait_ok=1
4308
+ job_usage_wait_ok "$branch" || wait_ok=0
4309
+ ra="$USAGE_RESUME_AT"
4310
+ when="the reset"
4311
+ [ -n "$ra" ] && when="the reset at ~$(stall_human_time "$ra")"
4312
+ [ "$USAGE_RESUME_REPAIRED" = "1" ] &&
4313
+ when="the assumed reset at ~$(stall_human_time "$ra") (the recorded reset time was lost)"
4314
+ if [ "$wait_ok" = "0" ]; then
4202
4315
  decision=wait-poller
4203
4316
  detail="usage limit reached; the resume poller resumes it after $when"
4204
4317
  notify paused "$branch" "$log_path" "usage limit reached — resumes automatically after $when"
4205
4318
  break
4206
4319
  fi
4207
4320
  usage_waiting=1
4321
+ # Captured once: the gate's own resume clears usage_resume_at later
4322
+ # in this same wait, and the bound below must not move with it.
4323
+ usage_wait_start="$(date +%s)"
4324
+ usage_wait_ra="$(job_int "$ra")" || usage_wait_ra=0
4325
+ [ "$usage_wait_ra" -gt "$usage_wait_start" ] || usage_wait_ra="$usage_wait_start"
4208
4326
  log "job: '$branch' is usage-paused — waiting in the job for $when"
4209
4327
  notify paused "$branch" "$log_path" "usage limit reached — waiting in the job for $when"
4210
4328
  fi
4211
4329
  sleep "$POLL_INTERVAL_SECS"
4212
4330
  usage_gate
4213
4331
  resume_paused_runs
4214
- if [ "$(registry_get "$branch" status)" = "running" ]; then
4332
+ status="$(registry_get "$branch" status)"
4333
+ if [ "$status" = "running" ]; then
4215
4334
  registry_set "$branch" pause_reason ""
4335
+ elif [ "$status" = "paused" ] &&
4336
+ [ "$(date +%s)" -gt $((usage_wait_ra + USAGE_CHECK_INTERVAL_SECS + POLL_INTERVAL_SECS + REMOTE_WAIT_MAX_SECS)) ]; then
4337
+ # The bound on the wait itself, self-hosted included. It is measured
4338
+ # from the gate's last chance to resume, not from the reset: the gate
4339
+ # is throttled to USAGE_CHECK_INTERVAL_SECS and runs after a
4340
+ # POLL_INTERVAL_SECS sleep, so its resume can lag the reset by both.
4341
+ # REMOTE_WAIT_MAX_SECS past that with no resume, the job hands the
4342
+ # run over rather than waiting on nothing.
4343
+ decision=wait-poller
4344
+ detail="usage limit reached; the in-job usage wait passed its bound (REMOTE_WAIT_MAX_SECS=${REMOTE_WAIT_MAX_SECS}s past the reset) without a resume"
4345
+ notify paused "$branch" "$log_path" "usage limit: the in-job wait passed ${REMOTE_WAIT_MAX_SECS}s after the reset without a resume — run /autonomous-sdlc-harness:branch-resume $branch to continue"
4346
+ break
4216
4347
  fi
4217
4348
  ;;
4218
4349
  *)