@bongos/core 1.20.33 → 1.20.34
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.bongos-core.json +19 -19
- package/docs/adr/0042-builder-self-deploy-ci-auto-merge.md +19 -5
- package/docs/adr/README.md +1 -1
- package/docs/module-api-changelog.md +3 -1
- package/docs/recipes/autobongos-windows-host.md +1 -1
- package/modules/autonomy/cadence.js +8 -0
- package/modules/autonomy/gauge.js +5 -1
- package/package-lock.json +2 -2
- package/package.json +1 -1
- package/release-notes.json +10 -0
- package/scripts/gds/autobongos-loop.js +37 -1
- package/scripts/gds/autobongos-run.js +38 -80
- package/src/module-api.js +1 -1
- package/tests/autobongos_cadence.mjs +27 -0
- package/tests/autobongos_loop.mjs +107 -94
package/.bongos-core.json
CHANGED
|
@@ -2,11 +2,11 @@
|
|
|
2
2
|
"artifact": "bongos-core",
|
|
3
3
|
"manifest_schema": 1,
|
|
4
4
|
"generator": "scripts/gds/package-core.js",
|
|
5
|
-
"core_version": "1.20.
|
|
6
|
-
"core_contract": "1.20.
|
|
7
|
-
"source_commit": "
|
|
5
|
+
"core_version": "1.20.34",
|
|
6
|
+
"core_contract": "1.20.34",
|
|
7
|
+
"source_commit": "b6e2dfd60ddc3de92e31e4ca330985ae253bedab",
|
|
8
8
|
"source_ref": "HEAD",
|
|
9
|
-
"built_at": "2026-09-30T22:
|
|
9
|
+
"built_at": "2026-09-30T22:16:46.560Z",
|
|
10
10
|
"redaction": {
|
|
11
11
|
"model": "docs-redacted+functional-verbatim",
|
|
12
12
|
"docs_redacted": 562,
|
|
@@ -17,7 +17,7 @@
|
|
|
17
17
|
"gate": "passed"
|
|
18
18
|
},
|
|
19
19
|
"file_count": 3259,
|
|
20
|
-
"tree_sha256": "
|
|
20
|
+
"tree_sha256": "91cd2d81b5a990188e0fc5b9230e3f322cf350277f64f07a5d3db549627845bf",
|
|
21
21
|
"files": [
|
|
22
22
|
{
|
|
23
23
|
"path": ".claude/skills/ask-for-help/SKILL.md",
|
|
@@ -617,7 +617,7 @@
|
|
|
617
617
|
{
|
|
618
618
|
"path": "docs/adr/0042-builder-self-deploy-ci-auto-merge.md",
|
|
619
619
|
"mode": "0000644",
|
|
620
|
-
"sha256": "
|
|
620
|
+
"sha256": "1a2c7e3b5ebeb50991f65dfe4032dab10de8be0848193c3daf9baae9aed4e314"
|
|
621
621
|
},
|
|
622
622
|
{
|
|
623
623
|
"path": "docs/adr/0043-git-ssh-trust-boundary-and-rank-floor-on-permission-paths.md",
|
|
@@ -2257,7 +2257,7 @@
|
|
|
2257
2257
|
{
|
|
2258
2258
|
"path": "docs/adr/README.md",
|
|
2259
2259
|
"mode": "0000644",
|
|
2260
|
-
"sha256": "
|
|
2260
|
+
"sha256": "17dbc3a335653ba36f2e415c44eced76d94011585477a079765dd1f5fad36d86"
|
|
2261
2261
|
},
|
|
2262
2262
|
{
|
|
2263
2263
|
"path": "docs/api-reference.md",
|
|
@@ -2802,7 +2802,7 @@
|
|
|
2802
2802
|
{
|
|
2803
2803
|
"path": "docs/module-api-changelog.md",
|
|
2804
2804
|
"mode": "0000644",
|
|
2805
|
-
"sha256": "
|
|
2805
|
+
"sha256": "974525f923ba04ab5b3fe9a0746a4ec6c859a30e4aebbb55a223e75e9c60689c"
|
|
2806
2806
|
},
|
|
2807
2807
|
{
|
|
2808
2808
|
"path": "docs/modules-contract.md",
|
|
@@ -2912,7 +2912,7 @@
|
|
|
2912
2912
|
{
|
|
2913
2913
|
"path": "docs/recipes/autobongos-windows-host.md",
|
|
2914
2914
|
"mode": "0000644",
|
|
2915
|
-
"sha256": "
|
|
2915
|
+
"sha256": "5e4821f62223ffc49cef22475f7c31d5bb798bde167ed03823931694f15060d7"
|
|
2916
2916
|
},
|
|
2917
2917
|
{
|
|
2918
2918
|
"path": "docs/recipes/bongos-cli-release.md",
|
|
@@ -4077,7 +4077,7 @@
|
|
|
4077
4077
|
{
|
|
4078
4078
|
"path": "modules/autonomy/cadence.js",
|
|
4079
4079
|
"mode": "0000644",
|
|
4080
|
-
"sha256": "
|
|
4080
|
+
"sha256": "e2ab1fd3381fa94af9966a036467ddcffe004cec6e4c07cc25da29b9f7aa4446"
|
|
4081
4081
|
},
|
|
4082
4082
|
{
|
|
4083
4083
|
"path": "modules/autonomy/db.js",
|
|
@@ -4097,7 +4097,7 @@
|
|
|
4097
4097
|
{
|
|
4098
4098
|
"path": "modules/autonomy/gauge.js",
|
|
4099
4099
|
"mode": "0000644",
|
|
4100
|
-
"sha256": "
|
|
4100
|
+
"sha256": "5cbbe803f8c977f49d715cb803fa7dc20bd1800e599659669c5fe28f4f9b9b20"
|
|
4101
4101
|
},
|
|
4102
4102
|
{
|
|
4103
4103
|
"path": "modules/autonomy/migrations/autonomy_001_autobongos_fence.sql",
|
|
@@ -8992,12 +8992,12 @@
|
|
|
8992
8992
|
{
|
|
8993
8993
|
"path": "package-lock.json",
|
|
8994
8994
|
"mode": "0000644",
|
|
8995
|
-
"sha256": "
|
|
8995
|
+
"sha256": "59289cb04f2457826ed40890631dffdbf24e079c5af32938989be59732703cd6"
|
|
8996
8996
|
},
|
|
8997
8997
|
{
|
|
8998
8998
|
"path": "package.json",
|
|
8999
8999
|
"mode": "0000644",
|
|
9000
|
-
"sha256": "
|
|
9000
|
+
"sha256": "95aa8bfbee50b46e7eb01de527366c0ed88c2d262b00f1550a7eb177cf90967d"
|
|
9001
9001
|
},
|
|
9002
9002
|
{
|
|
9003
9003
|
"path": "public-docs/index.html",
|
|
@@ -9017,7 +9017,7 @@
|
|
|
9017
9017
|
{
|
|
9018
9018
|
"path": "release-notes.json",
|
|
9019
9019
|
"mode": "0000644",
|
|
9020
|
-
"sha256": "
|
|
9020
|
+
"sha256": "f93660db46a480e8ed3fd8302937423a3b69843192b2e052ddd4c3e17e0614b0"
|
|
9021
9021
|
},
|
|
9022
9022
|
{
|
|
9023
9023
|
"path": "scripts/bongos-mcp.js",
|
|
@@ -9172,12 +9172,12 @@
|
|
|
9172
9172
|
{
|
|
9173
9173
|
"path": "scripts/gds/autobongos-loop.js",
|
|
9174
9174
|
"mode": "0000644",
|
|
9175
|
-
"sha256": "
|
|
9175
|
+
"sha256": "6aba5f9b059680ea802993d1d79dc24caa9e16de35fdc334344dbec27d82fad0"
|
|
9176
9176
|
},
|
|
9177
9177
|
{
|
|
9178
9178
|
"path": "scripts/gds/autobongos-run.js",
|
|
9179
9179
|
"mode": "0000644",
|
|
9180
|
-
"sha256": "
|
|
9180
|
+
"sha256": "ac0d77b54920e50205546db4269957fd6a0fc2f1ba1aca9c097abc920fa04f2a"
|
|
9181
9181
|
},
|
|
9182
9182
|
{
|
|
9183
9183
|
"path": "scripts/gds/autobongos-service.cmd",
|
|
@@ -11157,7 +11157,7 @@
|
|
|
11157
11157
|
{
|
|
11158
11158
|
"path": "src/module-api.js",
|
|
11159
11159
|
"mode": "0000644",
|
|
11160
|
-
"sha256": "
|
|
11160
|
+
"sha256": "df86f50ae0baa2f659cc454dfd08b57841fe02ae2704c4a40c425f36046d794c"
|
|
11161
11161
|
},
|
|
11162
11162
|
{
|
|
11163
11163
|
"path": "src/module-loader/catalog.js",
|
|
@@ -11602,7 +11602,7 @@
|
|
|
11602
11602
|
{
|
|
11603
11603
|
"path": "tests/autobongos_cadence.mjs",
|
|
11604
11604
|
"mode": "0000644",
|
|
11605
|
-
"sha256": "
|
|
11605
|
+
"sha256": "67b8d8868952172945c1db8f6fa2fe2e3a28212c49afc2b2378e0801df000441"
|
|
11606
11606
|
},
|
|
11607
11607
|
{
|
|
11608
11608
|
"path": "tests/autobongos_fence.mjs",
|
|
@@ -11617,7 +11617,7 @@
|
|
|
11617
11617
|
{
|
|
11618
11618
|
"path": "tests/autobongos_loop.mjs",
|
|
11619
11619
|
"mode": "0000644",
|
|
11620
|
-
"sha256": "
|
|
11620
|
+
"sha256": "91e92c46fe27a4382a762029418437642b038a77286de0977ac694397759a0b6"
|
|
11621
11621
|
},
|
|
11622
11622
|
{
|
|
11623
11623
|
"path": "tests/autobongos_service_ps1.mjs",
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
**Date:** 2026-06-07
|
|
4
4
|
**Context:** GDS-V3, task [#679](https://example.com/builders#/task/679) (cut prod deploy over to CI). Completes the [#597](https://example.com/builders#/task/597) / [ADR 0031](0031-cloud-dev-environments-for-builders.md) §6 goal — *no builder box holds the production droplet SSH key*. Extends [ADR 0031](0031-cloud-dev-environments-for-builders.md) §6; re-applies [ADR 0016](0016-trust-boundary-server-enforced-permissions.md) to CI.
|
|
5
|
-
**Status:** Accepted — **armed in production 2026-06-13** (interim, grader-off variant — task [#859](https://example.com/builders#/task/859); `config/deploy.json` `mode: ci`; see the *Arming record* below + session log [`<redacted>.md`](../session-logs/<redacted>.md)).
|
|
5
|
+
**Status:** Accepted — **armed in production 2026-06-13** (interim, grader-off variant — task [#859](https://example.com/builders#/task/859); `config/deploy.json` `mode: ci`; see the *Arming record* below + session log [`<redacted>.md`](../session-logs/<redacted>.md)).
|
|
6
6
|
**Amended 2026-09-30:** the CI grade-gate (precondition 1) is permanently retired, not deferred — see the *Amendment* at the end (task 1004456, blocker 1000103).
|
|
7
7
|
|
|
8
8
|
## Problem
|
|
9
9
|
|
|
@@ -35,7 +35,7 @@ With no human in the merge loop, **the CI checks *are* the authority**. The thre
|
|
|
35
35
|
|
|
36
36
|
1. **[`CODEOWNERS`](../../CODEOWNERS)** — a PR touching any gate surface (`.github/`, `config/deploy.json`, `ship.js`, the grader, `auth.js`/`route-rank-check.js`) **or any code that executes on the droplet at deploy** (`migrations/`, `scripts/migrate.sh`, `package.json`/`package-lock.json`, `scripts/deploy/` — `~/deploy.sh` runs migrations + `npm ci` from the merged tree, so a malicious migration or poisoned lockfile would otherwise auto-merge and run on prod) requires Archon review *even under auto-merge*, when branch protection has "Require review from Code Owners" on. Ordinary PRs (game, art, docs) carry no code owner and auto-merge freely — that's the point.
|
|
37
37
|
|
|
38
|
-
2. **The grader must RE-RUN in CI** — CODEOWNERS only catches *committed* changes to `ship.js`. A builder could run a **locally-edited, uncommitted** `ship.js` that skips the grade and opens an auto-merge PR anyway. An earlier draft of this ADR proposed a CI check that *reads the server-recorded grade* — but that is **theater**: `POST /tasks/:id/grade` records the grader scores *from the request body* (the server does not recompute them), so a tampered local ship.js can POST a fake passing grade. The only tamper-proof gate runs the grader **in CI, against the PR diff**, where the builder cannot touch it — [`.github/workflows/grade-gate.yml`](../../.github/workflows/grade-gate.yml) + [`scripts/gds/ci-grade.js`](../../scripts/gds/ci-grade.js) ([#855](https://example.com/builders#/task/855)), which reuse the exact production grading path (`ship.js runSubagentGrader` → the adversarial worker panel) and **fail closed**. It is **skip-safe** (green-but-not-gating) until the operator wires a grader auth secret AND marks it a required check — the same inert-until-armed pattern as `deploy-prod.yml`. This
|
|
38
|
+
2. **The grader must RE-RUN in CI** — CODEOWNERS only catches *committed* changes to `ship.js`. A builder could run a **locally-edited, uncommitted** `ship.js` that skips the grade and opens an auto-merge PR anyway. An earlier draft of this ADR proposed a CI check that *reads the server-recorded grade* — but that is **theater**: `POST /tasks/:id/grade` records the grader scores *from the request body* (the server does not recompute them), so a tampered local ship.js can POST a fake passing grade. The only tamper-proof gate runs the grader **in CI, against the PR diff**, where the builder cannot touch it — [`.github/workflows/grade-gate.yml`](../../.github/workflows/grade-gate.yml) + [`scripts/gds/ci-grade.js`](../../scripts/gds/ci-grade.js) ([#855](https://example.com/builders#/task/855)), which reuse the exact production grading path (`ship.js runSubagentGrader` → the adversarial worker panel) and **fail closed**. It is **skip-safe** (green-but-not-gating) until the operator wires a grader auth secret AND marks it a required check — the same inert-until-armed pattern as `deploy-prod.yml`. This was written as a **hard precondition** of arming `ci` mode. *(Superseded 2026-09-30: the CI grade-gate is permanently retired — see the Amendment below.)*
|
|
39
39
|
|
|
40
40
|
### Why not just hand builders the droplet key
|
|
41
41
|
|
|
@@ -43,7 +43,7 @@ Rejected. It puts a full shell on the production box (prod DB, every secret) on
|
|
|
43
43
|
|
|
44
44
|
## Cutover preconditions (do NOT arm `ci` until ALL hold)
|
|
45
45
|
|
|
46
|
-
1.
|
|
46
|
+
1. ~~**CI grade-gate armed + verified**~~ *(retired 2026-09-30 — see the Amendment below; never armed)* — the grader RE-RUN-in-CI check (`grade-gate.yml` + `ci-grade.js`, [#855](https://example.com/builders#/task/855)) is built and skip-safe but NOT yet a real gate. The operator must (a) add a grader auth secret (`ANTHROPIC_API_KEY` or `CLAUDE_CODE_OAUTH_TOKEN` — the grader spawns the `claude` CLI, ~4 model calls/PR), (b) verify a real grader-in-CI run passes a known-good PR and fails a known-bad one, and (c) mark `grade-gate` a required status check on `main`. *Without this, `ci` mode lets a tampered local ship.js bypass the quality grader.* (Built in [#855](https://example.com/builders#/task/855); arming/verification is part of the cutover.)
|
|
47
47
|
2. **Functional checks required** — the DB-free **unit suite** runs as a PR check ([`.github/workflows/pr-unit-tests.yml`](../../.github/workflows/pr-unit-tests.yml) + [`scripts/gds/run-unit-tests.js`](../../scripts/gds/run-unit-tests.js), 56 tests, [#856](https://example.com/builders#/task/856)); the operator marks it required at cutover. The **integration smoke** (full server + Postgres service container) is a follow-up ([#862](https://example.com/builders#/task/862)). The existing `secrets-scan` / `dep-audit` already run on PRs.
|
|
48
48
|
3. **Branch protection on `main`** — require a PR (block direct pushes), require the checks in (1)+(2) + Code-Owner review, enable auto-merge.
|
|
49
49
|
4. **`deploy-prod.yml` armed** — `PROD_SSH_KEY` + `DROPLET_KNOWN_HOSTS` secrets + `PROD_DEPLOY_VIA_CI=true` (a dedicated deploy keypair, not a personal key).
|
|
@@ -60,7 +60,7 @@ Until (1)–(5), the switch stays `laptop` and the operator lands work as today
|
|
|
60
60
|
|
|
61
61
|
## Follow-ups (filed as GDS tasks)
|
|
62
62
|
|
|
63
|
-
- CI **grade-gate** reading the server-recorded grade — **blocks cutover** (precondition 1).
|
|
63
|
+
- ~~CI **grade-gate** reading the server-recorded grade — **blocks cutover** (precondition 1).~~ Retired 2026-09-30 (Amendment below).
|
|
64
64
|
- PR-check workflows: unit suite + integration smoke (Postgres service) as required checks (precondition 2).
|
|
65
65
|
- Drop the per-builder `gh` dependency (server-mediated PR creation) + `doctor.js gh` check.
|
|
66
66
|
- Commit/version endpoint for a commit-pinned edge deploy confirmation.
|
|
@@ -79,4 +79,18 @@ during the provider outage. With the grader off everywhere, gating auto-merge on
|
|
|
79
79
|
|
|
80
80
|
What was set: `PROD_SSH_KEY` (a dedicated ed25519 deploy key, **forced-command `~/deploy.sh`** in the droplet's `authorized_keys`) + `DROPLET_KNOWN_HOSTS` secrets; repo var `PROD_DEPLOY_VIA_CI=true`; branch protection on `main` (require PR, required checks `unit`+`trufflehog`, require Code-Owner review, 0 ordinary approvals, block direct push for non-admins); repo auto-merge enabled; `config/deploy.json` → `ci`. `deploy-prod.yml` verified end-to-end (Actions → SSH → `~/deploy.sh` → public healthz) before the flip.
|
|
81
81
|
|
|
82
|
-
**Tighten-later (re-arm when the grader returns):** turn `GRADER_BYPASS_ENABLED` off
|
|
82
|
+
**Tighten-later (re-arm when the grader returns):** turn `GRADER_BYPASS_ENABLED` off ~~, re-introduce the `grade-gate` required check (precondition 1, needs an `ANTHROPIC_API_KEY` Actions secret)~~ *(the grade-gate leg is retired — Amendment below)*, and land the integration-smoke gate (task 862, precondition 2). **Rollback:** set `config/deploy.json` → `laptop` **and** unset the `PROD_DEPLOY_VIA_CI` repo var (admins retain direct main-push for this).
|
|
83
|
+
|
|
84
|
+
## Amendment — 2026-09-30: the CI grade-gate is permanently retired
|
|
85
|
+
|
|
86
|
+
**Owner ruling** (blocker 1000103, recorded under task 1004456): precondition 1 — the grader re-run in CI (`grade-gate.yml` + `scripts/gds/ci-grade.js`) — is **retired, not deferred**. It was never armed: `grade-gate.yml` does not exist and `ci-grade.js` has no caller. This ADR no longer promises it.
|
|
87
|
+
|
|
88
|
+
**What holds instead** is the layered model [ADR 0164](0164-implausible-pass-guard-substance-and-evidence.md) documents:
|
|
89
|
+
|
|
90
|
+
- **Grade time — honest-mistake hygiene.** The panel verdict stays *client-asserted* (`POST /tasks/:id/grade` records scores from the request body). The implausible-pass guard catches empty or evidence-free passes; it is not a tamper-proof gate and does not claim to be.
|
|
91
|
+
- **Land time — the binding authorities.** The required CI `unit` check, gate-review, Code-Owner review on gate surfaces, and the deploy-time main audit ([ADR 0043](0043-git-ssh-trust-boundary-and-rank-floor-on-permission-paths.md)) decide what reaches `main`, whatever the grade said.
|
|
92
|
+
- **Metic+ grade-time honesty is accepted honor-system.** The forgery classes this leaves uncaught are listed in ADR 0164 §3.
|
|
93
|
+
|
|
94
|
+
**Why B over landing the gate.** Across the 106-grade evaluation window (goal 1000056) there were zero known forged passes. A CI re-grade would roughly double grading spend per ship and add a new required check that strands every ship whenever the grader is down — the outage class later work spent two waves defusing. The honor-system residual is judged cheaper than that.
|
|
95
|
+
|
|
96
|
+
**Revisit if** a forged or tampered grade is ever found on `main`: the gate design in §*The auto-merge gate must be tamper-proof* (item 2) remains the reference for re-introducing it.
|
package/docs/adr/README.md
CHANGED
|
@@ -134,7 +134,7 @@ These 20 numbers are each shared by exactly two files. They are **accepted histo
|
|
|
134
134
|
| 0039 | [Setup-first onboarding UX: two-config selector + action-only Path checklist](0039-setup-first-onboarding-ux.md) | GDS / builder onboarding |
|
|
135
135
|
| 0040 | [Remote Control is the default browser-only on-ramp (reverses 0038's "RC is dead" verdict)](0040-remote-control-default-browser-onramp.md) | GDS / builder onboarding |
|
|
136
136
|
| 0041 | [Temporary global grader-bypass kill-switch (`GRADER_BYPASS_ENABLED`)](0041-temporary-grader-bypass-killswitch.md) | GDS / shipping pipeline |
|
|
137
|
-
| 0042 | [Builder self-deploy: CI auto-deploy on main + PR auto-merge (no builder box holds the prod key)](0042-builder-self-deploy-ci-auto-merge.md) | GDS / shipping pipeline |
|
|
137
|
+
| 0042 | [Builder self-deploy: CI auto-deploy on main + PR auto-merge (no builder box holds the prod key). **Amended 2026-09-30:** the CI grade-gate (precondition 1) is permanently retired in favour of ADR 0164's layered model](0042-builder-self-deploy-ci-auto-merge.md) | GDS / shipping pipeline |
|
|
138
138
|
| 0043 | [Git/SSH trust-boundary gap + a rank floor on permission-sensitive paths](0043-git-ssh-trust-boundary-and-rank-floor-on-permission-paths.md) | GDS / trust boundary |
|
|
139
139
|
| 0044-a | [Full Mediterranean palette replacement (not a split)](0044-mediterranean-palette-replacement.md) | Art pipeline / palette |
|
|
140
140
|
| 0044-b | [Per-box live game preview (the builder sandbox)](0044-per-box-live-game-preview.md) | GDS / builder experience |
|
|
@@ -2722,6 +2722,8 @@ is load-bearing: the script throws rather than guess if it is missing, and
|
|
|
2722
2722
|
1.20.32 — CI auto-patch (publish-on-merge, ADR 0161): carrier for merges
|
|
2723
2723
|
landed since 1.20.31 with no explicit bump. run 36781343271. (task 1002620)
|
|
2724
2724
|
1.20.33 — CI auto-patch (publish-on-merge, ADR 0161): carrier for merges
|
|
2725
|
-
landed since 1.20.32 with no explicit bump. run
|
|
2725
|
+
landed since 1.20.32 with no explicit bump. run 36783765458. (task 1002620)
|
|
2726
|
+
1.20.34 — CI auto-patch (publish-on-merge, ADR 0161): carrier for merges
|
|
2727
|
+
landed since 1.20.33 with no explicit bump. run 36784565937. (task 1002620)
|
|
2726
2728
|
---------------------------------------------------------------------------
|
|
2727
2729
|
```
|
|
@@ -178,7 +178,7 @@ It is the machine's working copy, not a place to edit.
|
|
|
178
178
|
| Runner line says **Not checking in** | is the PC on? `...\autobongos-service.ps1 status` on the box |
|
|
179
179
|
| **Checking in** but nothing ever ships | first the Gate page — is a goal allowlisted, and is the switch on? If both are on, the **gauge**: `autobongos-service.ps1 digest`, or `node scripts/gds/autonomy-gauge.js`. A `hold` with `unknown: true` means it cannot read your remaining capacity, and it will wait forever rather than guess |
|
|
180
180
|
| The log repeats `hold ... unknown: true` | the gauge has no reading it can vouch for. Open a **terminal** `claude` session on the box once — that writes it. A desktop-app session does not: the writer is a statusLine command, and only a terminal renders one |
|
|
181
|
-
| `hold` naming
|
|
181
|
+
| `hold` naming the **usage limit** | working as designed. A worker was cut off by the subscription limit, so the runner sleeps until the reset time and then resumes by itself. There is no fixed task cap (task 1004454) |
|
|
182
182
|
| Shipping stopped after a while | the log: the runner backs off after three failures and probes; the reason is in the log |
|
|
183
183
|
| It keeps restarting | `status` shows the last result; the `====` banners in the log mark each start |
|
|
184
184
|
|
|
@@ -110,6 +110,14 @@ function classifyEvent(row = {}) {
|
|
|
110
110
|
// counting it as such would hold a broken runner at full speed indefinitely.
|
|
111
111
|
if (row.verified === true) return { class: 'progress', detail: 'a task shipped and the ledger confirms it' };
|
|
112
112
|
if (row.ledger_shape === 'undetermined') return { class: 'transient', detail: 'the ledger could not be read back' };
|
|
113
|
+
// CUT OFF BY THE USAGE LIMIT (task 1004454). Checked before workerNeverRan,
|
|
114
|
+
// whose shape it shares (seconds long, nothing spent): the machine is fine,
|
|
115
|
+
// the window is spent, and the owner has said hitting it is acceptable. So it
|
|
116
|
+
// is a wait to the reset, never a step toward probing.
|
|
117
|
+
if (row.usage_limit) {
|
|
118
|
+
const resetAt = row.usage_limit.reset_at ?? null;
|
|
119
|
+
return { class: 'quiet', detail: `the usage limit was hit — sleeping to its reset${resetAt ? '' : ' (time not stated; a bounded wait)'}`, sleepUntil: resetAt };
|
|
120
|
+
}
|
|
113
121
|
// DID THE WORKER ACTUALLY RUN? (task 1004179, found by the 1003907 proof run.)
|
|
114
122
|
//
|
|
115
123
|
// Everything below this line assumes a worker that ran and met a real
|
|
@@ -283,7 +283,11 @@ function decideCapacity({ gauge, now, config = {} } = {}) {
|
|
|
283
283
|
// actually binds a heavy week — does not run at all. A stale weekly figure
|
|
284
284
|
// cannot stand in for it either: usage only grows, so an under-reported value
|
|
285
285
|
// makes the check MORE permissive, which is the wrong direction to be wrong in.
|
|
286
|
-
// Bounding the work done since `derivedSince` is what
|
|
286
|
+
// Bounding the work done since `derivedSince` is what replaced that pacing,
|
|
287
|
+
// until the owner ruled against a fixed cap (task 1004454, 2026-09-30): the
|
|
288
|
+
// runner now works on a derived reading and stops when a worker is actually cut
|
|
289
|
+
// off by the limit, sleeping to its reset. The field is kept on the decision so
|
|
290
|
+
// the log still says which boundary a derived reading stood on.
|
|
287
291
|
//
|
|
288
292
|
// The MOST RECENT boundary, not the oldest (task 1004376). A derived seven-day
|
|
289
293
|
// window starts up to a week back; counting from there would make the cap a
|
package/package-lock.json
CHANGED
|
@@ -1,12 +1,12 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@bongos/core",
|
|
3
|
-
"version": "1.20.
|
|
3
|
+
"version": "1.20.34",
|
|
4
4
|
"lockfileVersion": 3,
|
|
5
5
|
"requires": true,
|
|
6
6
|
"packages": {
|
|
7
7
|
"": {
|
|
8
8
|
"name": "@bongos/core",
|
|
9
|
-
"version": "1.20.
|
|
9
|
+
"version": "1.20.34",
|
|
10
10
|
"license": "AGPL-3.0-or-later",
|
|
11
11
|
"dependencies": {
|
|
12
12
|
"express": "^4.21.2",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@bongos/core",
|
|
3
|
-
"version": "1.20.
|
|
3
|
+
"version": "1.20.34",
|
|
4
4
|
"description": "Cloud Bongos — the AI-first build platform core (GDS + platform surfaces + module system), installed as a versioned dependency (ADR 0108).",
|
|
5
5
|
"license": "AGPL-3.0-or-later",
|
|
6
6
|
"main": "src/platform-server.js",
|
package/release-notes.json
CHANGED
|
@@ -8266,6 +8266,10 @@
|
|
|
8266
8266
|
}
|
|
8267
8267
|
],
|
|
8268
8268
|
"1.20.33": [
|
|
8269
|
+
{
|
|
8270
|
+
"id": "1004456",
|
|
8271
|
+
"text": "The self-deploy decision record no longer promises a CI re-grade check that was never built; it now records the owner's choice to retire it."
|
|
8272
|
+
},
|
|
8269
8273
|
{
|
|
8270
8274
|
"id": "1004396",
|
|
8271
8275
|
"text": "The unattended runner no longer gets stuck retrying a task it isn't allowed to take: it moves on to the next one, and only waits when a stuck ship is blocking everyone, resuming by itself once that lands."
|
|
@@ -8274,5 +8278,11 @@
|
|
|
8274
8278
|
"id": "1004449",
|
|
8275
8279
|
"text": "When a project update fails in a way the system can't fix by itself, it now opens one blocker for a person and posts one message on Discord, instead of failing quietly. When a later update succeeds, the blocker closes by itsel"
|
|
8276
8280
|
}
|
|
8281
|
+
],
|
|
8282
|
+
"1.20.34": [
|
|
8283
|
+
{
|
|
8284
|
+
"id": "1004454",
|
|
8285
|
+
"text": "The unattended runner no longer stops after two tasks: it keeps working while there's usage left, and if it hits the usage limit it sleeps until the limit resets and then carries on by itself."
|
|
8286
|
+
}
|
|
8277
8287
|
]
|
|
8278
8288
|
}
|
|
@@ -164,7 +164,43 @@ function classifyRun({ envelope = null, exitCode = 0, timedOut = false, spawnErr
|
|
|
164
164
|
return { outcome: 'claims_shipped', reason: parsed.verdict.what_happened, releaseClaim: false, verdict: parsed.verdict, costUsd: parsed.costUsd };
|
|
165
165
|
}
|
|
166
166
|
|
|
167
|
+
// usageLimitHit — was this worker cut off by the subscription's usage limit?
|
|
168
|
+
// (task 1004454)
|
|
169
|
+
//
|
|
170
|
+
// Owner ruling, 2026-09-30: no fixed cap on work; hitting the limit is fine, and
|
|
171
|
+
// the runner restarts once it resets. That makes the limit message the stop
|
|
172
|
+
// signal, so it has to be recognised — otherwise a limit-hit worker (seconds
|
|
173
|
+
// long, nothing spent) reads as workerNeverRan() and escalates the runner into
|
|
174
|
+
// probing for a condition the owner accepts.
|
|
175
|
+
//
|
|
176
|
+
// Only an ERROR envelope, or a run with no parseable envelope, can be a limit
|
|
177
|
+
// hit. A worker that FINISHED may well write about limits in its verdict (it
|
|
178
|
+
// might have fixed a rate-limit bug), and sleeping on that would stop the runner
|
|
179
|
+
// for doing its job.
|
|
180
|
+
//
|
|
181
|
+
// The CLI's machine form is `Claude AI usage limit reached|<epoch seconds>`; the
|
|
182
|
+
// human forms ("You've hit your limit · resets 3am", "5-hour limit reached ∙
|
|
183
|
+
// resets 3pm") carry a clock time in a zone we would have to guess, so they
|
|
184
|
+
// return resetAt: null and the caller picks a bounded wait instead of parsing one.
|
|
185
|
+
const LIMIT_EPOCH_RE = /usage limit reached\|(\d{9,11})/i;
|
|
186
|
+
const LIMIT_TEXT_RE = /usage limit reached|hit your (usage )?limit|\b(5|five)-hour limit reached|weekly limit reached|limit reached\s*[∙·•|-]\s*resets/i;
|
|
187
|
+
function usageLimitHit({ envelope = null, stderr = '' } = {}) {
|
|
188
|
+
let text = '';
|
|
189
|
+
let parsedEnvelope = null;
|
|
190
|
+
try { parsedEnvelope = typeof envelope === 'string' ? (envelope.trim() ? JSON.parse(envelope) : null) : envelope; } catch (_) { parsedEnvelope = null; }
|
|
191
|
+
if (parsedEnvelope && typeof parsedEnvelope === 'object') {
|
|
192
|
+
if (parsedEnvelope.is_error !== true) return null;
|
|
193
|
+
text = String(parsedEnvelope.result || '');
|
|
194
|
+
} else if (typeof envelope === 'string') {
|
|
195
|
+
text = envelope;
|
|
196
|
+
}
|
|
197
|
+
text = `${text}\n${stderr || ''}`;
|
|
198
|
+
const m = LIMIT_EPOCH_RE.exec(text);
|
|
199
|
+
if (m) return { resetAt: Number(m[1]) };
|
|
200
|
+
return LIMIT_TEXT_RE.test(text) ? { resetAt: null } : null;
|
|
201
|
+
}
|
|
202
|
+
|
|
167
203
|
module.exports = {
|
|
168
204
|
WORKER_VERDICT_SCHEMA, DEFAULT_MODEL, DEFAULT_WORKER_TIMEOUT_MS,
|
|
169
|
-
buildWorkerArgs, buildWorkerPrompt, parseWorkerEnvelope, classifyRun,
|
|
205
|
+
buildWorkerArgs, buildWorkerPrompt, parseWorkerEnvelope, classifyRun, usageLimitHit,
|
|
170
206
|
};
|
|
@@ -40,7 +40,7 @@ const os = require('node:os');
|
|
|
40
40
|
const seq = require('./sequence.js');
|
|
41
41
|
const gauge = require('./autonomy-gauge.js');
|
|
42
42
|
const {
|
|
43
|
-
buildWorkerArgs, buildWorkerPrompt, classifyRun, DEFAULT_MODEL, DEFAULT_WORKER_TIMEOUT_MS,
|
|
43
|
+
buildWorkerArgs, buildWorkerPrompt, classifyRun, usageLimitHit, DEFAULT_MODEL, DEFAULT_WORKER_TIMEOUT_MS,
|
|
44
44
|
} = require('./autobongos-loop.js');
|
|
45
45
|
const { verifyShip, failureReason } = require('./autobongos-verify.js');
|
|
46
46
|
const fence = require('../../modules/autonomy/fence.js');
|
|
@@ -95,55 +95,6 @@ function logPath() {
|
|
|
95
95
|
try { return require('../../src/instance-config.js').configPath('autobongos-runs.jsonl'); }
|
|
96
96
|
catch (_) { return path.join(REPO_ROOT, 'autobongos-runs.jsonl'); }
|
|
97
97
|
}
|
|
98
|
-
// countWorkedSince — how many tasks this runner has worked since an epoch-second
|
|
99
|
-
// mark. Reads the run log, because that is the only record that survives a
|
|
100
|
-
// restart, and the cap it feeds has to hold across one (a runner that forgets
|
|
101
|
-
// its count on every crash has no cap at all).
|
|
102
|
-
//
|
|
103
|
-
// BOUNDED ON BOTH AXES, because this runs every iteration of a loop that never
|
|
104
|
-
// exits and the log only grows. It reads the TAIL rather than the file, and it
|
|
105
|
-
// scans BACKWARDS and stops at the first row older than the mark — the rows that
|
|
106
|
-
// can possibly count are the newest ones, so the common case touches a handful
|
|
107
|
-
// of lines whatever the log's size.
|
|
108
|
-
//
|
|
109
|
-
// A torn or missing log counts as zero rather than refusing: the fence and the
|
|
110
|
-
// gauge are the gates, this is a bound on top of them, and a bound that bricks
|
|
111
|
-
// the runner when its own log is unreadable is worse than one that occasionally
|
|
112
|
-
// allows an extra task.
|
|
113
|
-
const WORKED_SCAN_TAIL_BYTES = 1024 * 1024;
|
|
114
|
-
const NEWLINE = String.fromCharCode(10);
|
|
115
|
-
|
|
116
|
-
function countWorkedSince(sinceEpochS) {
|
|
117
|
-
if (!Number.isFinite(sinceEpochS) || sinceEpochS <= 0) return 0;
|
|
118
|
-
let text;
|
|
119
|
-
try {
|
|
120
|
-
const p = logPath();
|
|
121
|
-
const { size } = fs.statSync(p);
|
|
122
|
-
const from = Math.max(0, size - WORKED_SCAN_TAIL_BYTES);
|
|
123
|
-
const fd = fs.openSync(p, 'r');
|
|
124
|
-
try {
|
|
125
|
-
const buf = Buffer.alloc(Math.min(size, WORKED_SCAN_TAIL_BYTES));
|
|
126
|
-
fs.readSync(fd, buf, 0, buf.length, from);
|
|
127
|
-
text = buf.toString('utf8');
|
|
128
|
-
} finally { fs.closeSync(fd); }
|
|
129
|
-
// A tail read can land mid-line; drop the first partial one.
|
|
130
|
-
if (from > 0) text = text.slice(text.indexOf(NEWLINE) + 1);
|
|
131
|
-
} catch (_) { return 0; }
|
|
132
|
-
|
|
133
|
-
const lines = text.split(NEWLINE);
|
|
134
|
-
let n = 0;
|
|
135
|
-
for (let i = lines.length - 1; i >= 0; i--) {
|
|
136
|
-
const line = lines[i];
|
|
137
|
-
if (!line.trim()) continue;
|
|
138
|
-
let row;
|
|
139
|
-
try { row = JSON.parse(line); } catch (_) { continue; } // a torn line costs one row, not the count
|
|
140
|
-
const at = Date.parse(row.at);
|
|
141
|
-
if (!Number.isFinite(at)) continue;
|
|
142
|
-
if (at / 1000 < sinceEpochS) break; // the log is append-ordered: nothing older can count
|
|
143
|
-
if (row.event === 'worked') n += 1;
|
|
144
|
-
}
|
|
145
|
-
return n;
|
|
146
|
-
}
|
|
147
98
|
|
|
148
99
|
function record(entry) {
|
|
149
100
|
const row = { at: new Date().toISOString(), ...entry };
|
|
@@ -401,9 +352,14 @@ const CLAIM_SET_ASIDE_MS = 30 * 60 * 1000;
|
|
|
401
352
|
// A bound on claim attempts in ONE pass, so a queue of refusals cannot turn one
|
|
402
353
|
// iteration into dozens of worktree creations.
|
|
403
354
|
const MAX_CLAIM_ATTEMPTS = 5;
|
|
355
|
+
// A limit message with no machine-readable reset (the CLI's human forms name a
|
|
356
|
+
// clock time in a zone we would have to guess). Half an hour is short against a
|
|
357
|
+
// five-hour window and long against a spawn that is cut off in two seconds, so a
|
|
358
|
+
// guess that is wrong costs one wasted spawn per half hour, not one per wake.
|
|
359
|
+
const LIMIT_UNKNOWN_RESET_MS = 30 * 60 * 1000;
|
|
404
360
|
|
|
405
361
|
function newRunnerState() {
|
|
406
|
-
return { setAside: new Map(), queueGate: null };
|
|
362
|
+
return { setAside: new Map(), queueGate: null, limitUntil: null };
|
|
407
363
|
}
|
|
408
364
|
const RUNNER_STATE = newRunnerState();
|
|
409
365
|
|
|
@@ -474,42 +430,30 @@ async function iteration(opts, deps = {}) {
|
|
|
474
430
|
return log({ event: 'hold', go: false, unknown: !!g.decision.unknown, reason: g.decision.reason, sleep_until: g.decision.sleepUntil ?? null });
|
|
475
431
|
}
|
|
476
432
|
|
|
477
|
-
// 1b.
|
|
478
|
-
//
|
|
479
|
-
//
|
|
480
|
-
//
|
|
481
|
-
//
|
|
482
|
-
//
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
//
|
|
488
|
-
// So the bound is a task count, not a percentage: on one derived window, do
|
|
489
|
-
// AUTOBONGOS_DERIVED_TASK_CAP tasks and then hold until a real reading exists.
|
|
490
|
-
// A terminal session writes one; the owner opening a terminal in the morning
|
|
491
|
-
// is what lifts it. Default 2 — each worker is wall-clock bounded at 90
|
|
492
|
-
// minutes, so two of them is well inside a five-hour window even at worst, and
|
|
493
|
-
// the cap only has to be small enough that being wrong is survivable.
|
|
494
|
-
if (g.decision.derived) {
|
|
495
|
-
const cap = Number(process.env.AUTOBONGOS_DERIVED_TASK_CAP ?? 2);
|
|
496
|
-
const since = Number(g.decision.derivedSince) || 0;
|
|
497
|
-
const done = countWorkedSince(since);
|
|
498
|
-
if (Number.isFinite(cap) && cap >= 0 && done >= cap) {
|
|
433
|
+
// 1b. The usage limit, once HIT (task 1004454). Owner ruling 2026-09-30: no
|
|
434
|
+
// fixed cap on how much work a reading buys — the gauge above is read before
|
|
435
|
+
// every task and a derived reading proceeds — and hitting the limit is
|
|
436
|
+
// acceptable, so long as the runner then waits for the reset and resumes by
|
|
437
|
+
// itself. A worker cut off by the limit records when it resets; until then no
|
|
438
|
+
// task is taken, because every worker spawned would be cut off the same way.
|
|
439
|
+
const state = deps.state || RUNNER_STATE;
|
|
440
|
+
const now = deps.now ? deps.now() : Date.now();
|
|
441
|
+
if (state.limitUntil) {
|
|
442
|
+
if (now < state.limitUntil) {
|
|
499
443
|
return log({
|
|
500
|
-
event: 'hold', go: false, unknown: false,
|
|
501
|
-
reason:
|
|
502
|
-
sleep_until:
|
|
444
|
+
event: 'hold', go: false, unknown: false, usage_limit: true,
|
|
445
|
+
reason: 'the subscription usage limit was hit — sleeping to its reset, then resuming by itself',
|
|
446
|
+
sleep_until: Math.round(state.limitUntil / 1000),
|
|
503
447
|
});
|
|
504
448
|
}
|
|
449
|
+
log({ event: 'usage_limit_reset', was_until: new Date(state.limitUntil).toISOString() });
|
|
450
|
+
state.limitUntil = null;
|
|
505
451
|
}
|
|
506
452
|
|
|
507
453
|
// 1c. A QUEUE-WIDE gate from an earlier pass (task 1004396). While the
|
|
508
454
|
// stranded task it named is still `confirmed`, every claim would be refused, so
|
|
509
455
|
// none is attempted. The wait wakes early when main moves (waitOrJump), which is
|
|
510
456
|
// exactly what the strand landing does.
|
|
511
|
-
const state = deps.state || RUNNER_STATE;
|
|
512
|
-
const now = deps.now ? deps.now() : Date.now();
|
|
513
457
|
if (state.queueGate) {
|
|
514
458
|
if (await queueGateHolds(state.queueGate, deps)) return log(queueGatedRow(state.queueGate, now));
|
|
515
459
|
log({ event: 'queue_gate_cleared', waited_on: state.queueGate.owed, waited_s: Math.max(0, Math.round((now - state.queueGate.since) / 1000)) });
|
|
@@ -592,6 +536,10 @@ async function iteration(opts, deps = {}) {
|
|
|
592
536
|
const started = Date.now();
|
|
593
537
|
const res = await spawnWork({ task, model: opts.model, timeoutMs: opts.timeoutMs, cwd: wtPath, deps });
|
|
594
538
|
const verdict = classifyRun(res);
|
|
539
|
+
// Cut off by the usage limit? Recorded here so the NEXT iteration sleeps to the
|
|
540
|
+
// reset instead of spawning a worker that will be cut off the same way.
|
|
541
|
+
const limit = verdict.outcome === 'claims_shipped' ? null : usageLimitHit({ envelope: res.envelope, stderr: res.stderr });
|
|
542
|
+
if (limit) state.limitUntil = limit.resetAt ? limit.resetAt * 1000 : now + LIMIT_UNKNOWN_RESET_MS;
|
|
595
543
|
|
|
596
544
|
// 6. VERIFY FROM THE LEDGER (task 1003903). Read the task back from Bongos and
|
|
597
545
|
// let IT say what happened — for every outcome, not only a claimed ship. The
|
|
@@ -623,6 +571,7 @@ async function iteration(opts, deps = {}) {
|
|
|
623
571
|
session_id: res.sessionId || null, duration_s: Math.round((Date.now() - started) / 1000),
|
|
624
572
|
cost_usd: verdict.costUsd ?? null,
|
|
625
573
|
output_truncated: !!res.truncated,
|
|
574
|
+
...(limit ? { usage_limit: { reset_at: limit.resetAt } } : {}),
|
|
626
575
|
undetermined_decisions: verdict.verdict ? verdict.verdict.undetermined_decisions : null,
|
|
627
576
|
// What the LEDGER says, kept separate from what the worker said, so the run
|
|
628
577
|
// log can be read afterwards without having to trust either one alone.
|
|
@@ -1022,6 +971,15 @@ async function forever(opts, deps = {}) {
|
|
|
1022
971
|
// corpse.
|
|
1023
972
|
await beat({ mode: state.mode, last_event: row.event, working_task_id: Number(row.task_id) || undefined });
|
|
1024
973
|
const seen = cadence.classifyEvent(row);
|
|
974
|
+
// A hold that names its reset SECOND waits until that second (task 1004454).
|
|
975
|
+
// classifyEvent carries it as an absolute epoch; the cadence has no clock of
|
|
976
|
+
// its own, so the conversion to a wait lives here beside the one clock read.
|
|
977
|
+
// Without it every hold waited the fixed idle and the "wake at the reset"
|
|
978
|
+
// the goal asks for was only ever as accurate as the poll.
|
|
979
|
+
if (Number.isFinite(Number(seen.sleepUntil)) && seen.sleepUntil !== null) {
|
|
980
|
+
const nowS = Math.floor((deps.now ? deps.now() : Date.now()) / 1000);
|
|
981
|
+
seen.sleepUntilS = Math.max(0, Number(seen.sleepUntil) - nowS);
|
|
982
|
+
}
|
|
1025
983
|
state = cadence.nextCadence(state, seen, opts.cadence);
|
|
1026
984
|
log({ event: 'cadence', mode: state.mode, consecutive_failures: state.consecutiveFailures, wait_s: state.waitS, class: seen.class, why: state.why });
|
|
1027
985
|
// A burst continues only while work is actually landing. Anything else ends
|
|
@@ -1082,8 +1040,8 @@ if (require.main === module) {
|
|
|
1082
1040
|
|
|
1083
1041
|
module.exports = {
|
|
1084
1042
|
readArgv, pickTask, iteration, record, logPath, spawnWorker, workerEnv, WORKER_ENV_ALLOW, resolveWorkerBin,
|
|
1085
|
-
worktreeName, readFence, reconcileLeftoverClaims, pollSignals, waitOrJump, forever, CLAIM_PREFIX,
|
|
1043
|
+
worktreeName, readFence, reconcileLeftoverClaims, pollSignals, waitOrJump, forever, CLAIM_PREFIX, codeDrifted, loadedHead, UPGRADE_EXIT_CODE, MIN_UPTIME_BEFORE_UPGRADE_MS,
|
|
1086
1044
|
heartbeatPath, writeHeartbeat, readHeartbeat, pidAlive, anotherRunnerIsAlive, HEARTBEAT_STALE_MS,
|
|
1087
1045
|
isRunnerTree, RUNNER_TREE_RE, postHeartbeat, RUNNER_STARTED_AT, failureReason,
|
|
1088
|
-
newRunnerState, QUEUE_GATED_EXIT, CLAIM_SET_ASIDE_MS, MAX_CLAIM_ATTEMPTS, owedTaskIds,
|
|
1046
|
+
newRunnerState, LIMIT_UNKNOWN_RESET_MS, QUEUE_GATED_EXIT, CLAIM_SET_ASIDE_MS, MAX_CLAIM_ATTEMPTS, owedTaskIds,
|
|
1089
1047
|
};
|
package/src/module-api.js
CHANGED
|
@@ -75,7 +75,7 @@ const { responsibilityFor, ROLE_RESPONSIBILITIES } = require('./role-responsibil
|
|
|
75
75
|
// MAJOR (see allowBoxScope below): passes the request through untouched.
|
|
76
76
|
function deprecatedNoopMiddleware(_req, _res, next) { next(); }
|
|
77
77
|
|
|
78
|
-
const CORE_VERSION = '1.20.
|
|
78
|
+
const CORE_VERSION = '1.20.34'; // CI auto-patch carrier (ADR 0161); changelog: docs/module-api-changelog.md
|
|
79
79
|
|
|
80
80
|
// A namespaced logger so a module's log lines are attributable + consistent.
|
|
81
81
|
// Usage: const log = api.logger('discord'); log.info('mounted');
|
|
@@ -293,6 +293,33 @@ function loopHarness({ rows, maxLoops = 6 }) {
|
|
|
293
293
|
return { deps, waits: emitted, recorded, opts: { goals: [], maxTasks: 3, maxLoops } };
|
|
294
294
|
}
|
|
295
295
|
|
|
296
|
+
// ── the usage limit: a wait to a known second, not a failure (task 1004454) ──
|
|
297
|
+
|
|
298
|
+
test('a worker cut off by the usage limit is quiet, not a failed launch', () => {
|
|
299
|
+
// It exits in seconds having spent nothing — exactly workerNeverRan()'s shape —
|
|
300
|
+
// but the machine is fine: the window is spent. Escalating it would put the
|
|
301
|
+
// runner into probing for a limit the owner has said is acceptable to hit.
|
|
302
|
+
const seen = classifyEvent({ event: 'worked', outcome: 'worker_failed', duration_s: 2, cost_usd: null, reason: 'x', usage_limit: { reset_at: 1790300000 } });
|
|
303
|
+
assert.equal(seen.class, 'quiet');
|
|
304
|
+
assert.equal(seen.sleepUntil, 1790300000);
|
|
305
|
+
});
|
|
306
|
+
|
|
307
|
+
test('a hold with a reset time makes the loop wait until THAT second, not a fixed idle', async () => {
|
|
308
|
+
const nowS = 1_790_290_000;
|
|
309
|
+
const h = loopHarness({ rows: [{ event: 'hold', go: false, reason: 'five-hour burn at the ceiling', sleep_until: nowS + 3600 }], maxLoops: 1 });
|
|
310
|
+
h.deps.now = () => nowS * 1000;
|
|
311
|
+
await runner.forever(h.opts, h.deps);
|
|
312
|
+
assert.equal(h.waits[0].waitS, 3600, 'the wake is the reset second the gauge named');
|
|
313
|
+
});
|
|
314
|
+
|
|
315
|
+
test('a hold whose reset has already passed waits no time at all', async () => {
|
|
316
|
+
const nowS = 1_790_290_000;
|
|
317
|
+
const h = loopHarness({ rows: [{ event: 'hold', go: false, reason: 'x', sleep_until: nowS - 5 }], maxLoops: 1 });
|
|
318
|
+
h.deps.now = () => nowS * 1000;
|
|
319
|
+
await runner.forever(h.opts, h.deps);
|
|
320
|
+
assert.equal(h.waits[0].waitS, 0);
|
|
321
|
+
});
|
|
322
|
+
|
|
296
323
|
test('PULLING THE NETWORK: the loop backs off and never returns', async () => {
|
|
297
324
|
const h = loopHarness({ rows: [{ event: 'pick_failed', reason: 'fetch failed ECONNREFUSED' }] });
|
|
298
325
|
await runner.forever(h.opts, h.deps);
|
|
@@ -11,7 +11,7 @@ import { createRequire } from 'node:module';
|
|
|
11
11
|
|
|
12
12
|
const require = createRequire(import.meta.url);
|
|
13
13
|
const {
|
|
14
|
-
WORKER_VERDICT_SCHEMA, buildWorkerArgs, buildWorkerPrompt, parseWorkerEnvelope, classifyRun,
|
|
14
|
+
WORKER_VERDICT_SCHEMA, buildWorkerArgs, buildWorkerPrompt, parseWorkerEnvelope, classifyRun, usageLimitHit,
|
|
15
15
|
} = require('../scripts/gds/autobongos-loop.js');
|
|
16
16
|
const runner = require('../scripts/gds/autobongos-run.js');
|
|
17
17
|
|
|
@@ -836,119 +836,132 @@ test('workerEnv passes Claude auth through, and still refuses the rest', () => {
|
|
|
836
836
|
assert.equal(env[leak], undefined, `${leak} must never reach a bypassPermissions worker`);
|
|
837
837
|
}
|
|
838
838
|
});
|
|
839
|
-
// ---
|
|
839
|
+
// --- usage is judged continuously; hitting the limit sleeps to its reset (task 1004454) ---
|
|
840
840
|
//
|
|
841
|
-
//
|
|
842
|
-
//
|
|
843
|
-
//
|
|
844
|
-
//
|
|
845
|
-
//
|
|
846
|
-
//
|
|
847
|
-
// run log, not from memory.
|
|
841
|
+
// Owner ruling, 2026-09-30: no fixed cap. The derived-gauge cap (task 1004178)
|
|
842
|
+
// held the runner after two tasks on an inferred reading — 776 holds in one
|
|
843
|
+
// 2026-09-26..29 run — for a risk the owner accepts: "if you hit usage limits
|
|
844
|
+
// that's fine, but then restart once the limit resets". So the gauge is still
|
|
845
|
+
// read before every task, a derived reading proceeds, and the limit itself is
|
|
846
|
+
// the stop: a worker cut off by it puts the runner to sleep until the reset.
|
|
848
847
|
|
|
849
848
|
import { mkdtempSync, writeFileSync, readFileSync } from 'node:fs';
|
|
850
849
|
import { tmpdir } from 'node:os';
|
|
851
850
|
import { join } from 'node:path';
|
|
852
851
|
|
|
853
|
-
|
|
854
|
-
const dir = mkdtempSync(join(tmpdir(), 'autobongos-
|
|
855
|
-
|
|
856
|
-
};
|
|
857
|
-
|
|
858
|
-
test('countWorkedSince counts only worked rows at or after the mark', () => {
|
|
859
|
-
const p = derivedLog();
|
|
860
|
-
process.env.AUTOBONGOS_LOG_FILE = p;
|
|
861
|
-
const at = (s) => new Date(s * 1000).toISOString();
|
|
862
|
-
writeFileSync(p, [
|
|
863
|
-
JSON.stringify({ at: at(1000), event: 'worked', task_id: '1' }), // before the mark
|
|
864
|
-
JSON.stringify({ at: at(3000), event: 'hold' }), // not a worked row
|
|
865
|
-
JSON.stringify({ at: at(3100), event: 'worked', task_id: '2' }),
|
|
866
|
-
'{ this line is torn', // costs one row, not the count
|
|
867
|
-
JSON.stringify({ at: at(3200), event: 'worked', task_id: '3' }),
|
|
868
|
-
].join('\n'));
|
|
869
|
-
assert.equal(runner.countWorkedSince(2000), 2);
|
|
870
|
-
assert.equal(runner.countWorkedSince(0), 0, 'a zero/absent mark counts nothing rather than everything');
|
|
871
|
-
delete process.env.AUTOBONGOS_LOG_FILE;
|
|
872
|
-
});
|
|
873
|
-
|
|
874
|
-
test('countWorkedSince treats an unreadable log as zero, not as a refusal', () => {
|
|
875
|
-
process.env.AUTOBONGOS_LOG_FILE = join(tmpdir(), 'autobongos-cap-nope', 'missing.jsonl');
|
|
876
|
-
assert.equal(runner.countWorkedSince(1), 0,
|
|
877
|
-
'the fence and the gauge are the gates; a bound that bricks the runner when its own log is gone is worse');
|
|
878
|
-
delete process.env.AUTOBONGOS_LOG_FILE;
|
|
879
|
-
});
|
|
880
|
-
|
|
881
|
-
test('a derived gauge HOLDS once the cap is reached, and the reason says why', async () => {
|
|
882
|
-
const p = derivedLog();
|
|
883
|
-
process.env.AUTOBONGOS_LOG_FILE = p;
|
|
884
|
-
process.env.AUTOBONGOS_DERIVED_TASK_CAP = '2';
|
|
852
|
+
test('a derived gauge keeps WORKING however many tasks it has already done — there is no cap', async () => {
|
|
853
|
+
const dir = mkdtempSync(join(tmpdir(), 'autobongos-nocap-'));
|
|
854
|
+
process.env.AUTOBONGOS_LOG_FILE = join(dir, 'runs.jsonl');
|
|
885
855
|
const at = (s) => new Date(s * 1000).toISOString();
|
|
886
|
-
|
|
887
|
-
|
|
888
|
-
|
|
889
|
-
|
|
890
|
-
|
|
856
|
+
const rows = [];
|
|
857
|
+
for (let i = 0; i < 6; i++) rows.push(JSON.stringify({ at: at(5100 + i), event: 'worked', task_id: String(i) }));
|
|
858
|
+
writeFileSync(process.env.AUTOBONGOS_LOG_FILE, rows.join('\n'));
|
|
859
|
+
let reachedPicker = false;
|
|
891
860
|
const events = [];
|
|
892
861
|
await runner.iteration({ goals: [], maxTasks: 1 }, {
|
|
862
|
+
state: runner.newRunnerState(),
|
|
893
863
|
readFence: async () => ({ raw: { enabled: true, goals: [{ goal_id: 7 }] }, graderBypassed: false }),
|
|
894
864
|
gauge: () => ({ decision: { go: true, unknown: false, derived: true, derivedSince: 5000, reason: 'derived' } }),
|
|
895
|
-
pickTask: async () => {
|
|
865
|
+
pickTask: async () => { reachedPicker = true; return { none: true, skipped: [] }; },
|
|
896
866
|
record: (row) => { events.push(row); return row; },
|
|
897
|
-
run: async () => { throw new Error('CAP BREACHED: ran a command'); },
|
|
898
|
-
spawnWorker: async () => { throw new Error('CAP BREACHED: spawned a worker'); },
|
|
899
867
|
});
|
|
900
|
-
assert.equal(
|
|
901
|
-
assert.
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
test('a derived gauge still WORKS while under the cap', async () => {
|
|
908
|
-
const p = derivedLog();
|
|
909
|
-
process.env.AUTOBONGOS_LOG_FILE = p;
|
|
910
|
-
process.env.AUTOBONGOS_DERIVED_TASK_CAP = '2';
|
|
911
|
-
writeFileSync(p, JSON.stringify({ at: new Date(5100 * 1000).toISOString(), event: 'worked', task_id: '1' }));
|
|
912
|
-
let reachedPicker = false;
|
|
868
|
+
assert.equal(reachedPicker, true, 'six tasks since the reset must not stop the seventh');
|
|
869
|
+
assert.ok(!events.some((e) => e.event === 'hold'), 'no hold on a derived reading');
|
|
870
|
+
delete process.env.AUTOBONGOS_LOG_FILE;
|
|
871
|
+
});
|
|
872
|
+
|
|
873
|
+
test('the gauge is still read before EVERY task, and a real hold still holds', async () => {
|
|
874
|
+
let reads = 0;
|
|
913
875
|
const events = [];
|
|
914
|
-
|
|
876
|
+
const d = {
|
|
877
|
+
state: runner.newRunnerState(),
|
|
915
878
|
readFence: async () => ({ raw: { enabled: true, goals: [{ goal_id: 7 }] }, graderBypassed: false }),
|
|
916
|
-
gauge: () =>
|
|
917
|
-
pickTask: async () => {
|
|
879
|
+
gauge: () => { reads += 1; return { decision: { go: false, unknown: false, reason: 'five-hour burn 80% is past the ceiling', sleepUntil: 1790218521 } }; },
|
|
880
|
+
pickTask: async () => { throw new Error('picked past a hold'); },
|
|
918
881
|
record: (row) => { events.push(row); return row; },
|
|
919
|
-
}
|
|
920
|
-
|
|
921
|
-
|
|
882
|
+
};
|
|
883
|
+
await runner.iteration({ goals: [], maxTasks: 1 }, d);
|
|
884
|
+
await runner.iteration({ goals: [], maxTasks: 1 }, d);
|
|
885
|
+
assert.equal(reads, 2);
|
|
886
|
+
assert.equal(events[1].event, 'hold');
|
|
887
|
+
assert.equal(events[1].sleep_until, 1790218521);
|
|
922
888
|
});
|
|
923
889
|
|
|
924
|
-
test('
|
|
925
|
-
const
|
|
926
|
-
|
|
927
|
-
const at = (s) => new Date(s * 1000).toISOString();
|
|
928
|
-
// 40k old rows, then the two that matter. A whole-file parse would touch
|
|
929
|
-
// every one of them on EVERY iteration of a loop that never exits.
|
|
930
|
-
const old = [];
|
|
931
|
-
for (let i = 0; i < 40000; i++) old.push(JSON.stringify({ at: at(1000 + i), event: 'worked', task_id: String(i) }));
|
|
932
|
-
old.push(JSON.stringify({ at: at(900000), event: 'worked', task_id: 'recent-1' }));
|
|
933
|
-
old.push(JSON.stringify({ at: at(900100), event: 'worked', task_id: 'recent-2' }));
|
|
934
|
-
writeFileSync(p, old.join('\n'));
|
|
935
|
-
const t0 = Date.now();
|
|
936
|
-
assert.equal(runner.countWorkedSince(800000), 2, 'only the rows at or after the mark count');
|
|
937
|
-
assert.ok(Date.now() - t0 < 500, 'and it must not walk the whole log to say so');
|
|
938
|
-
delete process.env.AUTOBONGOS_LOG_FILE;
|
|
890
|
+
test('usageLimitHit reads the reset second out of the CLI\'s own limit message', () => {
|
|
891
|
+
const env = JSON.stringify({ type: 'result', subtype: 'success', is_error: true, result: 'Claude AI usage limit reached|1790300000' });
|
|
892
|
+
assert.deepEqual(usageLimitHit({ envelope: env }), { resetAt: 1790300000 });
|
|
939
893
|
});
|
|
940
894
|
|
|
941
|
-
test('
|
|
942
|
-
const
|
|
943
|
-
|
|
944
|
-
|
|
945
|
-
|
|
946
|
-
|
|
947
|
-
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
|
|
951
|
-
|
|
895
|
+
test('usageLimitHit recognises a limit message with no machine-readable time, and says so', () => {
|
|
896
|
+
const env = JSON.stringify({ type: 'result', is_error: true, result: "You've hit your limit · resets 3am (America/New_York)" });
|
|
897
|
+
assert.deepEqual(usageLimitHit({ envelope: env }), { resetAt: null });
|
|
898
|
+
assert.deepEqual(usageLimitHit({ envelope: '', stderr: '5-hour limit reached ∙ resets 3pm' }), { resetAt: null });
|
|
899
|
+
});
|
|
900
|
+
|
|
901
|
+
test('usageLimitHit ignores the word "limit" in a worker that FINISHED', () => {
|
|
902
|
+
// A worker that fixed a GitHub rate-limit bug writes about rate limits in its
|
|
903
|
+
// verdict. Only an ERROR envelope, or a run with no envelope at all, can be a
|
|
904
|
+
// limit hit — anything else would put the runner to sleep for doing its job.
|
|
905
|
+
const env = JSON.stringify({ type: 'result', is_error: false, result: 'Claude AI usage limit reached|1790300000 is the banner text I fixed', structured_output: { outcome: 'shipped', what_happened: 'fixed the usage limit reached banner' } });
|
|
906
|
+
assert.equal(usageLimitHit({ envelope: env }), null);
|
|
907
|
+
assert.equal(usageLimitHit({ envelope: JSON.stringify({ is_error: true, result: 'API Error: 500' }) }), null);
|
|
908
|
+
});
|
|
909
|
+
|
|
910
|
+
function limitHarness(state, now, envelopeText) {
|
|
911
|
+
const rows = [];
|
|
912
|
+
const deps = {
|
|
913
|
+
state, now: () => now.t,
|
|
914
|
+
gauge: () => ({ decision: { go: true, unknown: false, derived: true, derivedSince: 1, reason: 'derived' } }),
|
|
915
|
+
readFence: async () => ({ raw: { enabled: true, goals: [{ goal_id: 1000119 }] }, graderBypassed: false }),
|
|
916
|
+
api: {},
|
|
917
|
+
verifyDeps: {
|
|
918
|
+
task: { id: 4242, status: 'active', updated_at: new Date().toISOString() },
|
|
919
|
+
probeArtifact: async () => ({ checked: true, onMain: false }),
|
|
920
|
+
fileBlocker: async () => ({ filed: false }),
|
|
921
|
+
},
|
|
922
|
+
release: async () => ({ ok: true, code: 0, stdout: '', stderr: '' }),
|
|
923
|
+
pickTask: async () => ({ task: { id: 4242, title: 't', kind: 'feature', description: 'd'.repeat(300), goal_id: 1000119, touches: [] }, goalId: 1000119, skipped: [] }),
|
|
924
|
+
worktreeName: () => 'autobongos-4242-abc123',
|
|
925
|
+
record: (r) => { rows.push(r); return r; },
|
|
926
|
+
run: async () => ({ ok: true, code: 0, stdout: '', stderr: '' }),
|
|
927
|
+
spawnWorker: async () => ({ envelope: envelopeText, exitCode: 1, sessionId: 's' }),
|
|
928
|
+
};
|
|
929
|
+
return { deps, rows };
|
|
930
|
+
}
|
|
931
|
+
|
|
932
|
+
test('a worker cut off by the usage limit puts the runner to sleep until the reset, then it resumes by itself', async () => {
|
|
933
|
+
const state = runner.newRunnerState();
|
|
934
|
+
const now = { t: 1_790_290_000_000 };
|
|
935
|
+
const env = JSON.stringify({ type: 'result', is_error: true, result: 'Claude AI usage limit reached|1790300000' });
|
|
936
|
+
const h = limitHarness(state, now, env);
|
|
937
|
+
|
|
938
|
+
const worked = await runner.iteration(opts, h.deps);
|
|
939
|
+
assert.equal(worked.event, 'worked');
|
|
940
|
+
assert.deepEqual(worked.usage_limit, { reset_at: 1790300000 }, 'the row says the limit was hit and when it resets');
|
|
941
|
+
|
|
942
|
+
let picked = false;
|
|
943
|
+
const asleep = await runner.iteration(opts, { ...h.deps, pickTask: async () => { picked = true; return { none: true }; } });
|
|
944
|
+
assert.equal(asleep.event, 'hold');
|
|
945
|
+
assert.equal(asleep.sleep_until, 1790300000, 'it sleeps to the reset second the CLI named');
|
|
946
|
+
assert.match(asleep.reason, /usage limit/);
|
|
947
|
+
assert.equal(picked, false, 'no task is taken while the limit is spent');
|
|
948
|
+
|
|
949
|
+
now.t = 1_790_300_001_000;
|
|
950
|
+
const awake = await runner.iteration(opts, { ...h.deps, pickTask: async () => { picked = true; return { none: true, skipped: [] }; } });
|
|
951
|
+
assert.equal(picked, true, 'past the reset it picks again without anyone touching it');
|
|
952
|
+
assert.equal(awake.event, 'nothing_claimable');
|
|
953
|
+
assert.equal(state.limitUntil, null);
|
|
954
|
+
});
|
|
955
|
+
|
|
956
|
+
test('a limit hit with no stated reset sleeps a bounded while and then tries again', async () => {
|
|
957
|
+
const state = runner.newRunnerState();
|
|
958
|
+
const now = { t: 2_000_000_000_000 };
|
|
959
|
+
const h = limitHarness(state, now, JSON.stringify({ is_error: true, result: '5-hour limit reached ∙ resets 3pm' }));
|
|
960
|
+
const worked = await runner.iteration(opts, h.deps);
|
|
961
|
+
assert.deepEqual(worked.usage_limit, { reset_at: null });
|
|
962
|
+
const asleep = await runner.iteration(opts, h.deps);
|
|
963
|
+
assert.equal(asleep.event, 'hold');
|
|
964
|
+
assert.equal(asleep.sleep_until, Math.round((now.t + runner.LIMIT_UNKNOWN_RESET_MS) / 1000));
|
|
952
965
|
});
|
|
953
966
|
|
|
954
967
|
// --- a --forever runner must be able to take a fix (task 1004374) -----------
|