offpeak 0.2.1__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. {offpeak-0.2.1 → offpeak-0.2.2}/.github/workflows/nightly.yml +8 -1
  2. offpeak-0.2.2/.github/workflows/settle.yml +56 -0
  3. {offpeak-0.2.1 → offpeak-0.2.2}/PKG-INFO +1 -1
  4. {offpeak-0.2.1 → offpeak-0.2.2}/docs/night-board.md +42 -2
  5. {offpeak-0.2.1 → offpeak-0.2.2}/docs/quickstart.md +16 -0
  6. {offpeak-0.2.1 → offpeak-0.2.2}/pyproject.toml +1 -1
  7. offpeak-0.2.2/receipts/2026-08-22-mechanics-1.json +32 -0
  8. offpeak-0.2.2/receipts/2026-08-22-mechanics-2.json +28 -0
  9. offpeak-0.2.2/receipts/2026-08-22-mechanics-3.json +27 -0
  10. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/__init__.py +1 -1
  11. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/venues/openai_batch.py +30 -3
  12. offpeak-0.2.2/tests/test_mechanics_run.py +133 -0
  13. {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_night_report.py +222 -25
  14. {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_run.py +71 -1
  15. offpeak-0.2.2/tests/test_settle_report.py +127 -0
  16. offpeak-0.2.2/tools/mechanics_run.py +295 -0
  17. {offpeak-0.2.1 → offpeak-0.2.2}/tools/night_report.py +216 -29
  18. offpeak-0.2.2/tools/settle_report.py +144 -0
  19. {offpeak-0.2.1 → offpeak-0.2.2}/.github/workflows/ci.yml +0 -0
  20. {offpeak-0.2.1 → offpeak-0.2.2}/.github/workflows/docs.yml +0 -0
  21. {offpeak-0.2.1 → offpeak-0.2.2}/.github/workflows/publish.yml +0 -0
  22. {offpeak-0.2.1 → offpeak-0.2.2}/.gitignore +0 -0
  23. {offpeak-0.2.1 → offpeak-0.2.2}/CONTRIBUTING.md +0 -0
  24. {offpeak-0.2.1 → offpeak-0.2.2}/LICENSE +0 -0
  25. {offpeak-0.2.1 → offpeak-0.2.2}/README.md +0 -0
  26. {offpeak-0.2.1 → offpeak-0.2.2}/SPEC.md +0 -0
  27. {offpeak-0.2.1 → offpeak-0.2.2}/docs/index.md +0 -0
  28. {offpeak-0.2.1 → offpeak-0.2.2}/docs/reference.md +0 -0
  29. {offpeak-0.2.1 → offpeak-0.2.2}/docs/roadmap.md +0 -0
  30. {offpeak-0.2.1 → offpeak-0.2.2}/mkdocs.yml +0 -0
  31. {offpeak-0.2.1 → offpeak-0.2.2}/mkdocs_hooks.py +0 -0
  32. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/__main__.py +0 -0
  33. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/client.py +0 -0
  34. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/deadline.py +0 -0
  35. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/job.py +0 -0
  36. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/prices.py +0 -0
  37. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/quote.py +0 -0
  38. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/venues/__init__.py +0 -0
  39. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/venues/anthropic_batch.py +0 -0
  40. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/venues/base.py +0 -0
  41. {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/venues/groq_batch.py +0 -0
  42. {offpeak-0.2.1 → offpeak-0.2.2}/tests/conftest.py +0 -0
  43. {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_deadline.py +0 -0
  44. {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_groq_venue.py +0 -0
  45. {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_job_receipt.py +0 -0
  46. {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_prices.py +0 -0
  47. {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_quote.py +0 -0
@@ -20,8 +20,11 @@ on:
20
20
  permissions:
21
21
  contents: write
22
22
 
23
+ # Everything that pushes board-data shares one queue. Two workflows writing
24
+ # different files on the same branch still race at the push, and a rejected
25
+ # push loses a night's mark.
23
26
  concurrency:
24
- group: night-board
27
+ group: board-data
25
28
  cancel-in-progress: false
26
29
 
27
30
  jobs:
@@ -60,6 +63,10 @@ jobs:
60
63
  run: pip install "gridstatus>=0.36"
61
64
 
62
65
  - name: Generate
66
+ # EIA_API_KEY is optional: absent, the US carbon columns are recorded
67
+ # unavailable and every other leg runs exactly as before.
68
+ env:
69
+ EIA_API_KEY: ${{ secrets.EIA_API_KEY }}
63
70
  run: |
64
71
  python tools/night_report.py \
65
72
  --mode "${{ steps.pick.outputs.mode }}" \
@@ -0,0 +1,56 @@
1
+ name: Settled runs
2
+
3
+ # Publishes the receipts in `receipts/` onto the board-data branch as
4
+ # `nightly/SETTLED.md`. Manual only: a settlement is a deliberate act, and the
5
+ # ledger should move when someone decides it moves, not on every push.
6
+ #
7
+ # board-data is written by CI and never by hand — the same rule the night board
8
+ # follows, for the same reason: main is protected and a bot push there would be
9
+ # rejected, and a ledger anyone can hand-edit is not a ledger.
10
+
11
+ on:
12
+ workflow_dispatch:
13
+
14
+ permissions:
15
+ contents: write
16
+
17
+ # Shared with the night board: everything that pushes board-data queues here.
18
+ concurrency:
19
+ group: board-data
20
+ cancel-in-progress: false
21
+
22
+ jobs:
23
+ settle:
24
+ runs-on: ubuntu-latest
25
+ steps:
26
+ - name: Check out the generator and the receipts (main)
27
+ uses: actions/checkout@v7
28
+
29
+ - uses: actions/setup-python@v7
30
+ with:
31
+ python-version: "3.12"
32
+
33
+ - name: Check out the ledger (board-data)
34
+ uses: actions/checkout@v7
35
+ with:
36
+ ref: board-data
37
+ path: board
38
+
39
+ - name: Render
40
+ run: |
41
+ python tools/settle_report.py \
42
+ --receipts receipts \
43
+ --outdir board/nightly
44
+
45
+ - name: Commit to board-data
46
+ working-directory: board
47
+ run: |
48
+ git config user.name "github-actions[bot]"
49
+ git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
50
+ git add nightly
51
+ if git diff --staged --quiet; then
52
+ echo "nothing changed — the ledger already says this"
53
+ exit 0
54
+ fi
55
+ git commit -m "board: settled runs $(date -u +%Y-%m-%d)"
56
+ git push
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: offpeak
3
- Version: 0.2.1
3
+ Version: 0.2.2
4
4
  Summary: Deadline-priced inference: give AI jobs a deadline and run them on the cheapest venue — provider batch tiers (−50%) today. Same model, same tokens, a different hour.
5
5
  Project-URL: Homepage, https://github.com/offpeak-ai/offpeak
6
6
  Project-URL: Repository, https://github.com/offpeak-ai/offpeak
@@ -28,6 +28,13 @@ night of 2026-08-20 the ERCOT Houston hub marked **3.94x** between its evening
28
28
  peak and the trough that followed, against the 4x a venue charges for the same
29
29
  hour of impatience.
30
30
 
31
+ The two sides of the grid do not move together, which is the point of marking
32
+ both. On 2026-08-18 and 08-19, CAISO's carbon ran **cleaner at the evening peak
33
+ than in the small hours** — spreads of 0.76x and 0.73x — because the sun that
34
+ serves the California evening has set by midnight. Cheap hours are not
35
+ automatically clean hours, and a board that only recorded price would have
36
+ implied otherwise.
37
+
31
38
  **[→ Read the board](https://github.com/offpeak-ai/offpeak/blob/board-data/nightly/BOARD.md)**
32
39
 
33
40
  ## How it works
@@ -49,6 +56,23 @@ Output lands on the
49
56
  `nightly/BOARD.md` plus the raw JSON per night — because `main` is protected and
50
57
  would reject a nightly bot push.
51
58
 
59
+ ## Settled runs are a different ledger
60
+
61
+ `BOARD.md` observes; it spends nothing at any venue. Runs that actually
62
+ executed and actually billed go in `nightly/SETTLED.md` on the same branch,
63
+ written by
64
+ [`tools/settle_report.py`](https://github.com/offpeak-ai/offpeak/blob/main/tools/settle_report.py)
65
+ from the receipts in
66
+ [`receipts/`](https://github.com/offpeak-ai/offpeak/tree/main/receipts) — never
67
+ by hand — and published by a manual workflow, because a settlement is a
68
+ deliberate act.
69
+
70
+ Every settled row carries its **scale**, and the column is not decoration. A
71
+ few dozen jobs proving the mechanics end to end and a production book are both
72
+ real settlements and are not the same evidence. A ledger that lets a reader
73
+ confuse them is doing marketing rather than accounting, so the scale is printed
74
+ before the money is.
75
+
52
76
  ## Sources
53
77
 
54
78
  - **Carbon** — [NESO carbon intensity](https://api.carbonintensity.org.uk),
@@ -57,6 +81,10 @@ would reject a nightly bot push.
57
81
  rates, GB region C, keyless.
58
82
  - **Power, US** — CAISO SP15 and ERCOT Houston day-ahead hourly, via
59
83
  [gridstatus](https://github.com/gridstatus/gridstatus), keyless.
84
+ - **Carbon, US** — [EIA-930](https://www.eia.gov/electricity/gridmonitor/)
85
+ hourly generation by fuel for the CAISO and ERCOT balancing authorities,
86
+ through the EIA Hourly Grid Monitor. Needs a free API key; without one the
87
+ column records itself unavailable and nothing else changes.
60
88
  - **Tokens** — the published price sheets, not a measurement:
61
89
  [OpenAI](https://developers.openai.com/api/docs/pricing) and
62
90
  [Anthropic](https://platform.claude.com/docs/en/about-claude/pricing).
@@ -66,9 +94,21 @@ any venue.
66
94
 
67
95
  ## Honest limits
68
96
 
69
- - **Four zones, two of them thin.** GB carbon and GB power are half-hourly and
97
+ - **Four zones, unevenly covered.** GB carbon and GB power are half-hourly and
70
98
  complete. CAISO SP15 and ERCOT Houston are **day-ahead hourly** prices, not
71
- settled real-time ones, and they are power only — no US carbon leg yet.
99
+ settled real-time ones.
100
+ - **US carbon is derived, GB carbon is measured.** NESO publishes an intensity;
101
+ EIA does not. The US columns are computed from EIA-930's hourly generation
102
+ mix times EIA's own CO2 coefficients and fleet heat rates — every input
103
+ published, the product an estimate, and marked `"basis": "derived"` in the
104
+ record so it is never confused with a measurement. It counts generation, not
105
+ consumption: imports and the carbon already stored in a battery are outside
106
+ what the method can see, and the share of generation EIA files under "other"
107
+ is reported per night rather than averaged in.
108
+ - **EIA runs about a day behind.** The 06:30Z mark usually lands before EIA has
109
+ published the night it is marking, so the US carbon columns are often empty
110
+ at first sight and fill in on a later re-mark. A column that is not there yet
111
+ is recorded as unavailable, never as zero.
72
112
  - **The token column is published, not observed.** The 2.0x and the 4x are read
73
113
  off price sheets; only the grid columns are measurements. A published number
74
114
  and a marked one are different kinds of claim, and the board should not blur
@@ -60,6 +60,22 @@ print(q.spread_usd, q.spread_pct)
60
60
 
61
61
  `Quote.basis` reports the provenance of every figure.
62
62
 
63
+ !!! warning "Reasoning models spend the ceiling before they speak"
64
+ On models that reason before answering — OpenAI's gpt-5 family, the
65
+ o-series — `max_tokens` caps **reasoning plus visible output**, and the
66
+ reasoning goes first. Set it too low and the job bills a full ceiling of
67
+ reasoning tokens and returns an empty string: a `Result` that is
68
+ technically ok, costs real money, and says nothing.
69
+
70
+ This is not hypothetical. A real batch here ran 24 jobs at
71
+ `max_tokens=16`, billed 374 output tokens, and returned 24 empty strings.
72
+ Give a reasoning model room — hundreds of tokens, not dozens — and price
73
+ the ceiling you actually set, which is what `quote()` does.
74
+
75
+ `offpeak` sends the ceiling under whichever name the venue wants
76
+ (`max_completion_tokens` where the model demands it), but it cannot make a
77
+ ceiling large enough to answer in.
78
+
63
79
  ### If you know roughly what it will write
64
80
 
65
81
  A floor is honest but not always useful. When you do have a sense of the output
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
4
4
 
5
5
  [project]
6
6
  name = "offpeak"
7
- version = "0.2.1"
7
+ version = "0.2.2"
8
8
  description = "Deadline-priced inference: give AI jobs a deadline and run them on the cheapest venue — provider batch tiers (−50%) today. Same model, same tokens, a different hour."
9
9
  readme = "README.md"
10
10
  license = "Apache-2.0"
@@ -0,0 +1,32 @@
1
+ {
2
+ "run_id": "2026-08-22-mechanics-1",
3
+ "scale": "mechanics proof (48 jobs, two venues)",
4
+ "settled_utc": "2026-08-21T21:35:46-07:00",
5
+ "deadline": "06:00",
6
+ "price_sheet": "2026-08-21",
7
+ "offpeak_version": "0.2.2.dev0",
8
+ "jobs": 48,
9
+ "ok": 24,
10
+ "failed": 24,
11
+ "fell_back": 0,
12
+ "sla_met": 24,
13
+ "input_tokens": 787,
14
+ "output_tokens": 155,
15
+ "list_usd": 0.001562,
16
+ "paid_usd": 0.000781,
17
+ "captured_usd": 0.000781,
18
+ "captured_pct": 50.0,
19
+ "left_on_table_usd": 0.0,
20
+ "by_venue": {
21
+ "anthropic:batch": 24,
22
+ "openai:batch": 24
23
+ },
24
+ "failure_kinds": [
25
+ "HTTP 400"
26
+ ],
27
+ "notes": [
28
+ "The first genuine settlement. Twenty-four short public-domain lines, classified in one word each, through both venues' cheapest lanes on a 06:00 deadline.",
29
+ "The Anthropic leg settled 24/24 at the batch tier. The OpenAI leg was rejected job by job with HTTP 400 'max_tokens is not supported with this model; use max_completion_tokens instead' - no tokens were processed, so nothing was billed there.",
30
+ "The sync fallback did not rescue those jobs: it rescues jobs that never came back, and these came back as failures. Fixed in the driver; run 2 is the same book re-run."
31
+ ]
32
+ }
@@ -0,0 +1,28 @@
1
+ {
2
+ "run_id": "2026-08-22-mechanics-2",
3
+ "scale": "mechanics proof (24 jobs, OpenAI, ceiling too low)",
4
+ "settled_utc": "2026-08-21T21:39:39-07:00",
5
+ "deadline": "06:00",
6
+ "price_sheet": "2026-08-21",
7
+ "offpeak_version": "0.2.2.dev0",
8
+ "jobs": 24,
9
+ "ok": 24,
10
+ "failed": 0,
11
+ "fell_back": 0,
12
+ "sla_met": 24,
13
+ "input_tokens": 728,
14
+ "output_tokens": 374,
15
+ "list_usd": 0.0005943999999999998,
16
+ "paid_usd": 0.0002971999999999999,
17
+ "captured_usd": 0.0002971999999999999,
18
+ "captured_pct": 50.0,
19
+ "left_on_table_usd": 0.0,
20
+ "by_venue": {
21
+ "openai:batch": 24
22
+ },
23
+ "notes": [
24
+ "The OpenAI leg re-run after the driver learned to send max_completion_tokens for the gpt-5 family. It settled: 24/24, billed, SLA met.",
25
+ "It also returned twenty-four empty strings. gpt-5.6 reasons before it answers and the 16-token ceiling was spent entirely on reasoning, leaving nothing to say. The money here bought 374 output tokens of nothing.",
26
+ "Published rather than quietly dropped: a settlement that billed for no usable output is exactly the kind of thing a receipt exists to make visible."
27
+ ]
28
+ }
@@ -0,0 +1,27 @@
1
+ {
2
+ "run_id": "2026-08-22-mechanics-3",
3
+ "scale": "mechanics proof (24 jobs, OpenAI, ceiling sized to the model)",
4
+ "settled_utc": "2026-08-21T21:42:51-07:00",
5
+ "deadline": "06:00",
6
+ "price_sheet": "2026-08-21",
7
+ "offpeak_version": "0.2.2.dev0",
8
+ "jobs": 24,
9
+ "ok": 24,
10
+ "failed": 0,
11
+ "fell_back": 0,
12
+ "sla_met": 24,
13
+ "input_tokens": 728,
14
+ "output_tokens": 971,
15
+ "list_usd": 0.0013108,
16
+ "paid_usd": 0.0006554,
17
+ "captured_usd": 0.0006554,
18
+ "captured_pct": 50.0,
19
+ "left_on_table_usd": 0.0,
20
+ "by_venue": {
21
+ "openai:batch": 24
22
+ },
23
+ "notes": [
24
+ "The same twenty-four lines once more, with a 256-token ceiling that leaves a reasoning model room to answer in.",
25
+ "24/24 settled with real one-word answers, 971 output tokens, zero fallbacks, every SLA met."
26
+ ]
27
+ }
@@ -18,7 +18,7 @@ from .prices import format_usd
18
18
  from .quote import Quote, VenueQuote, quote
19
19
  from .venues.base import BatchState, Venue
20
20
 
21
- __version__ = "0.2.1"
21
+ __version__ = "0.2.2"
22
22
 
23
23
  __all__ = [
24
24
  "job",
@@ -11,17 +11,44 @@ import json
11
11
  from ..job import Job, Result
12
12
  from .base import BatchState, Venue
13
13
 
14
- __all__ = ["OpenAIBatch", "build_jsonl", "parse_output_line"]
14
+ __all__ = ["OpenAIBatch", "build_jsonl", "parse_output_line", "body_params"]
15
15
 
16
16
  _MODEL_PREFIXES = ("gpt-", "o1", "o3", "o4", "chatgpt-")
17
17
  _ENDPOINT = "/v1/chat/completions"
18
18
 
19
+ # OpenAI's newer families reject ``max_tokens`` outright — "Unsupported
20
+ # parameter: 'max_tokens' is not supported with this model. Use
21
+ # 'max_completion_tokens' instead." — and a batch of a thousand jobs discovers
22
+ # this one HTTP 400 at a time, hours after submission.
23
+ #
24
+ # Translating is the driver's job. A caller says ``max_tokens`` once and it
25
+ # means the same thing at every venue; the venue that spells it differently is
26
+ # the venue's problem, not the caller's.
27
+ _MAX_COMPLETION_TOKENS_PREFIXES = ("o1", "o3", "o4", "gpt-5")
28
+
29
+
30
+ def body_params(job: Job) -> dict:
31
+ """The job's params as this venue's API spells them.
32
+
33
+ An explicit ``max_completion_tokens`` always wins: the caller who reached
34
+ for the provider's own name for the field meant it, and sending both would
35
+ be rejected.
36
+ """
37
+ params = dict(job.params)
38
+ if "max_tokens" not in params:
39
+ return params
40
+ if not job.model.startswith(_MAX_COMPLETION_TOKENS_PREFIXES):
41
+ return params
42
+ ceiling = params.pop("max_tokens")
43
+ params.setdefault("max_completion_tokens", ceiling)
44
+ return params
45
+
19
46
 
20
47
  def build_jsonl(jobs: list[Job]) -> bytes:
21
48
  """Render *jobs* as OpenAI Batch API JSONL (one request per line)."""
22
49
  lines = []
23
50
  for j in jobs:
24
- body = {"model": j.model, "messages": j.messages, **j.params}
51
+ body = {"model": j.model, "messages": j.messages, **body_params(j)}
25
52
  lines.append(
26
53
  json.dumps(
27
54
  {"custom_id": j.id, "method": "POST", "url": _ENDPOINT, "body": body},
@@ -129,7 +156,7 @@ class OpenAIBatch(Venue):
129
156
  def run_sync(self, job: Job) -> Result:
130
157
  try:
131
158
  response = self.client.chat.completions.create(
132
- model=job.model, messages=job.messages, **job.params
159
+ model=job.model, messages=job.messages, **body_params(job)
133
160
  )
134
161
  except Exception as exc: # noqa: BLE001
135
162
  return Result(job=job, error=str(exc))
@@ -0,0 +1,133 @@
1
+ """The capped settlement runner — everything that happens before money moves.
2
+
3
+ These tests never submit anything: `quote()` makes no API calls, and both paths
4
+ exercised here stop before the venues are touched.
5
+ """
6
+
7
+ import importlib.util
8
+ import json
9
+ from pathlib import Path
10
+
11
+ import pytest
12
+
13
+ _spec = importlib.util.spec_from_file_location(
14
+ "mechanics_run", Path(__file__).resolve().parent.parent / "tools" / "mechanics_run.py"
15
+ )
16
+ mr = importlib.util.module_from_spec(_spec)
17
+ _spec.loader.exec_module(mr)
18
+
19
+
20
+ class TestBook:
21
+ def test_one_lane_per_model_over_the_same_lines(self):
22
+ jobs = mr.build_book(["claude-haiku-4-5", "gpt-5.6-luna"])
23
+ assert len(jobs) == 2 * len(mr.LINES)
24
+ assert {j.model for j in jobs} == {"claude-haiku-4-5", "gpt-5.6-luna"}
25
+
26
+ def test_every_job_carries_a_ceiling(self):
27
+ jobs = mr.build_book(["gpt-5.6-luna"], max_tokens=64)
28
+ assert all(j.params["max_tokens"] == 64 for j in jobs)
29
+
30
+ def test_the_default_ceiling_leaves_a_reasoning_model_room_to_answer(self):
31
+ # A ceiling smaller than the reasoning buys a bill and an empty string.
32
+ assert mr.DEFAULT_MAX_TOKENS >= 128
33
+
34
+ def test_the_lines_are_the_work_not_filler(self):
35
+ jobs = mr.build_book(["gpt-5.6-luna"])
36
+ assert mr.LINES[0] in jobs[0].messages[0]["content"]
37
+
38
+
39
+ class TestCapGate:
40
+ def _out(self, tmp_path):
41
+ return str(tmp_path / "run")
42
+
43
+ def test_a_book_under_the_cap_passes_and_still_submits_nothing_on_dry_run(
44
+ self, tmp_path, capsys
45
+ ):
46
+ rc = mr.main(["--out", self._out(tmp_path), "--dry-run"])
47
+ out = capsys.readouterr().out
48
+ assert rc == 0
49
+ assert "under cap" in out
50
+ assert not (tmp_path / "run" / "handles.jsonl").exists()
51
+
52
+ def test_a_book_over_the_cap_aborts_before_submitting(self, tmp_path, capsys):
53
+ rc = mr.main(["--out", self._out(tmp_path), "--cap", "0.0000001"])
54
+ assert rc == 2
55
+ assert "ABORT" in capsys.readouterr().out
56
+ assert not (tmp_path / "run" / "handles.jsonl").exists()
57
+
58
+ def test_the_cap_counts_what_the_session_already_exposed(self, tmp_path, capsys):
59
+ # The cap is on the total, so a small book still trips it once enough
60
+ # has been spent around it.
61
+ rc = mr.main(
62
+ ["--out", self._out(tmp_path), "--models", "gpt-5.6-luna",
63
+ "--cap", "0.01", "--already-spent", "0.0099"]
64
+ )
65
+ assert rc == 2
66
+ assert "already exposed" in capsys.readouterr().out
67
+
68
+ def test_the_gate_prices_the_ceiling_not_a_hope(self, tmp_path, capsys):
69
+ # The quoted worst case must scale with the ceiling actually set.
70
+ mr.main(["--out", self._out(tmp_path), "--max-tokens", "16", "--dry-run"])
71
+ small = capsys.readouterr().out
72
+ mr.main(["--out", str(tmp_path / "run2"), "--max-tokens", "512", "--dry-run"])
73
+ large = capsys.readouterr().out
74
+
75
+ def worst(text):
76
+ line = next(ln for ln in text.splitlines() if ln.startswith("cap check"))
77
+ return float(line.split("$")[1].split()[0])
78
+
79
+ assert worst(large) > worst(small)
80
+
81
+
82
+ class TestRefusesToResubmit:
83
+ def test_an_existing_handle_log_stops_the_run(self, tmp_path, capsys):
84
+ out = tmp_path / "run"
85
+ out.mkdir()
86
+ (out / "handles.jsonl").write_text(
87
+ json.dumps({"venue": "openai:batch", "handle": "batch_x", "jobs": 1}) + "\n"
88
+ )
89
+ rc = mr.main(["--out", str(out)])
90
+ assert rc == 3
91
+ assert "refusing to re-submit" in capsys.readouterr().out
92
+
93
+
94
+ class TestCancelPath:
95
+ def test_cancelling_with_no_log_is_not_an_error(self, tmp_path, capsys):
96
+ rc = mr.main(["--out", str(tmp_path / "nothing-here"), "--cancel"])
97
+ assert rc == 0
98
+ assert "no handles recorded" in capsys.readouterr().out
99
+
100
+ def test_every_recorded_handle_is_cancelled_at_its_own_venue(self, tmp_path, capsys):
101
+ log = tmp_path / "handles.jsonl"
102
+ log.write_text(
103
+ json.dumps({"venue": "openai:batch", "handle": "batch_x", "jobs": 1}) + "\n"
104
+ + json.dumps({"venue": "anthropic:batch", "handle": "msgbatch_y", "jobs": 1}) + "\n"
105
+ )
106
+ cancelled = []
107
+
108
+ class FakeVenue:
109
+ def __init__(self, name):
110
+ self.name = name
111
+
112
+ def cancel(self, handle):
113
+ cancelled.append((self.name, handle))
114
+
115
+ mr.cancel_all(log, [FakeVenue("openai:batch"), FakeVenue("anthropic:batch")])
116
+ assert cancelled == [("openai:batch", "batch_x"), ("anthropic:batch", "msgbatch_y")]
117
+
118
+ def test_a_handle_from_an_unconfigured_venue_is_reported_not_skipped_silently(
119
+ self, tmp_path, capsys
120
+ ):
121
+ log = tmp_path / "handles.jsonl"
122
+ log.write_text(json.dumps({"venue": "groq:batch", "handle": "g_1", "jobs": 1}) + "\n")
123
+ mr.cancel_all(log, [])
124
+ assert "no venue to cancel with" in capsys.readouterr().out
125
+
126
+
127
+ @pytest.mark.parametrize("flag,attr,value", [
128
+ ("--cap", "cap", 0.25),
129
+ ("--max-tokens", "max_tokens", 32),
130
+ ])
131
+ def test_the_guard_rails_are_all_settable(flag, attr, value):
132
+ args = mr.parse_args([flag, str(value)])
133
+ assert getattr(args, attr) == value