offpeak 0.2.1__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {offpeak-0.2.1 → offpeak-0.2.2}/.github/workflows/nightly.yml +8 -1
- offpeak-0.2.2/.github/workflows/settle.yml +56 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/PKG-INFO +1 -1
- {offpeak-0.2.1 → offpeak-0.2.2}/docs/night-board.md +42 -2
- {offpeak-0.2.1 → offpeak-0.2.2}/docs/quickstart.md +16 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/pyproject.toml +1 -1
- offpeak-0.2.2/receipts/2026-08-22-mechanics-1.json +32 -0
- offpeak-0.2.2/receipts/2026-08-22-mechanics-2.json +28 -0
- offpeak-0.2.2/receipts/2026-08-22-mechanics-3.json +27 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/__init__.py +1 -1
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/venues/openai_batch.py +30 -3
- offpeak-0.2.2/tests/test_mechanics_run.py +133 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_night_report.py +222 -25
- {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_run.py +71 -1
- offpeak-0.2.2/tests/test_settle_report.py +127 -0
- offpeak-0.2.2/tools/mechanics_run.py +295 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/tools/night_report.py +216 -29
- offpeak-0.2.2/tools/settle_report.py +144 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/.github/workflows/ci.yml +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/.github/workflows/docs.yml +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/.github/workflows/publish.yml +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/.gitignore +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/CONTRIBUTING.md +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/LICENSE +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/README.md +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/SPEC.md +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/docs/index.md +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/docs/reference.md +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/docs/roadmap.md +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/mkdocs.yml +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/mkdocs_hooks.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/__main__.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/client.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/deadline.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/job.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/prices.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/quote.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/venues/__init__.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/venues/anthropic_batch.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/venues/base.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/src/offpeak/venues/groq_batch.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/tests/conftest.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_deadline.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_groq_venue.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_job_receipt.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_prices.py +0 -0
- {offpeak-0.2.1 → offpeak-0.2.2}/tests/test_quote.py +0 -0
|
@@ -20,8 +20,11 @@ on:
|
|
|
20
20
|
permissions:
|
|
21
21
|
contents: write
|
|
22
22
|
|
|
23
|
+
# Everything that pushes board-data shares one queue. Two workflows writing
|
|
24
|
+
# different files on the same branch still race at the push, and a rejected
|
|
25
|
+
# push loses a night's mark.
|
|
23
26
|
concurrency:
|
|
24
|
-
group:
|
|
27
|
+
group: board-data
|
|
25
28
|
cancel-in-progress: false
|
|
26
29
|
|
|
27
30
|
jobs:
|
|
@@ -60,6 +63,10 @@ jobs:
|
|
|
60
63
|
run: pip install "gridstatus>=0.36"
|
|
61
64
|
|
|
62
65
|
- name: Generate
|
|
66
|
+
# EIA_API_KEY is optional: absent, the US carbon columns are recorded
|
|
67
|
+
# unavailable and every other leg runs exactly as before.
|
|
68
|
+
env:
|
|
69
|
+
EIA_API_KEY: ${{ secrets.EIA_API_KEY }}
|
|
63
70
|
run: |
|
|
64
71
|
python tools/night_report.py \
|
|
65
72
|
--mode "${{ steps.pick.outputs.mode }}" \
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
name: Settled runs
|
|
2
|
+
|
|
3
|
+
# Publishes the receipts in `receipts/` onto the board-data branch as
|
|
4
|
+
# `nightly/SETTLED.md`. Manual only: a settlement is a deliberate act, and the
|
|
5
|
+
# ledger should move when someone decides it moves, not on every push.
|
|
6
|
+
#
|
|
7
|
+
# board-data is written by CI and never by hand — the same rule the night board
|
|
8
|
+
# follows, for the same reason: main is protected and a bot push there would be
|
|
9
|
+
# rejected, and a ledger anyone can hand-edit is not a ledger.
|
|
10
|
+
|
|
11
|
+
on:
|
|
12
|
+
workflow_dispatch:
|
|
13
|
+
|
|
14
|
+
permissions:
|
|
15
|
+
contents: write
|
|
16
|
+
|
|
17
|
+
# Shared with the night board: everything that pushes board-data queues here.
|
|
18
|
+
concurrency:
|
|
19
|
+
group: board-data
|
|
20
|
+
cancel-in-progress: false
|
|
21
|
+
|
|
22
|
+
jobs:
|
|
23
|
+
settle:
|
|
24
|
+
runs-on: ubuntu-latest
|
|
25
|
+
steps:
|
|
26
|
+
- name: Check out the generator and the receipts (main)
|
|
27
|
+
uses: actions/checkout@v7
|
|
28
|
+
|
|
29
|
+
- uses: actions/setup-python@v7
|
|
30
|
+
with:
|
|
31
|
+
python-version: "3.12"
|
|
32
|
+
|
|
33
|
+
- name: Check out the ledger (board-data)
|
|
34
|
+
uses: actions/checkout@v7
|
|
35
|
+
with:
|
|
36
|
+
ref: board-data
|
|
37
|
+
path: board
|
|
38
|
+
|
|
39
|
+
- name: Render
|
|
40
|
+
run: |
|
|
41
|
+
python tools/settle_report.py \
|
|
42
|
+
--receipts receipts \
|
|
43
|
+
--outdir board/nightly
|
|
44
|
+
|
|
45
|
+
- name: Commit to board-data
|
|
46
|
+
working-directory: board
|
|
47
|
+
run: |
|
|
48
|
+
git config user.name "github-actions[bot]"
|
|
49
|
+
git config user.email "41898282+github-actions[bot]@users.noreply.github.com"
|
|
50
|
+
git add nightly
|
|
51
|
+
if git diff --staged --quiet; then
|
|
52
|
+
echo "nothing changed — the ledger already says this"
|
|
53
|
+
exit 0
|
|
54
|
+
fi
|
|
55
|
+
git commit -m "board: settled runs $(date -u +%Y-%m-%d)"
|
|
56
|
+
git push
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: offpeak
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.2
|
|
4
4
|
Summary: Deadline-priced inference: give AI jobs a deadline and run them on the cheapest venue — provider batch tiers (−50%) today. Same model, same tokens, a different hour.
|
|
5
5
|
Project-URL: Homepage, https://github.com/offpeak-ai/offpeak
|
|
6
6
|
Project-URL: Repository, https://github.com/offpeak-ai/offpeak
|
|
@@ -28,6 +28,13 @@ night of 2026-08-20 the ERCOT Houston hub marked **3.94x** between its evening
|
|
|
28
28
|
peak and the trough that followed, against the 4x a venue charges for the same
|
|
29
29
|
hour of impatience.
|
|
30
30
|
|
|
31
|
+
The two sides of the grid do not move together, which is the point of marking
|
|
32
|
+
both. On 2026-08-18 and 08-19, CAISO's carbon ran **cleaner at the evening peak
|
|
33
|
+
than in the small hours** — spreads of 0.76x and 0.73x — because the sun that
|
|
34
|
+
serves the California evening has set by midnight. Cheap hours are not
|
|
35
|
+
automatically clean hours, and a board that only recorded price would have
|
|
36
|
+
implied otherwise.
|
|
37
|
+
|
|
31
38
|
**[→ Read the board](https://github.com/offpeak-ai/offpeak/blob/board-data/nightly/BOARD.md)**
|
|
32
39
|
|
|
33
40
|
## How it works
|
|
@@ -49,6 +56,23 @@ Output lands on the
|
|
|
49
56
|
`nightly/BOARD.md` plus the raw JSON per night — because `main` is protected and
|
|
50
57
|
would reject a nightly bot push.
|
|
51
58
|
|
|
59
|
+
## Settled runs are a different ledger
|
|
60
|
+
|
|
61
|
+
`BOARD.md` observes; it spends nothing at any venue. Runs that actually
|
|
62
|
+
executed and actually billed go in `nightly/SETTLED.md` on the same branch,
|
|
63
|
+
written by
|
|
64
|
+
[`tools/settle_report.py`](https://github.com/offpeak-ai/offpeak/blob/main/tools/settle_report.py)
|
|
65
|
+
from the receipts in
|
|
66
|
+
[`receipts/`](https://github.com/offpeak-ai/offpeak/tree/main/receipts) — never
|
|
67
|
+
by hand — and published by a manual workflow, because a settlement is a
|
|
68
|
+
deliberate act.
|
|
69
|
+
|
|
70
|
+
Every settled row carries its **scale**, and the column is not decoration. A
|
|
71
|
+
few dozen jobs proving the mechanics end to end and a production book are both
|
|
72
|
+
real settlements and are not the same evidence. A ledger that lets a reader
|
|
73
|
+
confuse them is doing marketing rather than accounting, so the scale is printed
|
|
74
|
+
before the money is.
|
|
75
|
+
|
|
52
76
|
## Sources
|
|
53
77
|
|
|
54
78
|
- **Carbon** — [NESO carbon intensity](https://api.carbonintensity.org.uk),
|
|
@@ -57,6 +81,10 @@ would reject a nightly bot push.
|
|
|
57
81
|
rates, GB region C, keyless.
|
|
58
82
|
- **Power, US** — CAISO SP15 and ERCOT Houston day-ahead hourly, via
|
|
59
83
|
[gridstatus](https://github.com/gridstatus/gridstatus), keyless.
|
|
84
|
+
- **Carbon, US** — [EIA-930](https://www.eia.gov/electricity/gridmonitor/)
|
|
85
|
+
hourly generation by fuel for the CAISO and ERCOT balancing authorities,
|
|
86
|
+
through the EIA Hourly Grid Monitor. Needs a free API key; without one the
|
|
87
|
+
column records itself unavailable and nothing else changes.
|
|
60
88
|
- **Tokens** — the published price sheets, not a measurement:
|
|
61
89
|
[OpenAI](https://developers.openai.com/api/docs/pricing) and
|
|
62
90
|
[Anthropic](https://platform.claude.com/docs/en/about-claude/pricing).
|
|
@@ -66,9 +94,21 @@ any venue.
|
|
|
66
94
|
|
|
67
95
|
## Honest limits
|
|
68
96
|
|
|
69
|
-
- **Four zones,
|
|
97
|
+
- **Four zones, unevenly covered.** GB carbon and GB power are half-hourly and
|
|
70
98
|
complete. CAISO SP15 and ERCOT Houston are **day-ahead hourly** prices, not
|
|
71
|
-
settled real-time ones
|
|
99
|
+
settled real-time ones.
|
|
100
|
+
- **US carbon is derived, GB carbon is measured.** NESO publishes an intensity;
|
|
101
|
+
EIA does not. The US columns are computed from EIA-930's hourly generation
|
|
102
|
+
mix times EIA's own CO2 coefficients and fleet heat rates — every input
|
|
103
|
+
published, the product an estimate, and marked `"basis": "derived"` in the
|
|
104
|
+
record so it is never confused with a measurement. It counts generation, not
|
|
105
|
+
consumption: imports and the carbon already stored in a battery are outside
|
|
106
|
+
what the method can see, and the share of generation EIA files under "other"
|
|
107
|
+
is reported per night rather than averaged in.
|
|
108
|
+
- **EIA runs about a day behind.** The 06:30Z mark usually lands before EIA has
|
|
109
|
+
published the night it is marking, so the US carbon columns are often empty
|
|
110
|
+
at first sight and fill in on a later re-mark. A column that is not there yet
|
|
111
|
+
is recorded as unavailable, never as zero.
|
|
72
112
|
- **The token column is published, not observed.** The 2.0x and the 4x are read
|
|
73
113
|
off price sheets; only the grid columns are measurements. A published number
|
|
74
114
|
and a marked one are different kinds of claim, and the board should not blur
|
|
@@ -60,6 +60,22 @@ print(q.spread_usd, q.spread_pct)
|
|
|
60
60
|
|
|
61
61
|
`Quote.basis` reports the provenance of every figure.
|
|
62
62
|
|
|
63
|
+
!!! warning "Reasoning models spend the ceiling before they speak"
|
|
64
|
+
On models that reason before answering — OpenAI's gpt-5 family, the
|
|
65
|
+
o-series — `max_tokens` caps **reasoning plus visible output**, and the
|
|
66
|
+
reasoning goes first. Set it too low and the job bills a full ceiling of
|
|
67
|
+
reasoning tokens and returns an empty string: a `Result` that is
|
|
68
|
+
technically ok, costs real money, and says nothing.
|
|
69
|
+
|
|
70
|
+
This is not hypothetical. A real batch here ran 24 jobs at
|
|
71
|
+
`max_tokens=16`, billed 374 output tokens, and returned 24 empty strings.
|
|
72
|
+
Give a reasoning model room — hundreds of tokens, not dozens — and price
|
|
73
|
+
the ceiling you actually set, which is what `quote()` does.
|
|
74
|
+
|
|
75
|
+
`offpeak` sends the ceiling under whichever name the venue wants
|
|
76
|
+
(`max_completion_tokens` where the model demands it), but it cannot make a
|
|
77
|
+
ceiling large enough to answer in.
|
|
78
|
+
|
|
63
79
|
### If you know roughly what it will write
|
|
64
80
|
|
|
65
81
|
A floor is honest but not always useful. When you do have a sense of the output
|
|
@@ -4,7 +4,7 @@ build-backend = "hatchling.build"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "offpeak"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.2"
|
|
8
8
|
description = "Deadline-priced inference: give AI jobs a deadline and run them on the cheapest venue — provider batch tiers (−50%) today. Same model, same tokens, a different hour."
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "Apache-2.0"
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
{
|
|
2
|
+
"run_id": "2026-08-22-mechanics-1",
|
|
3
|
+
"scale": "mechanics proof (48 jobs, two venues)",
|
|
4
|
+
"settled_utc": "2026-08-21T21:35:46-07:00",
|
|
5
|
+
"deadline": "06:00",
|
|
6
|
+
"price_sheet": "2026-08-21",
|
|
7
|
+
"offpeak_version": "0.2.2.dev0",
|
|
8
|
+
"jobs": 48,
|
|
9
|
+
"ok": 24,
|
|
10
|
+
"failed": 24,
|
|
11
|
+
"fell_back": 0,
|
|
12
|
+
"sla_met": 24,
|
|
13
|
+
"input_tokens": 787,
|
|
14
|
+
"output_tokens": 155,
|
|
15
|
+
"list_usd": 0.001562,
|
|
16
|
+
"paid_usd": 0.000781,
|
|
17
|
+
"captured_usd": 0.000781,
|
|
18
|
+
"captured_pct": 50.0,
|
|
19
|
+
"left_on_table_usd": 0.0,
|
|
20
|
+
"by_venue": {
|
|
21
|
+
"anthropic:batch": 24,
|
|
22
|
+
"openai:batch": 24
|
|
23
|
+
},
|
|
24
|
+
"failure_kinds": [
|
|
25
|
+
"HTTP 400"
|
|
26
|
+
],
|
|
27
|
+
"notes": [
|
|
28
|
+
"The first genuine settlement. Twenty-four short public-domain lines, classified in one word each, through both venues' cheapest lanes on a 06:00 deadline.",
|
|
29
|
+
"The Anthropic leg settled 24/24 at the batch tier. The OpenAI leg was rejected job by job with HTTP 400 'max_tokens is not supported with this model; use max_completion_tokens instead' - no tokens were processed, so nothing was billed there.",
|
|
30
|
+
"The sync fallback did not rescue those jobs: it rescues jobs that never came back, and these came back as failures. Fixed in the driver; run 2 is the same book re-run."
|
|
31
|
+
]
|
|
32
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
{
|
|
2
|
+
"run_id": "2026-08-22-mechanics-2",
|
|
3
|
+
"scale": "mechanics proof (24 jobs, OpenAI, ceiling too low)",
|
|
4
|
+
"settled_utc": "2026-08-21T21:39:39-07:00",
|
|
5
|
+
"deadline": "06:00",
|
|
6
|
+
"price_sheet": "2026-08-21",
|
|
7
|
+
"offpeak_version": "0.2.2.dev0",
|
|
8
|
+
"jobs": 24,
|
|
9
|
+
"ok": 24,
|
|
10
|
+
"failed": 0,
|
|
11
|
+
"fell_back": 0,
|
|
12
|
+
"sla_met": 24,
|
|
13
|
+
"input_tokens": 728,
|
|
14
|
+
"output_tokens": 374,
|
|
15
|
+
"list_usd": 0.0005943999999999998,
|
|
16
|
+
"paid_usd": 0.0002971999999999999,
|
|
17
|
+
"captured_usd": 0.0002971999999999999,
|
|
18
|
+
"captured_pct": 50.0,
|
|
19
|
+
"left_on_table_usd": 0.0,
|
|
20
|
+
"by_venue": {
|
|
21
|
+
"openai:batch": 24
|
|
22
|
+
},
|
|
23
|
+
"notes": [
|
|
24
|
+
"The OpenAI leg re-run after the driver learned to send max_completion_tokens for the gpt-5 family. It settled: 24/24, billed, SLA met.",
|
|
25
|
+
"It also returned twenty-four empty strings. gpt-5.6 reasons before it answers and the 16-token ceiling was spent entirely on reasoning, leaving nothing to say. The money here bought 374 output tokens of nothing.",
|
|
26
|
+
"Published rather than quietly dropped: a settlement that billed for no usable output is exactly the kind of thing a receipt exists to make visible."
|
|
27
|
+
]
|
|
28
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
{
|
|
2
|
+
"run_id": "2026-08-22-mechanics-3",
|
|
3
|
+
"scale": "mechanics proof (24 jobs, OpenAI, ceiling sized to the model)",
|
|
4
|
+
"settled_utc": "2026-08-21T21:42:51-07:00",
|
|
5
|
+
"deadline": "06:00",
|
|
6
|
+
"price_sheet": "2026-08-21",
|
|
7
|
+
"offpeak_version": "0.2.2.dev0",
|
|
8
|
+
"jobs": 24,
|
|
9
|
+
"ok": 24,
|
|
10
|
+
"failed": 0,
|
|
11
|
+
"fell_back": 0,
|
|
12
|
+
"sla_met": 24,
|
|
13
|
+
"input_tokens": 728,
|
|
14
|
+
"output_tokens": 971,
|
|
15
|
+
"list_usd": 0.0013108,
|
|
16
|
+
"paid_usd": 0.0006554,
|
|
17
|
+
"captured_usd": 0.0006554,
|
|
18
|
+
"captured_pct": 50.0,
|
|
19
|
+
"left_on_table_usd": 0.0,
|
|
20
|
+
"by_venue": {
|
|
21
|
+
"openai:batch": 24
|
|
22
|
+
},
|
|
23
|
+
"notes": [
|
|
24
|
+
"The same twenty-four lines once more, with a 256-token ceiling that leaves a reasoning model room to answer in.",
|
|
25
|
+
"24/24 settled with real one-word answers, 971 output tokens, zero fallbacks, every SLA met."
|
|
26
|
+
]
|
|
27
|
+
}
|
|
@@ -11,17 +11,44 @@ import json
|
|
|
11
11
|
from ..job import Job, Result
|
|
12
12
|
from .base import BatchState, Venue
|
|
13
13
|
|
|
14
|
-
__all__ = ["OpenAIBatch", "build_jsonl", "parse_output_line"]
|
|
14
|
+
__all__ = ["OpenAIBatch", "build_jsonl", "parse_output_line", "body_params"]
|
|
15
15
|
|
|
16
16
|
_MODEL_PREFIXES = ("gpt-", "o1", "o3", "o4", "chatgpt-")
|
|
17
17
|
_ENDPOINT = "/v1/chat/completions"
|
|
18
18
|
|
|
19
|
+
# OpenAI's newer families reject ``max_tokens`` outright — "Unsupported
|
|
20
|
+
# parameter: 'max_tokens' is not supported with this model. Use
|
|
21
|
+
# 'max_completion_tokens' instead." — and a batch of a thousand jobs discovers
|
|
22
|
+
# this one HTTP 400 at a time, hours after submission.
|
|
23
|
+
#
|
|
24
|
+
# Translating is the driver's job. A caller says ``max_tokens`` once and it
|
|
25
|
+
# means the same thing at every venue; the venue that spells it differently is
|
|
26
|
+
# the venue's problem, not the caller's.
|
|
27
|
+
_MAX_COMPLETION_TOKENS_PREFIXES = ("o1", "o3", "o4", "gpt-5")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def body_params(job: Job) -> dict:
|
|
31
|
+
"""The job's params as this venue's API spells them.
|
|
32
|
+
|
|
33
|
+
An explicit ``max_completion_tokens`` always wins: the caller who reached
|
|
34
|
+
for the provider's own name for the field meant it, and sending both would
|
|
35
|
+
be rejected.
|
|
36
|
+
"""
|
|
37
|
+
params = dict(job.params)
|
|
38
|
+
if "max_tokens" not in params:
|
|
39
|
+
return params
|
|
40
|
+
if not job.model.startswith(_MAX_COMPLETION_TOKENS_PREFIXES):
|
|
41
|
+
return params
|
|
42
|
+
ceiling = params.pop("max_tokens")
|
|
43
|
+
params.setdefault("max_completion_tokens", ceiling)
|
|
44
|
+
return params
|
|
45
|
+
|
|
19
46
|
|
|
20
47
|
def build_jsonl(jobs: list[Job]) -> bytes:
|
|
21
48
|
"""Render *jobs* as OpenAI Batch API JSONL (one request per line)."""
|
|
22
49
|
lines = []
|
|
23
50
|
for j in jobs:
|
|
24
|
-
body = {"model": j.model, "messages": j.messages, **j
|
|
51
|
+
body = {"model": j.model, "messages": j.messages, **body_params(j)}
|
|
25
52
|
lines.append(
|
|
26
53
|
json.dumps(
|
|
27
54
|
{"custom_id": j.id, "method": "POST", "url": _ENDPOINT, "body": body},
|
|
@@ -129,7 +156,7 @@ class OpenAIBatch(Venue):
|
|
|
129
156
|
def run_sync(self, job: Job) -> Result:
|
|
130
157
|
try:
|
|
131
158
|
response = self.client.chat.completions.create(
|
|
132
|
-
model=job.model, messages=job.messages, **job
|
|
159
|
+
model=job.model, messages=job.messages, **body_params(job)
|
|
133
160
|
)
|
|
134
161
|
except Exception as exc: # noqa: BLE001
|
|
135
162
|
return Result(job=job, error=str(exc))
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
"""The capped settlement runner — everything that happens before money moves.
|
|
2
|
+
|
|
3
|
+
These tests never submit anything: `quote()` makes no API calls, and both paths
|
|
4
|
+
exercised here stop before the venues are touched.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import importlib.util
|
|
8
|
+
import json
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
import pytest
|
|
12
|
+
|
|
13
|
+
_spec = importlib.util.spec_from_file_location(
|
|
14
|
+
"mechanics_run", Path(__file__).resolve().parent.parent / "tools" / "mechanics_run.py"
|
|
15
|
+
)
|
|
16
|
+
mr = importlib.util.module_from_spec(_spec)
|
|
17
|
+
_spec.loader.exec_module(mr)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class TestBook:
|
|
21
|
+
def test_one_lane_per_model_over_the_same_lines(self):
|
|
22
|
+
jobs = mr.build_book(["claude-haiku-4-5", "gpt-5.6-luna"])
|
|
23
|
+
assert len(jobs) == 2 * len(mr.LINES)
|
|
24
|
+
assert {j.model for j in jobs} == {"claude-haiku-4-5", "gpt-5.6-luna"}
|
|
25
|
+
|
|
26
|
+
def test_every_job_carries_a_ceiling(self):
|
|
27
|
+
jobs = mr.build_book(["gpt-5.6-luna"], max_tokens=64)
|
|
28
|
+
assert all(j.params["max_tokens"] == 64 for j in jobs)
|
|
29
|
+
|
|
30
|
+
def test_the_default_ceiling_leaves_a_reasoning_model_room_to_answer(self):
|
|
31
|
+
# A ceiling smaller than the reasoning buys a bill and an empty string.
|
|
32
|
+
assert mr.DEFAULT_MAX_TOKENS >= 128
|
|
33
|
+
|
|
34
|
+
def test_the_lines_are_the_work_not_filler(self):
|
|
35
|
+
jobs = mr.build_book(["gpt-5.6-luna"])
|
|
36
|
+
assert mr.LINES[0] in jobs[0].messages[0]["content"]
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
class TestCapGate:
|
|
40
|
+
def _out(self, tmp_path):
|
|
41
|
+
return str(tmp_path / "run")
|
|
42
|
+
|
|
43
|
+
def test_a_book_under_the_cap_passes_and_still_submits_nothing_on_dry_run(
|
|
44
|
+
self, tmp_path, capsys
|
|
45
|
+
):
|
|
46
|
+
rc = mr.main(["--out", self._out(tmp_path), "--dry-run"])
|
|
47
|
+
out = capsys.readouterr().out
|
|
48
|
+
assert rc == 0
|
|
49
|
+
assert "under cap" in out
|
|
50
|
+
assert not (tmp_path / "run" / "handles.jsonl").exists()
|
|
51
|
+
|
|
52
|
+
def test_a_book_over_the_cap_aborts_before_submitting(self, tmp_path, capsys):
|
|
53
|
+
rc = mr.main(["--out", self._out(tmp_path), "--cap", "0.0000001"])
|
|
54
|
+
assert rc == 2
|
|
55
|
+
assert "ABORT" in capsys.readouterr().out
|
|
56
|
+
assert not (tmp_path / "run" / "handles.jsonl").exists()
|
|
57
|
+
|
|
58
|
+
def test_the_cap_counts_what_the_session_already_exposed(self, tmp_path, capsys):
|
|
59
|
+
# The cap is on the total, so a small book still trips it once enough
|
|
60
|
+
# has been spent around it.
|
|
61
|
+
rc = mr.main(
|
|
62
|
+
["--out", self._out(tmp_path), "--models", "gpt-5.6-luna",
|
|
63
|
+
"--cap", "0.01", "--already-spent", "0.0099"]
|
|
64
|
+
)
|
|
65
|
+
assert rc == 2
|
|
66
|
+
assert "already exposed" in capsys.readouterr().out
|
|
67
|
+
|
|
68
|
+
def test_the_gate_prices_the_ceiling_not_a_hope(self, tmp_path, capsys):
|
|
69
|
+
# The quoted worst case must scale with the ceiling actually set.
|
|
70
|
+
mr.main(["--out", self._out(tmp_path), "--max-tokens", "16", "--dry-run"])
|
|
71
|
+
small = capsys.readouterr().out
|
|
72
|
+
mr.main(["--out", str(tmp_path / "run2"), "--max-tokens", "512", "--dry-run"])
|
|
73
|
+
large = capsys.readouterr().out
|
|
74
|
+
|
|
75
|
+
def worst(text):
|
|
76
|
+
line = next(ln for ln in text.splitlines() if ln.startswith("cap check"))
|
|
77
|
+
return float(line.split("$")[1].split()[0])
|
|
78
|
+
|
|
79
|
+
assert worst(large) > worst(small)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class TestRefusesToResubmit:
|
|
83
|
+
def test_an_existing_handle_log_stops_the_run(self, tmp_path, capsys):
|
|
84
|
+
out = tmp_path / "run"
|
|
85
|
+
out.mkdir()
|
|
86
|
+
(out / "handles.jsonl").write_text(
|
|
87
|
+
json.dumps({"venue": "openai:batch", "handle": "batch_x", "jobs": 1}) + "\n"
|
|
88
|
+
)
|
|
89
|
+
rc = mr.main(["--out", str(out)])
|
|
90
|
+
assert rc == 3
|
|
91
|
+
assert "refusing to re-submit" in capsys.readouterr().out
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
class TestCancelPath:
|
|
95
|
+
def test_cancelling_with_no_log_is_not_an_error(self, tmp_path, capsys):
|
|
96
|
+
rc = mr.main(["--out", str(tmp_path / "nothing-here"), "--cancel"])
|
|
97
|
+
assert rc == 0
|
|
98
|
+
assert "no handles recorded" in capsys.readouterr().out
|
|
99
|
+
|
|
100
|
+
def test_every_recorded_handle_is_cancelled_at_its_own_venue(self, tmp_path, capsys):
|
|
101
|
+
log = tmp_path / "handles.jsonl"
|
|
102
|
+
log.write_text(
|
|
103
|
+
json.dumps({"venue": "openai:batch", "handle": "batch_x", "jobs": 1}) + "\n"
|
|
104
|
+
+ json.dumps({"venue": "anthropic:batch", "handle": "msgbatch_y", "jobs": 1}) + "\n"
|
|
105
|
+
)
|
|
106
|
+
cancelled = []
|
|
107
|
+
|
|
108
|
+
class FakeVenue:
|
|
109
|
+
def __init__(self, name):
|
|
110
|
+
self.name = name
|
|
111
|
+
|
|
112
|
+
def cancel(self, handle):
|
|
113
|
+
cancelled.append((self.name, handle))
|
|
114
|
+
|
|
115
|
+
mr.cancel_all(log, [FakeVenue("openai:batch"), FakeVenue("anthropic:batch")])
|
|
116
|
+
assert cancelled == [("openai:batch", "batch_x"), ("anthropic:batch", "msgbatch_y")]
|
|
117
|
+
|
|
118
|
+
def test_a_handle_from_an_unconfigured_venue_is_reported_not_skipped_silently(
|
|
119
|
+
self, tmp_path, capsys
|
|
120
|
+
):
|
|
121
|
+
log = tmp_path / "handles.jsonl"
|
|
122
|
+
log.write_text(json.dumps({"venue": "groq:batch", "handle": "g_1", "jobs": 1}) + "\n")
|
|
123
|
+
mr.cancel_all(log, [])
|
|
124
|
+
assert "no venue to cancel with" in capsys.readouterr().out
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
@pytest.mark.parametrize("flag,attr,value", [
|
|
128
|
+
("--cap", "cap", 0.25),
|
|
129
|
+
("--max-tokens", "max_tokens", 32),
|
|
130
|
+
])
|
|
131
|
+
def test_the_guard_rails_are_all_settable(flag, attr, value):
|
|
132
|
+
args = mr.parse_args([flag, str(value)])
|
|
133
|
+
assert getattr(args, attr) == value
|