offpeak 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- offpeak-0.2.1/.github/workflows/docs.yml +54 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/.github/workflows/nightly.yml +6 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/.gitignore +4 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/PKG-INFO +22 -9
- {offpeak-0.2.0 → offpeak-0.2.1}/README.md +15 -8
- offpeak-0.2.1/docs/index.md +39 -0
- offpeak-0.2.1/docs/night-board.md +83 -0
- offpeak-0.2.1/docs/quickstart.md +136 -0
- offpeak-0.2.1/docs/reference.md +48 -0
- offpeak-0.2.1/docs/roadmap.md +54 -0
- offpeak-0.2.1/mkdocs.yml +56 -0
- offpeak-0.2.1/mkdocs_hooks.py +21 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/pyproject.toml +3 -1
- {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/__init__.py +1 -1
- offpeak-0.2.1/src/offpeak/prices.py +251 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/quote.py +83 -8
- offpeak-0.2.1/src/offpeak/venues/groq_batch.py +122 -0
- offpeak-0.2.1/tests/test_groq_venue.py +146 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/tests/test_night_report.py +114 -1
- offpeak-0.2.1/tests/test_prices.py +155 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/tests/test_quote.py +82 -6
- {offpeak-0.2.0 → offpeak-0.2.1}/tools/night_report.py +154 -14
- offpeak-0.2.0/src/offpeak/prices.py +0 -98
- {offpeak-0.2.0 → offpeak-0.2.1}/.github/workflows/ci.yml +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/.github/workflows/publish.yml +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/CONTRIBUTING.md +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/LICENSE +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/SPEC.md +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/__main__.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/client.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/deadline.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/job.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/venues/__init__.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/venues/anthropic_batch.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/venues/base.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/venues/openai_batch.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/tests/conftest.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/tests/test_deadline.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/tests/test_job_receipt.py +0 -0
- {offpeak-0.2.0 → offpeak-0.2.1}/tests/test_run.py +0 -0
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
name: Docs
|
|
2
|
+
|
|
3
|
+
# Builds the mkdocs site and deploys it to GitHub Pages. Pages is configured
|
|
4
|
+
# with build_type=workflow, so this workflow is the only publisher — there is
|
|
5
|
+
# no gh-pages branch to keep in sync.
|
|
6
|
+
|
|
7
|
+
on:
|
|
8
|
+
push:
|
|
9
|
+
branches: [main]
|
|
10
|
+
paths:
|
|
11
|
+
- "docs/**"
|
|
12
|
+
- "src/**"
|
|
13
|
+
- "mkdocs.yml"
|
|
14
|
+
- "mkdocs_hooks.py"
|
|
15
|
+
- "SPEC.md"
|
|
16
|
+
- "pyproject.toml"
|
|
17
|
+
- ".github/workflows/docs.yml"
|
|
18
|
+
workflow_dispatch:
|
|
19
|
+
|
|
20
|
+
permissions:
|
|
21
|
+
contents: read
|
|
22
|
+
pages: write
|
|
23
|
+
id-token: write
|
|
24
|
+
|
|
25
|
+
concurrency:
|
|
26
|
+
group: pages
|
|
27
|
+
cancel-in-progress: false
|
|
28
|
+
|
|
29
|
+
jobs:
|
|
30
|
+
build:
|
|
31
|
+
runs-on: ubuntu-latest
|
|
32
|
+
steps:
|
|
33
|
+
- uses: actions/checkout@v7
|
|
34
|
+
- uses: actions/setup-python@v7
|
|
35
|
+
with:
|
|
36
|
+
python-version: "3.12"
|
|
37
|
+
- name: Install
|
|
38
|
+
run: pip install -e ".[docs]"
|
|
39
|
+
- name: Build
|
|
40
|
+
run: mkdocs build --strict
|
|
41
|
+
- uses: actions/configure-pages@v6
|
|
42
|
+
- uses: actions/upload-pages-artifact@v5
|
|
43
|
+
with:
|
|
44
|
+
path: site
|
|
45
|
+
|
|
46
|
+
deploy:
|
|
47
|
+
needs: build
|
|
48
|
+
runs-on: ubuntu-latest
|
|
49
|
+
environment:
|
|
50
|
+
name: github-pages
|
|
51
|
+
url: ${{ steps.deployment.outputs.page_url }}
|
|
52
|
+
steps:
|
|
53
|
+
- id: deployment
|
|
54
|
+
uses: actions/deploy-pages@v5
|
|
@@ -54,10 +54,16 @@ jobs:
|
|
|
54
54
|
echo "mode=$mode" >> "$GITHUB_OUTPUT"
|
|
55
55
|
echo "running the $mode pass"
|
|
56
56
|
|
|
57
|
+
- name: Install US-zone data source
|
|
58
|
+
# gridstatus is an Action-only dependency: the SDK does not need it and
|
|
59
|
+
# the generator degrades to GB-only without it.
|
|
60
|
+
run: pip install "gridstatus>=0.36"
|
|
61
|
+
|
|
57
62
|
- name: Generate
|
|
58
63
|
run: |
|
|
59
64
|
python tools/night_report.py \
|
|
60
65
|
--mode "${{ steps.pick.outputs.mode }}" \
|
|
66
|
+
--us-zones \
|
|
61
67
|
--outdir board/nightly
|
|
62
68
|
|
|
63
69
|
- name: Commit to board-data
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: offpeak
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Deadline-priced inference: give AI jobs a deadline and run them on the cheapest venue — provider batch tiers (−50%) today. Same model, same tokens, a different hour.
|
|
5
5
|
Project-URL: Homepage, https://github.com/offpeak-ai/offpeak
|
|
6
6
|
Project-URL: Repository, https://github.com/offpeak-ai/offpeak
|
|
@@ -31,6 +31,12 @@ Requires-Dist: build; extra == 'dev'
|
|
|
31
31
|
Requires-Dist: pytest>=8; extra == 'dev'
|
|
32
32
|
Requires-Dist: ruff>=0.6; extra == 'dev'
|
|
33
33
|
Requires-Dist: twine; extra == 'dev'
|
|
34
|
+
Provides-Extra: docs
|
|
35
|
+
Requires-Dist: mkdocs-material<10,>=9.5; extra == 'docs'
|
|
36
|
+
Requires-Dist: mkdocs<2,>=1.6; extra == 'docs'
|
|
37
|
+
Requires-Dist: mkdocstrings[python]>=0.26; extra == 'docs'
|
|
38
|
+
Provides-Extra: groq
|
|
39
|
+
Requires-Dist: groq>=0.11; extra == 'groq'
|
|
34
40
|
Provides-Extra: openai
|
|
35
41
|
Requires-Dist: openai>=1.50; extra == 'openai'
|
|
36
42
|
Description-Content-Type: text/markdown
|
|
@@ -66,10 +72,12 @@ tokens 12,410,332 in · 3,104,551 out
|
|
|
66
72
|
list $27.93
|
|
67
73
|
paid $14.02
|
|
68
74
|
captured $13.91 (49.8%)
|
|
69
|
-
prices snapshot 2026-08 — override via offpeak.prices
|
|
75
|
+
prices snapshot 2026-08-21 — override via offpeak.prices
|
|
70
76
|
───────────────────────────────────────────────
|
|
71
77
|
```
|
|
72
78
|
|
|
79
|
+
**[Documentation](https://offpeak-ai.github.io/offpeak/)** · [Quickstart](https://offpeak-ai.github.io/offpeak/quickstart/) · [Spec](https://offpeak-ai.github.io/offpeak/spec/) · [Roadmap](https://offpeak-ai.github.io/offpeak/roadmap/)
|
|
80
|
+
|
|
73
81
|
## What it does
|
|
74
82
|
|
|
75
83
|
- **Know the price before you spend it.** `quote(jobs, deadline=...)` prices a run against the published sheets with no API calls and no key — list versus batch, per venue, plus what the wait is worth.
|
|
@@ -103,18 +111,21 @@ jobs 5000 across 1 venue(s)
|
|
|
103
111
|
deadline 2026-08-21 21:11 PDT (24.0h out)
|
|
104
112
|
tokens 4,000,000 in · 1,000,000 out
|
|
105
113
|
|
|
106
|
-
openai:batch 5000 job(s) list $
|
|
114
|
+
openai:batch 5000 job(s) list $2.00 batch $1.00 save $1.00 (50.0%)
|
|
107
115
|
|
|
108
|
-
list $
|
|
109
|
-
batch $
|
|
110
|
-
save $
|
|
116
|
+
list $2.00 (run now, synchronously)
|
|
117
|
+
batch $1.00 (run by the deadline)
|
|
118
|
+
save $1.00 (50.0%)
|
|
119
|
+
risk deadline is inside the 24h batch window — the SLA rests on the sync fallback, which pays list
|
|
111
120
|
basis input explicit; output explicit
|
|
112
|
-
prices snapshot 2026-08 — estimate only, not a bill
|
|
121
|
+
prices snapshot 2026-08-21 — estimate only, not a bill
|
|
113
122
|
───────────────────────────────────────────────
|
|
114
123
|
```
|
|
115
124
|
|
|
116
125
|
From Python, `offpeak.quote(jobs, deadline="06:00")` takes the same jobs you would pass to `run()`. Token counts come from the job where it knows them (`metadata={"input_tokens": ..., "output_tokens": ...}`, or `max_tokens` as an output ceiling) and are a labeled chars/4 estimate where it does not — every figure reports its provenance in `basis`, and a quote with no output signal is marked a **floor**, not an estimate.
|
|
117
126
|
|
|
127
|
+
If you do know roughly what the model will write, say so and get a priced number instead — `quote(jobs, deadline=..., assumed_output_ratio=0.25)`, or `metadata={"expected_output_tokens": 300}` on a single job. Those quotes are marked **EST**, distinct from a floor. Both are opt-in: absent one, `offpeak` assumes nothing on your behalf.
|
|
128
|
+
|
|
118
129
|
## Deadlines
|
|
119
130
|
|
|
120
131
|
Deadlines are how software says "this can wait" — the full semantics live in [SPEC.md](SPEC.md).
|
|
@@ -152,13 +163,15 @@ import offpeak
|
|
|
152
163
|
offpeak.prices.register_price("my-fine-tune", input_per_m=4.0, output_per_m=16.0)
|
|
153
164
|
```
|
|
154
165
|
|
|
155
|
-
Unknown models settle with `cost = None` rather than a guess.
|
|
166
|
+
Unknown models settle with `cost = None` rather than a guess. The sheet also carries what the venues charge for *urgency* — `get_fast_price()`, `urgency_spread()` — and flags list prices that are promotional, with the date and the price they decay to: `promo_decay("gpt-5.6-sol")` is `(1.25, 1.5)` after 2026-11-21.
|
|
156
167
|
|
|
157
168
|
## What this is (and the roadmap)
|
|
158
169
|
|
|
159
170
|
`offpeak` is the open client and spec for a simple claim: **intelligence has a time value**. A large share of AI work — embeddings, evals, backfills, report generation, overnight agents — has no human waiting on it, and the venues already price that patience at −50%. This library is the missing workflow.
|
|
160
171
|
|
|
161
|
-
The **[
|
|
172
|
+
The token side is wider than the headline discount. Patience is priced at −50% — a 2.0x spread — and haste is priced too: hold the model and venue constant and `gpt-5.6-sol` costs **$8.00 / $40.00** per 1M on OpenAI's fast tier against **$2.00 / $10.00** on its batch tier, a **4x intra-venue urgency spread** for the hour alone ([source](https://developers.openai.com/api/docs/pricing); sol's standard rate is promotional at least through 2026-11-21, and both tiers are defined off it, so the ratio outlives the prices). It is data, not prose: `offpeak.prices.urgency_spread("gpt-5.6-sol")` returns `4.0`.
|
|
173
|
+
|
|
174
|
+
The **[night board](https://github.com/offpeak-ai/offpeak/blob/board-data/nightly/BOARD.md)** marks the same claim against open grid data every night — GB power and carbon plus CAISO SP15 and ERCOT Houston peak/off-peak spreads, alongside the published token spreads. ERCOT Houston marked 3.94x on the night of 2026-08-20; a venue charges 4x for the same impatience.
|
|
162
175
|
|
|
163
176
|
The roadmap follows the same interface upward: more venues (Google batch, spot capacity, off-peak windows on your own GPUs), queue-latency forecasting instead of a fixed risk buffer, portfolio placement across venues, energy- and carbon-aware scheduling with per-job receipts. The venue interface (`offpeak.Venue`) is deliberately the extension point — a venue is anywhere deferred work can run.
|
|
164
177
|
|
|
@@ -29,10 +29,12 @@ tokens 12,410,332 in · 3,104,551 out
|
|
|
29
29
|
list $27.93
|
|
30
30
|
paid $14.02
|
|
31
31
|
captured $13.91 (49.8%)
|
|
32
|
-
prices snapshot 2026-08 — override via offpeak.prices
|
|
32
|
+
prices snapshot 2026-08-21 — override via offpeak.prices
|
|
33
33
|
───────────────────────────────────────────────
|
|
34
34
|
```
|
|
35
35
|
|
|
36
|
+
**[Documentation](https://offpeak-ai.github.io/offpeak/)** · [Quickstart](https://offpeak-ai.github.io/offpeak/quickstart/) · [Spec](https://offpeak-ai.github.io/offpeak/spec/) · [Roadmap](https://offpeak-ai.github.io/offpeak/roadmap/)
|
|
37
|
+
|
|
36
38
|
## What it does
|
|
37
39
|
|
|
38
40
|
- **Know the price before you spend it.** `quote(jobs, deadline=...)` prices a run against the published sheets with no API calls and no key — list versus batch, per venue, plus what the wait is worth.
|
|
@@ -66,18 +68,21 @@ jobs 5000 across 1 venue(s)
|
|
|
66
68
|
deadline 2026-08-21 21:11 PDT (24.0h out)
|
|
67
69
|
tokens 4,000,000 in · 1,000,000 out
|
|
68
70
|
|
|
69
|
-
openai:batch 5000 job(s) list $
|
|
71
|
+
openai:batch 5000 job(s) list $2.00 batch $1.00 save $1.00 (50.0%)
|
|
70
72
|
|
|
71
|
-
list $
|
|
72
|
-
batch $
|
|
73
|
-
save $
|
|
73
|
+
list $2.00 (run now, synchronously)
|
|
74
|
+
batch $1.00 (run by the deadline)
|
|
75
|
+
save $1.00 (50.0%)
|
|
76
|
+
risk deadline is inside the 24h batch window — the SLA rests on the sync fallback, which pays list
|
|
74
77
|
basis input explicit; output explicit
|
|
75
|
-
prices snapshot 2026-08 — estimate only, not a bill
|
|
78
|
+
prices snapshot 2026-08-21 — estimate only, not a bill
|
|
76
79
|
───────────────────────────────────────────────
|
|
77
80
|
```
|
|
78
81
|
|
|
79
82
|
From Python, `offpeak.quote(jobs, deadline="06:00")` takes the same jobs you would pass to `run()`. Token counts come from the job where it knows them (`metadata={"input_tokens": ..., "output_tokens": ...}`, or `max_tokens` as an output ceiling) and are a labeled chars/4 estimate where it does not — every figure reports its provenance in `basis`, and a quote with no output signal is marked a **floor**, not an estimate.
|
|
80
83
|
|
|
84
|
+
If you do know roughly what the model will write, say so and get a priced number instead — `quote(jobs, deadline=..., assumed_output_ratio=0.25)`, or `metadata={"expected_output_tokens": 300}` on a single job. Those quotes are marked **EST**, distinct from a floor. Both are opt-in: absent one, `offpeak` assumes nothing on your behalf.
|
|
85
|
+
|
|
81
86
|
## Deadlines
|
|
82
87
|
|
|
83
88
|
Deadlines are how software says "this can wait" — the full semantics live in [SPEC.md](SPEC.md).
|
|
@@ -115,13 +120,15 @@ import offpeak
|
|
|
115
120
|
offpeak.prices.register_price("my-fine-tune", input_per_m=4.0, output_per_m=16.0)
|
|
116
121
|
```
|
|
117
122
|
|
|
118
|
-
Unknown models settle with `cost = None` rather than a guess.
|
|
123
|
+
Unknown models settle with `cost = None` rather than a guess. The sheet also carries what the venues charge for *urgency* — `get_fast_price()`, `urgency_spread()` — and flags list prices that are promotional, with the date and the price they decay to: `promo_decay("gpt-5.6-sol")` is `(1.25, 1.5)` after 2026-11-21.
|
|
119
124
|
|
|
120
125
|
## What this is (and the roadmap)
|
|
121
126
|
|
|
122
127
|
`offpeak` is the open client and spec for a simple claim: **intelligence has a time value**. A large share of AI work — embeddings, evals, backfills, report generation, overnight agents — has no human waiting on it, and the venues already price that patience at −50%. This library is the missing workflow.
|
|
123
128
|
|
|
124
|
-
The **[
|
|
129
|
+
The token side is wider than the headline discount. Patience is priced at −50% — a 2.0x spread — and haste is priced too: hold the model and venue constant and `gpt-5.6-sol` costs **$8.00 / $40.00** per 1M on OpenAI's fast tier against **$2.00 / $10.00** on its batch tier, a **4x intra-venue urgency spread** for the hour alone ([source](https://developers.openai.com/api/docs/pricing); sol's standard rate is promotional at least through 2026-11-21, and both tiers are defined off it, so the ratio outlives the prices). It is data, not prose: `offpeak.prices.urgency_spread("gpt-5.6-sol")` returns `4.0`.
|
|
130
|
+
|
|
131
|
+
The **[night board](https://github.com/offpeak-ai/offpeak/blob/board-data/nightly/BOARD.md)** marks the same claim against open grid data every night — GB power and carbon plus CAISO SP15 and ERCOT Houston peak/off-peak spreads, alongside the published token spreads. ERCOT Houston marked 3.94x on the night of 2026-08-20; a venue charges 4x for the same impatience.
|
|
125
132
|
|
|
126
133
|
The roadmap follows the same interface upward: more venues (Google batch, spot capacity, off-peak windows on your own GPUs), queue-latency forecasting instead of a fixed risk buffer, portfolio placement across venues, energy- and carbon-aware scheduling with per-job receipts. The venue interface (`offpeak.Venue`) is deliberately the extension point — a venue is anywhere deferred work can run.
|
|
127
134
|
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# offpeak
|
|
2
|
+
|
|
3
|
+
**Deadline-priced inference.** Same model, same tokens, a different hour.
|
|
4
|
+
|
|
5
|
+
A large share of AI work — embeddings, evals, backfills, report generation,
|
|
6
|
+
overnight agents — has no human waiting on it. The providers already price that
|
|
7
|
+
patience: OpenAI and Anthropic both publish their batch tiers at **50% of
|
|
8
|
+
list**. `offpeak` is the workflow that collects the difference.
|
|
9
|
+
|
|
10
|
+
```python
|
|
11
|
+
import offpeak
|
|
12
|
+
|
|
13
|
+
jobs = [offpeak.job("claude-haiku-4-5", f"Summarize:\n\n{d}") for d in docs]
|
|
14
|
+
|
|
15
|
+
print(offpeak.quote(jobs, deadline="06:00")) # what is the wait worth?
|
|
16
|
+
results = offpeak.run(jobs, deadline="06:00") # collect it
|
|
17
|
+
print(offpeak.receipt(results)) # what it actually cost
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
- **[Quickstart](quickstart.md)** — install, quote, run, read the receipt.
|
|
21
|
+
- **[The night board](night-board.md)** — the same claim, marked nightly against open grid data.
|
|
22
|
+
- **[Spec](spec.md)** — deadline semantics, statuses, receipts.
|
|
23
|
+
- **[API reference](reference.md)** — every public symbol.
|
|
24
|
+
- **[Roadmap](roadmap.md)** — what exists, what does not, and what is being built.
|
|
25
|
+
|
|
26
|
+
## What it guarantees
|
|
27
|
+
|
|
28
|
+
**One `Result` per job, always.** Provider failures at submit, poll, cancel or
|
|
29
|
+
sync are captured, not raised. Affected jobs take the sync fallback where the
|
|
30
|
+
deadline still allows it, and otherwise return failed with the provider's
|
|
31
|
+
message attached. Exceptions are reserved for programming errors — a deadline
|
|
32
|
+
in the past, or a model no venue supports.
|
|
33
|
+
|
|
34
|
+
**Your keys, your perimeter.** `offpeak` talks straight to the providers with
|
|
35
|
+
your own credentials. There is no proxy and no third party in the data path.
|
|
36
|
+
|
|
37
|
+
**Receipts are arithmetic, not estimates.** Every figure traces to a published
|
|
38
|
+
price sheet, and a model that is not on one settles as `None` rather than a
|
|
39
|
+
guess.
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
# The night board
|
|
2
|
+
|
|
3
|
+
`offpeak` rests on a claim: **intelligence has a time value.** The token side of
|
|
4
|
+
that claim is already settled, and it is wider than the headline discount.
|
|
5
|
+
|
|
6
|
+
- **Patience is priced at −50%.** OpenAI and Anthropic both publish batch tiers
|
|
7
|
+
at half of list: a flat **2.0x** spread for work that can wait.
|
|
8
|
+
- **Haste is priced too.** Hold the model and the venue constant and read the
|
|
9
|
+
same sheet across its urgency tiers: `gpt-5.6-sol` is **$8.00 / $40.00** per
|
|
10
|
+
1M tokens on OpenAI's fast tier against **$2.00 / $10.00** on its batch tier
|
|
11
|
+
— a **4x intra-venue urgency spread**, on both legs, for the hour alone.
|
|
12
|
+
Source: [developers.openai.com/api/docs/pricing](https://developers.openai.com/api/docs/pricing).
|
|
13
|
+
|
|
14
|
+
!!! note "The promo caveat"
|
|
15
|
+
`gpt-5.6-sol`'s standard rate is promotional — the sheet says it runs *at
|
|
16
|
+
least* through **2026-11-21**, after which list is $5/$30. Fast and batch
|
|
17
|
+
are both defined off that list (2x and 0.5x), so the promo moves the
|
|
18
|
+
dollars and leaves the ratio: **the 4x is the durable figure, the prices
|
|
19
|
+
are the perishable ones.** Both are in the SDK rather than in prose —
|
|
20
|
+
`offpeak.prices.urgency_spread("gpt-5.6-sol")` returns `4.0`, and
|
|
21
|
+
`promo_decay()` returns the step-up the date will bring.
|
|
22
|
+
|
|
23
|
+
The night board marks the same claim against the other side of the trade: the
|
|
24
|
+
grid the compute runs on. Every night it records what power and carbon actually
|
|
25
|
+
did between the evening peak and the small hours, from open, keyless sources.
|
|
26
|
+
The grid's spread is not published anywhere — it has to be observed — and on the
|
|
27
|
+
night of 2026-08-20 the ERCOT Houston hub marked **3.94x** between its evening
|
|
28
|
+
peak and the trough that followed, against the 4x a venue charges for the same
|
|
29
|
+
hour of impatience.
|
|
30
|
+
|
|
31
|
+
**[→ Read the board](https://github.com/offpeak-ai/offpeak/blob/board-data/nightly/BOARD.md)**
|
|
32
|
+
|
|
33
|
+
## How it works
|
|
34
|
+
|
|
35
|
+
Two passes over the same night, run by
|
|
36
|
+
[a scheduled workflow](https://github.com/offpeak-ai/offpeak/blob/main/.github/workflows/nightly.yml):
|
|
37
|
+
|
|
38
|
+
| Pass | When | What it records |
|
|
39
|
+
|---|---|---|
|
|
40
|
+
| `quote` | 19:00Z | The night ahead — carbon **forecast**, day-ahead power |
|
|
41
|
+
| `mark` | 06:30Z | The night just finished — carbon **actuals** |
|
|
42
|
+
|
|
43
|
+
A night runs **16:00Z–07:00Z**: it opens with the 17:00 BST evening peak and
|
|
44
|
+
closes after the 00–05 BST trough, so one span carries both windows the board
|
|
45
|
+
compares.
|
|
46
|
+
|
|
47
|
+
Output lands on the
|
|
48
|
+
[`board-data` branch](https://github.com/offpeak-ai/offpeak/tree/board-data) —
|
|
49
|
+
`nightly/BOARD.md` plus the raw JSON per night — because `main` is protected and
|
|
50
|
+
would reject a nightly bot push.
|
|
51
|
+
|
|
52
|
+
## Sources
|
|
53
|
+
|
|
54
|
+
- **Carbon** — [NESO carbon intensity](https://api.carbonintensity.org.uk),
|
|
55
|
+
GB, keyless.
|
|
56
|
+
- **Power, GB** — [Octopus Agile](https://api.octopus.energy) day-ahead unit
|
|
57
|
+
rates, GB region C, keyless.
|
|
58
|
+
- **Power, US** — CAISO SP15 and ERCOT Houston day-ahead hourly, via
|
|
59
|
+
[gridstatus](https://github.com/gridstatus/gridstatus), keyless.
|
|
60
|
+
- **Tokens** — the published price sheets, not a measurement:
|
|
61
|
+
[OpenAI](https://developers.openai.com/api/docs/pricing) and
|
|
62
|
+
[Anthropic](https://platform.claude.com/docs/en/about-claude/pricing).
|
|
63
|
+
|
|
64
|
+
All are public and free. The board costs nothing to run and spends nothing at
|
|
65
|
+
any venue.
|
|
66
|
+
|
|
67
|
+
## Honest limits
|
|
68
|
+
|
|
69
|
+
- **Four zones, two of them thin.** GB carbon and GB power are half-hourly and
|
|
70
|
+
complete. CAISO SP15 and ERCOT Houston are **day-ahead hourly** prices, not
|
|
71
|
+
settled real-time ones, and they are power only — no US carbon leg yet.
|
|
72
|
+
- **The token column is published, not observed.** The 2.0x and the 4x are read
|
|
73
|
+
off price sheets; only the grid columns are measurements. A published number
|
|
74
|
+
and a marked one are different kinds of claim, and the board should not blur
|
|
75
|
+
them.
|
|
76
|
+
- **Observation, not advice.** The board records what the grid did. It does not
|
|
77
|
+
forecast, and `offpeak` does not currently schedule against it — the SDK
|
|
78
|
+
routes on published token prices alone.
|
|
79
|
+
- **Carbon actuals lag** roughly two hours. The 06:30Z mark clears the 00–05 BST
|
|
80
|
+
trough comfortably; the tail of the night can still be sparse.
|
|
81
|
+
- **A dead source costs its column, not the run.** Legs degrade independently,
|
|
82
|
+
and a night where both sources are down is recorded as unavailable rather than
|
|
83
|
+
guessed at.
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# Quickstart
|
|
2
|
+
|
|
3
|
+
## Install
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
pip install "offpeak[all]" # OpenAI + Anthropic venues
|
|
7
|
+
pip install "offpeak[anthropic]" # or just one
|
|
8
|
+
pip install "offpeak[openai]"
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
The core has zero dependencies; provider SDKs load only through the extras.
|
|
12
|
+
Venues read the standard environment variables (`OPENAI_API_KEY`,
|
|
13
|
+
`ANTHROPIC_API_KEY`), or take a configured client:
|
|
14
|
+
`OpenAIBatch(client=my_client)`.
|
|
15
|
+
|
|
16
|
+
## 1. Quote — before you spend anything
|
|
17
|
+
|
|
18
|
+
`quote()` makes **no API calls and needs no key**. It prices your jobs against
|
|
19
|
+
the bundled sheet: list versus batch, per venue.
|
|
20
|
+
|
|
21
|
+
```bash
|
|
22
|
+
python -m offpeak quote --model gpt-5.6-luna --input-tokens 800 --output-tokens 200 --jobs 5000
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
```
|
|
26
|
+
OFFPEAK QUOTE ─────────────────────────────────
|
|
27
|
+
jobs 5000 across 1 venue(s)
|
|
28
|
+
deadline 2026-08-21 21:11 PDT (24.0h out)
|
|
29
|
+
tokens 4,000,000 in · 1,000,000 out
|
|
30
|
+
|
|
31
|
+
openai:batch 5000 job(s) list $2.00 batch $1.00 save $1.00 (50.0%)
|
|
32
|
+
|
|
33
|
+
list $2.00 (run now, synchronously)
|
|
34
|
+
batch $1.00 (run by the deadline)
|
|
35
|
+
save $1.00 (50.0%)
|
|
36
|
+
risk deadline is inside the 24h batch window — the SLA rests on the sync fallback, which pays list
|
|
37
|
+
basis input explicit; output explicit
|
|
38
|
+
prices snapshot 2026-08-21 — estimate only, not a bill
|
|
39
|
+
───────────────────────────────────────────────
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
From Python, the same jobs you would pass to `run()`:
|
|
43
|
+
|
|
44
|
+
```python
|
|
45
|
+
q = offpeak.quote(jobs, deadline="06:00")
|
|
46
|
+
print(q.spread_usd, q.spread_pct)
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
!!! warning "Quotes that omit output are marked a floor"
|
|
50
|
+
Output costs more than input on every model on the sheet. If a job carries
|
|
51
|
+
no output-token signal, `quote()` prices its output at **zero** and labels
|
|
52
|
+
the whole quote a `FLOOR` — a stated floor is safer than an invented
|
|
53
|
+
number. Give it `max_tokens`, or explicit counts:
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
offpeak.job("claude-haiku-4-5", prompt, max_tokens=512)
|
|
57
|
+
offpeak.Job(model=..., messages=[...],
|
|
58
|
+
metadata={"input_tokens": 800, "output_tokens": 200})
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
`Quote.basis` reports the provenance of every figure.
|
|
62
|
+
|
|
63
|
+
### If you know roughly what it will write
|
|
64
|
+
|
|
65
|
+
A floor is honest but not always useful. When you do have a sense of the output
|
|
66
|
+
size, say so — and the quote prices it, marked `EST` rather than `FLOOR`:
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
# Across the run: assume each job writes a quarter of what it reads.
|
|
70
|
+
offpeak.quote(jobs, deadline="06:00", assumed_output_ratio=0.25)
|
|
71
|
+
|
|
72
|
+
# Or per job, which wins over a ratio and over max_tokens:
|
|
73
|
+
offpeak.Job(model=..., messages=[...],
|
|
74
|
+
metadata={"expected_output_tokens": 300})
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
```
|
|
78
|
+
EST 5000 job(s) priced on an assumed output size, not a measured one
|
|
79
|
+
the assumption is yours; the bill moves with what the model actually writes
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
Both are opt-in. Without one, nothing is assumed on your behalf: the default
|
|
83
|
+
stays the floor. The two marks mean different things and a quote can carry both
|
|
84
|
+
— `FLOOR` is understated by construction, `EST` can land either side of the
|
|
85
|
+
bill. `Quote.is_floor` and `Quote.is_estimated` are the same distinction in
|
|
86
|
+
code, and a ratio applies only to jobs with no signal of their own, so explicit
|
|
87
|
+
counts and `max_tokens` are never overridden by it.
|
|
88
|
+
|
|
89
|
+
## 2. Run — against a deadline
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
results = offpeak.run(jobs, deadline="06:00")
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Each job goes to the batch tier of a venue that supports its model. `offpeak`
|
|
96
|
+
polls until the work lands. If the batch has not completed by the time the
|
|
97
|
+
remaining window shrinks to the risk buffer, it cancels and re-runs the
|
|
98
|
+
stragglers synchronously at list price — you stated a deadline, and it is met.
|
|
99
|
+
|
|
100
|
+
Deadlines accept `"06:00"` (next occurrence), `"6h"`, `"90m"`, a `datetime`, a
|
|
101
|
+
`timedelta`, seconds, or an ISO 8601 string.
|
|
102
|
+
|
|
103
|
+
## 3. Read the receipt
|
|
104
|
+
|
|
105
|
+
```python
|
|
106
|
+
print(offpeak.receipt(results))
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
```
|
|
110
|
+
OFFPEAK SETTLEMENT ────────────────────────────
|
|
111
|
+
jobs 5000 (5000 ok, 120 sync fallback, 0 failed)
|
|
112
|
+
sla 5000/5000 met
|
|
113
|
+
venues anthropic:batch 3000 · openai:batch 2000
|
|
114
|
+
tokens 41,000,000 in · 3,200,000 out
|
|
115
|
+
list $2,469.00
|
|
116
|
+
paid $1,234.50
|
|
117
|
+
captured $1,234.50 (50.0%)
|
|
118
|
+
left $29.63 on the table (120 job(s) missed the batch tier)
|
|
119
|
+
prices snapshot 2026-08-21 — override via offpeak.prices
|
|
120
|
+
───────────────────────────────────────────────
|
|
121
|
+
```
|
|
122
|
+
|
|
123
|
+
`left on the table` is what the sync fallback gave up by missing the batch
|
|
124
|
+
tier. Per-job receipts render the same way — `print(results[0].receipt)` — with
|
|
125
|
+
sub-cent precision, so a small run reports what it cost rather than `$0.00`.
|
|
126
|
+
|
|
127
|
+
## Prices
|
|
128
|
+
|
|
129
|
+
The bundled sheet is a dated snapshot. Providers move prices; override at
|
|
130
|
+
runtime rather than waiting for a release:
|
|
131
|
+
|
|
132
|
+
```python
|
|
133
|
+
offpeak.prices.register_price("my-fine-tune", input_per_m=4.0, output_per_m=16.0)
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
Unknown models settle as `None`, never a guess.
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# API reference
|
|
2
|
+
|
|
3
|
+
Everything below is exported from the package root: `offpeak.run`,
|
|
4
|
+
`offpeak.Job`, and so on.
|
|
5
|
+
|
|
6
|
+
## Running work
|
|
7
|
+
|
|
8
|
+
::: offpeak.run
|
|
9
|
+
|
|
10
|
+
::: offpeak.quote
|
|
11
|
+
|
|
12
|
+
::: offpeak.receipt
|
|
13
|
+
|
|
14
|
+
::: offpeak.job
|
|
15
|
+
|
|
16
|
+
## Types
|
|
17
|
+
|
|
18
|
+
::: offpeak.Job
|
|
19
|
+
|
|
20
|
+
::: offpeak.Result
|
|
21
|
+
|
|
22
|
+
::: offpeak.Receipt
|
|
23
|
+
|
|
24
|
+
::: offpeak.Status
|
|
25
|
+
|
|
26
|
+
::: offpeak.Settlement
|
|
27
|
+
|
|
28
|
+
::: offpeak.Quote
|
|
29
|
+
|
|
30
|
+
::: offpeak.VenueQuote
|
|
31
|
+
|
|
32
|
+
## Deadlines
|
|
33
|
+
|
|
34
|
+
::: offpeak.parse_deadline
|
|
35
|
+
|
|
36
|
+
::: offpeak.seconds_until
|
|
37
|
+
|
|
38
|
+
## Venues
|
|
39
|
+
|
|
40
|
+
::: offpeak.Venue
|
|
41
|
+
|
|
42
|
+
::: offpeak.BatchState
|
|
43
|
+
|
|
44
|
+
::: offpeak.default_venues
|
|
45
|
+
|
|
46
|
+
## Prices
|
|
47
|
+
|
|
48
|
+
::: offpeak.prices
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# Roadmap
|
|
2
|
+
|
|
3
|
+
What exists, what does not, and what is being built. This page is meant to be
|
|
4
|
+
checkable — if something here is not true yet, it says so.
|
|
5
|
+
|
|
6
|
+
## What exists today
|
|
7
|
+
|
|
8
|
+
- **`run(jobs, deadline=...)`** across the OpenAI and Anthropic batch tiers,
|
|
9
|
+
with a sync fallback that protects the deadline.
|
|
10
|
+
- **`quote(jobs, deadline=...)`** — pre-trade pricing with no API calls.
|
|
11
|
+
- **Receipts and settlements** — list, paid, captured spread, and what a
|
|
12
|
+
fallback left on the table, as arithmetic against published price sheets.
|
|
13
|
+
- **A `Venue` interface** — the extension point. A venue is anywhere deferred
|
|
14
|
+
work can run.
|
|
15
|
+
- **[The night board](night-board.md)** — GB power and carbon, marked nightly.
|
|
16
|
+
|
|
17
|
+
## What does not exist yet
|
|
18
|
+
|
|
19
|
+
Stated plainly, because a roadmap that reads like a feature list is a
|
|
20
|
+
misleading one:
|
|
21
|
+
|
|
22
|
+
- **No queue-latency forecasting.** Deadline risk is a fixed buffer, not a
|
|
23
|
+
prediction. `offpeak` does not know how long a given venue's queue is; it
|
|
24
|
+
watches the clock and falls back.
|
|
25
|
+
- **No cross-venue portfolio placement.** Jobs route to the first venue that
|
|
26
|
+
supports the model, not to the cheapest or fastest across a portfolio.
|
|
27
|
+
- **No carbon-aware scheduling.** The night board observes the grid; the
|
|
28
|
+
scheduler does not read it.
|
|
29
|
+
- **No venues beyond the two batch tiers.** Google batch, spot capacity, and
|
|
30
|
+
off-peak windows on your own GPUs are interface-shaped but unwritten. A
|
|
31
|
+
Groq batch driver exists in the tree but is **untested against the live
|
|
32
|
+
API** — it is opt-in, excluded from the `all` extra, and not in
|
|
33
|
+
`default_venues()`.
|
|
34
|
+
|
|
35
|
+
## The hosted desk
|
|
36
|
+
|
|
37
|
+
A hosted desk that does the forecasting, cross-venue portfolio scheduling, and
|
|
38
|
+
SLA insurance at fleet scale — with payloads never leaving your perimeter — is
|
|
39
|
+
being built by the same team.
|
|
40
|
+
|
|
41
|
+
The intended seam is the one already in the library: a desk would be selected
|
|
42
|
+
per run, alongside the venues you already pass, so that moving from local
|
|
43
|
+
scheduling to hosted scheduling is a keyword argument rather than a rewrite.
|
|
44
|
+
|
|
45
|
+
!!! note "Not implemented"
|
|
46
|
+
That parameter does not exist in the public API today, and nothing in this
|
|
47
|
+
release accepts it. It is described here so the shape of the plan is
|
|
48
|
+
legible — not as something you can call. The SDK and the deadline spec stay
|
|
49
|
+
open, Apache-2.0, either way.
|
|
50
|
+
|
|
51
|
+
## The spec
|
|
52
|
+
|
|
53
|
+
Deadline semantics are versioned separately in [SPEC.md](spec.md), so a second
|
|
54
|
+
implementation can be written against them. Spec changes start as issues.
|
offpeak-0.2.1/mkdocs.yml
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
site_name: offpeak
|
|
2
|
+
site_description: Deadline-priced inference — same model, same tokens, a different hour.
|
|
3
|
+
site_url: https://offpeak-ai.github.io/offpeak/
|
|
4
|
+
repo_url: https://github.com/offpeak-ai/offpeak
|
|
5
|
+
repo_name: offpeak-ai/offpeak
|
|
6
|
+
edit_uri: edit/main/docs/
|
|
7
|
+
|
|
8
|
+
theme:
|
|
9
|
+
name: material
|
|
10
|
+
palette:
|
|
11
|
+
- media: "(prefers-color-scheme: light)"
|
|
12
|
+
scheme: default
|
|
13
|
+
primary: black
|
|
14
|
+
accent: indigo
|
|
15
|
+
toggle: {icon: material/weather-night, name: Switch to dark mode}
|
|
16
|
+
- media: "(prefers-color-scheme: dark)"
|
|
17
|
+
scheme: slate
|
|
18
|
+
primary: black
|
|
19
|
+
accent: indigo
|
|
20
|
+
toggle: {icon: material/weather-sunny, name: Switch to light mode}
|
|
21
|
+
features:
|
|
22
|
+
- navigation.sections
|
|
23
|
+
- navigation.top
|
|
24
|
+
- content.code.copy
|
|
25
|
+
|
|
26
|
+
nav:
|
|
27
|
+
- Home: index.md
|
|
28
|
+
- Quickstart: quickstart.md
|
|
29
|
+
- The night board: night-board.md
|
|
30
|
+
- Spec: spec.md
|
|
31
|
+
- API reference: reference.md
|
|
32
|
+
- Roadmap: roadmap.md
|
|
33
|
+
|
|
34
|
+
plugins:
|
|
35
|
+
- search
|
|
36
|
+
- mkdocstrings:
|
|
37
|
+
handlers:
|
|
38
|
+
python:
|
|
39
|
+
paths: [src]
|
|
40
|
+
options:
|
|
41
|
+
show_source: true
|
|
42
|
+
show_root_heading: true
|
|
43
|
+
heading_level: 3
|
|
44
|
+
docstring_style: sphinx
|
|
45
|
+
members_order: source
|
|
46
|
+
|
|
47
|
+
# SPEC.md lives at the repo root (it is the artifact people link to); this
|
|
48
|
+
# copies it in at build time rather than keeping a second copy that drifts.
|
|
49
|
+
hooks:
|
|
50
|
+
- mkdocs_hooks.py
|
|
51
|
+
|
|
52
|
+
markdown_extensions:
|
|
53
|
+
- admonition
|
|
54
|
+
- pymdownx.details
|
|
55
|
+
- pymdownx.superfences
|
|
56
|
+
- toc: {permalink: true}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
"""Build hook: mirror root-level docs into the site without duplicating them.
|
|
2
|
+
|
|
3
|
+
SPEC.md is the artifact people cite and link to, so it stays at the repo root.
|
|
4
|
+
Copying it in at build time means the site renders the same file the repo
|
|
5
|
+
serves, rather than a second copy that quietly drifts out of date.
|
|
6
|
+
|
|
7
|
+
docs/spec.md is generated and gitignored — do not edit it.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
ROOT = Path(__file__).resolve().parent
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def on_pre_build(config, **kwargs) -> None:
|
|
18
|
+
spec = ROOT / "SPEC.md"
|
|
19
|
+
if not spec.exists(): # pragma: no cover - repo always ships it
|
|
20
|
+
raise FileNotFoundError(f"SPEC.md missing at {spec}; the docs nav expects it")
|
|
21
|
+
(ROOT / "docs" / "spec.md").write_text(spec.read_text())
|