offpeak 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. offpeak-0.2.1/.github/workflows/docs.yml +54 -0
  2. {offpeak-0.2.0 → offpeak-0.2.1}/.github/workflows/nightly.yml +6 -0
  3. {offpeak-0.2.0 → offpeak-0.2.1}/.gitignore +4 -0
  4. {offpeak-0.2.0 → offpeak-0.2.1}/PKG-INFO +22 -9
  5. {offpeak-0.2.0 → offpeak-0.2.1}/README.md +15 -8
  6. offpeak-0.2.1/docs/index.md +39 -0
  7. offpeak-0.2.1/docs/night-board.md +83 -0
  8. offpeak-0.2.1/docs/quickstart.md +136 -0
  9. offpeak-0.2.1/docs/reference.md +48 -0
  10. offpeak-0.2.1/docs/roadmap.md +54 -0
  11. offpeak-0.2.1/mkdocs.yml +56 -0
  12. offpeak-0.2.1/mkdocs_hooks.py +21 -0
  13. {offpeak-0.2.0 → offpeak-0.2.1}/pyproject.toml +3 -1
  14. {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/__init__.py +1 -1
  15. offpeak-0.2.1/src/offpeak/prices.py +251 -0
  16. {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/quote.py +83 -8
  17. offpeak-0.2.1/src/offpeak/venues/groq_batch.py +122 -0
  18. offpeak-0.2.1/tests/test_groq_venue.py +146 -0
  19. {offpeak-0.2.0 → offpeak-0.2.1}/tests/test_night_report.py +114 -1
  20. offpeak-0.2.1/tests/test_prices.py +155 -0
  21. {offpeak-0.2.0 → offpeak-0.2.1}/tests/test_quote.py +82 -6
  22. {offpeak-0.2.0 → offpeak-0.2.1}/tools/night_report.py +154 -14
  23. offpeak-0.2.0/src/offpeak/prices.py +0 -98
  24. {offpeak-0.2.0 → offpeak-0.2.1}/.github/workflows/ci.yml +0 -0
  25. {offpeak-0.2.0 → offpeak-0.2.1}/.github/workflows/publish.yml +0 -0
  26. {offpeak-0.2.0 → offpeak-0.2.1}/CONTRIBUTING.md +0 -0
  27. {offpeak-0.2.0 → offpeak-0.2.1}/LICENSE +0 -0
  28. {offpeak-0.2.0 → offpeak-0.2.1}/SPEC.md +0 -0
  29. {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/__main__.py +0 -0
  30. {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/client.py +0 -0
  31. {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/deadline.py +0 -0
  32. {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/job.py +0 -0
  33. {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/venues/__init__.py +0 -0
  34. {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/venues/anthropic_batch.py +0 -0
  35. {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/venues/base.py +0 -0
  36. {offpeak-0.2.0 → offpeak-0.2.1}/src/offpeak/venues/openai_batch.py +0 -0
  37. {offpeak-0.2.0 → offpeak-0.2.1}/tests/conftest.py +0 -0
  38. {offpeak-0.2.0 → offpeak-0.2.1}/tests/test_deadline.py +0 -0
  39. {offpeak-0.2.0 → offpeak-0.2.1}/tests/test_job_receipt.py +0 -0
  40. {offpeak-0.2.0 → offpeak-0.2.1}/tests/test_run.py +0 -0
@@ -0,0 +1,54 @@
1
+ name: Docs
2
+
3
+ # Builds the mkdocs site and deploys it to GitHub Pages. Pages is configured
4
+ # with build_type=workflow, so this workflow is the only publisher — there is
5
+ # no gh-pages branch to keep in sync.
6
+
7
+ on:
8
+ push:
9
+ branches: [main]
10
+ paths:
11
+ - "docs/**"
12
+ - "src/**"
13
+ - "mkdocs.yml"
14
+ - "mkdocs_hooks.py"
15
+ - "SPEC.md"
16
+ - "pyproject.toml"
17
+ - ".github/workflows/docs.yml"
18
+ workflow_dispatch:
19
+
20
+ permissions:
21
+ contents: read
22
+ pages: write
23
+ id-token: write
24
+
25
+ concurrency:
26
+ group: pages
27
+ cancel-in-progress: false
28
+
29
+ jobs:
30
+ build:
31
+ runs-on: ubuntu-latest
32
+ steps:
33
+ - uses: actions/checkout@v7
34
+ - uses: actions/setup-python@v7
35
+ with:
36
+ python-version: "3.12"
37
+ - name: Install
38
+ run: pip install -e ".[docs]"
39
+ - name: Build
40
+ run: mkdocs build --strict
41
+ - uses: actions/configure-pages@v6
42
+ - uses: actions/upload-pages-artifact@v5
43
+ with:
44
+ path: site
45
+
46
+ deploy:
47
+ needs: build
48
+ runs-on: ubuntu-latest
49
+ environment:
50
+ name: github-pages
51
+ url: ${{ steps.deployment.outputs.page_url }}
52
+ steps:
53
+ - id: deployment
54
+ uses: actions/deploy-pages@v5
@@ -54,10 +54,16 @@ jobs:
54
54
  echo "mode=$mode" >> "$GITHUB_OUTPUT"
55
55
  echo "running the $mode pass"
56
56
 
57
+ - name: Install US-zone data source
58
+ # gridstatus is an Action-only dependency: the SDK does not need it and
59
+ # the generator degrades to GB-only without it.
60
+ run: pip install "gridstatus>=0.36"
61
+
57
62
  - name: Generate
58
63
  run: |
59
64
  python tools/night_report.py \
60
65
  --mode "${{ steps.pick.outputs.mode }}" \
66
+ --us-zones \
61
67
  --outdir board/nightly
62
68
 
63
69
  - name: Commit to board-data
@@ -23,3 +23,7 @@ htmlcov/
23
23
  # Claude Code (personal)
24
24
  .claude/settings.local.json
25
25
  HANDOFF.md
26
+
27
+ # Docs build output; docs/spec.md is copied from SPEC.md at build time
28
+ site/
29
+ docs/spec.md
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: offpeak
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Deadline-priced inference: give AI jobs a deadline and run them on the cheapest venue — provider batch tiers (−50%) today. Same model, same tokens, a different hour.
5
5
  Project-URL: Homepage, https://github.com/offpeak-ai/offpeak
6
6
  Project-URL: Repository, https://github.com/offpeak-ai/offpeak
@@ -31,6 +31,12 @@ Requires-Dist: build; extra == 'dev'
31
31
  Requires-Dist: pytest>=8; extra == 'dev'
32
32
  Requires-Dist: ruff>=0.6; extra == 'dev'
33
33
  Requires-Dist: twine; extra == 'dev'
34
+ Provides-Extra: docs
35
+ Requires-Dist: mkdocs-material<10,>=9.5; extra == 'docs'
36
+ Requires-Dist: mkdocs<2,>=1.6; extra == 'docs'
37
+ Requires-Dist: mkdocstrings[python]>=0.26; extra == 'docs'
38
+ Provides-Extra: groq
39
+ Requires-Dist: groq>=0.11; extra == 'groq'
34
40
  Provides-Extra: openai
35
41
  Requires-Dist: openai>=1.50; extra == 'openai'
36
42
  Description-Content-Type: text/markdown
@@ -66,10 +72,12 @@ tokens 12,410,332 in · 3,104,551 out
66
72
  list $27.93
67
73
  paid $14.02
68
74
  captured $13.91 (49.8%)
69
- prices snapshot 2026-08 — override via offpeak.prices
75
+ prices snapshot 2026-08-21 — override via offpeak.prices
70
76
  ───────────────────────────────────────────────
71
77
  ```
72
78
 
79
+ **[Documentation](https://offpeak-ai.github.io/offpeak/)** · [Quickstart](https://offpeak-ai.github.io/offpeak/quickstart/) · [Spec](https://offpeak-ai.github.io/offpeak/spec/) · [Roadmap](https://offpeak-ai.github.io/offpeak/roadmap/)
80
+
73
81
  ## What it does
74
82
 
75
83
  - **Know the price before you spend it.** `quote(jobs, deadline=...)` prices a run against the published sheets with no API calls and no key — list versus batch, per venue, plus what the wait is worth.
@@ -103,18 +111,21 @@ jobs 5000 across 1 venue(s)
103
111
  deadline 2026-08-21 21:11 PDT (24.0h out)
104
112
  tokens 4,000,000 in · 1,000,000 out
105
113
 
106
- openai:batch 5000 job(s) list $1.00 batch $0.50 save $0.50 (50.0%)
114
+ openai:batch 5000 job(s) list $2.00 batch $1.00 save $1.00 (50.0%)
107
115
 
108
- list $1.00 (run now, synchronously)
109
- batch $0.50 (run by the deadline)
110
- save $0.50 (50.0%)
116
+ list $2.00 (run now, synchronously)
117
+ batch $1.00 (run by the deadline)
118
+ save $1.00 (50.0%)
119
+ risk deadline is inside the 24h batch window — the SLA rests on the sync fallback, which pays list
111
120
  basis input explicit; output explicit
112
- prices snapshot 2026-08 — estimate only, not a bill
121
+ prices snapshot 2026-08-21 — estimate only, not a bill
113
122
  ───────────────────────────────────────────────
114
123
  ```
115
124
 
116
125
  From Python, `offpeak.quote(jobs, deadline="06:00")` takes the same jobs you would pass to `run()`. Token counts come from the job where it knows them (`metadata={"input_tokens": ..., "output_tokens": ...}`, or `max_tokens` as an output ceiling) and are a labeled chars/4 estimate where it does not — every figure reports its provenance in `basis`, and a quote with no output signal is marked a **floor**, not an estimate.
117
126
 
127
+ If you do know roughly what the model will write, say so and get a priced number instead — `quote(jobs, deadline=..., assumed_output_ratio=0.25)`, or `metadata={"expected_output_tokens": 300}` on a single job. Those quotes are marked **EST**, distinct from a floor. Both are opt-in: absent one, `offpeak` assumes nothing on your behalf.
128
+
118
129
  ## Deadlines
119
130
 
120
131
  Deadlines are how software says "this can wait" — the full semantics live in [SPEC.md](SPEC.md).
@@ -152,13 +163,15 @@ import offpeak
152
163
  offpeak.prices.register_price("my-fine-tune", input_per_m=4.0, output_per_m=16.0)
153
164
  ```
154
165
 
155
- Unknown models settle with `cost = None` rather than a guess.
166
+ Unknown models settle with `cost = None` rather than a guess. The sheet also carries what the venues charge for *urgency* — `get_fast_price()`, `urgency_spread()` — and flags list prices that are promotional, with the date and the price they decay to: `promo_decay("gpt-5.6-sol")` is `(1.25, 1.5)` after 2026-11-21.
156
167
 
157
168
  ## What this is (and the roadmap)
158
169
 
159
170
  `offpeak` is the open client and spec for a simple claim: **intelligence has a time value**. A large share of AI work — embeddings, evals, backfills, report generation, overnight agents — has no human waiting on it, and the venues already price that patience at −50%. This library is the missing workflow.
160
171
 
161
- The **[night board](https://github.com/offpeak-ai/offpeak/blob/board-data/nightly/BOARD.md)** marks the same claim against open grid data every night — power and carbon peak/off-peak spreads, alongside the 2.0x token spread the batch tiers already publish.
172
+ The token side is wider than the headline discount. Patience is priced at −50% — a 2.0x spread — and haste is priced too: hold the model and venue constant and `gpt-5.6-sol` costs **$8.00 / $40.00** per 1M on OpenAI's fast tier against **$2.00 / $10.00** on its batch tier, a **4x intra-venue urgency spread** for the hour alone ([source](https://developers.openai.com/api/docs/pricing); sol's standard rate is promotional at least through 2026-11-21, and both tiers are defined off it, so the ratio outlives the prices). It is data, not prose: `offpeak.prices.urgency_spread("gpt-5.6-sol")` returns `4.0`.
173
+
174
+ The **[night board](https://github.com/offpeak-ai/offpeak/blob/board-data/nightly/BOARD.md)** marks the same claim against open grid data every night — GB power and carbon plus CAISO SP15 and ERCOT Houston peak/off-peak spreads, alongside the published token spreads. ERCOT Houston marked 3.94x on the night of 2026-08-20; a venue charges 4x for the same impatience.
162
175
 
163
176
  The roadmap follows the same interface upward: more venues (Google batch, spot capacity, off-peak windows on your own GPUs), queue-latency forecasting instead of a fixed risk buffer, portfolio placement across venues, energy- and carbon-aware scheduling with per-job receipts. The venue interface (`offpeak.Venue`) is deliberately the extension point — a venue is anywhere deferred work can run.
164
177
 
@@ -29,10 +29,12 @@ tokens 12,410,332 in · 3,104,551 out
29
29
  list $27.93
30
30
  paid $14.02
31
31
  captured $13.91 (49.8%)
32
- prices snapshot 2026-08 — override via offpeak.prices
32
+ prices snapshot 2026-08-21 — override via offpeak.prices
33
33
  ───────────────────────────────────────────────
34
34
  ```
35
35
 
36
+ **[Documentation](https://offpeak-ai.github.io/offpeak/)** · [Quickstart](https://offpeak-ai.github.io/offpeak/quickstart/) · [Spec](https://offpeak-ai.github.io/offpeak/spec/) · [Roadmap](https://offpeak-ai.github.io/offpeak/roadmap/)
37
+
36
38
  ## What it does
37
39
 
38
40
  - **Know the price before you spend it.** `quote(jobs, deadline=...)` prices a run against the published sheets with no API calls and no key — list versus batch, per venue, plus what the wait is worth.
@@ -66,18 +68,21 @@ jobs 5000 across 1 venue(s)
66
68
  deadline 2026-08-21 21:11 PDT (24.0h out)
67
69
  tokens 4,000,000 in · 1,000,000 out
68
70
 
69
- openai:batch 5000 job(s) list $1.00 batch $0.50 save $0.50 (50.0%)
71
+ openai:batch 5000 job(s) list $2.00 batch $1.00 save $1.00 (50.0%)
70
72
 
71
- list $1.00 (run now, synchronously)
72
- batch $0.50 (run by the deadline)
73
- save $0.50 (50.0%)
73
+ list $2.00 (run now, synchronously)
74
+ batch $1.00 (run by the deadline)
75
+ save $1.00 (50.0%)
76
+ risk deadline is inside the 24h batch window — the SLA rests on the sync fallback, which pays list
74
77
  basis input explicit; output explicit
75
- prices snapshot 2026-08 — estimate only, not a bill
78
+ prices snapshot 2026-08-21 — estimate only, not a bill
76
79
  ───────────────────────────────────────────────
77
80
  ```
78
81
 
79
82
  From Python, `offpeak.quote(jobs, deadline="06:00")` takes the same jobs you would pass to `run()`. Token counts come from the job where it knows them (`metadata={"input_tokens": ..., "output_tokens": ...}`, or `max_tokens` as an output ceiling) and are a labeled chars/4 estimate where it does not — every figure reports its provenance in `basis`, and a quote with no output signal is marked a **floor**, not an estimate.
80
83
 
84
+ If you do know roughly what the model will write, say so and get a priced number instead — `quote(jobs, deadline=..., assumed_output_ratio=0.25)`, or `metadata={"expected_output_tokens": 300}` on a single job. Those quotes are marked **EST**, distinct from a floor. Both are opt-in: absent one, `offpeak` assumes nothing on your behalf.
85
+
81
86
  ## Deadlines
82
87
 
83
88
  Deadlines are how software says "this can wait" — the full semantics live in [SPEC.md](SPEC.md).
@@ -115,13 +120,15 @@ import offpeak
115
120
  offpeak.prices.register_price("my-fine-tune", input_per_m=4.0, output_per_m=16.0)
116
121
  ```
117
122
 
118
- Unknown models settle with `cost = None` rather than a guess.
123
+ Unknown models settle with `cost = None` rather than a guess. The sheet also carries what the venues charge for *urgency* — `get_fast_price()`, `urgency_spread()` — and flags list prices that are promotional, with the date and the price they decay to: `promo_decay("gpt-5.6-sol")` is `(1.25, 1.5)` after 2026-11-21.
119
124
 
120
125
  ## What this is (and the roadmap)
121
126
 
122
127
  `offpeak` is the open client and spec for a simple claim: **intelligence has a time value**. A large share of AI work — embeddings, evals, backfills, report generation, overnight agents — has no human waiting on it, and the venues already price that patience at −50%. This library is the missing workflow.
123
128
 
124
- The **[night board](https://github.com/offpeak-ai/offpeak/blob/board-data/nightly/BOARD.md)** marks the same claim against open grid data every night — power and carbon peak/off-peak spreads, alongside the 2.0x token spread the batch tiers already publish.
129
+ The token side is wider than the headline discount. Patience is priced at −50% — a 2.0x spread — and haste is priced too: hold the model and venue constant and `gpt-5.6-sol` costs **$8.00 / $40.00** per 1M on OpenAI's fast tier against **$2.00 / $10.00** on its batch tier, a **4x intra-venue urgency spread** for the hour alone ([source](https://developers.openai.com/api/docs/pricing); sol's standard rate is promotional at least through 2026-11-21, and both tiers are defined off it, so the ratio outlives the prices). It is data, not prose: `offpeak.prices.urgency_spread("gpt-5.6-sol")` returns `4.0`.
130
+
131
+ The **[night board](https://github.com/offpeak-ai/offpeak/blob/board-data/nightly/BOARD.md)** marks the same claim against open grid data every night — GB power and carbon plus CAISO SP15 and ERCOT Houston peak/off-peak spreads, alongside the published token spreads. ERCOT Houston marked 3.94x on the night of 2026-08-20; a venue charges 4x for the same impatience.
125
132
 
126
133
  The roadmap follows the same interface upward: more venues (Google batch, spot capacity, off-peak windows on your own GPUs), queue-latency forecasting instead of a fixed risk buffer, portfolio placement across venues, energy- and carbon-aware scheduling with per-job receipts. The venue interface (`offpeak.Venue`) is deliberately the extension point — a venue is anywhere deferred work can run.
127
134
 
@@ -0,0 +1,39 @@
1
+ # offpeak
2
+
3
+ **Deadline-priced inference.** Same model, same tokens, a different hour.
4
+
5
+ A large share of AI work — embeddings, evals, backfills, report generation,
6
+ overnight agents — has no human waiting on it. The providers already price that
7
+ patience: OpenAI and Anthropic both publish their batch tiers at **50% of
8
+ list**. `offpeak` is the workflow that collects the difference.
9
+
10
+ ```python
11
+ import offpeak
12
+
13
+ jobs = [offpeak.job("claude-haiku-4-5", f"Summarize:\n\n{d}") for d in docs]
14
+
15
+ print(offpeak.quote(jobs, deadline="06:00")) # what is the wait worth?
16
+ results = offpeak.run(jobs, deadline="06:00") # collect it
17
+ print(offpeak.receipt(results)) # what it actually cost
18
+ ```
19
+
20
+ - **[Quickstart](quickstart.md)** — install, quote, run, read the receipt.
21
+ - **[The night board](night-board.md)** — the same claim, marked nightly against open grid data.
22
+ - **[Spec](spec.md)** — deadline semantics, statuses, receipts.
23
+ - **[API reference](reference.md)** — every public symbol.
24
+ - **[Roadmap](roadmap.md)** — what exists, what does not, and what is being built.
25
+
26
+ ## What it guarantees
27
+
28
+ **One `Result` per job, always.** Provider failures at submit, poll, cancel or
29
+ sync are captured, not raised. Affected jobs take the sync fallback where the
30
+ deadline still allows it, and otherwise return failed with the provider's
31
+ message attached. Exceptions are reserved for programming errors — a deadline
32
+ in the past, or a model no venue supports.
33
+
34
+ **Your keys, your perimeter.** `offpeak` talks straight to the providers with
35
+ your own credentials. There is no proxy and no third party in the data path.
36
+
37
+ **Receipts are arithmetic, not estimates.** Every figure traces to a published
38
+ price sheet, and a model that is not on one settles as `None` rather than a
39
+ guess.
@@ -0,0 +1,83 @@
1
+ # The night board
2
+
3
+ `offpeak` rests on a claim: **intelligence has a time value.** The token side of
4
+ that claim is already settled, and it is wider than the headline discount.
5
+
6
+ - **Patience is priced at −50%.** OpenAI and Anthropic both publish batch tiers
7
+ at half of list: a flat **2.0x** spread for work that can wait.
8
+ - **Haste is priced too.** Hold the model and the venue constant and read the
9
+ same sheet across its urgency tiers: `gpt-5.6-sol` is **$8.00 / $40.00** per
10
+ 1M tokens on OpenAI's fast tier against **$2.00 / $10.00** on its batch tier
11
+ — a **4x intra-venue urgency spread**, on both legs, for the hour alone.
12
+ Source: [developers.openai.com/api/docs/pricing](https://developers.openai.com/api/docs/pricing).
13
+
14
+ !!! note "The promo caveat"
15
+ `gpt-5.6-sol`'s standard rate is promotional — the sheet says it runs *at
16
+ least* through **2026-11-21**, after which list is $5/$30. Fast and batch
17
+ are both defined off that list (2x and 0.5x), so the promo moves the
18
+ dollars and leaves the ratio: **the 4x is the durable figure, the prices
19
+ are the perishable ones.** Both are in the SDK rather than in prose —
20
+ `offpeak.prices.urgency_spread("gpt-5.6-sol")` returns `4.0`, and
21
+ `promo_decay()` returns the step-up the date will bring.
22
+
23
+ The night board marks the same claim against the other side of the trade: the
24
+ grid the compute runs on. Every night it records what power and carbon actually
25
+ did between the evening peak and the small hours, from open, keyless sources.
26
+ The grid's spread is not published anywhere — it has to be observed — and on the
27
+ night of 2026-08-20 the ERCOT Houston hub marked **3.94x** between its evening
28
+ peak and the trough that followed, against the 4x a venue charges for the same
29
+ hour of impatience.
30
+
31
+ **[→ Read the board](https://github.com/offpeak-ai/offpeak/blob/board-data/nightly/BOARD.md)**
32
+
33
+ ## How it works
34
+
35
+ Two passes over the same night, run by
36
+ [a scheduled workflow](https://github.com/offpeak-ai/offpeak/blob/main/.github/workflows/nightly.yml):
37
+
38
+ | Pass | When | What it records |
39
+ |---|---|---|
40
+ | `quote` | 19:00Z | The night ahead — carbon **forecast**, day-ahead power |
41
+ | `mark` | 06:30Z | The night just finished — carbon **actuals** |
42
+
43
+ A night runs **16:00Z–07:00Z**: it opens with the 17:00 BST evening peak and
44
+ closes after the 00–05 BST trough, so one span carries both windows the board
45
+ compares.
46
+
47
+ Output lands on the
48
+ [`board-data` branch](https://github.com/offpeak-ai/offpeak/tree/board-data) —
49
+ `nightly/BOARD.md` plus the raw JSON per night — because `main` is protected and
50
+ would reject a nightly bot push.
51
+
52
+ ## Sources
53
+
54
+ - **Carbon** — [NESO carbon intensity](https://api.carbonintensity.org.uk),
55
+ GB, keyless.
56
+ - **Power, GB** — [Octopus Agile](https://api.octopus.energy) day-ahead unit
57
+ rates, GB region C, keyless.
58
+ - **Power, US** — CAISO SP15 and ERCOT Houston day-ahead hourly, via
59
+ [gridstatus](https://github.com/gridstatus/gridstatus), keyless.
60
+ - **Tokens** — the published price sheets, not a measurement:
61
+ [OpenAI](https://developers.openai.com/api/docs/pricing) and
62
+ [Anthropic](https://platform.claude.com/docs/en/about-claude/pricing).
63
+
64
+ All are public and free. The board costs nothing to run and spends nothing at
65
+ any venue.
66
+
67
+ ## Honest limits
68
+
69
+ - **Four zones, two of them thin.** GB carbon and GB power are half-hourly and
70
+ complete. CAISO SP15 and ERCOT Houston are **day-ahead hourly** prices, not
71
+ settled real-time ones, and they are power only — no US carbon leg yet.
72
+ - **The token column is published, not observed.** The 2.0x and the 4x are read
73
+ off price sheets; only the grid columns are measurements. A published number
74
+ and a marked one are different kinds of claim, and the board should not blur
75
+ them.
76
+ - **Observation, not advice.** The board records what the grid did. It does not
77
+ forecast, and `offpeak` does not currently schedule against it — the SDK
78
+ routes on published token prices alone.
79
+ - **Carbon actuals lag** roughly two hours. The 06:30Z mark clears the 00–05 BST
80
+ trough comfortably; the tail of the night can still be sparse.
81
+ - **A dead source costs its column, not the run.** Legs degrade independently,
82
+ and a night where both sources are down is recorded as unavailable rather than
83
+ guessed at.
@@ -0,0 +1,136 @@
1
+ # Quickstart
2
+
3
+ ## Install
4
+
5
+ ```bash
6
+ pip install "offpeak[all]" # OpenAI + Anthropic venues
7
+ pip install "offpeak[anthropic]" # or just one
8
+ pip install "offpeak[openai]"
9
+ ```
10
+
11
+ The core has zero dependencies; provider SDKs load only through the extras.
12
+ Venues read the standard environment variables (`OPENAI_API_KEY`,
13
+ `ANTHROPIC_API_KEY`), or take a configured client:
14
+ `OpenAIBatch(client=my_client)`.
15
+
16
+ ## 1. Quote — before you spend anything
17
+
18
+ `quote()` makes **no API calls and needs no key**. It prices your jobs against
19
+ the bundled sheet: list versus batch, per venue.
20
+
21
+ ```bash
22
+ python -m offpeak quote --model gpt-5.6-luna --input-tokens 800 --output-tokens 200 --jobs 5000
23
+ ```
24
+
25
+ ```
26
+ OFFPEAK QUOTE ─────────────────────────────────
27
+ jobs 5000 across 1 venue(s)
28
+ deadline 2026-08-21 21:11 PDT (24.0h out)
29
+ tokens 4,000,000 in · 1,000,000 out
30
+
31
+ openai:batch 5000 job(s) list $2.00 batch $1.00 save $1.00 (50.0%)
32
+
33
+ list $2.00 (run now, synchronously)
34
+ batch $1.00 (run by the deadline)
35
+ save $1.00 (50.0%)
36
+ risk deadline is inside the 24h batch window — the SLA rests on the sync fallback, which pays list
37
+ basis input explicit; output explicit
38
+ prices snapshot 2026-08-21 — estimate only, not a bill
39
+ ───────────────────────────────────────────────
40
+ ```
41
+
42
+ From Python, the same jobs you would pass to `run()`:
43
+
44
+ ```python
45
+ q = offpeak.quote(jobs, deadline="06:00")
46
+ print(q.spread_usd, q.spread_pct)
47
+ ```
48
+
49
+ !!! warning "Quotes that omit output are marked a floor"
50
+ Output costs more than input on every model on the sheet. If a job carries
51
+ no output-token signal, `quote()` prices its output at **zero** and labels
52
+ the whole quote a `FLOOR` — a stated floor is safer than an invented
53
+ number. Give it `max_tokens`, or explicit counts:
54
+
55
+ ```python
56
+ offpeak.job("claude-haiku-4-5", prompt, max_tokens=512)
57
+ offpeak.Job(model=..., messages=[...],
58
+ metadata={"input_tokens": 800, "output_tokens": 200})
59
+ ```
60
+
61
+ `Quote.basis` reports the provenance of every figure.
62
+
63
+ ### If you know roughly what it will write
64
+
65
+ A floor is honest but not always useful. When you do have a sense of the output
66
+ size, say so — and the quote prices it, marked `EST` rather than `FLOOR`:
67
+
68
+ ```python
69
+ # Across the run: assume each job writes a quarter of what it reads.
70
+ offpeak.quote(jobs, deadline="06:00", assumed_output_ratio=0.25)
71
+
72
+ # Or per job, which wins over a ratio and over max_tokens:
73
+ offpeak.Job(model=..., messages=[...],
74
+ metadata={"expected_output_tokens": 300})
75
+ ```
76
+
77
+ ```
78
+ EST 5000 job(s) priced on an assumed output size, not a measured one
79
+ the assumption is yours; the bill moves with what the model actually writes
80
+ ```
81
+
82
+ Both are opt-in. Without one, nothing is assumed on your behalf: the default
83
+ stays the floor. The two marks mean different things and a quote can carry both
84
+ — `FLOOR` is understated by construction, `EST` can land either side of the
85
+ bill. `Quote.is_floor` and `Quote.is_estimated` are the same distinction in
86
+ code, and a ratio applies only to jobs with no signal of their own, so explicit
87
+ counts and `max_tokens` are never overridden by it.
88
+
89
+ ## 2. Run — against a deadline
90
+
91
+ ```python
92
+ results = offpeak.run(jobs, deadline="06:00")
93
+ ```
94
+
95
+ Each job goes to the batch tier of a venue that supports its model. `offpeak`
96
+ polls until the work lands. If the batch has not completed by the time the
97
+ remaining window shrinks to the risk buffer, it cancels and re-runs the
98
+ stragglers synchronously at list price — you stated a deadline, and it is met.
99
+
100
+ Deadlines accept `"06:00"` (next occurrence), `"6h"`, `"90m"`, a `datetime`, a
101
+ `timedelta`, seconds, or an ISO 8601 string.
102
+
103
+ ## 3. Read the receipt
104
+
105
+ ```python
106
+ print(offpeak.receipt(results))
107
+ ```
108
+
109
+ ```
110
+ OFFPEAK SETTLEMENT ────────────────────────────
111
+ jobs 5000 (5000 ok, 120 sync fallback, 0 failed)
112
+ sla 5000/5000 met
113
+ venues anthropic:batch 3000 · openai:batch 2000
114
+ tokens 41,000,000 in · 3,200,000 out
115
+ list $2,469.00
116
+ paid $1,234.50
117
+ captured $1,234.50 (50.0%)
118
+ left $29.63 on the table (120 job(s) missed the batch tier)
119
+ prices snapshot 2026-08-21 — override via offpeak.prices
120
+ ───────────────────────────────────────────────
121
+ ```
122
+
123
+ `left on the table` is what the sync fallback gave up by missing the batch
124
+ tier. Per-job receipts render the same way — `print(results[0].receipt)` — with
125
+ sub-cent precision, so a small run reports what it cost rather than `$0.00`.
126
+
127
+ ## Prices
128
+
129
+ The bundled sheet is a dated snapshot. Providers move prices; override at
130
+ runtime rather than waiting for a release:
131
+
132
+ ```python
133
+ offpeak.prices.register_price("my-fine-tune", input_per_m=4.0, output_per_m=16.0)
134
+ ```
135
+
136
+ Unknown models settle as `None`, never a guess.
@@ -0,0 +1,48 @@
1
+ # API reference
2
+
3
+ Everything below is exported from the package root: `offpeak.run`,
4
+ `offpeak.Job`, and so on.
5
+
6
+ ## Running work
7
+
8
+ ::: offpeak.run
9
+
10
+ ::: offpeak.quote
11
+
12
+ ::: offpeak.receipt
13
+
14
+ ::: offpeak.job
15
+
16
+ ## Types
17
+
18
+ ::: offpeak.Job
19
+
20
+ ::: offpeak.Result
21
+
22
+ ::: offpeak.Receipt
23
+
24
+ ::: offpeak.Status
25
+
26
+ ::: offpeak.Settlement
27
+
28
+ ::: offpeak.Quote
29
+
30
+ ::: offpeak.VenueQuote
31
+
32
+ ## Deadlines
33
+
34
+ ::: offpeak.parse_deadline
35
+
36
+ ::: offpeak.seconds_until
37
+
38
+ ## Venues
39
+
40
+ ::: offpeak.Venue
41
+
42
+ ::: offpeak.BatchState
43
+
44
+ ::: offpeak.default_venues
45
+
46
+ ## Prices
47
+
48
+ ::: offpeak.prices
@@ -0,0 +1,54 @@
1
+ # Roadmap
2
+
3
+ What exists, what does not, and what is being built. This page is meant to be
4
+ checkable — if something here is not true yet, it says so.
5
+
6
+ ## What exists today
7
+
8
+ - **`run(jobs, deadline=...)`** across the OpenAI and Anthropic batch tiers,
9
+ with a sync fallback that protects the deadline.
10
+ - **`quote(jobs, deadline=...)`** — pre-trade pricing with no API calls.
11
+ - **Receipts and settlements** — list, paid, captured spread, and what a
12
+ fallback left on the table, as arithmetic against published price sheets.
13
+ - **A `Venue` interface** — the extension point. A venue is anywhere deferred
14
+ work can run.
15
+ - **[The night board](night-board.md)** — GB power and carbon, marked nightly.
16
+
17
+ ## What does not exist yet
18
+
19
+ Stated plainly, because a roadmap that reads like a feature list is a
20
+ misleading one:
21
+
22
+ - **No queue-latency forecasting.** Deadline risk is a fixed buffer, not a
23
+ prediction. `offpeak` does not know how long a given venue's queue is; it
24
+ watches the clock and falls back.
25
+ - **No cross-venue portfolio placement.** Jobs route to the first venue that
26
+ supports the model, not to the cheapest or fastest across a portfolio.
27
+ - **No carbon-aware scheduling.** The night board observes the grid; the
28
+ scheduler does not read it.
29
+ - **No venues beyond the two batch tiers.** Google batch, spot capacity, and
30
+ off-peak windows on your own GPUs are interface-shaped but unwritten. A
31
+ Groq batch driver exists in the tree but is **untested against the live
32
+ API** — it is opt-in, excluded from the `all` extra, and not in
33
+ `default_venues()`.
34
+
35
+ ## The hosted desk
36
+
37
+ A hosted desk that does the forecasting, cross-venue portfolio scheduling, and
38
+ SLA insurance at fleet scale — with payloads never leaving your perimeter — is
39
+ being built by the same team.
40
+
41
+ The intended seam is the one already in the library: a desk would be selected
42
+ per run, alongside the venues you already pass, so that moving from local
43
+ scheduling to hosted scheduling is a keyword argument rather than a rewrite.
44
+
45
+ !!! note "Not implemented"
46
+ That parameter does not exist in the public API today, and nothing in this
47
+ release accepts it. It is described here so the shape of the plan is
48
+ legible — not as something you can call. The SDK and the deadline spec stay
49
+ open, Apache-2.0, either way.
50
+
51
+ ## The spec
52
+
53
+ Deadline semantics are versioned separately in [SPEC.md](spec.md), so a second
54
+ implementation can be written against them. Spec changes start as issues.
@@ -0,0 +1,56 @@
1
+ site_name: offpeak
2
+ site_description: Deadline-priced inference — same model, same tokens, a different hour.
3
+ site_url: https://offpeak-ai.github.io/offpeak/
4
+ repo_url: https://github.com/offpeak-ai/offpeak
5
+ repo_name: offpeak-ai/offpeak
6
+ edit_uri: edit/main/docs/
7
+
8
+ theme:
9
+ name: material
10
+ palette:
11
+ - media: "(prefers-color-scheme: light)"
12
+ scheme: default
13
+ primary: black
14
+ accent: indigo
15
+ toggle: {icon: material/weather-night, name: Switch to dark mode}
16
+ - media: "(prefers-color-scheme: dark)"
17
+ scheme: slate
18
+ primary: black
19
+ accent: indigo
20
+ toggle: {icon: material/weather-sunny, name: Switch to light mode}
21
+ features:
22
+ - navigation.sections
23
+ - navigation.top
24
+ - content.code.copy
25
+
26
+ nav:
27
+ - Home: index.md
28
+ - Quickstart: quickstart.md
29
+ - The night board: night-board.md
30
+ - Spec: spec.md
31
+ - API reference: reference.md
32
+ - Roadmap: roadmap.md
33
+
34
+ plugins:
35
+ - search
36
+ - mkdocstrings:
37
+ handlers:
38
+ python:
39
+ paths: [src]
40
+ options:
41
+ show_source: true
42
+ show_root_heading: true
43
+ heading_level: 3
44
+ docstring_style: sphinx
45
+ members_order: source
46
+
47
+ # SPEC.md lives at the repo root (it is the artifact people link to); this
48
+ # copies it in at build time rather than keeping a second copy that drifts.
49
+ hooks:
50
+ - mkdocs_hooks.py
51
+
52
+ markdown_extensions:
53
+ - admonition
54
+ - pymdownx.details
55
+ - pymdownx.superfences
56
+ - toc: {permalink: true}
@@ -0,0 +1,21 @@
1
+ """Build hook: mirror root-level docs into the site without duplicating them.
2
+
3
+ SPEC.md is the artifact people cite and link to, so it stays at the repo root.
4
+ Copying it in at build time means the site renders the same file the repo
5
+ serves, rather than a second copy that quietly drifts out of date.
6
+
7
+ docs/spec.md is generated and gitignored — do not edit it.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ from pathlib import Path
13
+
14
+ ROOT = Path(__file__).resolve().parent
15
+
16
+
17
+ def on_pre_build(config, **kwargs) -> None:
18
+ spec = ROOT / "SPEC.md"
19
+ if not spec.exists(): # pragma: no cover - repo always ships it
20
+ raise FileNotFoundError(f"SPEC.md missing at {spec}; the docs nav expects it")
21
+ (ROOT / "docs" / "spec.md").write_text(spec.read_text())