tpu-runner 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tpu_runner-0.1.0/LICENSE +21 -0
- tpu_runner-0.1.0/PKG-INFO +140 -0
- tpu_runner-0.1.0/README.md +116 -0
- tpu_runner-0.1.0/pyproject.toml +39 -0
- tpu_runner-0.1.0/setup.cfg +4 -0
- tpu_runner-0.1.0/tpu_runner/Dockerfile +10 -0
- tpu_runner-0.1.0/tpu_runner/__init__.py +18 -0
- tpu_runner-0.1.0/tpu_runner/__main__.py +4 -0
- tpu_runner-0.1.0/tpu_runner/capacity_policy.py +248 -0
- tpu_runner-0.1.0/tpu_runner/cli.py +1344 -0
- tpu_runner-0.1.0/tpu_runner/cloudbuild.yaml +11 -0
- tpu_runner-0.1.0/tpu_runner/controller-requirements.txt +2 -0
- tpu_runner-0.1.0/tpu_runner/controller.py +1680 -0
- tpu_runner-0.1.0/tpu_runner/deploy.sh +263 -0
- tpu_runner-0.1.0/tpu_runner/deployment.example.yaml +57 -0
- tpu_runner-0.1.0/tpu_runner/distributed.py +1017 -0
- tpu_runner-0.1.0/tpu_runner/gcp.py +454 -0
- tpu_runner-0.1.0/tpu_runner/placement.py +137 -0
- tpu_runner-0.1.0/tpu_runner/region_pools.py +24 -0
- tpu_runner-0.1.0/tpu_runner/runtime.py +994 -0
- tpu_runner-0.1.0/tpu_runner/specs.py +527 -0
- tpu_runner-0.1.0/tpu_runner/startup.sh +49 -0
- tpu_runner-0.1.0/tpu_runner.egg-info/PKG-INFO +140 -0
- tpu_runner-0.1.0/tpu_runner.egg-info/SOURCES.txt +26 -0
- tpu_runner-0.1.0/tpu_runner.egg-info/dependency_links.txt +1 -0
- tpu_runner-0.1.0/tpu_runner.egg-info/entry_points.txt +2 -0
- tpu_runner-0.1.0/tpu_runner.egg-info/requires.txt +6 -0
- tpu_runner-0.1.0/tpu_runner.egg-info/top_level.txt +1 -0
tpu_runner-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 David Hidary
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tpu-runner
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Minimal Google Cloud TPU job runner
|
|
5
|
+
License-Expression: MIT
|
|
6
|
+
Keywords: gcp,google-cloud,tpu,orchestration
|
|
7
|
+
Classifier: Environment :: Console
|
|
8
|
+
Classifier: Intended Audience :: Developers
|
|
9
|
+
Classifier: Operating System :: OS Independent
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Classifier: Topic :: System :: Distributed Computing
|
|
15
|
+
Requires-Python: >=3.11
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Requires-Dist: google-cloud-firestore>=2.16
|
|
19
|
+
Requires-Dist: PyYAML>=6.0
|
|
20
|
+
Provides-Extra: release
|
|
21
|
+
Requires-Dist: build<2,>=1; extra == "release"
|
|
22
|
+
Requires-Dist: twine<7,>=6; extra == "release"
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# TPU Runner
|
|
26
|
+
|
|
27
|
+
`tpu-runner` is a job orchestrator designed to **maximize utilization of your Google Cloud TPU allocation.**
|
|
28
|
+
|
|
29
|
+
It automatically **races capacity** requests across compatible TPU types and regions, **assigns jobs** by priority + FIFO, and **retries jobs** interrupted by Spot preemptions. It also **minimizes inter-region transfer costs**, cleans the workspace for new jobs, and lets related jobs reuse local caches (e.g. for previously compiled XLA artifacts or software environments).
|
|
30
|
+
|
|
31
|
+
We use Firestore for queue and orchestration state, GCS for source bundles and job artifacts, and a Cloud Run controller to manage TPU capacity and execution.
|
|
32
|
+
|
|
33
|
+
## Install
|
|
34
|
+
|
|
35
|
+
Install `tpu-runner` as a standalone CLI:
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
uv tool install tpu-runner
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
For local development, run this from the repository root:
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
uv tool install --editable .
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
Configure `gcloud`:
|
|
48
|
+
|
|
49
|
+
```bash
|
|
50
|
+
gcloud auth login
|
|
51
|
+
gcloud auth application-default login
|
|
52
|
+
gcloud config set project YOUR_PROJECT_ID
|
|
53
|
+
gcloud alpha compute tpus --help >/dev/null
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
## Set up the runner
|
|
57
|
+
|
|
58
|
+
Create an example deployment:
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
tpu-runner init
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Edit `deployment.yaml` with your project ID, existing Secret Manager secret names, and the TPU types, zones, maximum counts, runtime versions, and chip limits you are willing to use. Counts are provisioning ceilings; TPU Runner scales capacity up and down with demand and keeps idle capacity only when `keep_warm` is enabled.
|
|
65
|
+
|
|
66
|
+
The default `ssh_transport: direct` creates public-IP TPUs; set it to `iap` to create private-IP TPUs and reach them through IAP tunnels instead.
|
|
67
|
+
|
|
68
|
+
Validate and deploy:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
tpu-runner validate-fleet deployment.yaml
|
|
72
|
+
tpu-runner deploy deployment.yaml
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
`deploy` enables the required APIs and creates or updates the runner bucket, Firestore database, service accounts and IAM, SSH identity, worker startup script, controller image, and Cloud Run controller job.
|
|
76
|
+
|
|
77
|
+
## Submit work
|
|
78
|
+
|
|
79
|
+
In the root of the code you want to run, create `job.yaml`:
|
|
80
|
+
|
|
81
|
+
```yaml
|
|
82
|
+
jobs:
|
|
83
|
+
- tpu: [v4-64, v6e-64]
|
|
84
|
+
buckets:
|
|
85
|
+
- gs://my-training-us-central2
|
|
86
|
+
- gs://my-training-us-east1
|
|
87
|
+
bundle: .
|
|
88
|
+
priority: high
|
|
89
|
+
caches:
|
|
90
|
+
- key: pip
|
|
91
|
+
path: .cache/pip
|
|
92
|
+
env:
|
|
93
|
+
PIP_CACHE_DIR: .cache/pip
|
|
94
|
+
WANDB_PROJECT: my-project
|
|
95
|
+
command: python3 -m pip install -r requirements.txt && touch "$PIP_CACHE_DIR/.ready" && python3 train.py --data "$JOB_BUCKET/data" --checkpoints "$CHECKPOINT_GCS_DIR"
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
- `id` is optional. When omitted, submission generates and prints one; use that
|
|
99
|
+
ID with `watch`, `logs`, and `cancel`.
|
|
100
|
+
- `tpu` may contain one or several compatible TPU types to race.
|
|
101
|
+
- Omit `zone` and `tpu_name` to race all compatible fleet capacity. Set `zone`
|
|
102
|
+
to use one zone, or `tpu_name` to use one exact declared TPU.
|
|
103
|
+
- Create one listed bucket in each candidate region and mirror required data at
|
|
104
|
+
the same object paths. For a single-region job, list one bucket. The winning
|
|
105
|
+
region's bucket becomes `JOB_BUCKET`, and retries remain pinned to that
|
|
106
|
+
region.
|
|
107
|
+
- `bundle` is a local directory relative to `job.yaml`. TPU Runner archives and
|
|
108
|
+
uploads it; use `.tpu-runnerignore` to exclude files.
|
|
109
|
+
- `priority` may be `low`, `normal`, or `high`.
|
|
110
|
+
- Before a new job starts, TPU Runner clears the previous runner workspace and
|
|
111
|
+
shared memory, then extracts the new bundle into a fresh directory. A directory
|
|
112
|
+
declared under `caches` is preserved when it is marked `.ready` and the next job
|
|
113
|
+
on that worker declares the same `key`. Configure the relevant tool, such as
|
|
114
|
+
pip, to write to the cache path. Incomplete caches are discarded, and all
|
|
115
|
+
caches disappear when the TPU is deleted. Use `CHECKPOINT_GCS_DIR` for
|
|
116
|
+
durable checkpoints.
|
|
117
|
+
- `command` runs independently on every TPU worker. Other runner-provided
|
|
118
|
+
variables include `JOB_ID`, `ATTEMPT_ID`, `TPU_WORKER_HOST`,
|
|
119
|
+
`TPU_WORKER_COUNT`, `JOB_DIR`, and `ATTEMPT_GCS_DIR`.
|
|
120
|
+
|
|
121
|
+
Submit and watch the job:
|
|
122
|
+
|
|
123
|
+
```bash
|
|
124
|
+
tpu-runner validate-jobs job.yaml
|
|
125
|
+
tpu-runner submit job.yaml
|
|
126
|
+
tpu-runner watch JOB_ID
|
|
127
|
+
tpu-runner logs JOB_ID
|
|
128
|
+
tpu-runner cancel JOB_ID
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
Spot preemption and recognized infrastructure failures return a job to
|
|
133
|
+
`pending` with the same region and checkpoint directory. Application and setup
|
|
134
|
+
failures are terminal. TPU Runner schedules higher-priority jobs first. Within
|
|
135
|
+
each priority, it considers the most constrained jobs first and moves flexible
|
|
136
|
+
jobs to alternative idle TPUs when that allows more jobs to run.
|
|
137
|
+
|
|
138
|
+
Use `tpu-runner --help` for a list of all commands, or use a specific command such as `tpu-runner submit --help` for its options.
|
|
139
|
+
|
|
140
|
+
PRs and feature requests are welcome. A big thank you to the [Google TPU Research Cloud (TRC) program](https://sites.research.google/trc/) for inspiring this work.
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
# TPU Runner
|
|
2
|
+
|
|
3
|
+
`tpu-runner` is a job orchestrator designed to **maximize utilization of your Google Cloud TPU allocation.**
|
|
4
|
+
|
|
5
|
+
It automatically **races capacity** requests across compatible TPU types and regions, **assigns jobs** by priority + FIFO, and **retries jobs** interrupted by Spot preemptions. It also **minimizes inter-region transfer costs**, cleans the workspace for new jobs, and lets related jobs reuse local caches (e.g. for previously compiled XLA artifacts or software environments).
|
|
6
|
+
|
|
7
|
+
We use Firestore for queue and orchestration state, GCS for source bundles and job artifacts, and a Cloud Run controller to manage TPU capacity and execution.
|
|
8
|
+
|
|
9
|
+
## Install
|
|
10
|
+
|
|
11
|
+
Install `tpu-runner` as a standalone CLI:
|
|
12
|
+
|
|
13
|
+
```bash
|
|
14
|
+
uv tool install tpu-runner
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
For local development, run this from the repository root:
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
uv tool install --editable .
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
Configure `gcloud`:
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
gcloud auth login
|
|
27
|
+
gcloud auth application-default login
|
|
28
|
+
gcloud config set project YOUR_PROJECT_ID
|
|
29
|
+
gcloud alpha compute tpus --help >/dev/null
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
## Set up the runner
|
|
33
|
+
|
|
34
|
+
Create an example deployment:
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
tpu-runner init
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
Edit `deployment.yaml` with your project ID, existing Secret Manager secret names, and the TPU types, zones, maximum counts, runtime versions, and chip limits you are willing to use. Counts are provisioning ceilings; TPU Runner scales capacity up and down with demand and keeps idle capacity only when `keep_warm` is enabled.
|
|
41
|
+
|
|
42
|
+
The default `ssh_transport: direct` creates public-IP TPUs; set it to `iap` to create private-IP TPUs and reach them through IAP tunnels instead.
|
|
43
|
+
|
|
44
|
+
Validate and deploy:
|
|
45
|
+
|
|
46
|
+
```bash
|
|
47
|
+
tpu-runner validate-fleet deployment.yaml
|
|
48
|
+
tpu-runner deploy deployment.yaml
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
`deploy` enables the required APIs and creates or updates the runner bucket, Firestore database, service accounts and IAM, SSH identity, worker startup script, controller image, and Cloud Run controller job.
|
|
52
|
+
|
|
53
|
+
## Submit work
|
|
54
|
+
|
|
55
|
+
In the root of the code you want to run, create `job.yaml`:
|
|
56
|
+
|
|
57
|
+
```yaml
|
|
58
|
+
jobs:
|
|
59
|
+
- tpu: [v4-64, v6e-64]
|
|
60
|
+
buckets:
|
|
61
|
+
- gs://my-training-us-central2
|
|
62
|
+
- gs://my-training-us-east1
|
|
63
|
+
bundle: .
|
|
64
|
+
priority: high
|
|
65
|
+
caches:
|
|
66
|
+
- key: pip
|
|
67
|
+
path: .cache/pip
|
|
68
|
+
env:
|
|
69
|
+
PIP_CACHE_DIR: .cache/pip
|
|
70
|
+
WANDB_PROJECT: my-project
|
|
71
|
+
command: python3 -m pip install -r requirements.txt && touch "$PIP_CACHE_DIR/.ready" && python3 train.py --data "$JOB_BUCKET/data" --checkpoints "$CHECKPOINT_GCS_DIR"
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
- `id` is optional. When omitted, submission generates and prints one; use that
|
|
75
|
+
ID with `watch`, `logs`, and `cancel`.
|
|
76
|
+
- `tpu` may contain one or several compatible TPU types to race.
|
|
77
|
+
- Omit `zone` and `tpu_name` to race all compatible fleet capacity. Set `zone`
|
|
78
|
+
to use one zone, or `tpu_name` to use one exact declared TPU.
|
|
79
|
+
- Create one listed bucket in each candidate region and mirror required data at
|
|
80
|
+
the same object paths. For a single-region job, list one bucket. The winning
|
|
81
|
+
region's bucket becomes `JOB_BUCKET`, and retries remain pinned to that
|
|
82
|
+
region.
|
|
83
|
+
- `bundle` is a local directory relative to `job.yaml`. TPU Runner archives and
|
|
84
|
+
uploads it; use `.tpu-runnerignore` to exclude files.
|
|
85
|
+
- `priority` may be `low`, `normal`, or `high`.
|
|
86
|
+
- Before a new job starts, TPU Runner clears the previous runner workspace and
|
|
87
|
+
shared memory, then extracts the new bundle into a fresh directory. A directory
|
|
88
|
+
declared under `caches` is preserved when it is marked `.ready` and the next job
|
|
89
|
+
on that worker declares the same `key`. Configure the relevant tool, such as
|
|
90
|
+
pip, to write to the cache path. Incomplete caches are discarded, and all
|
|
91
|
+
caches disappear when the TPU is deleted. Use `CHECKPOINT_GCS_DIR` for
|
|
92
|
+
durable checkpoints.
|
|
93
|
+
- `command` runs independently on every TPU worker. Other runner-provided
|
|
94
|
+
variables include `JOB_ID`, `ATTEMPT_ID`, `TPU_WORKER_HOST`,
|
|
95
|
+
`TPU_WORKER_COUNT`, `JOB_DIR`, and `ATTEMPT_GCS_DIR`.
|
|
96
|
+
|
|
97
|
+
Submit and watch the job:
|
|
98
|
+
|
|
99
|
+
```bash
|
|
100
|
+
tpu-runner validate-jobs job.yaml
|
|
101
|
+
tpu-runner submit job.yaml
|
|
102
|
+
tpu-runner watch JOB_ID
|
|
103
|
+
tpu-runner logs JOB_ID
|
|
104
|
+
tpu-runner cancel JOB_ID
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
Spot preemption and recognized infrastructure failures return a job to
|
|
109
|
+
`pending` with the same region and checkpoint directory. Application and setup
|
|
110
|
+
failures are terminal. TPU Runner schedules higher-priority jobs first. Within
|
|
111
|
+
each priority, it considers the most constrained jobs first and moves flexible
|
|
112
|
+
jobs to alternative idle TPUs when that allows more jobs to run.
|
|
113
|
+
|
|
114
|
+
Use `tpu-runner --help` for a list of all commands, or use a specific command such as `tpu-runner submit --help` for its options.
|
|
115
|
+
|
|
116
|
+
PRs and feature requests are welcome. A big thank you to the [Google TPU Research Cloud (TRC) program](https://sites.research.google/trc/) for inspiring this work.
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "tpu-runner"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Minimal Google Cloud TPU job runner"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
requires-python = ">=3.11"
|
|
13
|
+
dependencies = [
|
|
14
|
+
"google-cloud-firestore>=2.16",
|
|
15
|
+
"PyYAML>=6.0",
|
|
16
|
+
]
|
|
17
|
+
keywords = ["gcp", "google-cloud", "tpu", "orchestration"]
|
|
18
|
+
classifiers = [
|
|
19
|
+
"Environment :: Console",
|
|
20
|
+
"Intended Audience :: Developers",
|
|
21
|
+
"Operating System :: OS Independent",
|
|
22
|
+
"Programming Language :: Python :: 3",
|
|
23
|
+
"Programming Language :: Python :: 3.11",
|
|
24
|
+
"Programming Language :: Python :: 3.12",
|
|
25
|
+
"Programming Language :: Python :: 3.13",
|
|
26
|
+
"Topic :: System :: Distributed Computing",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
[project.optional-dependencies]
|
|
30
|
+
release = ["build>=1,<2", "twine>=6,<7"]
|
|
31
|
+
|
|
32
|
+
[project.scripts]
|
|
33
|
+
tpu-runner = "tpu_runner.cli:main"
|
|
34
|
+
|
|
35
|
+
[tool.setuptools.packages.find]
|
|
36
|
+
include = ["tpu_runner*"]
|
|
37
|
+
|
|
38
|
+
[tool.setuptools.package-data]
|
|
39
|
+
tpu_runner = ["*.sh", "*.txt", "*.yaml", "Dockerfile"]
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
FROM gcr.io/google.com/cloudsdktool/google-cloud-cli:slim
|
|
2
|
+
|
|
3
|
+
WORKDIR /app
|
|
4
|
+
COPY tpu_runner/controller-requirements.txt ./
|
|
5
|
+
COPY tpu_runner ./tpu_runner
|
|
6
|
+
|
|
7
|
+
RUN python3 -m pip install --no-cache-dir --break-system-packages \
|
|
8
|
+
-r controller-requirements.txt
|
|
9
|
+
|
|
10
|
+
ENTRYPOINT ["python3", "-m", "tpu_runner.cli"]
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""GCP TPU fleet and job orchestration helpers."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
|
|
5
|
+
from .specs import FleetSpec, JobSpec, load_fleet_spec, load_job_specs
|
|
6
|
+
|
|
7
|
+
try:
|
|
8
|
+
__version__ = version("tpu-runner")
|
|
9
|
+
except PackageNotFoundError: # Source tree imported without installation.
|
|
10
|
+
__version__ = "0+unknown"
|
|
11
|
+
|
|
12
|
+
__all__ = [
|
|
13
|
+
"FleetSpec",
|
|
14
|
+
"JobSpec",
|
|
15
|
+
"__version__",
|
|
16
|
+
"load_fleet_spec",
|
|
17
|
+
"load_job_specs",
|
|
18
|
+
]
|
|
@@ -0,0 +1,248 @@
|
|
|
1
|
+
"""Pure placement and capacity policy."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from .gcp import generated_resource_names
|
|
6
|
+
from .region_pools import region_is_in_pool
|
|
7
|
+
from .runtime import JobRecord, ResourceRecord
|
|
8
|
+
from .specs import FleetSpec, TPUEntry, region_from_zone
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
JOB_PRIORITY_RANK = {"low": 0, "normal": 1, "high": 2}
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def pending_job_accepts_entry(job: JobRecord, entry) -> bool:
|
|
15
|
+
if job.status != "pending":
|
|
16
|
+
return False
|
|
17
|
+
if job.spec.zone and job.spec.zone != entry.zone:
|
|
18
|
+
return False
|
|
19
|
+
entry_region = region_from_zone(entry.zone)
|
|
20
|
+
if not job.spec.accepts_region(entry_region):
|
|
21
|
+
return False
|
|
22
|
+
if job.spec.storage_region and (
|
|
23
|
+
not job.spec.region or not region_is_in_pool(entry_region, job.spec.region)
|
|
24
|
+
):
|
|
25
|
+
return False
|
|
26
|
+
if job.spec.tpu_name:
|
|
27
|
+
if entry.adopted:
|
|
28
|
+
return (
|
|
29
|
+
job.spec.accepts_tpu_type(entry.type)
|
|
30
|
+
and job.spec.tpu_name == entry.existing
|
|
31
|
+
)
|
|
32
|
+
return job.spec.accepts_tpu_type(entry.type) and any(
|
|
33
|
+
generated_resource_names(entry, ordinal)[1] == job.spec.tpu_name
|
|
34
|
+
for ordinal in range(1, entry.count + 1)
|
|
35
|
+
)
|
|
36
|
+
return (entry.adopted or entry.count > 0) and job.spec.accepts_tpu_type(
|
|
37
|
+
entry.type
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def pending_job_accepts_resource(job: JobRecord, resource: ResourceRecord) -> bool:
|
|
42
|
+
"""Match compatible idle capacity while preserving exact TPU pins."""
|
|
43
|
+
|
|
44
|
+
if job.status != "pending":
|
|
45
|
+
return False
|
|
46
|
+
if not job.spec.accepts_tpu_type(resource.tpu_type):
|
|
47
|
+
return False
|
|
48
|
+
if job.spec.zone and job.spec.zone != resource.zone:
|
|
49
|
+
return False
|
|
50
|
+
resource_region = region_from_zone(resource.zone)
|
|
51
|
+
if not job.spec.accepts_region(resource_region):
|
|
52
|
+
return False
|
|
53
|
+
if job.spec.storage_region and (
|
|
54
|
+
not job.spec.region or not region_is_in_pool(resource_region, job.spec.region)
|
|
55
|
+
):
|
|
56
|
+
return False
|
|
57
|
+
return not job.spec.tpu_name or job.spec.tpu_name == resource.tpu_name
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def resource_is_busy(resource: ResourceRecord) -> bool:
|
|
61
|
+
return bool(
|
|
62
|
+
resource.status == "busy"
|
|
63
|
+
or resource.current_job_id
|
|
64
|
+
or resource.current_attempt_id
|
|
65
|
+
)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def pending_job_entry_constraint_key(job: JobRecord, entries: list) -> tuple:
|
|
69
|
+
compatible = sum(pending_job_accepts_entry(job, entry) for entry in entries)
|
|
70
|
+
return (
|
|
71
|
+
compatible == 0,
|
|
72
|
+
-JOB_PRIORITY_RANK[job.spec.priority],
|
|
73
|
+
compatible,
|
|
74
|
+
job.submitted_at,
|
|
75
|
+
job.spec.id,
|
|
76
|
+
)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def pending_job_resource_constraint_key(
|
|
80
|
+
job: JobRecord,
|
|
81
|
+
resources: list[ResourceRecord],
|
|
82
|
+
) -> tuple:
|
|
83
|
+
compatible = sum(
|
|
84
|
+
pending_job_accepts_resource(job, resource) for resource in resources
|
|
85
|
+
)
|
|
86
|
+
return (
|
|
87
|
+
compatible == 0,
|
|
88
|
+
-JOB_PRIORITY_RANK[job.spec.priority],
|
|
89
|
+
compatible,
|
|
90
|
+
job.submitted_at,
|
|
91
|
+
job.spec.id,
|
|
92
|
+
)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def plan_idle_assignments(
|
|
96
|
+
jobs: list[JobRecord],
|
|
97
|
+
resources: list[ResourceRecord],
|
|
98
|
+
*,
|
|
99
|
+
entries: list[TPUEntry] | None = None,
|
|
100
|
+
) -> list[tuple[JobRecord, ResourceRecord]]:
|
|
101
|
+
"""Match idle resources without displacing an earlier pending job.
|
|
102
|
+
|
|
103
|
+
Jobs are considered by priority, constraint count, and FIFO order. An
|
|
104
|
+
augmenting path may move an earlier flexible job to another idle resource,
|
|
105
|
+
but it never drops that job merely to fit a later one.
|
|
106
|
+
"""
|
|
107
|
+
|
|
108
|
+
idle_resources = sorted(
|
|
109
|
+
(
|
|
110
|
+
resource
|
|
111
|
+
for resource in resources
|
|
112
|
+
if resource.status == "idle" and not resource_is_busy(resource)
|
|
113
|
+
),
|
|
114
|
+
key=lambda resource: (not resource.adopted, resource.id),
|
|
115
|
+
)
|
|
116
|
+
resources_by_id = {resource.id: resource for resource in idle_resources}
|
|
117
|
+
ordered_jobs = sorted(
|
|
118
|
+
(job for job in jobs if job.status == "pending"),
|
|
119
|
+
key=lambda job: (
|
|
120
|
+
pending_job_entry_constraint_key(job, entries)
|
|
121
|
+
if entries is not None
|
|
122
|
+
else pending_job_resource_constraint_key(job, idle_resources)
|
|
123
|
+
),
|
|
124
|
+
)
|
|
125
|
+
matched_by_resource: dict[str, JobRecord] = {}
|
|
126
|
+
|
|
127
|
+
def match(job: JobRecord, visited: set[str]) -> bool:
|
|
128
|
+
candidates = sorted(
|
|
129
|
+
(
|
|
130
|
+
resource
|
|
131
|
+
for resource in idle_resources
|
|
132
|
+
if resource.id not in visited
|
|
133
|
+
and pending_job_accepts_resource(job, resource)
|
|
134
|
+
),
|
|
135
|
+
key=lambda resource: (
|
|
136
|
+
resource.id in matched_by_resource,
|
|
137
|
+
not resource.adopted,
|
|
138
|
+
resource.id,
|
|
139
|
+
),
|
|
140
|
+
)
|
|
141
|
+
for resource in candidates:
|
|
142
|
+
visited.add(resource.id)
|
|
143
|
+
previous = matched_by_resource.get(resource.id)
|
|
144
|
+
if previous is None or match(previous, visited):
|
|
145
|
+
matched_by_resource[resource.id] = job
|
|
146
|
+
return True
|
|
147
|
+
return False
|
|
148
|
+
|
|
149
|
+
for job in ordered_jobs:
|
|
150
|
+
match(job, set())
|
|
151
|
+
|
|
152
|
+
resources_by_job = {
|
|
153
|
+
job.spec.id: resources_by_id[resource_id]
|
|
154
|
+
for resource_id, job in matched_by_resource.items()
|
|
155
|
+
}
|
|
156
|
+
return [
|
|
157
|
+
(job, resources_by_job[job.spec.id])
|
|
158
|
+
for job in ordered_jobs
|
|
159
|
+
if job.spec.id in resources_by_job
|
|
160
|
+
]
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def allocate_managed_pending_demand(
|
|
164
|
+
jobs: list[JobRecord],
|
|
165
|
+
*,
|
|
166
|
+
fleet: FleetSpec,
|
|
167
|
+
resources: tuple[ResourceRecord, ...],
|
|
168
|
+
) -> dict[str, int]:
|
|
169
|
+
"""Request every pending job in every compatible Spot entry.
|
|
170
|
+
|
|
171
|
+
These are capacity races, not duplicate executions. Assignment remains a
|
|
172
|
+
Firestore transaction, and the losing entry's demand disappears on the
|
|
173
|
+
next reconciliation. Every physical entry count remains a hard ceiling.
|
|
174
|
+
"""
|
|
175
|
+
|
|
176
|
+
entries = [entry for entry in fleet.tpus if not entry.adopted]
|
|
177
|
+
busy = {
|
|
178
|
+
entry.id: sum(
|
|
179
|
+
1
|
|
180
|
+
for resource in resources
|
|
181
|
+
if resource.fleet_entry_id == entry.id
|
|
182
|
+
and not resource.adopted
|
|
183
|
+
and resource_is_busy(resource)
|
|
184
|
+
)
|
|
185
|
+
for entry in entries
|
|
186
|
+
}
|
|
187
|
+
demand = {entry.id: 0 for entry in entries}
|
|
188
|
+
for job in sorted(
|
|
189
|
+
jobs,
|
|
190
|
+
key=lambda candidate: pending_job_entry_constraint_key(candidate, entries),
|
|
191
|
+
):
|
|
192
|
+
for entry in entries:
|
|
193
|
+
available = max(0, entry.count - busy[entry.id])
|
|
194
|
+
if demand[entry.id] >= available:
|
|
195
|
+
continue
|
|
196
|
+
if pending_job_accepts_entry(job, entry):
|
|
197
|
+
demand[entry.id] += 1
|
|
198
|
+
return demand
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def desired_managed_capacity_counts(
|
|
202
|
+
jobs: list[JobRecord],
|
|
203
|
+
*,
|
|
204
|
+
fleet: FleetSpec,
|
|
205
|
+
resources: tuple[ResourceRecord, ...],
|
|
206
|
+
) -> dict[str, int]:
|
|
207
|
+
"""Return busy plus raced pending demand, capped by physical ceilings."""
|
|
208
|
+
|
|
209
|
+
idle_adopted = [
|
|
210
|
+
resource
|
|
211
|
+
for resource in resources
|
|
212
|
+
if resource.adopted and resource.status == "idle"
|
|
213
|
+
]
|
|
214
|
+
adopted_job_ids = {
|
|
215
|
+
job.spec.id
|
|
216
|
+
for job, _ in plan_idle_assignments(
|
|
217
|
+
jobs,
|
|
218
|
+
idle_adopted,
|
|
219
|
+
entries=list(fleet.tpus),
|
|
220
|
+
)
|
|
221
|
+
}
|
|
222
|
+
jobs_requiring_managed = [
|
|
223
|
+
job
|
|
224
|
+
for job in jobs
|
|
225
|
+
if job.status == "pending" and job.spec.id not in adopted_job_ids
|
|
226
|
+
]
|
|
227
|
+
|
|
228
|
+
pending_demand = allocate_managed_pending_demand(
|
|
229
|
+
jobs_requiring_managed,
|
|
230
|
+
fleet=fleet,
|
|
231
|
+
resources=resources,
|
|
232
|
+
)
|
|
233
|
+
desired: dict[str, int] = {}
|
|
234
|
+
for entry in fleet.tpus:
|
|
235
|
+
if entry.adopted:
|
|
236
|
+
continue
|
|
237
|
+
busy_count = sum(
|
|
238
|
+
1
|
|
239
|
+
for resource in resources
|
|
240
|
+
if resource.fleet_entry_id == entry.id
|
|
241
|
+
and not resource.adopted
|
|
242
|
+
and resource_is_busy(resource)
|
|
243
|
+
)
|
|
244
|
+
desired[entry.id] = min(
|
|
245
|
+
entry.count,
|
|
246
|
+
max(int(entry.keep_warm), busy_count + pending_demand.get(entry.id, 0)),
|
|
247
|
+
)
|
|
248
|
+
return desired
|