tpu-runner 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 David Hidary
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,140 @@
1
+ Metadata-Version: 2.4
2
+ Name: tpu-runner
3
+ Version: 0.1.0
4
+ Summary: Minimal Google Cloud TPU job runner
5
+ License-Expression: MIT
6
+ Keywords: gcp,google-cloud,tpu,orchestration
7
+ Classifier: Environment :: Console
8
+ Classifier: Intended Audience :: Developers
9
+ Classifier: Operating System :: OS Independent
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: Programming Language :: Python :: 3.11
12
+ Classifier: Programming Language :: Python :: 3.12
13
+ Classifier: Programming Language :: Python :: 3.13
14
+ Classifier: Topic :: System :: Distributed Computing
15
+ Requires-Python: >=3.11
16
+ Description-Content-Type: text/markdown
17
+ License-File: LICENSE
18
+ Requires-Dist: google-cloud-firestore>=2.16
19
+ Requires-Dist: PyYAML>=6.0
20
+ Provides-Extra: release
21
+ Requires-Dist: build<2,>=1; extra == "release"
22
+ Requires-Dist: twine<7,>=6; extra == "release"
23
+ Dynamic: license-file
24
+
25
+ # TPU Runner
26
+
27
+ `tpu-runner` is a job orchestrator designed to **maximize utilization of your Google Cloud TPU allocation.**
28
+
29
+ It automatically **races capacity** requests across compatible TPU types and regions, **assigns jobs** by priority + FIFO, and **retries jobs** interrupted by Spot preemptions. It also **minimizes inter-region transfer costs**, cleans the workspace for new jobs, and lets related jobs reuse local caches (e.g. for previously compiled XLA artifacts or software environments).
30
+
31
+ We use Firestore for queue and orchestration state, GCS for source bundles and job artifacts, and a Cloud Run controller to manage TPU capacity and execution.
32
+
33
+ ## Install
34
+
35
+ Install `tpu-runner` as a standalone CLI:
36
+
37
+ ```bash
38
+ uv tool install tpu-runner
39
+ ```
40
+
41
+ For local development, run this from the repository root:
42
+
43
+ ```bash
44
+ uv tool install --editable .
45
+ ```
46
+
47
+ Configure `gcloud`:
48
+
49
+ ```bash
50
+ gcloud auth login
51
+ gcloud auth application-default login
52
+ gcloud config set project YOUR_PROJECT_ID
53
+ gcloud alpha compute tpus --help >/dev/null
54
+ ```
55
+
56
+ ## Set up the runner
57
+
58
+ Create an example deployment:
59
+
60
+ ```bash
61
+ tpu-runner init
62
+ ```
63
+
64
+ Edit `deployment.yaml` with your project ID, existing Secret Manager secret names, and the TPU types, zones, maximum counts, runtime versions, and chip limits you are willing to use. Counts are provisioning ceilings; TPU Runner scales capacity up and down with demand and keeps idle capacity only when `keep_warm` is enabled.
65
+
66
+ The default `ssh_transport: direct` creates public-IP TPUs; set it to `iap` to create private-IP TPUs and reach them through IAP tunnels instead.
67
+
68
+ Validate and deploy:
69
+
70
+ ```bash
71
+ tpu-runner validate-fleet deployment.yaml
72
+ tpu-runner deploy deployment.yaml
73
+ ```
74
+
75
+ `deploy` enables the required APIs and creates or updates the runner bucket, Firestore database, service accounts and IAM, SSH identity, worker startup script, controller image, and Cloud Run controller job.
76
+
77
+ ## Submit work
78
+
79
+ In the root of the code you want to run, create `job.yaml`:
80
+
81
+ ```yaml
82
+ jobs:
83
+ - tpu: [v4-64, v6e-64]
84
+ buckets:
85
+ - gs://my-training-us-central2
86
+ - gs://my-training-us-east1
87
+ bundle: .
88
+ priority: high
89
+ caches:
90
+ - key: pip
91
+ path: .cache/pip
92
+ env:
93
+ PIP_CACHE_DIR: .cache/pip
94
+ WANDB_PROJECT: my-project
95
+ command: python3 -m pip install -r requirements.txt && touch "$PIP_CACHE_DIR/.ready" && python3 train.py --data "$JOB_BUCKET/data" --checkpoints "$CHECKPOINT_GCS_DIR"
96
+ ```
97
+
98
+ - `id` is optional. When omitted, submission generates and prints one; use that
99
+ ID with `watch`, `logs`, and `cancel`.
100
+ - `tpu` may contain one or several compatible TPU types to race.
101
+ - Omit `zone` and `tpu_name` to race all compatible fleet capacity. Set `zone`
102
+ to use one zone, or `tpu_name` to use one exact declared TPU.
103
+ - Create one listed bucket in each candidate region and mirror required data at
104
+ the same object paths. For a single-region job, list one bucket. The winning
105
+ region's bucket becomes `JOB_BUCKET`, and retries remain pinned to that
106
+ region.
107
+ - `bundle` is a local directory relative to `job.yaml`. TPU Runner archives and
108
+ uploads it; use `.tpu-runnerignore` to exclude files.
109
+ - `priority` may be `low`, `normal`, or `high`.
110
+ - Before a new job starts, TPU Runner clears the previous runner workspace and
111
+ shared memory, then extracts the new bundle into a fresh directory. A directory
112
+ declared under `caches` is preserved when it is marked `.ready` and the next job
113
+ on that worker declares the same `key`. Configure the relevant tool, such as
114
+ pip, to write to the cache path. Incomplete caches are discarded, and all
115
+ caches disappear when the TPU is deleted. Use `CHECKPOINT_GCS_DIR` for
116
+ durable checkpoints.
117
+ - `command` runs independently on every TPU worker. Other runner-provided
118
+ variables include `JOB_ID`, `ATTEMPT_ID`, `TPU_WORKER_HOST`,
119
+ `TPU_WORKER_COUNT`, `JOB_DIR`, and `ATTEMPT_GCS_DIR`.
120
+
121
+ Submit and watch the job:
122
+
123
+ ```bash
124
+ tpu-runner validate-jobs job.yaml
125
+ tpu-runner submit job.yaml
126
+ tpu-runner watch JOB_ID
127
+ tpu-runner logs JOB_ID
128
+ tpu-runner cancel JOB_ID
129
+ ```
130
+
131
+
132
+ Spot preemption and recognized infrastructure failures return a job to
133
+ `pending` with the same region and checkpoint directory. Application and setup
134
+ failures are terminal. TPU Runner schedules higher-priority jobs first. Within
135
+ each priority, it considers the most constrained jobs first and moves flexible
136
+ jobs to alternative idle TPUs when that allows more jobs to run.
137
+
138
+ Use `tpu-runner --help` for a list of all commands, or use a specific command such as `tpu-runner submit --help` for its options.
139
+
140
+ PRs and feature requests are welcome. A big thank you to the [Google TPU Research Cloud (TRC) program](https://sites.research.google/trc/) for inspiring this work.
@@ -0,0 +1,116 @@
1
+ # TPU Runner
2
+
3
+ `tpu-runner` is a job orchestrator designed to **maximize utilization of your Google Cloud TPU allocation.**
4
+
5
+ It automatically **races capacity** requests across compatible TPU types and regions, **assigns jobs** by priority + FIFO, and **retries jobs** interrupted by Spot preemptions. It also **minimizes inter-region transfer costs**, cleans the workspace for new jobs, and lets related jobs reuse local caches (e.g. for previously compiled XLA artifacts or software environments).
6
+
7
+ We use Firestore for queue and orchestration state, GCS for source bundles and job artifacts, and a Cloud Run controller to manage TPU capacity and execution.
8
+
9
+ ## Install
10
+
11
+ Install `tpu-runner` as a standalone CLI:
12
+
13
+ ```bash
14
+ uv tool install tpu-runner
15
+ ```
16
+
17
+ For local development, run this from the repository root:
18
+
19
+ ```bash
20
+ uv tool install --editable .
21
+ ```
22
+
23
+ Configure `gcloud`:
24
+
25
+ ```bash
26
+ gcloud auth login
27
+ gcloud auth application-default login
28
+ gcloud config set project YOUR_PROJECT_ID
29
+ gcloud alpha compute tpus --help >/dev/null
30
+ ```
31
+
32
+ ## Set up the runner
33
+
34
+ Create an example deployment:
35
+
36
+ ```bash
37
+ tpu-runner init
38
+ ```
39
+
40
+ Edit `deployment.yaml` with your project ID, existing Secret Manager secret names, and the TPU types, zones, maximum counts, runtime versions, and chip limits you are willing to use. Counts are provisioning ceilings; TPU Runner scales capacity up and down with demand and keeps idle capacity only when `keep_warm` is enabled.
41
+
42
+ The default `ssh_transport: direct` creates public-IP TPUs; set it to `iap` to create private-IP TPUs and reach them through IAP tunnels instead.
43
+
44
+ Validate and deploy:
45
+
46
+ ```bash
47
+ tpu-runner validate-fleet deployment.yaml
48
+ tpu-runner deploy deployment.yaml
49
+ ```
50
+
51
+ `deploy` enables the required APIs and creates or updates the runner bucket, Firestore database, service accounts and IAM, SSH identity, worker startup script, controller image, and Cloud Run controller job.
52
+
53
+ ## Submit work
54
+
55
+ In the root of the code you want to run, create `job.yaml`:
56
+
57
+ ```yaml
58
+ jobs:
59
+ - tpu: [v4-64, v6e-64]
60
+ buckets:
61
+ - gs://my-training-us-central2
62
+ - gs://my-training-us-east1
63
+ bundle: .
64
+ priority: high
65
+ caches:
66
+ - key: pip
67
+ path: .cache/pip
68
+ env:
69
+ PIP_CACHE_DIR: .cache/pip
70
+ WANDB_PROJECT: my-project
71
+ command: python3 -m pip install -r requirements.txt && touch "$PIP_CACHE_DIR/.ready" && python3 train.py --data "$JOB_BUCKET/data" --checkpoints "$CHECKPOINT_GCS_DIR"
72
+ ```
73
+
74
+ - `id` is optional. When omitted, submission generates and prints one; use that
75
+ ID with `watch`, `logs`, and `cancel`.
76
+ - `tpu` may contain one or several compatible TPU types to race.
77
+ - Omit `zone` and `tpu_name` to race all compatible fleet capacity. Set `zone`
78
+ to use one zone, or `tpu_name` to use one exact declared TPU.
79
+ - Create one listed bucket in each candidate region and mirror required data at
80
+ the same object paths. For a single-region job, list one bucket. The winning
81
+ region's bucket becomes `JOB_BUCKET`, and retries remain pinned to that
82
+ region.
83
+ - `bundle` is a local directory relative to `job.yaml`. TPU Runner archives and
84
+ uploads it; use `.tpu-runnerignore` to exclude files.
85
+ - `priority` may be `low`, `normal`, or `high`.
86
+ - Before a new job starts, TPU Runner clears the previous runner workspace and
87
+ shared memory, then extracts the new bundle into a fresh directory. A directory
88
+ declared under `caches` is preserved when it is marked `.ready` and the next job
89
+ on that worker declares the same `key`. Configure the relevant tool, such as
90
+ pip, to write to the cache path. Incomplete caches are discarded, and all
91
+ caches disappear when the TPU is deleted. Use `CHECKPOINT_GCS_DIR` for
92
+ durable checkpoints.
93
+ - `command` runs independently on every TPU worker. Other runner-provided
94
+ variables include `JOB_ID`, `ATTEMPT_ID`, `TPU_WORKER_HOST`,
95
+ `TPU_WORKER_COUNT`, `JOB_DIR`, and `ATTEMPT_GCS_DIR`.
96
+
97
+ Submit and watch the job:
98
+
99
+ ```bash
100
+ tpu-runner validate-jobs job.yaml
101
+ tpu-runner submit job.yaml
102
+ tpu-runner watch JOB_ID
103
+ tpu-runner logs JOB_ID
104
+ tpu-runner cancel JOB_ID
105
+ ```
106
+
107
+
108
+ Spot preemption and recognized infrastructure failures return a job to
109
+ `pending` with the same region and checkpoint directory. Application and setup
110
+ failures are terminal. TPU Runner schedules higher-priority jobs first. Within
111
+ each priority, it considers the most constrained jobs first and moves flexible
112
+ jobs to alternative idle TPUs when that allows more jobs to run.
113
+
114
+ Use `tpu-runner --help` for a list of all commands, or use a specific command such as `tpu-runner submit --help` for its options.
115
+
116
+ PRs and feature requests are welcome. A big thank you to the [Google TPU Research Cloud (TRC) program](https://sites.research.google/trc/) for inspiring this work.
@@ -0,0 +1,39 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "tpu-runner"
7
+ version = "0.1.0"
8
+ description = "Minimal Google Cloud TPU job runner"
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ license-files = ["LICENSE"]
12
+ requires-python = ">=3.11"
13
+ dependencies = [
14
+ "google-cloud-firestore>=2.16",
15
+ "PyYAML>=6.0",
16
+ ]
17
+ keywords = ["gcp", "google-cloud", "tpu", "orchestration"]
18
+ classifiers = [
19
+ "Environment :: Console",
20
+ "Intended Audience :: Developers",
21
+ "Operating System :: OS Independent",
22
+ "Programming Language :: Python :: 3",
23
+ "Programming Language :: Python :: 3.11",
24
+ "Programming Language :: Python :: 3.12",
25
+ "Programming Language :: Python :: 3.13",
26
+ "Topic :: System :: Distributed Computing",
27
+ ]
28
+
29
+ [project.optional-dependencies]
30
+ release = ["build>=1,<2", "twine>=6,<7"]
31
+
32
+ [project.scripts]
33
+ tpu-runner = "tpu_runner.cli:main"
34
+
35
+ [tool.setuptools.packages.find]
36
+ include = ["tpu_runner*"]
37
+
38
+ [tool.setuptools.package-data]
39
+ tpu_runner = ["*.sh", "*.txt", "*.yaml", "Dockerfile"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,10 @@
1
+ FROM gcr.io/google.com/cloudsdktool/google-cloud-cli:slim
2
+
3
+ WORKDIR /app
4
+ COPY tpu_runner/controller-requirements.txt ./
5
+ COPY tpu_runner ./tpu_runner
6
+
7
+ RUN python3 -m pip install --no-cache-dir --break-system-packages \
8
+ -r controller-requirements.txt
9
+
10
+ ENTRYPOINT ["python3", "-m", "tpu_runner.cli"]
@@ -0,0 +1,18 @@
1
+ """GCP TPU fleet and job orchestration helpers."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+
5
+ from .specs import FleetSpec, JobSpec, load_fleet_spec, load_job_specs
6
+
7
+ try:
8
+ __version__ = version("tpu-runner")
9
+ except PackageNotFoundError: # Source tree imported without installation.
10
+ __version__ = "0+unknown"
11
+
12
+ __all__ = [
13
+ "FleetSpec",
14
+ "JobSpec",
15
+ "__version__",
16
+ "load_fleet_spec",
17
+ "load_job_specs",
18
+ ]
@@ -0,0 +1,4 @@
1
+ from .cli import main
2
+
3
+
4
+ raise SystemExit(main())
@@ -0,0 +1,248 @@
1
+ """Pure placement and capacity policy."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from .gcp import generated_resource_names
6
+ from .region_pools import region_is_in_pool
7
+ from .runtime import JobRecord, ResourceRecord
8
+ from .specs import FleetSpec, TPUEntry, region_from_zone
9
+
10
+
11
+ JOB_PRIORITY_RANK = {"low": 0, "normal": 1, "high": 2}
12
+
13
+
14
+ def pending_job_accepts_entry(job: JobRecord, entry) -> bool:
15
+ if job.status != "pending":
16
+ return False
17
+ if job.spec.zone and job.spec.zone != entry.zone:
18
+ return False
19
+ entry_region = region_from_zone(entry.zone)
20
+ if not job.spec.accepts_region(entry_region):
21
+ return False
22
+ if job.spec.storage_region and (
23
+ not job.spec.region or not region_is_in_pool(entry_region, job.spec.region)
24
+ ):
25
+ return False
26
+ if job.spec.tpu_name:
27
+ if entry.adopted:
28
+ return (
29
+ job.spec.accepts_tpu_type(entry.type)
30
+ and job.spec.tpu_name == entry.existing
31
+ )
32
+ return job.spec.accepts_tpu_type(entry.type) and any(
33
+ generated_resource_names(entry, ordinal)[1] == job.spec.tpu_name
34
+ for ordinal in range(1, entry.count + 1)
35
+ )
36
+ return (entry.adopted or entry.count > 0) and job.spec.accepts_tpu_type(
37
+ entry.type
38
+ )
39
+
40
+
41
+ def pending_job_accepts_resource(job: JobRecord, resource: ResourceRecord) -> bool:
42
+ """Match compatible idle capacity while preserving exact TPU pins."""
43
+
44
+ if job.status != "pending":
45
+ return False
46
+ if not job.spec.accepts_tpu_type(resource.tpu_type):
47
+ return False
48
+ if job.spec.zone and job.spec.zone != resource.zone:
49
+ return False
50
+ resource_region = region_from_zone(resource.zone)
51
+ if not job.spec.accepts_region(resource_region):
52
+ return False
53
+ if job.spec.storage_region and (
54
+ not job.spec.region or not region_is_in_pool(resource_region, job.spec.region)
55
+ ):
56
+ return False
57
+ return not job.spec.tpu_name or job.spec.tpu_name == resource.tpu_name
58
+
59
+
60
+ def resource_is_busy(resource: ResourceRecord) -> bool:
61
+ return bool(
62
+ resource.status == "busy"
63
+ or resource.current_job_id
64
+ or resource.current_attempt_id
65
+ )
66
+
67
+
68
+ def pending_job_entry_constraint_key(job: JobRecord, entries: list) -> tuple:
69
+ compatible = sum(pending_job_accepts_entry(job, entry) for entry in entries)
70
+ return (
71
+ compatible == 0,
72
+ -JOB_PRIORITY_RANK[job.spec.priority],
73
+ compatible,
74
+ job.submitted_at,
75
+ job.spec.id,
76
+ )
77
+
78
+
79
+ def pending_job_resource_constraint_key(
80
+ job: JobRecord,
81
+ resources: list[ResourceRecord],
82
+ ) -> tuple:
83
+ compatible = sum(
84
+ pending_job_accepts_resource(job, resource) for resource in resources
85
+ )
86
+ return (
87
+ compatible == 0,
88
+ -JOB_PRIORITY_RANK[job.spec.priority],
89
+ compatible,
90
+ job.submitted_at,
91
+ job.spec.id,
92
+ )
93
+
94
+
95
+ def plan_idle_assignments(
96
+ jobs: list[JobRecord],
97
+ resources: list[ResourceRecord],
98
+ *,
99
+ entries: list[TPUEntry] | None = None,
100
+ ) -> list[tuple[JobRecord, ResourceRecord]]:
101
+ """Match idle resources without displacing an earlier pending job.
102
+
103
+ Jobs are considered by priority, constraint count, and FIFO order. An
104
+ augmenting path may move an earlier flexible job to another idle resource,
105
+ but it never drops that job merely to fit a later one.
106
+ """
107
+
108
+ idle_resources = sorted(
109
+ (
110
+ resource
111
+ for resource in resources
112
+ if resource.status == "idle" and not resource_is_busy(resource)
113
+ ),
114
+ key=lambda resource: (not resource.adopted, resource.id),
115
+ )
116
+ resources_by_id = {resource.id: resource for resource in idle_resources}
117
+ ordered_jobs = sorted(
118
+ (job for job in jobs if job.status == "pending"),
119
+ key=lambda job: (
120
+ pending_job_entry_constraint_key(job, entries)
121
+ if entries is not None
122
+ else pending_job_resource_constraint_key(job, idle_resources)
123
+ ),
124
+ )
125
+ matched_by_resource: dict[str, JobRecord] = {}
126
+
127
+ def match(job: JobRecord, visited: set[str]) -> bool:
128
+ candidates = sorted(
129
+ (
130
+ resource
131
+ for resource in idle_resources
132
+ if resource.id not in visited
133
+ and pending_job_accepts_resource(job, resource)
134
+ ),
135
+ key=lambda resource: (
136
+ resource.id in matched_by_resource,
137
+ not resource.adopted,
138
+ resource.id,
139
+ ),
140
+ )
141
+ for resource in candidates:
142
+ visited.add(resource.id)
143
+ previous = matched_by_resource.get(resource.id)
144
+ if previous is None or match(previous, visited):
145
+ matched_by_resource[resource.id] = job
146
+ return True
147
+ return False
148
+
149
+ for job in ordered_jobs:
150
+ match(job, set())
151
+
152
+ resources_by_job = {
153
+ job.spec.id: resources_by_id[resource_id]
154
+ for resource_id, job in matched_by_resource.items()
155
+ }
156
+ return [
157
+ (job, resources_by_job[job.spec.id])
158
+ for job in ordered_jobs
159
+ if job.spec.id in resources_by_job
160
+ ]
161
+
162
+
163
+ def allocate_managed_pending_demand(
164
+ jobs: list[JobRecord],
165
+ *,
166
+ fleet: FleetSpec,
167
+ resources: tuple[ResourceRecord, ...],
168
+ ) -> dict[str, int]:
169
+ """Request every pending job in every compatible Spot entry.
170
+
171
+ These are capacity races, not duplicate executions. Assignment remains a
172
+ Firestore transaction, and the losing entry's demand disappears on the
173
+ next reconciliation. Every physical entry count remains a hard ceiling.
174
+ """
175
+
176
+ entries = [entry for entry in fleet.tpus if not entry.adopted]
177
+ busy = {
178
+ entry.id: sum(
179
+ 1
180
+ for resource in resources
181
+ if resource.fleet_entry_id == entry.id
182
+ and not resource.adopted
183
+ and resource_is_busy(resource)
184
+ )
185
+ for entry in entries
186
+ }
187
+ demand = {entry.id: 0 for entry in entries}
188
+ for job in sorted(
189
+ jobs,
190
+ key=lambda candidate: pending_job_entry_constraint_key(candidate, entries),
191
+ ):
192
+ for entry in entries:
193
+ available = max(0, entry.count - busy[entry.id])
194
+ if demand[entry.id] >= available:
195
+ continue
196
+ if pending_job_accepts_entry(job, entry):
197
+ demand[entry.id] += 1
198
+ return demand
199
+
200
+
201
+ def desired_managed_capacity_counts(
202
+ jobs: list[JobRecord],
203
+ *,
204
+ fleet: FleetSpec,
205
+ resources: tuple[ResourceRecord, ...],
206
+ ) -> dict[str, int]:
207
+ """Return busy plus raced pending demand, capped by physical ceilings."""
208
+
209
+ idle_adopted = [
210
+ resource
211
+ for resource in resources
212
+ if resource.adopted and resource.status == "idle"
213
+ ]
214
+ adopted_job_ids = {
215
+ job.spec.id
216
+ for job, _ in plan_idle_assignments(
217
+ jobs,
218
+ idle_adopted,
219
+ entries=list(fleet.tpus),
220
+ )
221
+ }
222
+ jobs_requiring_managed = [
223
+ job
224
+ for job in jobs
225
+ if job.status == "pending" and job.spec.id not in adopted_job_ids
226
+ ]
227
+
228
+ pending_demand = allocate_managed_pending_demand(
229
+ jobs_requiring_managed,
230
+ fleet=fleet,
231
+ resources=resources,
232
+ )
233
+ desired: dict[str, int] = {}
234
+ for entry in fleet.tpus:
235
+ if entry.adopted:
236
+ continue
237
+ busy_count = sum(
238
+ 1
239
+ for resource in resources
240
+ if resource.fleet_entry_id == entry.id
241
+ and not resource.adopted
242
+ and resource_is_busy(resource)
243
+ )
244
+ desired[entry.id] = min(
245
+ entry.count,
246
+ max(int(entry.keep_warm), busy_count + pending_demand.get(entry.id, 0)),
247
+ )
248
+ return desired