autoinference-utils 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,13 +8,13 @@ __pycache__/
8
8
 
9
9
  # Distribution / packaging
10
10
  .Python
11
+ node_modules/
11
12
  build/
12
13
  develop-eggs/
13
14
  dist/
14
15
  downloads/
15
16
  eggs/
16
17
  .eggs/
17
- lib/
18
18
  lib64/
19
19
  parts/
20
20
  sdist/
@@ -195,9 +195,9 @@ cython_debug/
195
195
  .abstra/
196
196
 
197
197
  # Visual Studio Code
198
- # Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
198
+ # Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
199
199
  # that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore
200
- # and can be added to the global gitignore or merged into this file. However, if you prefer,
200
+ # and can be added to the global gitignore or merged into this file. However, if you prefer,
201
201
  # you could uncomment the following to ignore the entire vscode folder
202
202
  # .vscode/
203
203
 
@@ -219,8 +219,6 @@ __marimo__/
219
219
  __MACOSX/
220
220
  .AppleDouble
221
221
  .LSOverride
222
- Icon[
223
- ]
224
222
 
225
223
  # Thumbnails
226
224
  ._*
@@ -242,4 +240,9 @@ Temporary Items
242
240
  .apdisk
243
241
 
244
242
  benchmark_results/
245
- docs/
243
+ docs/*
244
+ !docs/data-model.md
245
+ !docs/dogfood-feedback.md
246
+ !docs/main-architecture-explainer.md
247
+ !docs/recipe-seeding.md
248
+ !docs/sweep-learnings.md
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: autoinference-utils
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: Shared endpoint abstractions for autoinference deployments
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -238,9 +238,17 @@ class VLLMEndpoint(Endpoint):
238
238
  health_poll_interval: float = 5.0,
239
239
  health_request_timeout: float = 5.0,
240
240
  ):
241
- super().__init__(base_url=f"http://localhost:{worker_port}")
241
+ # In bench mode vLLM runs on worker_port + 10000 and a /bench proxy
242
+ # listens on the original worker_port; both fall through to the same
243
+ # port when bench mode is off.
244
+ self.bench_mode = os.environ.get(BENCH_MODE_ENV) == "1"
245
+ self.listen_port = worker_port
246
+ vllm_port = (
247
+ worker_port + BENCH_MODE_PORT_OFFSET if self.bench_mode else worker_port
248
+ )
249
+ super().__init__(base_url=f"http://localhost:{vllm_port}")
242
250
  self.model = model
243
- self.worker_port = worker_port
251
+ self.worker_port = vllm_port
244
252
  self.extra_server_args = (
245
253
  dict(extra_server_args) if extra_server_args else {}
246
254
  )
@@ -248,6 +256,10 @@ class VLLMEndpoint(Endpoint):
248
256
  self.health_poll_interval = health_poll_interval
249
257
  self.health_request_timeout = health_request_timeout
250
258
  self._proc: Optional[subprocess.Popen] = None
259
+ self._bench_server: Optional[http.server.ThreadingHTTPServer] = None
260
+
261
+ if os.environ.get(DUMMY_WEIGHTS_ENV) == "1":
262
+ self.extra_server_args.setdefault("--load-format", "dummy")
251
263
 
252
264
  def _build_cmd(self) -> list[str]:
253
265
  cmd = [
@@ -264,13 +276,6 @@ class VLLMEndpoint(Endpoint):
264
276
  return cmd
265
277
 
266
278
  def start(self):
267
- if (
268
- os.environ.get(BENCH_MODE_ENV) == "1"
269
- or os.environ.get(DUMMY_WEIGHTS_ENV) == "1"
270
- ):
271
- raise NotImplementedError(
272
- "bench_mode and dummy weights are not supported for VLLMEndpoint"
273
- )
274
279
  cmd = self._build_cmd()
275
280
  print(f"[vllm] starting: {shlex.join(cmd)}")
276
281
  self._proc = subprocess.Popen(cmd)
@@ -281,8 +286,16 @@ class VLLMEndpoint(Endpoint):
281
286
  poll_interval=self.health_poll_interval,
282
287
  request_timeout=self.health_request_timeout,
283
288
  )
289
+ if self.bench_mode:
290
+ self._bench_server = start_bench_proxy(
291
+ listen_port=self.listen_port,
292
+ upstream_port=self.worker_port,
293
+ )
284
294
 
285
295
  def stop(self):
296
+ if self._bench_server is not None:
297
+ self._bench_server.shutdown()
298
+ self._bench_server = None
286
299
  terminate_process(self._proc)
287
300
  self._proc = None
288
301
 
@@ -371,15 +384,23 @@ class RouterEndpoint(Endpoint):
371
384
  health_timeout: float = 10 * 60,
372
385
  health_poll_interval: float = 5.0,
373
386
  ):
374
- super().__init__(base_url=f"http://localhost:{router_port}")
387
+ self.bench_mode = os.environ.get(BENCH_MODE_ENV) == "1"
388
+ self.listen_port = router_port
389
+ actual_router_port = (
390
+ router_port + BENCH_MODE_PORT_OFFSET if self.bench_mode else router_port
391
+ )
392
+ super().__init__(base_url=f"http://localhost:{actual_router_port}")
375
393
  self.pd_config = list(pd_config)
376
- self.worker_port = worker_port
377
- self.router_port = router_port
394
+ self.worker_port = (
395
+ worker_port + BENCH_MODE_PORT_OFFSET if self.bench_mode else worker_port
396
+ )
397
+ self.router_port = actual_router_port
378
398
  self.prefill_bootstrap_port = prefill_bootstrap_port
379
399
  self.api_key = api_key
380
400
  self.health_timeout = health_timeout
381
401
  self.health_poll_interval = health_poll_interval
382
402
  self._proc: Optional[subprocess.Popen] = None
403
+ self._bench_server: Optional[http.server.ThreadingHTTPServer] = None
383
404
 
384
405
  def _build_cmd(self) -> list[str]:
385
406
  cmd = [
@@ -434,8 +455,16 @@ class RouterEndpoint(Endpoint):
434
455
  timeout=self.health_timeout,
435
456
  poll_interval=self.health_poll_interval,
436
457
  )
458
+ if self.bench_mode:
459
+ self._bench_server = start_bench_proxy(
460
+ listen_port=self.listen_port,
461
+ upstream_port=self.router_port,
462
+ )
437
463
 
438
464
  def stop(self):
465
+ if self._bench_server is not None:
466
+ self._bench_server.shutdown()
467
+ self._bench_server = None
439
468
  terminate_process(self._proc)
440
469
  self._proc = None
441
470
 
@@ -579,7 +608,7 @@ def run_bench(
579
608
  benchmark: str, args: list[str], target: str, output_dir: str
580
609
  ) -> tuple[int, bytes]:
581
610
  """Run a benchmark task (an ``invoke`` @task in
582
- autoinference.tools.benchmarks) as a subprocess. Used by the in-container
611
+ autoinference.benchmarks) as a subprocess. Used by the in-container
583
612
  /bench proxy and the Modal bench worker (--target mode). stdout/stderr
584
613
  stream to the container's own stdout so Modal logs capture them live;
585
614
  returns (rc, run_report.json body) at the end."""
@@ -591,7 +620,7 @@ def run_bench(
591
620
  {
592
621
  "ok": False,
593
622
  "error": (
594
- "autoinference.tools.benchmarks not importable in container"
623
+ "autoinference.benchmarks not importable in container"
595
624
  ),
596
625
  "volume_path": output_dir,
597
626
  }
@@ -629,7 +658,7 @@ def run_bench(
629
658
 
630
659
  def _benchmarks_search_root() -> str | None:
631
660
  try:
632
- import autoinference.tools.benchmarks.tasks as tasks_mod
661
+ import autoinference.benchmarks.tasks as tasks_mod
633
662
  except Exception:
634
663
  return None
635
664
  return os.path.dirname(os.path.abspath(tasks_mod.__file__))
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "autoinference-utils"
3
- version = "0.2.0"
3
+ version = "0.2.2"
4
4
  description = "Shared endpoint abstractions for autoinference deployments"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"