autoinference-utils 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,13 +8,13 @@ __pycache__/
8
8
 
9
9
  # Distribution / packaging
10
10
  .Python
11
+ node_modules/
11
12
  build/
12
13
  develop-eggs/
13
14
  dist/
14
15
  downloads/
15
16
  eggs/
16
17
  .eggs/
17
- lib/
18
18
  lib64/
19
19
  parts/
20
20
  sdist/
@@ -195,9 +195,9 @@ cython_debug/
195
195
  .abstra/
196
196
 
197
197
  # Visual Studio Code
198
- # Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
198
+ # Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
199
199
  # that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore
200
- # and can be added to the global gitignore or merged into this file. However, if you prefer,
200
+ # and can be added to the global gitignore or merged into this file. However, if you prefer,
201
201
  # you could uncomment the following to ignore the entire vscode folder
202
202
  # .vscode/
203
203
 
@@ -219,8 +219,6 @@ __marimo__/
219
219
  __MACOSX/
220
220
  .AppleDouble
221
221
  .LSOverride
222
- Icon[
223
- ]
224
222
 
225
223
  # Thumbnails
226
224
  ._*
@@ -242,4 +240,7 @@ Temporary Items
242
240
  .apdisk
243
241
 
244
242
  benchmark_results/
245
- docs/
243
+ docs/*
244
+ !docs/data-model.md
245
+ !docs/recipe-seeding.md
246
+ !docs/sweep-learnings.md
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: autoinference-utils
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Shared endpoint abstractions for autoinference deployments
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -371,15 +371,23 @@ class RouterEndpoint(Endpoint):
371
371
  health_timeout: float = 10 * 60,
372
372
  health_poll_interval: float = 5.0,
373
373
  ):
374
- super().__init__(base_url=f"http://localhost:{router_port}")
374
+ self.bench_mode = os.environ.get(BENCH_MODE_ENV) == "1"
375
+ self.listen_port = router_port
376
+ actual_router_port = (
377
+ router_port + BENCH_MODE_PORT_OFFSET if self.bench_mode else router_port
378
+ )
379
+ super().__init__(base_url=f"http://localhost:{actual_router_port}")
375
380
  self.pd_config = list(pd_config)
376
- self.worker_port = worker_port
377
- self.router_port = router_port
381
+ self.worker_port = (
382
+ worker_port + BENCH_MODE_PORT_OFFSET if self.bench_mode else worker_port
383
+ )
384
+ self.router_port = actual_router_port
378
385
  self.prefill_bootstrap_port = prefill_bootstrap_port
379
386
  self.api_key = api_key
380
387
  self.health_timeout = health_timeout
381
388
  self.health_poll_interval = health_poll_interval
382
389
  self._proc: Optional[subprocess.Popen] = None
390
+ self._bench_server: Optional[http.server.ThreadingHTTPServer] = None
383
391
 
384
392
  def _build_cmd(self) -> list[str]:
385
393
  cmd = [
@@ -434,8 +442,16 @@ class RouterEndpoint(Endpoint):
434
442
  timeout=self.health_timeout,
435
443
  poll_interval=self.health_poll_interval,
436
444
  )
445
+ if self.bench_mode:
446
+ self._bench_server = start_bench_proxy(
447
+ listen_port=self.listen_port,
448
+ upstream_port=self.router_port,
449
+ )
437
450
 
438
451
  def stop(self):
452
+ if self._bench_server is not None:
453
+ self._bench_server.shutdown()
454
+ self._bench_server = None
439
455
  terminate_process(self._proc)
440
456
  self._proc = None
441
457
 
@@ -579,7 +595,7 @@ def run_bench(
579
595
  benchmark: str, args: list[str], target: str, output_dir: str
580
596
  ) -> tuple[int, bytes]:
581
597
  """Run a benchmark task (an ``invoke`` @task in
582
- autoinference.tools.benchmarks) as a subprocess. Used by the in-container
598
+ autoinference.benchmarks) as a subprocess. Used by the in-container
583
599
  /bench proxy and the Modal bench worker (--target mode). stdout/stderr
584
600
  stream to the container's own stdout so Modal logs capture them live;
585
601
  returns (rc, run_report.json body) at the end."""
@@ -591,7 +607,7 @@ def run_bench(
591
607
  {
592
608
  "ok": False,
593
609
  "error": (
594
- "autoinference.tools.benchmarks not importable in container"
610
+ "autoinference.benchmarks not importable in container"
595
611
  ),
596
612
  "volume_path": output_dir,
597
613
  }
@@ -629,7 +645,7 @@ def run_bench(
629
645
 
630
646
  def _benchmarks_search_root() -> str | None:
631
647
  try:
632
- import autoinference.tools.benchmarks.tasks as tasks_mod
648
+ import autoinference.benchmarks.tasks as tasks_mod
633
649
  except Exception:
634
650
  return None
635
651
  return os.path.dirname(os.path.abspath(tasks_mod.__file__))
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "autoinference-utils"
3
- version = "0.2.0"
3
+ version = "0.2.1"
4
4
  description = "Shared endpoint abstractions for autoinference deployments"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"