autoinference-utils 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {autoinference_utils-0.2.0 → autoinference_utils-0.2.1}/.gitignore +7 -6
- {autoinference_utils-0.2.0 → autoinference_utils-0.2.1}/PKG-INFO +1 -1
- {autoinference_utils-0.2.0 → autoinference_utils-0.2.1}/autoinference_utils/endpoint.py +22 -6
- {autoinference_utils-0.2.0 → autoinference_utils-0.2.1}/pyproject.toml +1 -1
- {autoinference_utils-0.2.0 → autoinference_utils-0.2.1}/README.md +0 -0
- {autoinference_utils-0.2.0 → autoinference_utils-0.2.1}/autoinference_utils/__init__.py +0 -0
|
@@ -8,13 +8,13 @@ __pycache__/
|
|
|
8
8
|
|
|
9
9
|
# Distribution / packaging
|
|
10
10
|
.Python
|
|
11
|
+
node_modules/
|
|
11
12
|
build/
|
|
12
13
|
develop-eggs/
|
|
13
14
|
dist/
|
|
14
15
|
downloads/
|
|
15
16
|
eggs/
|
|
16
17
|
.eggs/
|
|
17
|
-
lib/
|
|
18
18
|
lib64/
|
|
19
19
|
parts/
|
|
20
20
|
sdist/
|
|
@@ -195,9 +195,9 @@ cython_debug/
|
|
|
195
195
|
.abstra/
|
|
196
196
|
|
|
197
197
|
# Visual Studio Code
|
|
198
|
-
# Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
|
|
198
|
+
# Visual Studio Code specific template is maintained in a separate VisualStudioCode.gitignore
|
|
199
199
|
# that can be found at https://github.com/github/gitignore/blob/main/Global/VisualStudioCode.gitignore
|
|
200
|
-
# and can be added to the global gitignore or merged into this file. However, if you prefer,
|
|
200
|
+
# and can be added to the global gitignore or merged into this file. However, if you prefer,
|
|
201
201
|
# you could uncomment the following to ignore the entire vscode folder
|
|
202
202
|
# .vscode/
|
|
203
203
|
|
|
@@ -219,8 +219,6 @@ __marimo__/
|
|
|
219
219
|
__MACOSX/
|
|
220
220
|
.AppleDouble
|
|
221
221
|
.LSOverride
|
|
222
|
-
Icon[
|
|
223
|
-
]
|
|
224
222
|
|
|
225
223
|
# Thumbnails
|
|
226
224
|
._*
|
|
@@ -242,4 +240,7 @@ Temporary Items
|
|
|
242
240
|
.apdisk
|
|
243
241
|
|
|
244
242
|
benchmark_results/
|
|
245
|
-
docs
|
|
243
|
+
docs/*
|
|
244
|
+
!docs/data-model.md
|
|
245
|
+
!docs/recipe-seeding.md
|
|
246
|
+
!docs/sweep-learnings.md
|
|
@@ -371,15 +371,23 @@ class RouterEndpoint(Endpoint):
|
|
|
371
371
|
health_timeout: float = 10 * 60,
|
|
372
372
|
health_poll_interval: float = 5.0,
|
|
373
373
|
):
|
|
374
|
-
|
|
374
|
+
self.bench_mode = os.environ.get(BENCH_MODE_ENV) == "1"
|
|
375
|
+
self.listen_port = router_port
|
|
376
|
+
actual_router_port = (
|
|
377
|
+
router_port + BENCH_MODE_PORT_OFFSET if self.bench_mode else router_port
|
|
378
|
+
)
|
|
379
|
+
super().__init__(base_url=f"http://localhost:{actual_router_port}")
|
|
375
380
|
self.pd_config = list(pd_config)
|
|
376
|
-
self.worker_port =
|
|
377
|
-
|
|
381
|
+
self.worker_port = (
|
|
382
|
+
worker_port + BENCH_MODE_PORT_OFFSET if self.bench_mode else worker_port
|
|
383
|
+
)
|
|
384
|
+
self.router_port = actual_router_port
|
|
378
385
|
self.prefill_bootstrap_port = prefill_bootstrap_port
|
|
379
386
|
self.api_key = api_key
|
|
380
387
|
self.health_timeout = health_timeout
|
|
381
388
|
self.health_poll_interval = health_poll_interval
|
|
382
389
|
self._proc: Optional[subprocess.Popen] = None
|
|
390
|
+
self._bench_server: Optional[http.server.ThreadingHTTPServer] = None
|
|
383
391
|
|
|
384
392
|
def _build_cmd(self) -> list[str]:
|
|
385
393
|
cmd = [
|
|
@@ -434,8 +442,16 @@ class RouterEndpoint(Endpoint):
|
|
|
434
442
|
timeout=self.health_timeout,
|
|
435
443
|
poll_interval=self.health_poll_interval,
|
|
436
444
|
)
|
|
445
|
+
if self.bench_mode:
|
|
446
|
+
self._bench_server = start_bench_proxy(
|
|
447
|
+
listen_port=self.listen_port,
|
|
448
|
+
upstream_port=self.router_port,
|
|
449
|
+
)
|
|
437
450
|
|
|
438
451
|
def stop(self):
|
|
452
|
+
if self._bench_server is not None:
|
|
453
|
+
self._bench_server.shutdown()
|
|
454
|
+
self._bench_server = None
|
|
439
455
|
terminate_process(self._proc)
|
|
440
456
|
self._proc = None
|
|
441
457
|
|
|
@@ -579,7 +595,7 @@ def run_bench(
|
|
|
579
595
|
benchmark: str, args: list[str], target: str, output_dir: str
|
|
580
596
|
) -> tuple[int, bytes]:
|
|
581
597
|
"""Run a benchmark task (an ``invoke`` @task in
|
|
582
|
-
autoinference.
|
|
598
|
+
autoinference.benchmarks) as a subprocess. Used by the in-container
|
|
583
599
|
/bench proxy and the Modal bench worker (--target mode). stdout/stderr
|
|
584
600
|
stream to the container's own stdout so Modal logs capture them live;
|
|
585
601
|
returns (rc, run_report.json body) at the end."""
|
|
@@ -591,7 +607,7 @@ def run_bench(
|
|
|
591
607
|
{
|
|
592
608
|
"ok": False,
|
|
593
609
|
"error": (
|
|
594
|
-
"autoinference.
|
|
610
|
+
"autoinference.benchmarks not importable in container"
|
|
595
611
|
),
|
|
596
612
|
"volume_path": output_dir,
|
|
597
613
|
}
|
|
@@ -629,7 +645,7 @@ def run_bench(
|
|
|
629
645
|
|
|
630
646
|
def _benchmarks_search_root() -> str | None:
|
|
631
647
|
try:
|
|
632
|
-
import autoinference.
|
|
648
|
+
import autoinference.benchmarks.tasks as tasks_mod
|
|
633
649
|
except Exception:
|
|
634
650
|
return None
|
|
635
651
|
return os.path.dirname(os.path.abspath(tasks_mod.__file__))
|
|
File without changes
|
|
File without changes
|