autoinference-utils 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: autoinference-utils
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Shared endpoint abstractions for autoinference deployments
5
5
  Requires-Python: >=3.10
6
6
  Description-Content-Type: text/markdown
@@ -26,6 +26,9 @@ BENCH_MODE_ENV = "ENDPOINT_BENCH_MODE"
26
26
  DUMMY_WEIGHTS_ENV = "ENDPOINT_DUMMY"
27
27
  BENCH_MODE_PORT_OFFSET = 10000
28
28
 
29
+ MODAL_FLASH_REQUEST_UUID_HEADER= "x-modal-flash-request-uuid"
30
+ MODAL_SESSION_ID_HEADER = "modal-session-id"
31
+
29
32
  # Endpoint class names whose servers emit OAI streaming chunks that
30
33
  # sglang.bench_serving can't parse. deploy_and_bench auto-injects
31
34
  # --openai-stream-compat for deployments importing any of these.
@@ -93,6 +96,7 @@ class SGLangEndpoint(Endpoint):
93
96
  health_timeout: float = 20 * 60,
94
97
  health_poll_interval: float = 5.0,
95
98
  health_request_timeout: float = 5.0,
99
+ log_requests_level: int = 0,
96
100
  ):
97
101
  # In bench mode SGLang runs on worker_port + 10000 and a /bench proxy
98
102
  # listens on the original worker_port; both fall through to the same
@@ -124,6 +128,13 @@ class SGLangEndpoint(Endpoint):
124
128
  self.health_timeout = health_timeout
125
129
  self.health_poll_interval = health_poll_interval
126
130
  self.health_request_timeout = health_request_timeout
131
+
132
+ # SGLang Log request level: -1 = disabled, 0 = metadata only, 1,2,3 in increasing verbosity
133
+ # Refer to https://docs.sglang.io/docs/advanced_features/server_arguments#logging for more details
134
+ if not (log_requests_level >= -1 and log_requests_level <= 3):
135
+ raise ValueError("log_requests_level must be between -1 and 3")
136
+ self.log_requests_level = log_requests_level
137
+
127
138
  self._proc: Optional[subprocess.Popen] = None
128
139
  self._bench_server: Optional[http.server.ThreadingHTTPServer] = None
129
140
 
@@ -139,6 +150,32 @@ class SGLangEndpoint(Endpoint):
139
150
  ):
140
151
  self.load_format = "dummy"
141
152
  self.extra_server_args.setdefault("--model-loader-extra-config", "{}")
153
+
154
+ # Request logging enabled, log only metadata by default
155
+ if self.log_requests_level >= 0:
156
+ # Enable request logging
157
+ if "--log-requests" not in self.extra_server_args:
158
+ self.extra_server_args.setdefault("--log-requests", "")
159
+ # Set request logging level
160
+ if "--log-requests-level" not in self.extra_server_args:
161
+ self.extra_server_args.setdefault("--log-requests-level", str(self.log_requests_level))
162
+ else:
163
+ print(f"[endpoint] --log-requests-level already set to {self.extra_server_args['--log-requests-level']} in server args")
164
+
165
+ # Add Modal-forwarded request headers to SGLang logged headers
166
+ sglang_log_request_headers = [
167
+ h.strip().lower()
168
+ for h in os.environ.get("SGLANG_LOG_REQUEST_HEADERS", "").split(",")
169
+ if h.strip()
170
+ ]
171
+ if MODAL_FLASH_REQUEST_UUID_HEADER not in sglang_log_request_headers:
172
+ sglang_log_request_headers.append(MODAL_FLASH_REQUEST_UUID_HEADER)
173
+ if MODAL_SESSION_ID_HEADER not in sglang_log_request_headers:
174
+ sglang_log_request_headers.append(MODAL_SESSION_ID_HEADER)
175
+
176
+ self.log_request_headers = ",".join(sglang_log_request_headers)
177
+ os.environ["SGLANG_LOG_REQUEST_HEADERS"] = self.log_request_headers
178
+
142
179
 
143
180
  def _build_cmd(self) -> list[str]:
144
181
  cmd = [
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "autoinference-utils"
3
- version = "0.2.2"
3
+ version = "0.2.3"
4
4
  description = "Shared endpoint abstractions for autoinference deployments"
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"