inferlab-integration-specialized-engine 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,454 @@
1
+ """Planning and rendering for the shared token-only Specialized Engine contract."""
2
+
3
+ from importlib.metadata import PackageNotFoundError, version
4
+ from typing import Annotated
5
+
6
+ from inferlab_adapter_sdk import (
7
+ AdapterErrorCode,
8
+ AdapterOperationError,
9
+ CaptureTargetRequirement,
10
+ CaptureWindowControlEndpoint,
11
+ CaptureWindowControlRequirement,
12
+ EndpointProtocol,
13
+ EndpointRequirement,
14
+ FrontendCoRendering,
15
+ FrontendGatewayComponent,
16
+ FrontendProcessRole,
17
+ GatewayFrontendBinding,
18
+ GatewayPlan,
19
+ GatewayTarget,
20
+ GatewayTargetEngine,
21
+ HttpActionSpec,
22
+ HttpMethod,
23
+ IntegrationIdentity,
24
+ Parallelism,
25
+ ParallelismAttention,
26
+ ParallelismExperts,
27
+ ParallelismOuter,
28
+ PlanServeInput,
29
+ PlanServeResult,
30
+ ProcessSpec,
31
+ ReadinessProbe,
32
+ ReadinessProbeHttp,
33
+ ReadinessProbeProcessAlive,
34
+ RenderedServeProcess,
35
+ RenderServeInput,
36
+ RenderServeResult,
37
+ RenderSource,
38
+ ServeProcessAllocationFrontend,
39
+ ServeProcessAllocationModelRank,
40
+ ServeReplicaRequirement,
41
+ ServeRoleKind,
42
+ ServeRoleLink,
43
+ ServeRoleLinkRequestRouting,
44
+ ServeRoleResult,
45
+ ServeTopology,
46
+ SettingValue,
47
+ effective_settings,
48
+ integration_identity,
49
+ rendered_frontend,
50
+ rendered_model_rank,
51
+ replica_id,
52
+ require_role,
53
+ split_serve_allocations,
54
+ validate_settings,
55
+ )
56
+ from pydantic import BaseModel, ConfigDict, Field
57
+
58
+ _ADAPTER_DISTRIBUTION = "inferlab-integration-specialized-engine"
59
+ _GATEWAY_BACKEND = "smg"
60
+ _GATEWAY_IMPLEMENTATION = "tokenspeed-smg"
61
+ _DEFERRED_WORKER_STARTUP_TIMEOUT_SECS = 2_147_483_647
62
+
63
+
64
+ class EngineContractSettings(BaseModel):
65
+ """Settings shared by every implementation of the token Engine contract."""
66
+
67
+ model_config = ConfigDict(extra="forbid")
68
+
69
+ default_max_output_tokens: Annotated[int, Field(ge=1)] = 16
70
+ max_num_batched_tokens: Annotated[int, Field(ge=1)] = 12_288
71
+
72
+
73
+ def _identity() -> IntegrationIdentity:
74
+ return integration_identity(
75
+ adapter_id="inferlab-specialized-engine",
76
+ adapter_distribution=_ADAPTER_DISTRIBUTION,
77
+ framework="specialized-engine",
78
+ framework_distribution=_ADAPTER_DISTRIBUTION,
79
+ module_file=__file__,
80
+ )
81
+
82
+
83
+ def _smg_version() -> str:
84
+ try:
85
+ return version("tokenspeed-smg")
86
+ except PackageNotFoundError:
87
+ return "unavailable"
88
+
89
+
90
+ def _pure_tp_parallelism(
91
+ parallelism: Parallelism,
92
+ error_code: AdapterErrorCode = AdapterErrorCode.invalid_settings,
93
+ ) -> tuple[Parallelism, int]:
94
+ outer = parallelism.outer
95
+ tensor_parallel_size = (
96
+ outer.tensor_parallel_size
97
+ if outer is not None and outer.tensor_parallel_size is not None
98
+ else 1
99
+ )
100
+ non_tp_values = [
101
+ outer.pipeline_parallel_size if outer is not None else None,
102
+ (parallelism.attention.data_parallel_size if parallelism.attention is not None else None),
103
+ (
104
+ parallelism.attention.context_parallel_size
105
+ if parallelism.attention is not None
106
+ else None
107
+ ),
108
+ (parallelism.experts.data_parallel_size if parallelism.experts is not None else None),
109
+ (parallelism.experts.expert_parallel_size if parallelism.experts is not None else None),
110
+ ]
111
+ if any(value not in {None, 1} for value in non_tp_values):
112
+ raise AdapterOperationError(
113
+ error_code,
114
+ "the Specialized Engine contract supports only tensor parallelism",
115
+ )
116
+
117
+ component_tp_values = [
118
+ (parallelism.attention.tensor_parallel_size if parallelism.attention is not None else None),
119
+ (parallelism.experts.tensor_parallel_size if parallelism.experts is not None else None),
120
+ (
121
+ parallelism.experts.dense_tensor_parallel_size
122
+ if parallelism.experts is not None
123
+ else None
124
+ ),
125
+ ]
126
+ if any(value is not None and value != tensor_parallel_size for value in component_tp_values):
127
+ raise AdapterOperationError(
128
+ error_code,
129
+ "attention and expert tensor parallelism must match outer tensor parallelism",
130
+ )
131
+
132
+ effective = Parallelism(
133
+ outer=ParallelismOuter(
134
+ tensor_parallel_size=tensor_parallel_size,
135
+ pipeline_parallel_size=1,
136
+ ),
137
+ attention=ParallelismAttention(
138
+ tensor_parallel_size=tensor_parallel_size,
139
+ data_parallel_size=1,
140
+ context_parallel_size=1,
141
+ ),
142
+ experts=ParallelismExperts(
143
+ tensor_parallel_size=tensor_parallel_size,
144
+ data_parallel_size=1,
145
+ expert_parallel_size=1,
146
+ dense_tensor_parallel_size=tensor_parallel_size,
147
+ ),
148
+ )
149
+ return effective, tensor_parallel_size
150
+
151
+
152
+ def _public_endpoint() -> EndpointRequirement:
153
+ return EndpointRequirement(
154
+ protocol=EndpointProtocol(),
155
+ completions_path="/v1/completions",
156
+ chat_completions_path="/v1/chat/completions",
157
+ prefix_cache_reset=HttpActionSpec(method=HttpMethod(), path="/flush_cache"),
158
+ )
159
+
160
+
161
+ def plan_serve(input: PlanServeInput) -> PlanServeResult:
162
+ """Plan one token Engine behind one SMG Gateway."""
163
+ if input.topology != ServeTopology.single:
164
+ raise AdapterOperationError(
165
+ AdapterErrorCode.invalid_settings,
166
+ "the Specialized Engine integration supports only single topology",
167
+ )
168
+ if input.gateway_backend != _GATEWAY_BACKEND:
169
+ raise AdapterOperationError(
170
+ AdapterErrorCode.invalid_settings,
171
+ "the Specialized Engine integration requires Gateway backend smg",
172
+ )
173
+ if input.pd_router_backend is not None:
174
+ raise AdapterOperationError(
175
+ AdapterErrorCode.invalid_settings,
176
+ "the Specialized Engine routed-single workflow must not select a P/D Router",
177
+ )
178
+ if input.kv_transfer is not None:
179
+ raise AdapterOperationError(
180
+ AdapterErrorCode.invalid_settings,
181
+ "single topology does not use KV transfer",
182
+ )
183
+ role = require_role(input, ServeRoleKind.serve)
184
+ if role.replica_count != 1:
185
+ raise AdapterOperationError(
186
+ AdapterErrorCode.invalid_settings,
187
+ "the Specialized Engine integration supports exactly one replica",
188
+ )
189
+ settings = validate_settings(EngineContractSettings, role.settings)
190
+ parallelism, tensor_parallel_size = _pure_tp_parallelism(role.parallelism)
191
+ role_result = ServeRoleResult(
192
+ id=role.id,
193
+ kind=role.kind,
194
+ declared_replica_count=role.replica_count,
195
+ effective_replica_count=role.replica_count,
196
+ effective_settings=effective_settings(settings),
197
+ effective_parallelism=parallelism,
198
+ public_endpoint=None,
199
+ )
200
+ gateway = GatewayPlan(
201
+ backend=_GATEWAY_BACKEND,
202
+ implementation=_GATEWAY_IMPLEMENTATION,
203
+ implementation_version=_smg_version(),
204
+ effective_settings={
205
+ "worker_protocol": SettingValue(root="tokenspeed_scheduler_v1"),
206
+ "policy": SettingValue(root="passthrough"),
207
+ "retries": SettingValue(root=False),
208
+ "circuit_breaker": SettingValue(root=False),
209
+ },
210
+ endpoint=_public_endpoint(),
211
+ readiness=ReadinessProbe(root=ReadinessProbeHttp(path="/readiness")),
212
+ ports=["prometheus"],
213
+ targets=[GatewayTarget(root=GatewayTargetEngine(role=role.id))],
214
+ render_inputs=[],
215
+ render_source=RenderSource.integration,
216
+ co_rendering=FrontendCoRendering(process_role=FrontendProcessRole()),
217
+ )
218
+ return PlanServeResult(
219
+ integration=_identity(),
220
+ roles=[role_result],
221
+ replicas=[
222
+ ServeReplicaRequirement(
223
+ id=replica_id(role, 0),
224
+ role_id=role.id,
225
+ replica_index=0,
226
+ device_count=tensor_parallel_size,
227
+ ports=[],
228
+ primary_ports=[],
229
+ primary_readiness=ReadinessProbe(root=ReadinessProbeProcessAlive()),
230
+ worker_readiness=ReadinessProbe(root=ReadinessProbeProcessAlive()),
231
+ capture_target=(
232
+ CaptureTargetRequirement(
233
+ window_control=CaptureWindowControlRequirement(
234
+ endpoint=CaptureWindowControlEndpoint.gateway,
235
+ start=HttpActionSpec(method=HttpMethod(), path="/start_profile"),
236
+ stop=HttpActionSpec(method=HttpMethod(), path="/stop_profile"),
237
+ )
238
+ )
239
+ if input.profiling
240
+ else None
241
+ ),
242
+ )
243
+ ],
244
+ links=[
245
+ ServeRoleLink(root=ServeRoleLinkRequestRouting(source="gateway", targets=[role.id]))
246
+ ],
247
+ gateway=gateway,
248
+ pd_router=None,
249
+ )
250
+
251
+
252
+ def _require_engine(
253
+ allocations: list[ServeProcessAllocationModelRank],
254
+ ) -> ServeProcessAllocationModelRank:
255
+ if len(allocations) != 1:
256
+ raise AdapterOperationError(
257
+ AdapterErrorCode.invalid_request,
258
+ "the Specialized Engine integration requires one model-rank allocation",
259
+ )
260
+ engine = allocations[0]
261
+ if (
262
+ engine.role_kind != ServeRoleKind.serve
263
+ or engine.replica != 0
264
+ or engine.rank != 0
265
+ or engine.rank_count != 1
266
+ ):
267
+ raise AdapterOperationError(
268
+ AdapterErrorCode.invalid_request,
269
+ "the Engine allocation must be serve replica 0 in one rank process",
270
+ )
271
+ effective_parallelism, tensor_parallel_size = _pure_tp_parallelism(
272
+ engine.effective_parallelism,
273
+ AdapterErrorCode.invalid_request,
274
+ )
275
+ if (
276
+ engine.effective_parallelism != effective_parallelism
277
+ or len(engine.devices) != tensor_parallel_size
278
+ ):
279
+ raise AdapterOperationError(
280
+ AdapterErrorCode.invalid_request,
281
+ "the Engine rank process must own one device per effective tensor-parallel rank",
282
+ )
283
+ if engine.endpoint is None:
284
+ raise AdapterOperationError(
285
+ AdapterErrorCode.invalid_request,
286
+ "the Engine allocation is missing its endpoint",
287
+ )
288
+ return engine
289
+
290
+
291
+ def _require_gateway(allocation: object, engine_role: str) -> ServeProcessAllocationFrontend:
292
+ if not isinstance(allocation, ServeProcessAllocationFrontend):
293
+ raise AdapterOperationError(
294
+ AdapterErrorCode.invalid_request,
295
+ "the routed-single workflow requires one Gateway frontend allocation",
296
+ )
297
+ if not isinstance(allocation.components.root, GatewayFrontendBinding):
298
+ raise AdapterOperationError(
299
+ AdapterErrorCode.invalid_request,
300
+ "the routed-single frontend must bind only [gateway]",
301
+ )
302
+ if allocation.components.root.root != [FrontendGatewayComponent()]:
303
+ raise AdapterOperationError(
304
+ AdapterErrorCode.invalid_request,
305
+ "the routed-single frontend must bind only [gateway]",
306
+ )
307
+ gateway = allocation.gateway
308
+ if (
309
+ gateway.backend != _GATEWAY_BACKEND
310
+ or gateway.implementation != _GATEWAY_IMPLEMENTATION
311
+ or gateway.implementation_version != _smg_version()
312
+ or gateway.render_source != RenderSource.integration
313
+ or allocation.pd_router is not None
314
+ or allocation.process_role != gateway.co_rendering.process_role
315
+ ):
316
+ raise AdapterOperationError(
317
+ AdapterErrorCode.invalid_request,
318
+ "the frontend allocation does not preserve the planned SMG Gateway",
319
+ )
320
+ targets = gateway.targets
321
+ if (
322
+ len(targets) != 1
323
+ or not isinstance(targets[0].root, GatewayTargetEngine)
324
+ or targets[0].root.role != engine_role
325
+ ):
326
+ raise AdapterOperationError(
327
+ AdapterErrorCode.invalid_request,
328
+ "the SMG Gateway must target the sole Engine role",
329
+ )
330
+ return allocation
331
+
332
+
333
+ def _render_engine(
334
+ input: RenderServeInput,
335
+ allocation: ServeProcessAllocationModelRank,
336
+ ) -> RenderedServeProcess:
337
+ endpoint = allocation.endpoint
338
+ if endpoint is None:
339
+ raise AdapterOperationError(
340
+ AdapterErrorCode.invalid_request,
341
+ "the Engine allocation is missing its endpoint",
342
+ )
343
+ settings = validate_settings(EngineContractSettings, allocation.effective_settings)
344
+ _, tensor_parallel_size = _pure_tp_parallelism(
345
+ allocation.effective_parallelism,
346
+ AdapterErrorCode.invalid_request,
347
+ )
348
+ return rendered_model_rank(
349
+ allocation,
350
+ ProcessSpec(
351
+ argv=[
352
+ "inferlab-token-engine",
353
+ "smg-worker",
354
+ "--listen",
355
+ f"{endpoint.host}:{endpoint.port}",
356
+ "--model",
357
+ allocation.model_locator,
358
+ "--served-model-name",
359
+ input.model.served_name,
360
+ "--tensor-parallel-size",
361
+ str(tensor_parallel_size),
362
+ "--default-max-output-tokens",
363
+ str(settings.default_max_output_tokens),
364
+ "--max-num-batched-tokens",
365
+ str(settings.max_num_batched_tokens),
366
+ ],
367
+ env={},
368
+ ),
369
+ )
370
+
371
+
372
+ def _render_gateway(
373
+ allocation: ServeProcessAllocationFrontend,
374
+ engine: ServeProcessAllocationModelRank,
375
+ ) -> RenderedServeProcess:
376
+ engine_endpoint = engine.endpoint
377
+ if engine_endpoint is None:
378
+ raise AdapterOperationError(
379
+ AdapterErrorCode.invalid_request,
380
+ "the Engine allocation is missing its endpoint",
381
+ )
382
+ prometheus = allocation.ports.get("prometheus")
383
+ if prometheus is None:
384
+ raise AdapterOperationError(
385
+ AdapterErrorCode.invalid_request,
386
+ "the SMG Gateway allocation is missing its Prometheus port",
387
+ )
388
+ return rendered_frontend(
389
+ allocation,
390
+ ProcessSpec(
391
+ argv=[
392
+ "smg",
393
+ "launch",
394
+ "--host",
395
+ "0.0.0.0",
396
+ "--port",
397
+ str(allocation.endpoint.port),
398
+ "--prometheus-port",
399
+ str(prometheus.port),
400
+ "--worker-startup-timeout-secs",
401
+ str(_DEFERRED_WORKER_STARTUP_TIMEOUT_SECS),
402
+ "--worker-urls",
403
+ f"grpc://{engine_endpoint.host}:{engine_endpoint.port}",
404
+ "--model-path",
405
+ engine.model_locator,
406
+ "--tokenizer-path",
407
+ engine.model_locator,
408
+ "--policy",
409
+ "passthrough",
410
+ "--disable-retries",
411
+ "--disable-circuit-breaker",
412
+ ],
413
+ env={},
414
+ ),
415
+ )
416
+
417
+
418
+ def render_serve(input: RenderServeInput) -> RenderServeResult:
419
+ if (
420
+ input.topology != ServeTopology.single
421
+ or input.gateway_backend != _GATEWAY_BACKEND
422
+ or input.pd_router_backend is not None
423
+ or input.kv_transfer is not None
424
+ ):
425
+ raise AdapterOperationError(
426
+ AdapterErrorCode.invalid_request,
427
+ "render input is not the planned routed-single SMG workflow",
428
+ )
429
+ allocations, model_allocations = split_serve_allocations(input.allocations)
430
+ if len(allocations) != 2:
431
+ raise AdapterOperationError(
432
+ AdapterErrorCode.invalid_request,
433
+ "the routed-single workflow requires one Engine and one Gateway allocation",
434
+ )
435
+ engine = _require_engine(model_allocations)
436
+ frontend_candidates = [
437
+ allocation
438
+ for allocation in allocations
439
+ if isinstance(allocation, ServeProcessAllocationFrontend)
440
+ ]
441
+ if len(frontend_candidates) != 1:
442
+ raise AdapterOperationError(
443
+ AdapterErrorCode.invalid_request,
444
+ "the routed-single workflow requires one Gateway allocation",
445
+ )
446
+ _require_gateway(frontend_candidates[0], engine.role)
447
+
448
+ processes: list[RenderedServeProcess] = []
449
+ for allocation in allocations:
450
+ if isinstance(allocation, ServeProcessAllocationModelRank):
451
+ processes.append(_render_engine(input, allocation))
452
+ elif isinstance(allocation, ServeProcessAllocationFrontend):
453
+ processes.append(_render_gateway(allocation, engine))
454
+ return RenderServeResult(integration=_identity(), processes=processes)
@@ -0,0 +1,11 @@
1
+ from inferlab_adapter_sdk import run_adapter
2
+
3
+ from . import plan_serve, render_serve
4
+
5
+
6
+ def main() -> None:
7
+ raise SystemExit(run_adapter(plan_serve, render_serve=render_serve))
8
+
9
+
10
+ if __name__ == "__main__":
11
+ main()
@@ -0,0 +1,33 @@
1
+ Metadata-Version: 2.4
2
+ Name: inferlab-integration-specialized-engine
3
+ Version: 0.2.0
4
+ Summary: Inferlab adapter integration for token-only Specialized Engines
5
+ Project-URL: Repository, https://github.com/Infer-Lab/InferLab
6
+ License-Expression: MIT
7
+ License-File: LICENSE
8
+ Requires-Python: >=3.12
9
+ Requires-Dist: inferlab-adapter-sdk==0.5.0
10
+ Description-Content-Type: text/markdown
11
+
12
+ # Inferlab Integration for Specialized Engines
13
+
14
+ This package connects Inferlab to a hardware-by-model Specialized Engine that
15
+ implements the canonical `inferlab-token-engine smg-worker` command. The Engine
16
+ accepts prompt token IDs and returns generated token IDs over SMG's worker
17
+ protocol. TokenSpeed SMG remains responsible for the public HTTP API,
18
+ tokenization, chat templates, detokenization, and response formatting.
19
+
20
+ The integration contains no model-, architecture-, hardware-, or
21
+ Engine-implementation-specific lowering. A downstream workspace supplies the
22
+ Rust Engine binary, SMG, model intent, source revision, locked environment, and
23
+ private placement bindings. Consequently, a new conforming Engine does not
24
+ need another Inferlab integration package.
25
+
26
+ The supported shape is deliberately closed: one `single` Engine replica in one
27
+ process behind one SMG Gateway. The process owns an arbitrary nonzero pure-TP
28
+ device set; attention, expert, and dense-expert tensor parallelism all equal
29
+ the outer TP width, while pipeline, data, context, and expert parallelism stay
30
+ at one. The contract remains serial and has no P/D Router, KV-transfer,
31
+ batching, or Engine-local profiling surface. InferLab can profile the Engine
32
+ process tree while TokenSpeed SMG exposes the capture-window
33
+ `POST /start_profile` and `POST /stop_profile` actions on the Gateway.
@@ -0,0 +1,7 @@
1
+ inferlab_integration_specialized_engine/__init__.py,sha256=nIhQRyuKpHDC0JynfJ3hk2GIqDCsZ2Ttx2dgLQhzdT4,16347
2
+ inferlab_integration_specialized_engine/__main__.py,sha256=e1uoS23_cXd5sMrRK_nSuUPgdz37Tx1xBhxqMlCYhBg,220
3
+ inferlab_integration_specialized_engine-0.2.0.dist-info/METADATA,sha256=v5AnpmvUcWT73awqu2hSmTEDhHi0jPinY-fAOFZDKDA,1731
4
+ inferlab_integration_specialized_engine-0.2.0.dist-info/WHEEL,sha256=mffPy8wBnZQn2VnJUU5jE99KsxaSfiyMHV9Yt0aLVxs,87
5
+ inferlab_integration_specialized_engine-0.2.0.dist-info/entry_points.txt,sha256=rXD9hNQ3ll_gLI7tVT5Dsu2WR9BApLuiOCiWHzALbZg,110
6
+ inferlab_integration_specialized_engine-0.2.0.dist-info/licenses/LICENSE,sha256=_sFBKPJJjTLuE3lL3Ir9HDHTTrB9zsZtMBAOvBfBWA4,1065
7
+ inferlab_integration_specialized_engine-0.2.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.30.1
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ inferlab-adapter-specialized-engine = inferlab_integration_specialized_engine.__main__:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Zihua Wu
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.