inferlab-integration-specialized-engine 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- inferlab_integration_specialized_engine/__init__.py +454 -0
- inferlab_integration_specialized_engine/__main__.py +11 -0
- inferlab_integration_specialized_engine-0.2.0.dist-info/METADATA +33 -0
- inferlab_integration_specialized_engine-0.2.0.dist-info/RECORD +7 -0
- inferlab_integration_specialized_engine-0.2.0.dist-info/WHEEL +4 -0
- inferlab_integration_specialized_engine-0.2.0.dist-info/entry_points.txt +2 -0
- inferlab_integration_specialized_engine-0.2.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,454 @@
|
|
|
1
|
+
"""Planning and rendering for the shared token-only Specialized Engine contract."""
|
|
2
|
+
|
|
3
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
4
|
+
from typing import Annotated
|
|
5
|
+
|
|
6
|
+
from inferlab_adapter_sdk import (
|
|
7
|
+
AdapterErrorCode,
|
|
8
|
+
AdapterOperationError,
|
|
9
|
+
CaptureTargetRequirement,
|
|
10
|
+
CaptureWindowControlEndpoint,
|
|
11
|
+
CaptureWindowControlRequirement,
|
|
12
|
+
EndpointProtocol,
|
|
13
|
+
EndpointRequirement,
|
|
14
|
+
FrontendCoRendering,
|
|
15
|
+
FrontendGatewayComponent,
|
|
16
|
+
FrontendProcessRole,
|
|
17
|
+
GatewayFrontendBinding,
|
|
18
|
+
GatewayPlan,
|
|
19
|
+
GatewayTarget,
|
|
20
|
+
GatewayTargetEngine,
|
|
21
|
+
HttpActionSpec,
|
|
22
|
+
HttpMethod,
|
|
23
|
+
IntegrationIdentity,
|
|
24
|
+
Parallelism,
|
|
25
|
+
ParallelismAttention,
|
|
26
|
+
ParallelismExperts,
|
|
27
|
+
ParallelismOuter,
|
|
28
|
+
PlanServeInput,
|
|
29
|
+
PlanServeResult,
|
|
30
|
+
ProcessSpec,
|
|
31
|
+
ReadinessProbe,
|
|
32
|
+
ReadinessProbeHttp,
|
|
33
|
+
ReadinessProbeProcessAlive,
|
|
34
|
+
RenderedServeProcess,
|
|
35
|
+
RenderServeInput,
|
|
36
|
+
RenderServeResult,
|
|
37
|
+
RenderSource,
|
|
38
|
+
ServeProcessAllocationFrontend,
|
|
39
|
+
ServeProcessAllocationModelRank,
|
|
40
|
+
ServeReplicaRequirement,
|
|
41
|
+
ServeRoleKind,
|
|
42
|
+
ServeRoleLink,
|
|
43
|
+
ServeRoleLinkRequestRouting,
|
|
44
|
+
ServeRoleResult,
|
|
45
|
+
ServeTopology,
|
|
46
|
+
SettingValue,
|
|
47
|
+
effective_settings,
|
|
48
|
+
integration_identity,
|
|
49
|
+
rendered_frontend,
|
|
50
|
+
rendered_model_rank,
|
|
51
|
+
replica_id,
|
|
52
|
+
require_role,
|
|
53
|
+
split_serve_allocations,
|
|
54
|
+
validate_settings,
|
|
55
|
+
)
|
|
56
|
+
from pydantic import BaseModel, ConfigDict, Field
|
|
57
|
+
|
|
58
|
+
_ADAPTER_DISTRIBUTION = "inferlab-integration-specialized-engine"
|
|
59
|
+
_GATEWAY_BACKEND = "smg"
|
|
60
|
+
_GATEWAY_IMPLEMENTATION = "tokenspeed-smg"
|
|
61
|
+
_DEFERRED_WORKER_STARTUP_TIMEOUT_SECS = 2_147_483_647
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class EngineContractSettings(BaseModel):
|
|
65
|
+
"""Settings shared by every implementation of the token Engine contract."""
|
|
66
|
+
|
|
67
|
+
model_config = ConfigDict(extra="forbid")
|
|
68
|
+
|
|
69
|
+
default_max_output_tokens: Annotated[int, Field(ge=1)] = 16
|
|
70
|
+
max_num_batched_tokens: Annotated[int, Field(ge=1)] = 12_288
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _identity() -> IntegrationIdentity:
|
|
74
|
+
return integration_identity(
|
|
75
|
+
adapter_id="inferlab-specialized-engine",
|
|
76
|
+
adapter_distribution=_ADAPTER_DISTRIBUTION,
|
|
77
|
+
framework="specialized-engine",
|
|
78
|
+
framework_distribution=_ADAPTER_DISTRIBUTION,
|
|
79
|
+
module_file=__file__,
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _smg_version() -> str:
|
|
84
|
+
try:
|
|
85
|
+
return version("tokenspeed-smg")
|
|
86
|
+
except PackageNotFoundError:
|
|
87
|
+
return "unavailable"
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _pure_tp_parallelism(
|
|
91
|
+
parallelism: Parallelism,
|
|
92
|
+
error_code: AdapterErrorCode = AdapterErrorCode.invalid_settings,
|
|
93
|
+
) -> tuple[Parallelism, int]:
|
|
94
|
+
outer = parallelism.outer
|
|
95
|
+
tensor_parallel_size = (
|
|
96
|
+
outer.tensor_parallel_size
|
|
97
|
+
if outer is not None and outer.tensor_parallel_size is not None
|
|
98
|
+
else 1
|
|
99
|
+
)
|
|
100
|
+
non_tp_values = [
|
|
101
|
+
outer.pipeline_parallel_size if outer is not None else None,
|
|
102
|
+
(parallelism.attention.data_parallel_size if parallelism.attention is not None else None),
|
|
103
|
+
(
|
|
104
|
+
parallelism.attention.context_parallel_size
|
|
105
|
+
if parallelism.attention is not None
|
|
106
|
+
else None
|
|
107
|
+
),
|
|
108
|
+
(parallelism.experts.data_parallel_size if parallelism.experts is not None else None),
|
|
109
|
+
(parallelism.experts.expert_parallel_size if parallelism.experts is not None else None),
|
|
110
|
+
]
|
|
111
|
+
if any(value not in {None, 1} for value in non_tp_values):
|
|
112
|
+
raise AdapterOperationError(
|
|
113
|
+
error_code,
|
|
114
|
+
"the Specialized Engine contract supports only tensor parallelism",
|
|
115
|
+
)
|
|
116
|
+
|
|
117
|
+
component_tp_values = [
|
|
118
|
+
(parallelism.attention.tensor_parallel_size if parallelism.attention is not None else None),
|
|
119
|
+
(parallelism.experts.tensor_parallel_size if parallelism.experts is not None else None),
|
|
120
|
+
(
|
|
121
|
+
parallelism.experts.dense_tensor_parallel_size
|
|
122
|
+
if parallelism.experts is not None
|
|
123
|
+
else None
|
|
124
|
+
),
|
|
125
|
+
]
|
|
126
|
+
if any(value is not None and value != tensor_parallel_size for value in component_tp_values):
|
|
127
|
+
raise AdapterOperationError(
|
|
128
|
+
error_code,
|
|
129
|
+
"attention and expert tensor parallelism must match outer tensor parallelism",
|
|
130
|
+
)
|
|
131
|
+
|
|
132
|
+
effective = Parallelism(
|
|
133
|
+
outer=ParallelismOuter(
|
|
134
|
+
tensor_parallel_size=tensor_parallel_size,
|
|
135
|
+
pipeline_parallel_size=1,
|
|
136
|
+
),
|
|
137
|
+
attention=ParallelismAttention(
|
|
138
|
+
tensor_parallel_size=tensor_parallel_size,
|
|
139
|
+
data_parallel_size=1,
|
|
140
|
+
context_parallel_size=1,
|
|
141
|
+
),
|
|
142
|
+
experts=ParallelismExperts(
|
|
143
|
+
tensor_parallel_size=tensor_parallel_size,
|
|
144
|
+
data_parallel_size=1,
|
|
145
|
+
expert_parallel_size=1,
|
|
146
|
+
dense_tensor_parallel_size=tensor_parallel_size,
|
|
147
|
+
),
|
|
148
|
+
)
|
|
149
|
+
return effective, tensor_parallel_size
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _public_endpoint() -> EndpointRequirement:
|
|
153
|
+
return EndpointRequirement(
|
|
154
|
+
protocol=EndpointProtocol(),
|
|
155
|
+
completions_path="/v1/completions",
|
|
156
|
+
chat_completions_path="/v1/chat/completions",
|
|
157
|
+
prefix_cache_reset=HttpActionSpec(method=HttpMethod(), path="/flush_cache"),
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def plan_serve(input: PlanServeInput) -> PlanServeResult:
|
|
162
|
+
"""Plan one token Engine behind one SMG Gateway."""
|
|
163
|
+
if input.topology != ServeTopology.single:
|
|
164
|
+
raise AdapterOperationError(
|
|
165
|
+
AdapterErrorCode.invalid_settings,
|
|
166
|
+
"the Specialized Engine integration supports only single topology",
|
|
167
|
+
)
|
|
168
|
+
if input.gateway_backend != _GATEWAY_BACKEND:
|
|
169
|
+
raise AdapterOperationError(
|
|
170
|
+
AdapterErrorCode.invalid_settings,
|
|
171
|
+
"the Specialized Engine integration requires Gateway backend smg",
|
|
172
|
+
)
|
|
173
|
+
if input.pd_router_backend is not None:
|
|
174
|
+
raise AdapterOperationError(
|
|
175
|
+
AdapterErrorCode.invalid_settings,
|
|
176
|
+
"the Specialized Engine routed-single workflow must not select a P/D Router",
|
|
177
|
+
)
|
|
178
|
+
if input.kv_transfer is not None:
|
|
179
|
+
raise AdapterOperationError(
|
|
180
|
+
AdapterErrorCode.invalid_settings,
|
|
181
|
+
"single topology does not use KV transfer",
|
|
182
|
+
)
|
|
183
|
+
role = require_role(input, ServeRoleKind.serve)
|
|
184
|
+
if role.replica_count != 1:
|
|
185
|
+
raise AdapterOperationError(
|
|
186
|
+
AdapterErrorCode.invalid_settings,
|
|
187
|
+
"the Specialized Engine integration supports exactly one replica",
|
|
188
|
+
)
|
|
189
|
+
settings = validate_settings(EngineContractSettings, role.settings)
|
|
190
|
+
parallelism, tensor_parallel_size = _pure_tp_parallelism(role.parallelism)
|
|
191
|
+
role_result = ServeRoleResult(
|
|
192
|
+
id=role.id,
|
|
193
|
+
kind=role.kind,
|
|
194
|
+
declared_replica_count=role.replica_count,
|
|
195
|
+
effective_replica_count=role.replica_count,
|
|
196
|
+
effective_settings=effective_settings(settings),
|
|
197
|
+
effective_parallelism=parallelism,
|
|
198
|
+
public_endpoint=None,
|
|
199
|
+
)
|
|
200
|
+
gateway = GatewayPlan(
|
|
201
|
+
backend=_GATEWAY_BACKEND,
|
|
202
|
+
implementation=_GATEWAY_IMPLEMENTATION,
|
|
203
|
+
implementation_version=_smg_version(),
|
|
204
|
+
effective_settings={
|
|
205
|
+
"worker_protocol": SettingValue(root="tokenspeed_scheduler_v1"),
|
|
206
|
+
"policy": SettingValue(root="passthrough"),
|
|
207
|
+
"retries": SettingValue(root=False),
|
|
208
|
+
"circuit_breaker": SettingValue(root=False),
|
|
209
|
+
},
|
|
210
|
+
endpoint=_public_endpoint(),
|
|
211
|
+
readiness=ReadinessProbe(root=ReadinessProbeHttp(path="/readiness")),
|
|
212
|
+
ports=["prometheus"],
|
|
213
|
+
targets=[GatewayTarget(root=GatewayTargetEngine(role=role.id))],
|
|
214
|
+
render_inputs=[],
|
|
215
|
+
render_source=RenderSource.integration,
|
|
216
|
+
co_rendering=FrontendCoRendering(process_role=FrontendProcessRole()),
|
|
217
|
+
)
|
|
218
|
+
return PlanServeResult(
|
|
219
|
+
integration=_identity(),
|
|
220
|
+
roles=[role_result],
|
|
221
|
+
replicas=[
|
|
222
|
+
ServeReplicaRequirement(
|
|
223
|
+
id=replica_id(role, 0),
|
|
224
|
+
role_id=role.id,
|
|
225
|
+
replica_index=0,
|
|
226
|
+
device_count=tensor_parallel_size,
|
|
227
|
+
ports=[],
|
|
228
|
+
primary_ports=[],
|
|
229
|
+
primary_readiness=ReadinessProbe(root=ReadinessProbeProcessAlive()),
|
|
230
|
+
worker_readiness=ReadinessProbe(root=ReadinessProbeProcessAlive()),
|
|
231
|
+
capture_target=(
|
|
232
|
+
CaptureTargetRequirement(
|
|
233
|
+
window_control=CaptureWindowControlRequirement(
|
|
234
|
+
endpoint=CaptureWindowControlEndpoint.gateway,
|
|
235
|
+
start=HttpActionSpec(method=HttpMethod(), path="/start_profile"),
|
|
236
|
+
stop=HttpActionSpec(method=HttpMethod(), path="/stop_profile"),
|
|
237
|
+
)
|
|
238
|
+
)
|
|
239
|
+
if input.profiling
|
|
240
|
+
else None
|
|
241
|
+
),
|
|
242
|
+
)
|
|
243
|
+
],
|
|
244
|
+
links=[
|
|
245
|
+
ServeRoleLink(root=ServeRoleLinkRequestRouting(source="gateway", targets=[role.id]))
|
|
246
|
+
],
|
|
247
|
+
gateway=gateway,
|
|
248
|
+
pd_router=None,
|
|
249
|
+
)
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _require_engine(
|
|
253
|
+
allocations: list[ServeProcessAllocationModelRank],
|
|
254
|
+
) -> ServeProcessAllocationModelRank:
|
|
255
|
+
if len(allocations) != 1:
|
|
256
|
+
raise AdapterOperationError(
|
|
257
|
+
AdapterErrorCode.invalid_request,
|
|
258
|
+
"the Specialized Engine integration requires one model-rank allocation",
|
|
259
|
+
)
|
|
260
|
+
engine = allocations[0]
|
|
261
|
+
if (
|
|
262
|
+
engine.role_kind != ServeRoleKind.serve
|
|
263
|
+
or engine.replica != 0
|
|
264
|
+
or engine.rank != 0
|
|
265
|
+
or engine.rank_count != 1
|
|
266
|
+
):
|
|
267
|
+
raise AdapterOperationError(
|
|
268
|
+
AdapterErrorCode.invalid_request,
|
|
269
|
+
"the Engine allocation must be serve replica 0 in one rank process",
|
|
270
|
+
)
|
|
271
|
+
effective_parallelism, tensor_parallel_size = _pure_tp_parallelism(
|
|
272
|
+
engine.effective_parallelism,
|
|
273
|
+
AdapterErrorCode.invalid_request,
|
|
274
|
+
)
|
|
275
|
+
if (
|
|
276
|
+
engine.effective_parallelism != effective_parallelism
|
|
277
|
+
or len(engine.devices) != tensor_parallel_size
|
|
278
|
+
):
|
|
279
|
+
raise AdapterOperationError(
|
|
280
|
+
AdapterErrorCode.invalid_request,
|
|
281
|
+
"the Engine rank process must own one device per effective tensor-parallel rank",
|
|
282
|
+
)
|
|
283
|
+
if engine.endpoint is None:
|
|
284
|
+
raise AdapterOperationError(
|
|
285
|
+
AdapterErrorCode.invalid_request,
|
|
286
|
+
"the Engine allocation is missing its endpoint",
|
|
287
|
+
)
|
|
288
|
+
return engine
|
|
289
|
+
|
|
290
|
+
|
|
291
|
+
def _require_gateway(allocation: object, engine_role: str) -> ServeProcessAllocationFrontend:
|
|
292
|
+
if not isinstance(allocation, ServeProcessAllocationFrontend):
|
|
293
|
+
raise AdapterOperationError(
|
|
294
|
+
AdapterErrorCode.invalid_request,
|
|
295
|
+
"the routed-single workflow requires one Gateway frontend allocation",
|
|
296
|
+
)
|
|
297
|
+
if not isinstance(allocation.components.root, GatewayFrontendBinding):
|
|
298
|
+
raise AdapterOperationError(
|
|
299
|
+
AdapterErrorCode.invalid_request,
|
|
300
|
+
"the routed-single frontend must bind only [gateway]",
|
|
301
|
+
)
|
|
302
|
+
if allocation.components.root.root != [FrontendGatewayComponent()]:
|
|
303
|
+
raise AdapterOperationError(
|
|
304
|
+
AdapterErrorCode.invalid_request,
|
|
305
|
+
"the routed-single frontend must bind only [gateway]",
|
|
306
|
+
)
|
|
307
|
+
gateway = allocation.gateway
|
|
308
|
+
if (
|
|
309
|
+
gateway.backend != _GATEWAY_BACKEND
|
|
310
|
+
or gateway.implementation != _GATEWAY_IMPLEMENTATION
|
|
311
|
+
or gateway.implementation_version != _smg_version()
|
|
312
|
+
or gateway.render_source != RenderSource.integration
|
|
313
|
+
or allocation.pd_router is not None
|
|
314
|
+
or allocation.process_role != gateway.co_rendering.process_role
|
|
315
|
+
):
|
|
316
|
+
raise AdapterOperationError(
|
|
317
|
+
AdapterErrorCode.invalid_request,
|
|
318
|
+
"the frontend allocation does not preserve the planned SMG Gateway",
|
|
319
|
+
)
|
|
320
|
+
targets = gateway.targets
|
|
321
|
+
if (
|
|
322
|
+
len(targets) != 1
|
|
323
|
+
or not isinstance(targets[0].root, GatewayTargetEngine)
|
|
324
|
+
or targets[0].root.role != engine_role
|
|
325
|
+
):
|
|
326
|
+
raise AdapterOperationError(
|
|
327
|
+
AdapterErrorCode.invalid_request,
|
|
328
|
+
"the SMG Gateway must target the sole Engine role",
|
|
329
|
+
)
|
|
330
|
+
return allocation
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def _render_engine(
|
|
334
|
+
input: RenderServeInput,
|
|
335
|
+
allocation: ServeProcessAllocationModelRank,
|
|
336
|
+
) -> RenderedServeProcess:
|
|
337
|
+
endpoint = allocation.endpoint
|
|
338
|
+
if endpoint is None:
|
|
339
|
+
raise AdapterOperationError(
|
|
340
|
+
AdapterErrorCode.invalid_request,
|
|
341
|
+
"the Engine allocation is missing its endpoint",
|
|
342
|
+
)
|
|
343
|
+
settings = validate_settings(EngineContractSettings, allocation.effective_settings)
|
|
344
|
+
_, tensor_parallel_size = _pure_tp_parallelism(
|
|
345
|
+
allocation.effective_parallelism,
|
|
346
|
+
AdapterErrorCode.invalid_request,
|
|
347
|
+
)
|
|
348
|
+
return rendered_model_rank(
|
|
349
|
+
allocation,
|
|
350
|
+
ProcessSpec(
|
|
351
|
+
argv=[
|
|
352
|
+
"inferlab-token-engine",
|
|
353
|
+
"smg-worker",
|
|
354
|
+
"--listen",
|
|
355
|
+
f"{endpoint.host}:{endpoint.port}",
|
|
356
|
+
"--model",
|
|
357
|
+
allocation.model_locator,
|
|
358
|
+
"--served-model-name",
|
|
359
|
+
input.model.served_name,
|
|
360
|
+
"--tensor-parallel-size",
|
|
361
|
+
str(tensor_parallel_size),
|
|
362
|
+
"--default-max-output-tokens",
|
|
363
|
+
str(settings.default_max_output_tokens),
|
|
364
|
+
"--max-num-batched-tokens",
|
|
365
|
+
str(settings.max_num_batched_tokens),
|
|
366
|
+
],
|
|
367
|
+
env={},
|
|
368
|
+
),
|
|
369
|
+
)
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def _render_gateway(
|
|
373
|
+
allocation: ServeProcessAllocationFrontend,
|
|
374
|
+
engine: ServeProcessAllocationModelRank,
|
|
375
|
+
) -> RenderedServeProcess:
|
|
376
|
+
engine_endpoint = engine.endpoint
|
|
377
|
+
if engine_endpoint is None:
|
|
378
|
+
raise AdapterOperationError(
|
|
379
|
+
AdapterErrorCode.invalid_request,
|
|
380
|
+
"the Engine allocation is missing its endpoint",
|
|
381
|
+
)
|
|
382
|
+
prometheus = allocation.ports.get("prometheus")
|
|
383
|
+
if prometheus is None:
|
|
384
|
+
raise AdapterOperationError(
|
|
385
|
+
AdapterErrorCode.invalid_request,
|
|
386
|
+
"the SMG Gateway allocation is missing its Prometheus port",
|
|
387
|
+
)
|
|
388
|
+
return rendered_frontend(
|
|
389
|
+
allocation,
|
|
390
|
+
ProcessSpec(
|
|
391
|
+
argv=[
|
|
392
|
+
"smg",
|
|
393
|
+
"launch",
|
|
394
|
+
"--host",
|
|
395
|
+
"0.0.0.0",
|
|
396
|
+
"--port",
|
|
397
|
+
str(allocation.endpoint.port),
|
|
398
|
+
"--prometheus-port",
|
|
399
|
+
str(prometheus.port),
|
|
400
|
+
"--worker-startup-timeout-secs",
|
|
401
|
+
str(_DEFERRED_WORKER_STARTUP_TIMEOUT_SECS),
|
|
402
|
+
"--worker-urls",
|
|
403
|
+
f"grpc://{engine_endpoint.host}:{engine_endpoint.port}",
|
|
404
|
+
"--model-path",
|
|
405
|
+
engine.model_locator,
|
|
406
|
+
"--tokenizer-path",
|
|
407
|
+
engine.model_locator,
|
|
408
|
+
"--policy",
|
|
409
|
+
"passthrough",
|
|
410
|
+
"--disable-retries",
|
|
411
|
+
"--disable-circuit-breaker",
|
|
412
|
+
],
|
|
413
|
+
env={},
|
|
414
|
+
),
|
|
415
|
+
)
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def render_serve(input: RenderServeInput) -> RenderServeResult:
|
|
419
|
+
if (
|
|
420
|
+
input.topology != ServeTopology.single
|
|
421
|
+
or input.gateway_backend != _GATEWAY_BACKEND
|
|
422
|
+
or input.pd_router_backend is not None
|
|
423
|
+
or input.kv_transfer is not None
|
|
424
|
+
):
|
|
425
|
+
raise AdapterOperationError(
|
|
426
|
+
AdapterErrorCode.invalid_request,
|
|
427
|
+
"render input is not the planned routed-single SMG workflow",
|
|
428
|
+
)
|
|
429
|
+
allocations, model_allocations = split_serve_allocations(input.allocations)
|
|
430
|
+
if len(allocations) != 2:
|
|
431
|
+
raise AdapterOperationError(
|
|
432
|
+
AdapterErrorCode.invalid_request,
|
|
433
|
+
"the routed-single workflow requires one Engine and one Gateway allocation",
|
|
434
|
+
)
|
|
435
|
+
engine = _require_engine(model_allocations)
|
|
436
|
+
frontend_candidates = [
|
|
437
|
+
allocation
|
|
438
|
+
for allocation in allocations
|
|
439
|
+
if isinstance(allocation, ServeProcessAllocationFrontend)
|
|
440
|
+
]
|
|
441
|
+
if len(frontend_candidates) != 1:
|
|
442
|
+
raise AdapterOperationError(
|
|
443
|
+
AdapterErrorCode.invalid_request,
|
|
444
|
+
"the routed-single workflow requires one Gateway allocation",
|
|
445
|
+
)
|
|
446
|
+
_require_gateway(frontend_candidates[0], engine.role)
|
|
447
|
+
|
|
448
|
+
processes: list[RenderedServeProcess] = []
|
|
449
|
+
for allocation in allocations:
|
|
450
|
+
if isinstance(allocation, ServeProcessAllocationModelRank):
|
|
451
|
+
processes.append(_render_engine(input, allocation))
|
|
452
|
+
elif isinstance(allocation, ServeProcessAllocationFrontend):
|
|
453
|
+
processes.append(_render_gateway(allocation, engine))
|
|
454
|
+
return RenderServeResult(integration=_identity(), processes=processes)
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: inferlab-integration-specialized-engine
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: Inferlab adapter integration for token-only Specialized Engines
|
|
5
|
+
Project-URL: Repository, https://github.com/Infer-Lab/InferLab
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
License-File: LICENSE
|
|
8
|
+
Requires-Python: >=3.12
|
|
9
|
+
Requires-Dist: inferlab-adapter-sdk==0.5.0
|
|
10
|
+
Description-Content-Type: text/markdown
|
|
11
|
+
|
|
12
|
+
# Inferlab Integration for Specialized Engines
|
|
13
|
+
|
|
14
|
+
This package connects Inferlab to a hardware-by-model Specialized Engine that
|
|
15
|
+
implements the canonical `inferlab-token-engine smg-worker` command. The Engine
|
|
16
|
+
accepts prompt token IDs and returns generated token IDs over SMG's worker
|
|
17
|
+
protocol. TokenSpeed SMG remains responsible for the public HTTP API,
|
|
18
|
+
tokenization, chat templates, detokenization, and response formatting.
|
|
19
|
+
|
|
20
|
+
The integration contains no model-, architecture-, hardware-, or
|
|
21
|
+
Engine-implementation-specific lowering. A downstream workspace supplies the
|
|
22
|
+
Rust Engine binary, SMG, model intent, source revision, locked environment, and
|
|
23
|
+
private placement bindings. Consequently, a new conforming Engine does not
|
|
24
|
+
need another Inferlab integration package.
|
|
25
|
+
|
|
26
|
+
The supported shape is deliberately closed: one `single` Engine replica in one
|
|
27
|
+
process behind one SMG Gateway. The process owns an arbitrary nonzero pure-TP
|
|
28
|
+
device set; attention, expert, and dense-expert tensor parallelism all equal
|
|
29
|
+
the outer TP width, while pipeline, data, context, and expert parallelism stay
|
|
30
|
+
at one. The contract remains serial and has no P/D Router, KV-transfer,
|
|
31
|
+
batching, or Engine-local profiling surface. InferLab can profile the Engine
|
|
32
|
+
process tree while TokenSpeed SMG exposes the capture-window
|
|
33
|
+
`POST /start_profile` and `POST /stop_profile` actions on the Gateway.
|
|
@@ -0,0 +1,7 @@
|
|
|
1
|
+
inferlab_integration_specialized_engine/__init__.py,sha256=nIhQRyuKpHDC0JynfJ3hk2GIqDCsZ2Ttx2dgLQhzdT4,16347
|
|
2
|
+
inferlab_integration_specialized_engine/__main__.py,sha256=e1uoS23_cXd5sMrRK_nSuUPgdz37Tx1xBhxqMlCYhBg,220
|
|
3
|
+
inferlab_integration_specialized_engine-0.2.0.dist-info/METADATA,sha256=v5AnpmvUcWT73awqu2hSmTEDhHi0jPinY-fAOFZDKDA,1731
|
|
4
|
+
inferlab_integration_specialized_engine-0.2.0.dist-info/WHEEL,sha256=mffPy8wBnZQn2VnJUU5jE99KsxaSfiyMHV9Yt0aLVxs,87
|
|
5
|
+
inferlab_integration_specialized_engine-0.2.0.dist-info/entry_points.txt,sha256=rXD9hNQ3ll_gLI7tVT5Dsu2WR9BApLuiOCiWHzALbZg,110
|
|
6
|
+
inferlab_integration_specialized_engine-0.2.0.dist-info/licenses/LICENSE,sha256=_sFBKPJJjTLuE3lL3Ir9HDHTTrB9zsZtMBAOvBfBWA4,1065
|
|
7
|
+
inferlab_integration_specialized_engine-0.2.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Zihua Wu
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|