PyPI - tpu-inference - Versions diffs - 0.12.0.dev20251207__tar.gz → 0.12.0.dev20251219__tar.gz - Mend

tpu-inference 0.12.0.dev20251207tar.gz → 0.12.0.dev20251219tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.

Potentially problematic release.

This version of tpu-inference might be problematic. Click here for more details.

Files changed (193) hide show

{tpu_inference-0.12.0.dev20251207/tpu_inference.egg-info → tpu_inference-0.12.0.dev20251219}/PKG-INFO RENAMED Viewed

@@ -1,6 +1,6 @@
 Metadata-Version: 2.4
 Name: tpu_inference
-Version: 0.12.0.dev20251207
+Version: 0.12.0.dev20251219
 Author: tpu_inference Contributors
 Classifier: Development Status :: 3 - Alpha
 Classifier: Intended Audience :: Developers
@@ -53,14 +53,12 @@ Dynamic: requires-python
 ---
-_Upcoming Events_ 🔥
-- Join us at the [PyTorch Conference, October 22-23](https://events.linuxfoundation.org/pytorch-conference/) in San Francisco!
-- Join us at [Ray Summit, November 3-5](https://www.anyscale.com/ray-summit/2025) in San Francisco!
-- Join us at [JAX DevLab on November 18th](https://rsvp.withgoogle.com/events/devlab-fall-2025) in Sunnyvale!
 _Latest News_ 🔥
+- [Pytorch Conference](https://pytorchconference.sched.com/event/27QCh/sponsored-session-everything-everywhere-all-at-once-vllm-hardware-optionality-with-spotify-and-google-brittany-rockwell-google-shireen-kheradpey-spotify) Learn how Spotify uses vLLM with both GPUs and TPUs to drive down costs and improve user experience.
+- Check back soon for a recording of our session at [Ray Summit, November 3-5](https://www.anyscale.com/ray-summit/2025) in San Francisco!
+- Check back soon for a recording of our session at [JAX DevLab on November 18th](https://rsvp.withgoogle.com/events/devlab-fall-2025) in Sunnyvale!
 - [2025/10] [vLLM TPU: A New Unified Backend Supporting PyTorch and JAX on TPU](https://blog.vllm.ai/2025/10/16/vllm-tpu.html)
 <details>

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/README.md RENAMED Viewed

@@ -11,14 +11,12 @@
 ---
-_Upcoming Events_ 🔥
-- Join us at the [PyTorch Conference, October 22-23](https://events.linuxfoundation.org/pytorch-conference/) in San Francisco!
-- Join us at [Ray Summit, November 3-5](https://www.anyscale.com/ray-summit/2025) in San Francisco!
-- Join us at [JAX DevLab on November 18th](https://rsvp.withgoogle.com/events/devlab-fall-2025) in Sunnyvale!
 _Latest News_ 🔥
+- [Pytorch Conference](https://pytorchconference.sched.com/event/27QCh/sponsored-session-everything-everywhere-all-at-once-vllm-hardware-optionality-with-spotify-and-google-brittany-rockwell-google-shireen-kheradpey-spotify) Learn how Spotify uses vLLM with both GPUs and TPUs to drive down costs and improve user experience.
+- Check back soon for a recording of our session at [Ray Summit, November 3-5](https://www.anyscale.com/ray-summit/2025) in San Francisco!
+- Check back soon for a recording of our session at [JAX DevLab on November 18th](https://rsvp.withgoogle.com/events/devlab-fall-2025) in Sunnyvale!
 - [2025/10] [vLLM TPU: A New Unified Backend Supporting PyTorch and JAX on TPU](https://blog.vllm.ai/2025/10/16/vllm-tpu.html)
 <details>

tpu_inference-0.12.0.dev20251219/tests/kernels/gmm_test.py ADDED Viewed

@@ -0,0 +1,191 @@
+import jax
+import jax.numpy as jnp
+from absl.testing import absltest, parameterized
+from jax._src import test_util as jtu
+from tpu_inference.kernels.megablox.gmm import gmm
+jax.config.parse_flags_with_absl()
+def quantize_tensor(x: jax.Array,
+                    dtype: jnp.dtype,
+                    axis: int = -1,
+                    block_size: int = 256):
+    if jnp.issubdtype(dtype, jnp.integer):
+        dtype_info = jnp.iinfo(dtype)
+        max_val = int(dtype_info.max)
+        min_val = int(dtype_info.min)
+    else:
+        dtype_info = jnp.finfo(dtype)
+        max_val = float(dtype_info.max)
+        min_val = float(dtype_info.min)
+    orig_shape = x.shape
+    blocked_shape = orig_shape[:axis] + (-1,
+                                         block_size) + orig_shape[axis + 1:]
+    x_blocked = x.reshape(blocked_shape)
+    x_blocked_abs_max = jnp.max(jnp.abs(x_blocked),
+                                axis=axis + 1,
+                                keepdims=True)
+    scale = x_blocked_abs_max / max_val
+    x_blocked_q = jnp.clip(x_blocked / scale, min_val, max_val).astype(dtype)
+    x_q = x_blocked_q.reshape(orig_shape)
+    scale = scale.squeeze(axis=axis + 1).astype(jnp.float32)
+    return x_q, scale
+def reference_gmm(
+    lhs: jax.Array,
+    rhs: jax.Array,
+    group_sizes: jax.Array,
+    rhs_scale: jax.Array | None = None,
+    rhs_bias: jax.Array | None = None,
+    group_offset: jax.Array | None = None,
+):
+    num_groups, out_size, in_size = rhs.shape
+    assert lhs.shape[1] == in_size
+    if group_offset is None:
+        group_offset = jnp.array(0, dtype=jnp.int32)
+    start = group_sizes[:group_offset].sum()
+    group_sizes = group_sizes[group_offset:]
+    assert len(group_sizes) == num_groups
+    if rhs_scale is not None:
+        num_blocks = rhs_scale.shape[1]
+    else:
+        num_blocks = 1
+    block_size = in_size // num_blocks
+    gmm_out = [jnp.zeros((start, out_size), lhs.dtype)]
+    for group in range(num_groups):
+        end = start + group_sizes[group]
+        lhs_slice = lhs[start:end]
+        rhs_slice = rhs[group]
+        out = 0
+        for block in range(num_blocks):
+            block_start = block * block_size
+            block_end = block_start + block_size
+            lhs_block = lhs_slice[:, block_start:block_end].astype(jnp.float32)
+            rhs_block = rhs_slice[:, block_start:block_end].astype(jnp.float32)
+            acc = jnp.einsum("bd,hd->bh", lhs_block, rhs_block)
+            if rhs_scale is not None:
+                acc *= rhs_scale[group][block]
+            out += acc
+        if rhs_bias is not None:
+            out = out + rhs_bias[group]
+        gmm_out.append(out.astype(lhs.dtype))
+        start = end
+    return jnp.concat(gmm_out, axis=0)
+@jtu.with_config(jax_numpy_dtype_promotion="standard")
+class GmmTest(jtu.JaxTestCase):
+    @parameterized.product(
+        batch_size=[128],
+        in_size=[1024],
+        out_size=[1024],
+        num_groups=[16, 32],
+        has_bias=[True, False],
+    )
+    def test_gmm(self, batch_size, in_size, out_size, num_groups, has_bias):
+        key = jax.random.key(0)
+        lhs = jax.random.normal(key, (batch_size, in_size), dtype=jnp.bfloat16)
+        rhs = jax.random.normal(key, (num_groups, out_size, in_size),
+                                dtype=jnp.bfloat16)
+        rhs_bias = None
+        if has_bias:
+            rhs_bias = jax.random.normal(key, (num_groups, 1, out_size),
+                                         dtype=jnp.bfloat16)
+        group_sizes = jax.random.randint(key, (num_groups, ),
+                                         0,
+                                         batch_size,
+                                         dtype=jnp.int32)
+        expected = reference_gmm(lhs, rhs, group_sizes, rhs_bias=rhs_bias)
+        actual = gmm(
+            lhs,
+            rhs,
+            group_sizes,
+            rhs_bias=rhs_bias,
+            transpose_rhs=True,
+            preferred_element_type=jnp.bfloat16,
+        )
+        self.assertArraysAllClose(actual, expected)
+    @parameterized.product(
+        batch_size=[128],
+        in_size=[1024],
+        out_size=[1024],
+        num_groups=[16, 32],
+        has_bias=[True, False],
+        weight_dtype=[jnp.int8, jnp.float8_e5m2, jnp.float4_e2m1fn],
+        block_size=[256, 512],
+    )
+    def test_gmm_weight_quantized(
+        self,
+        batch_size,
+        in_size,
+        out_size,
+        num_groups,
+        has_bias,
+        weight_dtype,
+        block_size,
+    ):
+        if weight_dtype == jnp.float4_e2m1fn and not jtu.is_device_tpu_at_least(
+                version=7):
+            self.skipTest("Expect TPUv7+")
+        key = jax.random.key(0)
+        lhs = jax.random.normal(key, (batch_size, in_size), dtype=jnp.bfloat16)
+        rhs = jax.random.normal(key, (num_groups, out_size, in_size),
+                                dtype=jnp.bfloat16)
+        rhs_q, rhs_scale = quantize_tensor(rhs,
+                                           weight_dtype,
+                                           axis=2,
+                                           block_size=block_size)
+        rhs_scale = jnp.swapaxes(rhs_scale, 1, 2)
+        rhs_scale = jnp.expand_dims(rhs_scale, axis=2)
+        rhs_bias = None
+        if has_bias:
+            rhs_bias = jax.random.normal(key, (num_groups, 1, out_size),
+                                         dtype=jnp.bfloat16)
+        group_sizes = jax.random.randint(key, (num_groups, ),
+                                         0,
+                                         batch_size,
+                                         dtype=jnp.int32)
+        expected = reference_gmm(lhs,
+                                 rhs_q,
+                                 group_sizes,
+                                 rhs_scale=rhs_scale,
+                                 rhs_bias=rhs_bias)
+        actual = gmm(
+            lhs,
+            rhs_q,
+            group_sizes,
+            rhs_scale=rhs_scale,
+            rhs_bias=rhs_bias,
+            transpose_rhs=True,
+            preferred_element_type=jnp.bfloat16,
+        )
+        self.assertArraysAllClose(actual, expected, atol=3e-1, rtol=3e-1)
+if __name__ == "__main__":
+    absltest.main(testLoader=jtu.JaxTestLoader())

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/tests/kernels/quantized_matmul_kernel_test.py RENAMED Viewed

@@ -1,7 +1,5 @@
 # SPDX-License-Identifier: Apache-2.0
-import functools
 import jax
 import jax.numpy as jnp
 from absl.testing import absltest, parameterized
@@ -10,6 +8,7 @@ from jax._src import test_util as jtu
 from tpu_inference.kernels.quantized_matmul import (kernel, tuned_block_sizes,
                                                     util)
+xla_quantized_matmul = kernel.xla_quantized_matmul
 quantized_matmul_kernel = kernel.quantized_matmul_kernel
 quantize_tensor = util.quantize_tensor
 get_tuned_block_sizes = tuned_block_sizes.get_tuned_block_sizes
@@ -17,37 +16,6 @@ get_tuned_block_sizes = tuned_block_sizes.get_tuned_block_sizes
 jax.config.parse_flags_with_absl()
-@functools.partial(jax.jit, static_argnames=["quantize_activation"])
-def reference_quantized_matmul(
-    x: jax.Array,
-    w_q: jax.Array,
-    w_scale: jax.Array,
-    quantize_activation=True,
-):
-    if quantize_activation:
-        acc_dtype = jnp.float32
-        if quantize_activation and jnp.issubdtype(w_q.dtype, jnp.integer):
-            acc_dtype = jnp.int32
-        x_q, x_scale = quantize_tensor(x, w_q.dtype)
-        out = jax.lax.dot_general(
-            x_q,
-            w_q,
-            dimension_numbers=(((1, ), (1, )), ((), ())),
-            preferred_element_type=acc_dtype,
-        ).astype(jnp.float32)
-        out *= x_scale
-    else:
-        out = jax.lax.dot_general(
-            x,
-            w_q,
-            dimension_numbers=(((1, ), (1, )), ((), ())),
-            preferred_element_type=jnp.float32,
-        )
-    out *= jnp.expand_dims(w_scale, 0)
-    return out.astype(x.dtype)
 @jtu.with_config(jax_numpy_dtype_promotion="standard")
 class QuantizedMatmulKernelTest(jtu.JaxTestCase):
@@ -94,7 +62,7 @@ class QuantizedMatmulKernelTest(jtu.JaxTestCase):
             x_q_dtype=x_q_dtype,
             tuned_value=tuned_value,
         )
-        expected = reference_quantized_matmul(
+        expected = xla_quantized_matmul(
             x, w_q, w_scale, quantize_activation=quantize_activation)
         self.assertAllClose(output,

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/tests/kernels/ragged_paged_attention_kernel_v3_hd64_test.py RENAMED Viewed

@@ -176,7 +176,9 @@ class RaggedPagedAttentionHeadDim64KernelTest(jtu.JaxTestCase):
         )
         output = output[:cu_q_lens[distribution[-1]]]
-        dtype_bits = dtypes.bit_width(jnp.dtype(kv_dtype))
+        dtype_bits = (dtypes.bit_width(jnp.dtype(kv_dtype)) if hasattr(
+            dtypes, "bit_width") else dtypes.itemsize_bits(
+                jnp.dtype(kv_dtype)))
         tols = {
             32: 0.15,
             16: 0.2,

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/tests/kernels/ragged_paged_attention_kernel_v3_test.py RENAMED Viewed

@@ -162,7 +162,9 @@ class RaggedPagedAttentionKernelTest(jtu.JaxTestCase):
         )
         output = output[:cu_q_lens[distribution[-1]]]
-        dtype_bits = dtypes.bit_width(jnp.dtype(kv_dtype))
+        dtype_bits = (dtypes.bit_width(jnp.dtype(kv_dtype)) if hasattr(
+            dtypes, "bit_width") else dtypes.itemsize_bits(
+                jnp.dtype(kv_dtype)))
         tols = {
             32: 0.15,
             16: 0.2,

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/tests/lora/test_layers.py RENAMED Viewed

@@ -18,7 +18,7 @@ from vllm.lora.layers import (BaseLayerWithLoRA, ColumnParallelLinearWithLoRA,
                               ReplicatedLinearWithLoRA,
                               RowParallelLinearWithLoRA)
 # yapf: enable
-from vllm.lora.models import LoRALayerWeights, PackedLoRALayerWeights
+from vllm.lora.lora_weights import LoRALayerWeights, PackedLoRALayerWeights
 from vllm.lora.punica_wrapper import get_punica_wrapper
 from vllm.model_executor.layers.linear import (ColumnParallelLinear,
                                                MergedColumnParallelLinear,
@@ -499,9 +499,13 @@ def _create_random_linear_parallel_layer(layer_type, vllm_config, mesh):
     return linear, lora_linear
+def _get_devices():
+    return jax.devices()
 def _create_mesh():
     axis_names = ("data", "model")
-    devices = jax.devices()
+    devices = _get_devices()
     mesh_shape = (1, len(devices))
     mesh = jax.make_mesh(mesh_shape, axis_names, devices=devices)
     return mesh
@@ -513,7 +517,7 @@ def _verify_lora_linear_layer(linear, lora_linear):
         # BaseLinearLayerWithLoRA.weight property guarantees this.
         # if len(devices) != 1, `reorder_concatenated_tensor_for_sharding` function may reorder the out_features dimension of the weight matrix.
         # So the below check will fail.
-        if len(jax.devices()) == 1:
+        if len(_get_devices()) == 1:
             assert torch.equal(linear.weight.data,
                                lora_linear.weight.to('cpu'))

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/tests/lora/test_lora.py RENAMED Viewed

@@ -29,7 +29,7 @@ def setup_vllm(num_loras: int, tp: int = 1) -> vllm.LLM:
 # For multi-chip test, we only use TP=2 because the base model Qwen/Qwen2.5-3B-Instruct has 2 kv heads and the current attention kernel requires it to be divisible by tp_size.
-TP = [2] if os.environ.get("USE_V6E8_QUEUE", False) else [1]
+TP = [2] if os.environ.get("TEST_LORA_TP", False) else [1]
 @pytest.mark.parametrize("tp", TP)

tpu_inference-0.12.0.dev20251219/tests/lora/test_lora_perf.py ADDED Viewed

@@ -0,0 +1,53 @@
+import os
+import time
+import pytest
+import vllm
+from vllm.lora.request import LoRARequest
+TP = [2] if os.environ.get("USE_V6E8_QUEUE", False) else [1]
+@pytest.mark.parametrize("tp", TP)
+def test_lora_performance(tp):
+    prompt = "What is 1+1? \n"
+    llm_without_lora = vllm.LLM(
+        model="Qwen/Qwen2.5-3B-Instruct",
+        max_model_len=256,
+        max_num_batched_tokens=64,
+        max_num_seqs=8,
+        tensor_parallel_size=tp,
+    )
+    start_time = time.time()
+    llm_without_lora.generate(
+        prompt,
+        sampling_params=vllm.SamplingParams(max_tokens=16, temperature=0),
+    )[0].outputs[0].text
+    base_time = time.time() - start_time
+    del llm_without_lora
+    # Waiting for TPUs to be released
+    time.sleep(10)
+    llm_with_lora = vllm.LLM(model="Qwen/Qwen2.5-3B-Instruct",
+                             max_model_len=256,
+                             max_num_batched_tokens=64,
+                             max_num_seqs=8,
+                             tensor_parallel_size=tp,
+                             enable_lora=True,
+                             max_loras=1,
+                             max_lora_rank=8)
+    lora_request = LoRARequest(
+        "lora_adapter_2", 2,
+        "Username6568/Qwen2.5-3B-Instruct-1_plus_1_equals_2_adapter")
+    start_time = time.time()
+    llm_with_lora.generate(prompt,
+                           sampling_params=vllm.SamplingParams(max_tokens=16,
+                                                               temperature=0),
+                           lora_request=lora_request)[0].outputs[0].text
+    lora_time = time.time() - start_time
+    print(f"Base time: {base_time}, LoRA time: {lora_time}")
+    assert (base_time /
+            lora_time) < 8, f"Base time: {base_time}, LoRA time: {lora_time}"
+    del llm_with_lora

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/tests/test_envs.py RENAMED Viewed

@@ -60,6 +60,7 @@ def test_boolean_env_vars(monkeypatch: pytest.MonkeyPatch):
     monkeypatch.setenv("SKIP_JAX_PRECOMPILE", "0")
     monkeypatch.setenv("VLLM_XLA_CHECK_RECOMPILATION", "0")
     monkeypatch.setenv("NEW_MODEL_DESIGN", "0")
+    monkeypatch.setenv("ENABLE_QUANTIZED_MATMUL_KERNEL", "0")
     monkeypatch.setenv("USE_MOE_EP_KERNEL", "0")
     # Test SKIP_JAX_PRECOMPILE (default False)
@@ -86,6 +87,82 @@ def test_boolean_env_vars(monkeypatch: pytest.MonkeyPatch):
     monkeypatch.setenv("USE_MOE_EP_KERNEL", "1")
     assert envs.USE_MOE_EP_KERNEL is True
+    # Test ENABLE_QUANTIZED_MATMUL_KERNEL (default False)
+    assert envs.ENABLE_QUANTIZED_MATMUL_KERNEL is False
+    monkeypatch.setenv("ENABLE_QUANTIZED_MATMUL_KERNEL", "1")
+    assert envs.ENABLE_QUANTIZED_MATMUL_KERNEL is True
+def test_boolean_env_vars_string_values(monkeypatch: pytest.MonkeyPatch):
+    """Test that boolean env vars accept string values like 'True' and 'False'"""
+    # Test NEW_MODEL_DESIGN with string "True"
+    monkeypatch.setenv("NEW_MODEL_DESIGN", "True")
+    assert envs.NEW_MODEL_DESIGN is True
+    monkeypatch.setenv("NEW_MODEL_DESIGN", "true")
+    assert envs.NEW_MODEL_DESIGN is True
+    monkeypatch.setenv("NEW_MODEL_DESIGN", "False")
+    assert envs.NEW_MODEL_DESIGN is False
+    monkeypatch.setenv("NEW_MODEL_DESIGN", "false")
+    assert envs.NEW_MODEL_DESIGN is False
+    # Test SKIP_JAX_PRECOMPILE with string values
+    monkeypatch.setenv("SKIP_JAX_PRECOMPILE", "True")
+    assert envs.SKIP_JAX_PRECOMPILE is True
+    monkeypatch.setenv("SKIP_JAX_PRECOMPILE", "false")
+    assert envs.SKIP_JAX_PRECOMPILE is False
+    # Test VLLM_XLA_CHECK_RECOMPILATION with string values
+    monkeypatch.setenv("VLLM_XLA_CHECK_RECOMPILATION", "TRUE")
+    assert envs.VLLM_XLA_CHECK_RECOMPILATION is True
+    monkeypatch.setenv("VLLM_XLA_CHECK_RECOMPILATION", "FALSE")
+    assert envs.VLLM_XLA_CHECK_RECOMPILATION is False
+    # Test USE_MOE_EP_KERNEL with string values
+    monkeypatch.setenv("USE_MOE_EP_KERNEL", "true")
+    assert envs.USE_MOE_EP_KERNEL is True
+    monkeypatch.setenv("USE_MOE_EP_KERNEL", "False")
+    assert envs.USE_MOE_EP_KERNEL is False
+def test_boolean_env_vars_invalid_values(monkeypatch: pytest.MonkeyPatch):
+    """Test that boolean env vars raise errors for invalid values"""
+    # Test invalid value for NEW_MODEL_DESIGN
+    monkeypatch.setenv("NEW_MODEL_DESIGN", "yes")
+    with pytest.raises(
+            ValueError,
+            match="Invalid boolean value 'yes' for NEW_MODEL_DESIGN"):
+        _ = envs.NEW_MODEL_DESIGN
+    monkeypatch.setenv("NEW_MODEL_DESIGN", "2")
+    with pytest.raises(ValueError,
+                       match="Invalid boolean value '2' for NEW_MODEL_DESIGN"):
+        _ = envs.NEW_MODEL_DESIGN
+    # Test invalid value for SKIP_JAX_PRECOMPILE
+    monkeypatch.setenv("SKIP_JAX_PRECOMPILE", "invalid")
+    with pytest.raises(
+            ValueError,
+            match="Invalid boolean value 'invalid' for SKIP_JAX_PRECOMPILE"):
+        _ = envs.SKIP_JAX_PRECOMPILE
+def test_boolean_env_vars_empty_string(monkeypatch: pytest.MonkeyPatch):
+    """Test that empty string returns default value"""
+    monkeypatch.setenv("NEW_MODEL_DESIGN", "")
+    assert envs.NEW_MODEL_DESIGN is False  # Should return default
+    monkeypatch.setenv("SKIP_JAX_PRECOMPILE", "")
+    assert envs.SKIP_JAX_PRECOMPILE is False  # Should return default
 def test_integer_env_vars(monkeypatch: pytest.MonkeyPatch):
     # Ensure clean environment for integer vars by setting to defaults
@@ -179,7 +256,7 @@ def test_disaggregated_serving_env_vars(monkeypatch: pytest.MonkeyPatch):
 def test_model_impl_type_default(monkeypatch: pytest.MonkeyPatch):
     monkeypatch.delenv("MODEL_IMPL_TYPE", raising=False)
-    assert envs.MODEL_IMPL_TYPE == "flax_nnx"
+    assert envs.MODEL_IMPL_TYPE == "auto"
 def test_cache_preserves_values_across_env_changes(

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/tpu_inference/distributed/tpu_connector.py RENAMED Viewed

@@ -694,9 +694,9 @@ class TPUConnectorWorker:
 def get_uuid() -> int:
     int128 = uuid4().int
-    # Must be 64-bit int, otherwise vllm output encoder would raise error.
-    int64 = int128 >> 64
-    return int64
+    # Must be less than 64-bit int, otherwise vllm output encoder would raise error.
+    # use 50 bit to avoid GO trunk the int when doing JSon serialization
+    return int128 >> 78
 @jax.jit

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/tpu_inference/envs.py RENAMED Viewed

@@ -16,7 +16,7 @@ if TYPE_CHECKING:
     DECODE_SLICES: str = ""
     SKIP_JAX_PRECOMPILE: bool = False
     VLLM_XLA_CHECK_RECOMPILATION: bool = False
-    MODEL_IMPL_TYPE: str = "flax_nnx"
+    MODEL_IMPL_TYPE: str = "auto"
     NEW_MODEL_DESIGN: bool = False
     PHASED_PROFILING_DIR: str = ""
     PYTHON_TRACER_LEVEL: int = 1
@@ -24,6 +24,7 @@ if TYPE_CHECKING:
     NUM_SLICES: int = 1
     RAY_USAGE_STATS_ENABLED: str = "0"
     VLLM_USE_RAY_COMPILED_DAG_CHANNEL_TYPE: str = "shm"
+    ENABLE_QUANTIZED_MATMUL_KERNEL: bool = False
 def env_with_choices(
@@ -69,6 +70,34 @@ def env_with_choices(
     return _get_validated_env
+def env_bool(env_name: str, default: bool = False) -> Callable[[], bool]:
+    """
+    Accepts both numeric strings ("0", "1") and boolean strings
+    ("true", "false", "True", "False").
+    Args:
+        env_name: Name of the environment variable
+        default: Default boolean value if not set
+    """
+    def _get_bool_env() -> bool:
+        value = os.getenv(env_name)
+        if value is None or value == "":
+            return default
+        value_lower = value.lower()
+        if value_lower in ("true", "1"):
+            return True
+        elif value_lower in ("false", "0"):
+            return False
+        else:
+            raise ValueError(
+                f"Invalid boolean value '{value}' for {env_name}. "
+                f"Valid options: '0', '1', 'true', 'false', 'True', 'False'.")
+    return _get_bool_env
 environment_variables: dict[str, Callable[[], Any]] = {
     # JAX platform selection (e.g., "tpu", "cpu", "proxy")
     "JAX_PLATFORMS":
@@ -93,17 +122,17 @@ environment_variables: dict[str, Callable[[], Any]] = {
     lambda: os.getenv("DECODE_SLICES", ""),
     # Skip JAX precompilation step during initialization
     "SKIP_JAX_PRECOMPILE":
-    lambda: bool(int(os.getenv("SKIP_JAX_PRECOMPILE") or "0")),
+    env_bool("SKIP_JAX_PRECOMPILE", default=False),
     # Check for XLA recompilation during execution
     "VLLM_XLA_CHECK_RECOMPILATION":
-    lambda: bool(int(os.getenv("VLLM_XLA_CHECK_RECOMPILATION") or "0")),
+    env_bool("VLLM_XLA_CHECK_RECOMPILATION", default=False),
     # Model implementation type (e.g., "flax_nnx")
     "MODEL_IMPL_TYPE":
-    env_with_choices("MODEL_IMPL_TYPE", "flax_nnx",
-                     ["vllm", "flax_nnx", "jetpack"]),
+    env_with_choices("MODEL_IMPL_TYPE", "auto",
+                     ["auto", "vllm", "flax_nnx", "jetpack"]),
     # Enable new experimental model design
     "NEW_MODEL_DESIGN":
-    lambda: bool(int(os.getenv("NEW_MODEL_DESIGN") or "0")),
+    env_bool("NEW_MODEL_DESIGN", default=False),
     # Directory to store phased profiling output
     "PHASED_PROFILING_DIR":
     lambda: os.getenv("PHASED_PROFILING_DIR", ""),
@@ -112,7 +141,7 @@ environment_variables: dict[str, Callable[[], Any]] = {
     lambda: int(os.getenv("PYTHON_TRACER_LEVEL") or "1"),
     # Use custom expert-parallel kernel for MoE (Mixture of Experts)
     "USE_MOE_EP_KERNEL":
-    lambda: bool(int(os.getenv("USE_MOE_EP_KERNEL") or "0")),
+    env_bool("USE_MOE_EP_KERNEL", default=False),
     # Number of TPU slices for multi-slice mesh
     "NUM_SLICES":
     lambda: int(os.getenv("NUM_SLICES") or "1"),
@@ -122,6 +151,8 @@ environment_variables: dict[str, Callable[[], Any]] = {
     # Ray compiled DAG channel type for TPU
     "VLLM_USE_RAY_COMPILED_DAG_CHANNEL_TYPE":
     env_with_choices("VLLM_USE_RAY_COMPILED_DAG_CHANNEL_TYPE", "shm", ["shm"]),
+    "ENABLE_QUANTIZED_MATMUL_KERNEL":
+    lambda: bool(int(os.getenv("ENABLE_QUANTIZED_MATMUL_KERNEL") or "0")),
 }

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/tpu_inference/executors/ray_distributed_executor.py RENAMED Viewed

@@ -145,6 +145,9 @@ class RayDistributedExecutor(RayDistributedExecutorV1):
                 device_str: node['Resources'][device_str]
             } for node in ray_nodes]
         else:
+            assert pp_size == len(
+                ray_nodes
+            ), f"Cannot use PP across hosts, please set --pipeline-parallel-size to 1 or {len(ray_nodes)}"
             num_devices_per_pp_rank = self.vllm_config.sharding_config.total_devices
             placement_group_specs = [{
                 device_str: num_devices_per_pp_rank

{tpu_inference-0.12.0.dev20251207 → tpu_inference-0.12.0.dev20251219}/tpu_inference/kernels/collectives/all_gather_matmul.py RENAMED Viewed

@@ -540,12 +540,16 @@ def get_vmem_estimate_bytes(
     """Returns the total vmem bytes used by the kernel."""
     m_per_device = m // tp_size
     n_per_device = n // tp_size
-    y_vmem_bytes = n_per_device * k * dtypes.bit_width(y_dtype) // 8
+    y_vmem_bytes = (n_per_device * k * (dtypes.bit_width(y_dtype) if hasattr(
+        dtypes, "bit_width") else dtypes.itemsize_bits(y_dtype)) // 8)
     total_bytes = (
-        2 * m_per_device * k * dtypes.bit_width(x_dtype) //
-        8  # x_vmem_scratch_ref
+        2 * m_per_device * k *
+        (dtypes.bit_width(x_dtype) if hasattr(dtypes, "bit_width") else
+         dtypes.itemsize_bits(x_dtype)) // 8  # x_vmem_scratch_ref
         + y_vmem_bytes  # y_vmem_scratch_ref
-        + 2 * m * bn * dtypes.bit_width(out_dtype) // 8  # o_vmem_scratch_ref
+        + 2 * m * bn *
+        (dtypes.bit_width(out_dtype) if hasattr(dtypes, "bit_width") else
+         dtypes.itemsize_bits(out_dtype)) // 8  # o_vmem_scratch_ref
         + acc_bytes  # acc_vmem_scratch_ref, jnp.float32
     )
     return total_bytes
@@ -639,8 +643,10 @@ def all_gather_matmul(
     # NOTE(chengjiyao): acc buffer is not used in the grid_k == 1 case.
     if grid_k == 1:
         acc_shape = (8, 128)
-    acc_bytes = acc_shape[0] * acc_shape[1] * dtypes.bit_width(
-        jnp.float32) // 8
+    acc_bytes = (
+        acc_shape[0] *
+        acc_shape[1] * (dtypes.bit_width(jnp.float32) if hasattr(
+            dtypes, "bit_width") else dtypes.itemsize_bits(jnp.float32)) // 8)
     y_vmem_shape = (n_per_device, k) if rhs_transpose else (k, n_per_device)
     estimated_vmem_bytes = get_vmem_estimate_bytes(
         m,

tpu-inference 0.12.0.dev20251207__tar.gz → 0.12.0.dev20251219__tar.gz

Potentially problematic release.

tpu-inference 0.12.0.dev20251207tar.gz → 0.12.0.dev20251219tar.gz