flash-rt 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- flash_rt/__init__.py +107 -0
- flash_rt/_extensions.py +119 -0
- flash_rt/amd/__init__.py +9 -0
- flash_rt/amd/core/__init__.py +0 -0
- flash_rt/amd/core/hip_buffer.py +176 -0
- flash_rt/amd/core/hip_graph.py +102 -0
- flash_rt/amd/frontends/__init__.py +1 -0
- flash_rt/amd/frontends/torch/__init__.py +1 -0
- flash_rt/amd/frontends/torch/groot_n17.py +1013 -0
- flash_rt/amd/frontends/torch/pi05.py +1431 -0
- flash_rt/amd/hardware/__init__.py +1 -0
- flash_rt/amd/hardware/cdna4/__init__.py +1 -0
- flash_rt/amd/hardware/cdna4/attn_backend.py +338 -0
- flash_rt/amd/hardware/cdna4/attn_backend_aiter.py +410 -0
- flash_rt/amd/hardware/cdna4/attn_backend_groot_n17.py +297 -0
- flash_rt/amd/models/__init__.py +1 -0
- flash_rt/amd/models/groot_n17/__init__.py +1 -0
- flash_rt/amd/models/groot_n17/pipeline.py +1058 -0
- flash_rt/amd/models/pi05/__init__.py +1 -0
- flash_rt/amd/models/pi05/pipeline.py +1860 -0
- flash_rt/api.py +1144 -0
- flash_rt/catalog/__init__.py +39 -0
- flash_rt/catalog/binding.py +412 -0
- flash_rt/catalog/bindings/cosmos3_video_pipeline.yaml +94 -0
- flash_rt/catalog/bindings/groot_n16_dit.yaml +23 -0
- flash_rt/catalog/bindings/groot_n16_llm.yaml +21 -0
- flash_rt/catalog/bindings/groot_n16_pipeline.yaml +107 -0
- flash_rt/catalog/bindings/groot_n16_tick.yaml +26 -0
- flash_rt/catalog/bindings/groot_n16_vision.yaml +23 -0
- flash_rt/catalog/bindings/groot_n17_pipeline.yaml +117 -0
- flash_rt/catalog/bindings/lingbot_vla_pipeline.yaml +106 -0
- flash_rt/catalog/bindings/motus_tick.yaml +120 -0
- flash_rt/catalog/bindings/nexn2_pipeline.yaml +112 -0
- flash_rt/catalog/bindings/pi05.yaml +29 -0
- flash_rt/catalog/bindings/pi05_prefix.yaml +21 -0
- flash_rt/catalog/bindings/pi05_tick.yaml +94 -0
- flash_rt/catalog/bindings/pi05_vision.yaml +23 -0
- flash_rt/catalog/bindings/qwen25_15b.yaml +22 -0
- flash_rt/catalog/bindings/qwen36_27b_pipeline.yaml +130 -0
- flash_rt/catalog/bindings/qwen3_8b.yaml +22 -0
- flash_rt/catalog/bindings/qwen3_8b_pipeline.yaml +98 -0
- flash_rt/catalog/bindings/qwen3_vl_8b_pipeline.yaml +131 -0
- flash_rt/catalog/bindings/qwen3_vl_8b_text.yaml +22 -0
- flash_rt/catalog/bindings/qwen3_vl_8b_vision.yaml +23 -0
- flash_rt/catalog/bindings/smolvla_base.yaml +21 -0
- flash_rt/catalog/bindings/smolvla_expert.yaml +21 -0
- flash_rt/catalog/bindings/smolvla_pipeline.yaml +93 -0
- flash_rt/catalog/bindings/smolvla_tick.yaml +27 -0
- flash_rt/catalog/bindings/smolvla_vision.yaml +23 -0
- flash_rt/catalog/bindings/wan22_video_pipeline.yaml +119 -0
- flash_rt/catalog/registry.py +108 -0
- flash_rt/catalog/structures/__init__.py +0 -0
- flash_rt/catalog/structures/adaln_producer/__init__.py +0 -0
- flash_rt/catalog/structures/adaln_producer/reference.py +40 -0
- flash_rt/catalog/structures/adaln_producer/structure.yaml +106 -0
- flash_rt/catalog/structures/attention_core/__init__.py +0 -0
- flash_rt/catalog/structures/attention_core/reference.py +36 -0
- flash_rt/catalog/structures/attention_core/structure.yaml +72 -0
- flash_rt/catalog/structures/autoregressive_decode_pipeline/structure.yaml +82 -0
- flash_rt/catalog/structures/cadence_static/__init__.py +0 -0
- flash_rt/catalog/structures/cadence_static/reference.py +24 -0
- flash_rt/catalog/structures/cadence_static/structure.yaml +56 -0
- flash_rt/catalog/structures/decoder_block/__init__.py +0 -0
- flash_rt/catalog/structures/decoder_block/reference.py +38 -0
- flash_rt/catalog/structures/decoder_block/structure.yaml +86 -0
- flash_rt/catalog/structures/decoder_ffn/__init__.py +0 -0
- flash_rt/catalog/structures/decoder_ffn/reference.py +64 -0
- flash_rt/catalog/structures/decoder_ffn/structure.yaml +44 -0
- flash_rt/catalog/structures/gated_delta_core/reference.py +54 -0
- flash_rt/catalog/structures/gated_delta_core/structure.yaml +60 -0
- flash_rt/catalog/structures/linear_proj/__init__.py +3 -0
- flash_rt/catalog/structures/linear_proj/reference.py +35 -0
- flash_rt/catalog/structures/linear_proj/structure.yaml +74 -0
- flash_rt/catalog/structures/modnorm_qkv_chain/__init__.py +1 -0
- flash_rt/catalog/structures/modnorm_qkv_chain/reference.py +39 -0
- flash_rt/catalog/structures/modnorm_qkv_chain/structure.yaml +59 -0
- flash_rt/catalog/structures/norm_fused/__init__.py +0 -0
- flash_rt/catalog/structures/norm_fused/reference.py +26 -0
- flash_rt/catalog/structures/norm_fused/structure.yaml +50 -0
- flash_rt/catalog/structures/patch_projection/reference.py +15 -0
- flash_rt/catalog/structures/patch_projection/structure.yaml +53 -0
- flash_rt/catalog/structures/qk_norm_rope/__init__.py +3 -0
- flash_rt/catalog/structures/qk_norm_rope/reference.py +114 -0
- flash_rt/catalog/structures/qk_norm_rope/structure.yaml +83 -0
- flash_rt/catalog/structures/qkv_pack/__init__.py +0 -0
- flash_rt/catalog/structures/qkv_pack/reference.py +33 -0
- flash_rt/catalog/structures/qkv_pack/structure.yaml +65 -0
- flash_rt/catalog/structures/qkv_rope/__init__.py +1 -0
- flash_rt/catalog/structures/qkv_rope/reference.py +39 -0
- flash_rt/catalog/structures/qkv_rope/structure.yaml +55 -0
- flash_rt/catalog/structures/video_generation_pipeline/__init__.py +2 -0
- flash_rt/catalog/structures/video_generation_pipeline/structure.yaml +81 -0
- flash_rt/catalog/structures/vision_ffn/__init__.py +0 -0
- flash_rt/catalog/structures/vision_ffn/reference.py +35 -0
- flash_rt/catalog/structures/vision_ffn/structure.yaml +42 -0
- flash_rt/catalog/structures/vla_tick_pipeline/__init__.py +7 -0
- flash_rt/catalog/structures/vla_tick_pipeline/structure.yaml +82 -0
- flash_rt/configs/__init__.py +0 -0
- flash_rt/configs/cosmos3_edge.yaml +21 -0
- flash_rt/configs/cosmos3_video.yaml +24 -0
- flash_rt/configs/groot.yaml +73 -0
- flash_rt/configs/groot_n17.yaml +53 -0
- flash_rt/configs/hyvla.yaml +65 -0
- flash_rt/configs/ltx25.yaml +41 -0
- flash_rt/configs/motus.yaml +85 -0
- flash_rt/configs/nexn2.yaml +79 -0
- flash_rt/configs/pi0.yaml +38 -0
- flash_rt/configs/pi05.yaml +38 -0
- flash_rt/configs/qwen36.yaml +68 -0
- flash_rt/configs/wan22_ti2v_5b.yaml +24 -0
- flash_rt/core/__init__.py +0 -0
- flash_rt/core/calibration.py +301 -0
- flash_rt/core/calibration_api.py +70 -0
- flash_rt/core/config.py +96 -0
- flash_rt/core/context.py +47 -0
- flash_rt/core/cuda_buffer.py +189 -0
- flash_rt/core/cuda_graph.py +81 -0
- flash_rt/core/parity.py +37 -0
- flash_rt/core/precision_spec.py +164 -0
- flash_rt/core/quant/__init__.py +0 -0
- flash_rt/core/quant/calibrator.py +170 -0
- flash_rt/core/quantization.py +73 -0
- flash_rt/core/rl/__init__.py +75 -0
- flash_rt/core/rl/acp_tags.py +51 -0
- flash_rt/core/rl/advantage.py +163 -0
- flash_rt/core/rl/cfg_sampler.py +72 -0
- flash_rt/core/rl/reward.py +233 -0
- flash_rt/core/rl/value_function.py +198 -0
- flash_rt/core/thor_frontend_utils.py +152 -0
- flash_rt/core/utils/__init__.py +0 -0
- flash_rt/core/utils/actions.py +19 -0
- flash_rt/core/utils/hardware.py +50 -0
- flash_rt/core/utils/norm_stats.py +359 -0
- flash_rt/core/utils/pi05_prompt.py +35 -0
- flash_rt/core/weights/__init__.py +0 -0
- flash_rt/core/weights/loader.py +135 -0
- flash_rt/core/weights/transformer.py +691 -0
- flash_rt/core/weights/weight_cache.py +147 -0
- flash_rt/datasets/__init__.py +11 -0
- flash_rt/datasets/libero.py +306 -0
- flash_rt/executors/__init__.py +6 -0
- flash_rt/executors/fp4_utils.py +241 -0
- flash_rt/executors/fp4_utils_cb.py +207 -0
- flash_rt/executors/jax_weights.py +270 -0
- flash_rt/executors/torch_weights.py +500 -0
- flash_rt/executors/weight_loader.py +331 -0
- flash_rt/frontends/__init__.py +8 -0
- flash_rt/frontends/_fp8_layout.py +32 -0
- flash_rt/frontends/jax/__init__.py +1 -0
- flash_rt/frontends/jax/_pi05_thor_spec.py +52 -0
- flash_rt/frontends/jax/_pi0_thor_spec.py +32 -0
- flash_rt/frontends/jax/_thor_spec_common.py +114 -0
- flash_rt/frontends/jax/pi05_rtx.py +576 -0
- flash_rt/frontends/jax/pi05_thor.py +2768 -0
- flash_rt/frontends/jax/pi05_thor_fp4.py +879 -0
- flash_rt/frontends/jax/pi0_rtx.py +483 -0
- flash_rt/frontends/jax/pi0_thor.py +1425 -0
- flash_rt/frontends/jax/pi0fast.py +1337 -0
- flash_rt/frontends/jetson_pi/__init__.py +12 -0
- flash_rt/frontends/jetson_pi/llm.py +261 -0
- flash_rt/frontends/jetson_pi/mllm.py +289 -0
- flash_rt/frontends/jetson_pi/pi0.py +420 -0
- flash_rt/frontends/torch/__init__.py +1 -0
- flash_rt/frontends/torch/_chameleon_quant.py +251 -0
- flash_rt/frontends/torch/_chameleon_rtx_sm87_spec.py +85 -0
- flash_rt/frontends/torch/_chameleon_thor_spec.py +92 -0
- flash_rt/frontends/torch/_cosmos3_edge_thor_spec.py +177 -0
- flash_rt/frontends/torch/_groot_n17_rtx_spec.py +13 -0
- flash_rt/frontends/torch/_groot_n17_thor_spec.py +406 -0
- flash_rt/frontends/torch/_groot_thor_spec.py +105 -0
- flash_rt/frontends/torch/_higgs_audio_v3_bf16.py +374 -0
- flash_rt/frontends/torch/_higgs_audio_v3_fp8.py +464 -0
- flash_rt/frontends/torch/_hyvla_thor_spec.py +196 -0
- flash_rt/frontends/torch/_lingbot_thor_spec.py +321 -0
- flash_rt/frontends/torch/_motus_rtx_spec.py +47 -0
- flash_rt/frontends/torch/_nexn2_rtx_decode.py +1621 -0
- flash_rt/frontends/torch/_nexn2_rtx_forward.py +1815 -0
- flash_rt/frontends/torch/_nexn2_rtx_nvfp4_weights.py +416 -0
- flash_rt/frontends/torch/_pi05_thor_spec.py +100 -0
- flash_rt/frontends/torch/_pi0_thor_spec.py +81 -0
- flash_rt/frontends/torch/_qwen36_rtx_dflash_forward.py +940 -0
- flash_rt/frontends/torch/_qwen36_rtx_dflash_weights.py +396 -0
- flash_rt/frontends/torch/_qwen36_rtx_nvfp4_weights.py +741 -0
- flash_rt/frontends/torch/_qwen36_rtx_turboquant.py +872 -0
- flash_rt/frontends/torch/_qwen36_rtx_weights.py +411 -0
- flash_rt/frontends/torch/_qwen3_rtx_nvfp4_weights.py +575 -0
- flash_rt/frontends/torch/_qwen3_vl_bf16_weights.py +191 -0
- flash_rt/frontends/torch/_qwen3_vl_fp8_weights.py +245 -0
- flash_rt/frontends/torch/_qwen3_vl_geometry.py +337 -0
- flash_rt/frontends/torch/_qwen3_vl_vision_rtx.py +625 -0
- flash_rt/frontends/torch/_template/attention.py +124 -0
- flash_rt/frontends/torch/_template/frontend.py +330 -0
- flash_rt/frontends/torch/_template/pipeline.py +263 -0
- flash_rt/frontends/torch/_template/weights_spec.py +215 -0
- flash_rt/frontends/torch/_thor_spec_common.py +147 -0
- flash_rt/frontends/torch/chameleon_rtx_sm87.py +721 -0
- flash_rt/frontends/torch/chameleon_thor.py +911 -0
- flash_rt/frontends/torch/cosmos3_edge_thor.py +571 -0
- flash_rt/frontends/torch/cosmos3_video_rtx.py +130 -0
- flash_rt/frontends/torch/groot_n17_rtx.py +152 -0
- flash_rt/frontends/torch/groot_n17_rtx_fp16.py +655 -0
- flash_rt/frontends/torch/groot_n17_rtx_fp8.py +582 -0
- flash_rt/frontends/torch/groot_n17_rtx_sm89.py +609 -0
- flash_rt/frontends/torch/groot_n17_rtx_sm89_fp16.py +652 -0
- flash_rt/frontends/torch/groot_n17_thor.py +1965 -0
- flash_rt/frontends/torch/groot_n17_thor_fp16.py +49 -0
- flash_rt/frontends/torch/groot_n17_thor_fp4.py +165 -0
- flash_rt/frontends/torch/groot_n17_thor_fp8.py +780 -0
- flash_rt/frontends/torch/groot_rtx.py +1876 -0
- flash_rt/frontends/torch/groot_rtx_fp16.py +1162 -0
- flash_rt/frontends/torch/groot_thor.py +3623 -0
- flash_rt/frontends/torch/groot_thor_fp16.py +28 -0
- flash_rt/frontends/torch/higgs_audio_v3_rtx.py +601 -0
- flash_rt/frontends/torch/hyvla_orin.py +306 -0
- flash_rt/frontends/torch/hyvla_thor.py +683 -0
- flash_rt/frontends/torch/lingbot_thor.py +116 -0
- flash_rt/frontends/torch/ltx25_rtx.py +378 -0
- flash_rt/frontends/torch/motus_rtx.py +1562 -0
- flash_rt/frontends/torch/nexn2_rtx.py +310 -0
- flash_rt/frontends/torch/pi05_rtx.py +1949 -0
- flash_rt/frontends/torch/pi05_rtx_fp16.py +1806 -0
- flash_rt/frontends/torch/pi05_thor.py +3083 -0
- flash_rt/frontends/torch/pi05_thor_fp4.py +1500 -0
- flash_rt/frontends/torch/pi0_rtx.py +957 -0
- flash_rt/frontends/torch/pi0_thor.py +1405 -0
- flash_rt/frontends/torch/pi0fast.py +1402 -0
- flash_rt/frontends/torch/qwen36_moe.py +426 -0
- flash_rt/frontends/torch/qwen36_moe_rtx.py +20 -0
- flash_rt/frontends/torch/qwen36_rtx.py +12388 -0
- flash_rt/frontends/torch/qwen36_spark.py +200 -0
- flash_rt/frontends/torch/qwen36_thor.py +1320 -0
- flash_rt/frontends/torch/qwen3_rtx.py +2166 -0
- flash_rt/frontends/torch/qwen3_vl_fp8_sm89.py +912 -0
- flash_rt/frontends/torch/qwen3_vl_fp8_sm89_multimodal.py +456 -0
- flash_rt/frontends/torch/qwen3_vl_rtx.py +642 -0
- flash_rt/frontends/torch/qwen3_vl_rtx_bf16.py +1031 -0
- flash_rt/frontends/torch/qwen3_vl_thor.py +856 -0
- flash_rt/frontends/torch/wan22_rtx.py +477 -0
- flash_rt/hardware/__init__.py +311 -0
- flash_rt/hardware/backend.py +407 -0
- flash_rt/hardware/blackwell/__init__.py +12 -0
- flash_rt/hardware/rtx/__init__.py +37 -0
- flash_rt/hardware/rtx/attn_backend.py +856 -0
- flash_rt/hardware/rtx/attn_backend_batched_pi05.py +303 -0
- flash_rt/hardware/rtx/attn_backend_chameleon.py +237 -0
- flash_rt/hardware/rtx/attn_backend_groot.py +447 -0
- flash_rt/hardware/rtx/attn_backend_groot_n17.py +251 -0
- flash_rt/hardware/rtx/attn_backend_groot_n17_backbone.py +191 -0
- flash_rt/hardware/rtx/attn_backend_motus.py +128 -0
- flash_rt/hardware/rtx/attn_backend_nexn2.py +409 -0
- flash_rt/hardware/rtx/attn_backend_qwen3.py +520 -0
- flash_rt/hardware/rtx/attn_backend_qwen36.py +272 -0
- flash_rt/hardware/thor/__init__.py +9 -0
- flash_rt/hardware/thor/attn_backend.py +559 -0
- flash_rt/hardware/thor/attn_backend_chameleon.py +362 -0
- flash_rt/hardware/thor/attn_backend_groot.py +328 -0
- flash_rt/hardware/thor/attn_backend_groot_n17.py +423 -0
- flash_rt/hardware/thor/attn_backend_qwen3.py +229 -0
- flash_rt/hardware/thor/attn_backend_qwen36.py +530 -0
- flash_rt/hardware/thor/fa4_backend.py +117 -0
- flash_rt/hardware/thor/shared_primitives.py +727 -0
- flash_rt/hardware/thor/shared_primitives_batched.py +178 -0
- flash_rt/hardware/thor/shared_primitives_fp4.py +512 -0
- flash_rt/hardware/thor/vqgan_trt_backend.py +187 -0
- flash_rt/models/__init__.py +11 -0
- flash_rt/models/chameleon/__init__.py +18 -0
- flash_rt/models/chameleon/pipeline_rtx.py +305 -0
- flash_rt/models/chameleon/pipeline_thor.py +1126 -0
- flash_rt/models/chameleon/vqvae_hf.py +124 -0
- flash_rt/models/cosmos3_edge/__init__.py +37 -0
- flash_rt/models/cosmos3_edge/action_only_official.py +3447 -0
- flash_rt/models/cosmos3_edge/boundary_dump.py +151 -0
- flash_rt/models/cosmos3_edge/denoise_ref.py +325 -0
- flash_rt/models/cosmos3_edge/dump_replay.py +195 -0
- flash_rt/models/cosmos3_edge/layer_ref.py +1040 -0
- flash_rt/models/cosmos3_edge/pipeline_thor.py +549 -0
- flash_rt/models/cosmos3_edge/static_engine.py +346 -0
- flash_rt/models/cosmos3_edge/static_unipc.py +234 -0
- flash_rt/models/cosmos3_edge/vae_native.py +304 -0
- flash_rt/models/cosmos3_edge/weights.py +91 -0
- flash_rt/models/cosmos3_reasoner/__init__.py +1 -0
- flash_rt/models/cosmos3_reasoner/pipeline_thor.py +691 -0
- flash_rt/models/cosmos3_video/__init__.py +6 -0
- flash_rt/models/cosmos3_video/fm_solvers_unipc.py +808 -0
- flash_rt/models/cosmos3_video/kernels/__init__.py +22 -0
- flash_rt/models/cosmos3_video/kernels/csrc/bindings.cpp +17 -0
- flash_rt/models/cosmos3_video/kernels/csrc/fused_qk_norm_rope.cu +61 -0
- flash_rt/models/cosmos3_video/kernels/setup.py +33 -0
- flash_rt/models/cosmos3_video/pipeline_rtx.py +234 -0
- flash_rt/models/groot/__init__.py +32 -0
- flash_rt/models/groot/embodiments.py +69 -0
- flash_rt/models/groot/pipeline_rtx.py +1179 -0
- flash_rt/models/groot/pipeline_rtx_fp16.py +1034 -0
- flash_rt/models/groot/pipeline_thor.py +981 -0
- flash_rt/models/groot_n17/__init__.py +50 -0
- flash_rt/models/groot_n17/calibration.py +439 -0
- flash_rt/models/groot_n17/embodiments.py +38 -0
- flash_rt/models/groot_n17/mrope_table.py +178 -0
- flash_rt/models/groot_n17/pipeline_rtx.py +11 -0
- flash_rt/models/groot_n17/pipeline_rtx_fp16.py +836 -0
- flash_rt/models/groot_n17/pipeline_rtx_fp8.py +413 -0
- flash_rt/models/groot_n17/pipeline_rtx_sm89.py +541 -0
- flash_rt/models/groot_n17/pipeline_thor.py +1386 -0
- flash_rt/models/higgs_audio_v3/__init__.py +14 -0
- flash_rt/models/higgs_audio_v3/_codec/__init__.py +0 -0
- flash_rt/models/higgs_audio_v3/_codec/env_guard.py +42 -0
- flash_rt/models/higgs_audio_v3/_codec/tokenizer_config.json +129 -0
- flash_rt/models/higgs_audio_v3/_codec/tokenizer_model.py +940 -0
- flash_rt/models/higgs_audio_v3/codec.py +81 -0
- flash_rt/models/higgs_audio_v3/pipeline_rtx.py +64 -0
- flash_rt/models/hyvla/__init__.py +1 -0
- flash_rt/models/hyvla/pipeline_orin.py +430 -0
- flash_rt/models/hyvla/pipeline_thor.py +572 -0
- flash_rt/models/lingbot/__init__.py +17 -0
- flash_rt/models/lingbot/_csrc_loader.py +70 -0
- flash_rt/models/lingbot/buffer_binder.py +156 -0
- flash_rt/models/lingbot/calibration.py +163 -0
- flash_rt/models/lingbot/forward.py +784 -0
- flash_rt/models/lingbot/fp4_ops.py +90 -0
- flash_rt/models/lingbot/graph_runner.py +265 -0
- flash_rt/models/lingbot/kernel_ops.py +1487 -0
- flash_rt/models/lingbot/mixed_attention.py +793 -0
- flash_rt/models/lingbot/norms.py +113 -0
- flash_rt/models/lingbot/pipeline_thor.py +169 -0
- flash_rt/models/lingbot/rope_adapter.py +156 -0
- flash_rt/models/lingbot/sample_actions.py +394 -0
- flash_rt/models/lingbot/vit.py +486 -0
- flash_rt/models/lingbot/vit_rope_adapter.py +247 -0
- flash_rt/models/ltx25/__init__.py +16 -0
- flash_rt/models/ltx25/_attn_swap.py +244 -0
- flash_rt/models/ltx25/_nvfp4_ffn_swap.py +301 -0
- flash_rt/models/ltx25/_resident_graph.py +206 -0
- flash_rt/models/melband_roformer/__init__.py +13 -0
- flash_rt/models/melband_roformer/pipeline.py +329 -0
- flash_rt/models/minimax_remover/__init__.py +27 -0
- flash_rt/models/minimax_remover/_attention.py +428 -0
- flash_rt/models/minimax_remover/_fp8_linear.py +426 -0
- flash_rt/models/minimax_remover/_fp8_manual_denoise.py +298 -0
- flash_rt/models/minimax_remover/_fp8_pipeline.py +617 -0
- flash_rt/models/minimax_remover/_kern_block.py +282 -0
- flash_rt/models/minimax_remover/_kernels.py +282 -0
- flash_rt/models/minimax_remover/_manual_denoise.py +413 -0
- flash_rt/models/minimax_remover/_nvfp4_linear.py +236 -0
- flash_rt/models/minimax_remover/_triton_flash_attn.py +139 -0
- flash_rt/models/minimax_remover/_utils.py +94 -0
- flash_rt/models/minimax_remover/_vae_nvfp4.py +569 -0
- flash_rt/models/minimax_remover/_vae_opt.py +880 -0
- flash_rt/models/minimax_remover/pipeline.py +209 -0
- flash_rt/models/motus/__init__.py +0 -0
- flash_rt/models/motus/_action_ffn_v6t_install.py +162 -0
- flash_rt/models/motus/_action_und_qkv_fp8_swap.py +266 -0
- flash_rt/models/motus/_attn_swap.py +224 -0
- flash_rt/models/motus/_awq_fp8_swap.py +525 -0
- flash_rt/models/motus/_easycache_swap.py +279 -0
- flash_rt/models/motus/_ffn_swap.py +225 -0
- flash_rt/models/motus/_fp8_swap.py +635 -0
- flash_rt/models/motus/_graph_capture.py +203 -0
- flash_rt/models/motus/_handtuned_fp8_dispatch.py +235 -0
- flash_rt/models/motus/_kv_cache_swap.py +539 -0
- flash_rt/models/motus/_linear_swap.py +248 -0
- flash_rt/models/motus/_mixcache_swap.py +276 -0
- flash_rt/models/motus/_modulate_fuse_swap.py +1649 -0
- flash_rt/models/motus/_motus_nvfp4_ffn_video_swap.py +677 -0
- flash_rt/models/motus/_norm_swap.py +240 -0
- flash_rt/models/motus/_rope_swap.py +226 -0
- flash_rt/models/motus/_stream.py +22 -0
- flash_rt/models/motus/_taylorseer_swap.py +275 -0
- flash_rt/models/motus/_teacache_swap.py +210 -0
- flash_rt/models/motus/_tinyfp8_dispatch_install.py +170 -0
- flash_rt/models/motus/_und_ffn_v5t_install.py +180 -0
- flash_rt/models/motus/_vae_fp4_swap.py +849 -0
- flash_rt/models/motus/_vae_fp8_resample_swap.py +534 -0
- flash_rt/models/motus/_vae_fp8_swap.py +1082 -0
- flash_rt/models/motus/_vae_swap.py +80 -0
- flash_rt/models/motus/_vae_time_conv_fp8_swap.py +292 -0
- flash_rt/models/motus/_wan_qkv_fuse_swap.py +905 -0
- flash_rt/models/motus/pipeline_rtx.py +1164 -0
- flash_rt/models/nexn2/__init__.py +17 -0
- flash_rt/models/nexn2/pipeline_rtx.py +137 -0
- flash_rt/models/omnivoice/__init__.py +30 -0
- flash_rt/models/omnivoice/pipeline_rtx.py +546 -0
- flash_rt/models/pi0/__init__.py +9 -0
- flash_rt/models/pi0/pipeline_rtx.py +1110 -0
- flash_rt/models/pi0/pipeline_thor.py +434 -0
- flash_rt/models/pi05/__init__.py +28 -0
- flash_rt/models/pi05/pipeline_rtx.py +2209 -0
- flash_rt/models/pi05/pipeline_rtx_batched.py +1188 -0
- flash_rt/models/pi05/pipeline_rtx_cfg.py +657 -0
- flash_rt/models/pi05/pipeline_rtx_cfg_batched.py +435 -0
- flash_rt/models/pi05/pipeline_rtx_fp16.py +2276 -0
- flash_rt/models/pi05/pipeline_thor.py +929 -0
- flash_rt/models/pi05/pipeline_thor_batched.py +346 -0
- flash_rt/models/pi05/pipeline_thor_cfg.py +238 -0
- flash_rt/models/pi05/pipeline_thor_cfg_batched.py +180 -0
- flash_rt/models/pi05/runtime_export.py +449 -0
- flash_rt/models/pi0fast/__init__.py +1 -0
- flash_rt/models/pi0fast/pipeline.py +840 -0
- flash_rt/models/qwen3/__init__.py +11 -0
- flash_rt/models/qwen3/pipeline_rtx.py +90 -0
- flash_rt/models/qwen36/__init__.py +27 -0
- flash_rt/models/qwen36/pipeline_rtx.py +159 -0
- flash_rt/models/qwen3_vl/__init__.py +19 -0
- flash_rt/models/qwen3_vl/pipeline_rtx.py +145 -0
- flash_rt/models/wan22/__init__.py +1 -0
- flash_rt/models/wan22/pipeline_rtx.py +28 -0
- flash_rt/npu/__init__.py +6 -0
- flash_rt/npu/core/__init__.py +0 -0
- flash_rt/npu/core/abi.py +81 -0
- flash_rt/npu/core/acl_runtime.py +165 -0
- flash_rt/npu/core/decode_attention.py +26 -0
- flash_rt/npu/core/decoder_int8.py +399 -0
- flash_rt/npu/core/device.py +73 -0
- flash_rt/npu/core/gu_int8.py +99 -0
- flash_rt/npu/core/linear.py +283 -0
- flash_rt/npu/core/native_kernels.py +269 -0
- flash_rt/npu/core/npu_graph.py +68 -0
- flash_rt/npu/frontends/__init__.py +0 -0
- flash_rt/npu/frontends/torch/__init__.py +0 -0
- flash_rt/npu/frontends/torch/pi05.py +438 -0
- flash_rt/npu/hardware/__init__.py +9 -0
- flash_rt/npu/models/__init__.py +1 -0
- flash_rt/npu/models/pi05/__init__.py +1 -0
- flash_rt/npu/models/pi05/attention.py +128 -0
- flash_rt/npu/models/pi05/captured.py +212 -0
- flash_rt/npu/models/pi05/fast.py +666 -0
- flash_rt/npu/models/pi05/pipeline.py +336 -0
- flash_rt/npu/models/pi05/quantization.py +235 -0
- flash_rt/npu/verify.py +110 -0
- flash_rt/py.typed +0 -0
- flash_rt/refs/__init__.py +17 -0
- flash_rt/refs/pi05_cfg_reference.py +310 -0
- flash_rt/runtime/__init__.py +47 -0
- flash_rt/runtime/cuda_libraries.py +108 -0
- flash_rt/runtime/exec.py +53 -0
- flash_rt/runtime/export.py +520 -0
- flash_rt/runtime/provider.py +113 -0
- flash_rt/runtime/rtc.py +261 -0
- flash_rt/runtime/rtc_temporal_fusion.py +545 -0
- flash_rt/runtime/vlash.py +420 -0
- flash_rt/subgraphs/__init__.py +43 -0
- flash_rt/subgraphs/capture.py +179 -0
- flash_rt/subgraphs/pi05/__init__.py +3 -0
- flash_rt/subgraphs/pi05/context_action.py +67 -0
- flash_rt/subgraphs/pi05/rtc_prefix.py +79 -0
- flash_rt/subgraphs/pi05/rtc_vjp_guided.py +96 -0
- flash_rt/subgraphs/pi05/stage_plans.py +89 -0
- flash_rt/subgraphs/pi05/vlash.py +29 -0
- flash_rt/subgraphs/stage_plan.py +214 -0
- flash_rt/utils/__init__.py +1 -0
- flash_rt/utils/paligemma_tokenizer.py +135 -0
- flash_rt-0.2.0.dist-info/METADATA +1341 -0
- flash_rt-0.2.0.dist-info/RECORD +455 -0
- flash_rt-0.2.0.dist-info/WHEEL +5 -0
- flash_rt-0.2.0.dist-info/licenses/LICENSE +202 -0
- flash_rt-0.2.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""FlashRT -- Nex-N2-mini model pipelines.
|
|
2
|
+
|
|
3
|
+
pipeline_rtx.py - Nexn2Dims (static dims) + Nexn2Pipeline (BF16 HF
|
|
4
|
+
reference, used only for kernelized=False).
|
|
5
|
+
|
|
6
|
+
Nex-N2-mini is the MoE sibling of the dense qwen36 family
|
|
7
|
+
(architectures=Qwen3_5MoeForConditionalGeneration / model_type=qwen3_5_moe):
|
|
8
|
+
hybrid Gated-DeltaNet + softmax-attention with a fine-grained 256-expert MoE
|
|
9
|
+
FFN. The production NVFP4 kernel forward/decode (CUDA-graph, chunked
|
|
10
|
+
long-context prefill) lives in flash_rt.frontends.torch._nexn2_rtx_*.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from flash_rt.models.nexn2.pipeline_rtx import Nexn2Pipeline
|
|
14
|
+
|
|
15
|
+
__all__ = [
|
|
16
|
+
'Nexn2Pipeline',
|
|
17
|
+
]
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""FlashRT -- Nex-N2-mini static dims + BF16 reference pipeline.
|
|
2
|
+
|
|
3
|
+
This module holds the static dimension constants (``Nexn2Dims``) and the
|
|
4
|
+
``Nexn2Pipeline`` BF16-eager wrapper around the HF reference model. The
|
|
5
|
+
reference pipeline is used **only** when the frontend is constructed with
|
|
6
|
+
``kernelized=False`` (the correctness baseline for the golden cosine fixture);
|
|
7
|
+
the production path -- NVFP4 kernels, CUDA-graph decode, chunked long-context
|
|
8
|
+
prefill -- lives in the frontend forward/decode modules
|
|
9
|
+
(``flash_rt.frontends.torch._nexn2_rtx_{forward,decode}``), not here.
|
|
10
|
+
|
|
11
|
+
Architecture summary (Nex-N2-mini = model_type qwen3_5_moe)::
|
|
12
|
+
|
|
13
|
+
[input_ids]
|
|
14
|
+
|
|
|
15
|
+
v embed_tokens (BF16, vocab=248320, hidden=2048)
|
|
16
|
+
v
|
|
17
|
+
40 decoder layers, alternating linear-attn (3) + full-attn (1):
|
|
18
|
+
layer 0,1,2: linear_attention (Gated DeltaNet, conv1d k=4,
|
|
19
|
+
16 K-heads / 32 V-heads)
|
|
20
|
+
layer 3: full_attention (GQA 16Q/2KV, head_dim=256,
|
|
21
|
+
output_gate, partial RoPE 0.25)
|
|
22
|
+
layer 4..39: same pattern repeats (linear x3, full x1) ...
|
|
23
|
+
|
|
|
24
|
+
v per layer: RMSNorm -> attn (linear or full)
|
|
25
|
+
v + residual -> RMSNorm -> MoE FFN -> residual
|
|
26
|
+
v MoE: 256 experts, top-8 routed + 1 shared expert
|
|
27
|
+
v
|
|
28
|
+
v final RMSNorm -> lm_head (BF16, untied)
|
|
29
|
+
v
|
|
30
|
+
[logits: (B, S, 248320)]
|
|
31
|
+
|
|
32
|
+
The config declares one MTP (multi-token-prediction) layer, but the released
|
|
33
|
+
Nex-N2-mini checkpoint ships no MTP tensors, so speculative decode is not wired.
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
from __future__ import annotations
|
|
37
|
+
|
|
38
|
+
from dataclasses import dataclass, field
|
|
39
|
+
from typing import Any
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
@dataclass
|
|
43
|
+
class Nexn2Dims:
|
|
44
|
+
"""Static dimension constants for Nex-N2-mini.
|
|
45
|
+
|
|
46
|
+
Source: config.json:text_config (model_type=qwen3_5_moe_text). Fixed
|
|
47
|
+
for the mini (35B-A3B) variant; if another size is added later this
|
|
48
|
+
becomes a per-checkpoint loader instead of a class-level constant.
|
|
49
|
+
"""
|
|
50
|
+
hidden: int = 2048
|
|
51
|
+
num_layers: int = 40
|
|
52
|
+
full_attn_period: int = 4 # full at indices 3, 7, ..., 39
|
|
53
|
+
vocab_size: int = 248320
|
|
54
|
+
rms_norm_eps: float = 1e-6
|
|
55
|
+
|
|
56
|
+
# full-attention sites (10 layers)
|
|
57
|
+
full_q_heads: int = 16
|
|
58
|
+
full_kv_heads: int = 2 # GQA 8:1
|
|
59
|
+
full_head_dim: int = 256
|
|
60
|
+
partial_rotary_factor: float = 0.25 # rotary_dim = 64
|
|
61
|
+
rope_theta: float = 1.0e7
|
|
62
|
+
mrope_section: tuple[int, ...] = (11, 11, 10)
|
|
63
|
+
|
|
64
|
+
# linear-attention sites (30 layers, Gated DeltaNet)
|
|
65
|
+
lin_k_heads: int = 16
|
|
66
|
+
lin_v_heads: int = 32 # differs from qwen36 (48)
|
|
67
|
+
lin_head_dim: int = 128
|
|
68
|
+
lin_conv_kernel: int = 4
|
|
69
|
+
|
|
70
|
+
# MoE FFN (every layer)
|
|
71
|
+
moe_num_experts: int = 256
|
|
72
|
+
moe_experts_per_tok: int = 8
|
|
73
|
+
moe_intermediate: int = 512
|
|
74
|
+
shared_expert_intermediate: int = 512
|
|
75
|
+
|
|
76
|
+
# MTP head
|
|
77
|
+
mtp_layers: int = 1
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class Nexn2Pipeline:
|
|
81
|
+
"""BF16 HF reference pipeline for Nex-N2-mini.
|
|
82
|
+
|
|
83
|
+
Hosts an HF reference model and delegates ``forward`` / ``generate``. Used
|
|
84
|
+
only by the frontend's ``kernelized=False`` path (the correctness baseline);
|
|
85
|
+
the production NVFP4 kernel forward/decode lives in
|
|
86
|
+
``flash_rt.frontends.torch._nexn2_rtx_{forward,decode}``.
|
|
87
|
+
"""
|
|
88
|
+
|
|
89
|
+
DIMS = Nexn2Dims()
|
|
90
|
+
|
|
91
|
+
def __init__(self, hf_model: Any) -> None:
|
|
92
|
+
"""Wrap an HF model object (the qwen3_5_moe auto-loader output)."""
|
|
93
|
+
self.hf = hf_model
|
|
94
|
+
self.config = hf_model.config
|
|
95
|
+
text_cfg = getattr(self.config, 'text_config', self.config)
|
|
96
|
+
# Sanity-check the dim assumptions against the checkpoint config.
|
|
97
|
+
assert text_cfg.hidden_size == self.DIMS.hidden, (
|
|
98
|
+
f'expected hidden={self.DIMS.hidden}, got {text_cfg.hidden_size}'
|
|
99
|
+
)
|
|
100
|
+
assert text_cfg.num_hidden_layers == self.DIMS.num_layers
|
|
101
|
+
assert text_cfg.head_dim == self.DIMS.full_head_dim
|
|
102
|
+
assert text_cfg.num_experts == self.DIMS.moe_num_experts
|
|
103
|
+
assert (
|
|
104
|
+
text_cfg.layer_types.count('full_attention')
|
|
105
|
+
== self.DIMS.num_layers // self.DIMS.full_attn_period
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
def forward(self, input_ids):
|
|
109
|
+
"""Single forward pass: token IDs -> logits (BF16 HF reference).
|
|
110
|
+
|
|
111
|
+
Args:
|
|
112
|
+
input_ids: (B, S) torch.long on cuda.
|
|
113
|
+
|
|
114
|
+
Returns:
|
|
115
|
+
logits: (B, S, vocab_size) bf16 on cuda.
|
|
116
|
+
"""
|
|
117
|
+
import torch # local import; pipeline_rtx is import-time-light.
|
|
118
|
+
with torch.no_grad():
|
|
119
|
+
out = self.hf(
|
|
120
|
+
input_ids=input_ids, use_cache=False, return_dict=True,
|
|
121
|
+
)
|
|
122
|
+
return out.logits
|
|
123
|
+
|
|
124
|
+
def generate(self, input_ids, *, max_new_tokens: int, do_sample: bool = False):
|
|
125
|
+
"""Greedy/sampled autoregressive generate (BF16 HF reference path).
|
|
126
|
+
|
|
127
|
+
The production decode (CUDA-graph, on-device argmax) is in the
|
|
128
|
+
frontend's kernelized path, not here.
|
|
129
|
+
"""
|
|
130
|
+
import torch
|
|
131
|
+
with torch.no_grad():
|
|
132
|
+
return self.hf.generate(
|
|
133
|
+
input_ids=input_ids,
|
|
134
|
+
max_new_tokens=max_new_tokens,
|
|
135
|
+
do_sample=do_sample,
|
|
136
|
+
use_cache=True,
|
|
137
|
+
)
|
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""FlashRT — OmniVoice TTS model pipeline.
|
|
2
|
+
|
|
3
|
+
Mixed BF16 CFG + FP4 noCFG acceleration preserves audio quality at 5.0x
|
|
4
|
+
throughput on Blackwell SM120 GPUs.
|
|
5
|
+
|
|
6
|
+
Per the unified API contract:
|
|
7
|
+
inject() — patch OmniVoice model for FlashRT acceleration
|
|
8
|
+
free_encoder() — release encoder weights (~600 MB VRAM saved)
|
|
9
|
+
eject() — restore original forward and generate methods
|
|
10
|
+
|
|
11
|
+
See docs/PERFORMANCE_OMNIVOICE.md for performance specifications.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from flash_rt.models.omnivoice.pipeline_rtx import (
|
|
15
|
+
FlashRTLlm,
|
|
16
|
+
FlashRTLlmBF16,
|
|
17
|
+
inject,
|
|
18
|
+
free_encoder,
|
|
19
|
+
eject,
|
|
20
|
+
_check_kernels,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"FlashRTLlm",
|
|
25
|
+
"FlashRTLlmBF16",
|
|
26
|
+
"inject",
|
|
27
|
+
"free_encoder",
|
|
28
|
+
"eject",
|
|
29
|
+
"_check_kernels",
|
|
30
|
+
]
|