# DeepSeek V4 NVFP4 vLLM + CuTeDSL NVFP4 MoE Kernel FROM vllm/vllm-openai:nightly-x86_64 # Remove broken nixl_ep (built against CUDA 12, image is CUDA 13) RUN pip uninstall -y nixl-ep; rm -rf /usr/local/lib/python3.12/dist-packages/nixl_ep RUN apt-get update && apt-get install -y git screen cmake libcusolver-dev-13-0 libcusparse-dev-13-0 libcublas-dev-13-0 libcurand-dev-13-0 libcufft-dev-13-0 libnvjitlink-dev-13-0 && rm -rf /var/lib/apt/lists/* # Remove the broken symlink if it exists RUN rm -f /usr/local/cuda/lib64/libcudrt.so.12 ENV CUDA_HOME=/usr/local/cuda ENV TORCH_CUDA_ARCH_LIST="10.0" # Install CuTeDSL (NVFP4 block-scaled GEMM kernel framework) RUN pip install nvidia-cutlass-dsl==4.5.0 nvidia-cutlass-dsl-libs-base==4.5.0 ARG CACHE_BUSTER=${TIMESTAMP} # Copy the NVFP4 mega_moe Python kernel (no C++ build needed) COPY src/ /root/nvfp4-megamoe-kernel/src/ COPY pyproject.toml /root/nvfp4-megamoe-kernel/pyproject.toml RUN cd /root/nvfp4-megamoe-kernel && pip install -e . # Copy the CuTeDSL kernel and bridge layer COPY cutedsl/ /root/nvfp4-megamoe-kernel/cutedsl/ ENV PYTHONPATH="/root/nvfp4-megamoe-kernel:${PYTHONPATH}" # Patch vLLM — overwrite model files and register architecture ARG VLLM_MODELS_DIR=/usr/local/lib/python3.12/dist-packages/vllm/model_executor/models ARG VLLM_LAYERS_DIR=/usr/local/lib/python3.12/dist-packages/vllm/model_executor/layers COPY vllm/patches/deepseek_v4.py ${VLLM_MODELS_DIR}/deepseek_v4.py COPY vllm/patches/deepseek_v4_attention.py ${VLLM_LAYERS_DIR}/deepseek_v4_attention.py COPY vllm/nvfp4_cutedsl.py ${VLLM_MODELS_DIR}/nvfp4_cutedsl.py RUN sed -i 's/"DeepseekV32ForCausalLM": ("deepseek_v2", "DeepseekV3ForCausalLM"),/"DeepseekV32ForCausalLM": ("deepseek_v2", "DeepseekV3ForCausalLM"),\n "DeepseekV4ForCausalLM": ("deepseek_v4", "DeepseekV4ForCausalLM"),/' \ ${VLLM_MODELS_DIR}/registry.py # Patch process_weights_after_loading to call model._post_quant_fix() after quant setup ARG VLLM_LOADER_DIR=/usr/local/lib/python3.12/dist-packages/vllm/model_executor/model_loader RUN python3 -c " import re path = '${VLLM_LOADER_DIR}/utils.py'.replace('\$', '') with open(path) as f: src = f.read() # Add _post_quant_fix() call at end of process_weights_after_loading old = ' if model_config.quantization == \"torchao\":' new = ''' # Custom: allow models to run post-quant-init fixes if hasattr(model, '_post_quant_fix'): model._post_quant_fix() if model_config.quantization == \"torchao\":''' src = src.replace(old, new, 1) with open(path, 'w') as f: f.write(src) print('Patched process_weights_after_loading') " # Verify RUN python3 -c "import torch; print(f'PyTorch {torch.__version__} OK')" && \ python3 -c "import vllm; print('vLLM OK')" && \ python3 -c "import nvfp4_megamoe_kernel; print('NVFP4 kernel OK')" && \ python3 -c "import cutlass; print('CuTeDSL OK')"