vllm
A high-throughput and memory-efficient inference and serving engine for LLMs
File Explorer
Download Latest Version (.zip)Showing a partial file list β download the zip above to see everything.
- cluster.sh
- models.yaml
- pipeline-disagg.yaml
- run-slurm-disagg-test.sh
- run_xPyD_disagg.slurm
- vllm_disagg.sh
- test-intel.yaml
- amd.yaml
- ascend_npu.yaml
- cpu.yaml
- gh200.yaml
- intel.yaml
- image_build.sh
- image_build.yaml
- image_build_arm64.sh
- image_build_cpu.sh
- image_build_cpu_arm64.sh
- image_build_hpu.sh
- image_build_torch_nightly.sh
- image_build_xpu.sh
- basic_correctness_intel.yaml
- benchmarks_intel.yaml
- engine_intel.yaml
- expert_parallelism_intel.yaml
- kernels_intel.yaml
- lm_eval_intel.yaml
- lora_intel.yaml
- misc_intel.yaml
- model_executor_intel.yaml
- model_runner_v2_intel.yaml
- models_distributed_intel.yaml
- models_multimodal_intel.yaml
- quantization.yaml
- samplers_intel.yaml
- test-intel.yaml
- DeepSeek-V2-Lite-Chat.yaml
- Meta-Llama-3-70B-Instruct-FBGEMM-nonuniform.yaml
- Meta-Llama-3-70B-Instruct.yaml
- Meta-Llama-3-8B-Instruct-Channelwise-compressed-tensors.yaml
- Meta-Llama-3-8B-Instruct-FBGEMM-nonuniform.yaml
- Meta-Llama-3-8B-Instruct-FP8-compressed-tensors.yaml
- Meta-Llama-3-8B-Instruct-FP8.yaml
- Meta-Llama-3-8B-Instruct-INT8-compressed-tensors-asym.yaml
- Meta-Llama-3-8B-Instruct-INT8-compressed-tensors.yaml
- Meta-Llama-3-8B-Instruct-nonuniform-compressed-tensors.yaml
- Meta-Llama-3-8B-Instruct.yaml
- Meta-Llama-3-8B-QQQ.yaml
- Meta-Llama-3.2-1B-Instruct-FP8-compressed-tensors.yaml
- Meta-Llama-3.2-1B-Instruct-INT8-compressed-tensors.yaml
- Meta-Llama-4-Maverick-17B-128E-Instruct-FP8-MM.yaml
- Meta-Llama-4-Maverick-17B-128E-Instruct-FP8.yaml
- Minitron-4B-Base-FP8.yaml
- Mixtral-8x22B-Instruct-v0.1-FP8-Dynamic.yaml
- Mixtral-8x7B-Instruct-v0.1-FP8.yaml
- Mixtral-8x7B-Instruct-v0.1.yaml
- models-large-hopper.txt
- models-large-rocm-fp8.txt
- models-large-rocm.txt
- models-large.txt
- models-mm-large-h100.txt
- models-mm-small.txt
- models-small-rocm.txt
- models-small.txt
- NVIDIA-Nemotron-3-Nano-30B-A3B-BF16.yaml
- NVIDIA-Nemotron-3-Nano-30B-A3B-FP8.yaml
- Qwen1.5-MoE-W4A16-compressed-tensors.yaml
- Qwen2-1.5B-Instruct-FP8W8.yaml
- Qwen2-1.5B-Instruct-INT8-compressed-tensors.yaml
- Qwen2-57B-A14-Instruct.yaml
- Qwen2.5-1.5B-Instruct.yaml
- Qwen2.5-VL-3B-Instruct-FP8-dynamic.yaml
- Qwen2.5-VL-7B-Instruct.yaml
- Qwen3-235B-A22B-Instruct-2507-FP8.yaml
- conftest.py
- run-lm-eval-chartqa-vllm-vlm-baseline.sh
- run-lm-eval-gsm-hf-baseline.sh
- run-lm-eval-gsm-vllm-baseline.sh
- run-lm-eval-mmlupro-vllm-baseline.sh
- test_lm_eval_correctness.py
- compare-json-results.py
- convert-results-json-to-markdown.py
- launch-server.sh
- run-performance-benchmarks.sh
- genai-perf-tests.json
- latency-tests-arm64-cpu.json
- latency-tests-cpu.json
- latency-tests-hpu.json
- latency-tests.json
- nightly-tests.json
- serving-tests-arm64-cpu.json
- serving-tests-cpu-asr.json
- serving-tests-cpu-embed.json
- serving-tests-cpu-text.json
- serving-tests-cpu.json
- serving-tests-hpu.json
- serving-tests.json
- throughput-tests-arm64-cpu.json
- throughput-tests-cpu.json
- throughput-tests-hpu.json
- throughput-tests.json
- performance-benchmarks-descriptions.md
- README.md
- run-amd-test.sh
- run-cpu-compatibility-test.sh
- run-cpu-distributed-smoke-test.sh
- run-cpu-test-arm.sh
- run-cpu-test-ppc64le.sh
- run-cpu-test-s390x.sh
- run-cpu-test.sh
- run-gh200-test.sh
- run-hpu-test.sh
- run-intel-ci-test.sh
- run-intel-test.sh
- run-npu-test.sh
- run-tpu-v1-test-part2.sh
- run-tpu-v1-test.sh
- manylinux.sh
- select-python.sh
- build-ci-base.sh
- build-test-image.sh
- refresh-base-image.sh
- smoke-test-image.sh
- deepseek_v2_lite_ep_eplb.sh
- deepseek_v2_lite_prefetch_offload.sh
- qwen30b_a3b_fp8_block_ep_eplb.sh
- qwen30b_a3b_fp8_dp4_async_eplb.sh
- qwen3_next_mtp_async_eplb.sh
- run-bfcl-eval.sh
- cleanup_docker.sh
- config_v6e_1.env
- docker_run_bm.sh
- quantized_v6e_1.env
- run_bm.sh
- create-xpu-ecr-manifest.sh
- publish-triton-shim.sh
- push-nightly-builds-xpu.sh
- annotate-build-artifact.sh
- annotate-image-build.sh
- annotate-rocm-release.sh
- build-macos-wheel.sh
- cache-rocm-base-wheels.sh
- check-ray-compatibility.sh
- cherry-pick-from-milestone.sh
- ci-bake-rocm.sh
- ci-clean-log.sh
- ci-fetch-log.sh
- cleanup-nightly-builds.sh
- detect-manylinux-tag.py
- docker-build-metadata-args.sh
- generate-and-upload-nightly-index.sh
- generate-nightly-index.py
- install-kv-connectors.sh
- publish-release-images.sh
- push-nightly-builds-rocm.sh
- push-nightly-builds.sh
- rerun-test.sh
- run-benchmarks.sh
- run-multi-node-test.sh
- run-rust-frontend-cargo-ci.sh
- trigger-ci-build.sh
- upload-nightly-wheels.sh
- upload-release-wheels-pypi.sh
- upload-rocm-wheels.sh
- attention.yaml
- basic_correctness.yaml
- benchmarks.yaml
- compile.yaml
- cuda.yaml
- disaggregated.yaml
- disaggregated_mooncake.yaml
- distributed.yaml
- docker.yaml
- e2e_integration.yaml
- engine.yaml
- entrypoints.yaml
- expert_parallelism.yaml
- fault_tolerance.yaml
- jit_monitor.yaml
- kernels.yaml
- lm_eval.yaml
- lora.yaml
- misc.yaml
- model_executor.yaml
- model_runner_v2.yaml
- models_basic.yaml
- models_distributed.yaml
- models_language.yaml
- models_multimodal.yaml
- plugins.yaml
- pytorch.yaml
- quantization.yaml
- ray_compat.yaml
- rust_frontend.yaml
- rust_frontend_cargo.yaml
- samplers.yaml
- spec_decode.yaml
- torch_abi.yaml
- weight_loading.yaml
- .pipeline_gen_v2
- check-torch-abi.py
- check-wheel-size.py
- ci_config.yaml
- ci_config_intel.yaml
- ci_config_rocm.yaml
- release-pipeline.yaml
- test-amd.yaml
- test-pipeline.yaml
- SKILL.md
- config.yaml
- 100-documentation.yml
- 200-installation.yml
- 300-usage.yml
- 400-bug-report.yml
- 450-ci-failure.yml
- 500-feature-request.yml
- 700-performance-discussion.yml
- 750-RFC.yml
- config.yml
- actionlint.json
- markdownlint.json
- mypy.json
- shellcheck.json
- build.sh
- create_release.js
- cuda-install.sh
- env.sh
- pytorch-install.sh
- run_ci_command.py
- test_run_ci_command.py
- add_label_automerge.yml
- buf.yml
- issue_autolabel.yml
- macos-smoke-test.yml
- new_pr_bot.yml
- notify-ci-authorized.yml
- pre-commit.yml
- record-ci-approval.yml
- run-ci-command.yml
- stale.yml
- actionlint.yaml
- CODEOWNERS
- dependabot.yml
- FUNDING.yml
- mergify.yml
- PULL_REQUEST_TEMPLATE.md
- scale-config.yml
- mla_decode.yaml
- mla_fa4_fp8_output.yaml
- mla_mixed_batch.yaml
- mla_prefill.yaml
- mla_sparse_decode.yaml
- mla_sparse_masked_mha_vs_mqa.yaml
- mla_sparse_mha_vs_mqa.yaml
- mla_sparse_prefill.yaml
- reorder_threshold.yaml
- speculative_decode.yaml
- standard_attention.yaml
- standard_decode.yaml
- standard_prefill.yaml
- __init__.py
- batch_spec.py
- benchmark.py
- common.py
- mla_runner.py
- README.md
- runner.py
- auto_tune.sh
- batch_auto_tune.sh
- README.md
- utils.py
- w8a8_benchmarks.py
- weight_shapes.py
- layernorm_rms_benchmarks.py
- merge_attn_states_benchmarks.py
- silu_mul_block_quant_benchmark.py
- benchmark_cpu_attn.py
- benchmark_cpu_fused_moe.py
- benchmark_fp8_block_dense_gemm.py
- README.md
- __init__.py
- bench_ir_ops.py
- shapes.py
- __init__.py
- bench_concat_mla_q.py
- bench_cp_gather_fp8.py
- benchmark_2d_silu_mul_fp8_quant.py
- benchmark_activation.py
- benchmark_block_fp8_gemm.py
- benchmark_cp_gather.py
- benchmark_cutlass_moe_fp8.py
- benchmark_cutlass_moe_nvfp4.py
- benchmark_device_communicators.py
- benchmark_flydsl_moe_w4a16.py
- benchmark_fp8_gemm.py
- benchmark_fused_collective.py
- benchmark_fused_moe_lora_one_shot.py
- benchmark_fused_q_cutedsl.py
- benchmark_fused_topk.py
- benchmark_grouped_gemm_cutlass.py
- benchmark_inkling_qkvr_prep.py
- benchmark_int8_gemm.py
- benchmark_k3_cutedsl_residual.py
- benchmark_kimi_k3_gemm_rs_ar.py
- benchmark_kimi_k3_kda_decode.py
- benchmark_kimi_k3_latent_moe_tail.py
- benchmark_kimi_k3_sp_collectives.py
- benchmark_layernorm.py
- benchmark_lora.py
- benchmark_machete.py
- benchmark_marlin.py
- benchmark_mla_k_concat.py
- benchmark_moe.py
- benchmark_moe_align_block_size.py
- benchmark_moe_defaults.py
- benchmark_moe_permute_unpermute.py
- benchmark_mrope.py
- benchmark_mxfp4_qutlass.py
- benchmark_nvfp4_gemm.py
- benchmark_nvfp4_quant.py
- benchmark_nvfp4_qutlass.py
- benchmark_paged_attention.py
- benchmark_per_token_group_quant.py
- benchmark_per_token_quant_fp8.py
- benchmark_quant.py
- benchmark_rdna_hybrid_w4a16_gemm.py
- benchmark_relu_squared.py
- benchmark_reshape_and_cache.py
- benchmark_reshape_and_cache_flash.py
- benchmark_rmsnorm.py
- benchmark_rope.py
- benchmark_router_gemm.py
- benchmark_selective_state_update.py
- benchmark_shapes.py
- benchmark_silu_mul_fp8_quant.py
- benchmark_thinking_budget.py
- benchmark_trtllm_decode_attention.py
- benchmark_trtllm_prefill_attention.py
- benchmark_vit_aiter_fp8_attn.py
- benchmark_vit_bilinear_pos_embed.py
- benchmark_vit_fp8_attn.py
- benchmark_w8a8_block_fp8.py
- graph_machete_bench.py
- requirements.txt
- utils.py
- weight_shapes.py
- bench_dataset.py
- bench_utils.py
- benchmark_serving_multi_turn.py
- convert_sharegpt_to_openai.py
- generate_multi_turn.json
- README.md
- requirements.txt
- benchmark_hashing.py
- e2e_decode_speedup.py
- structured_schema_1.json
- __init__.py
- backend_request_func.py
- benchmark_batch_invariance.py
- benchmark_block_pool.py
- benchmark_hash.py
- benchmark_hidden_state_extraction.py
- benchmark_latency.py
- benchmark_long_document_qa_throughput.py
- benchmark_ngram_proposer.py
- benchmark_pin_memory.py
- benchmark_prefix_block_hash.py
- benchmark_prefix_caching.py
- benchmark_prioritization.py
- benchmark_serving.py
- benchmark_serving_structured_output.py
- benchmark_throughput.py
- benchmark_topk_topp.py
- benchmark_utils.py
- kv_cache_watermark.sh
- README.md
- run_structured_output_benchmark.sh
- sonnet.txt
- deepgemm.cmake
- flashkda.cmake
- flashmla.cmake
- fmha_sm100.cmake
- qutlass.cmake
- tml_fa4.cmake
- triton_kernels.cmake
- vllm_flash_attn.cmake
- pytorch_stable_string.patch
- cpu_extension.cmake
- hipify.py
- utils.cmake
- attention_dtypes.h
- attention_generic.cuh
- dtype_bfloat16.cuh
- dtype_float16.cuh
- dtype_float32.cuh
- dtype_fp8.cuh
- batch_invariant.hpp
- exception.hpp
- registration.h
- scalar_type.hpp
- cpu_micro_gemm_amx.hpp
- cpu_micro_gemm_impl.hpp
- cpu_micro_gemm_int8_neon.hpp
- cpu_micro_gemm_neon.hpp
- cpu_micro_gemm_rvv.hpp
- cpu_micro_gemm_vec.hpp
- cpu_micro_gemm_vsx.hpp
- blas_gemm.h
- bmm.cpp
- common.h
- conv.cpp
- decode.cpp
- extend.cpp
- fla.cpp
- flash_attn.h
- gemm.cpp
- gemm.h
- gemm_fp8.cpp
- gemm_int4.cpp
- gemm_int8.cpp
- mla_cache.cpp
- moe.cpp
- moe.h
- moe_fp8.cpp
- moe_int4.cpp
- moe_int8.cpp
- vec.h
- vec_pack.h
- activation.cpp
- activation_lut_bf16.cpp
- cpu_arch_macros.h
- cpu_attn.cpp
- cpu_attn_amx.hpp
- cpu_attn_fp8.hpp
- cpu_attn_impl.hpp
- cpu_attn_neon.hpp
- cpu_attn_neon_bfmmla.hpp
- cpu_attn_rvv.hpp
- cpu_attn_vec.hpp
- cpu_attn_vec16.hpp
- cpu_attn_vsx.hpp
- cpu_attn_vxe.hpp
- cpu_fused_moe.cpp
- cpu_fused_moe_activations.hpp
- cpu_fused_moe_int8.cpp
- cpu_tanhf_neon.hpp
- cpu_types.hpp
- cpu_types_arm.hpp
- cpu_types_riscv.hpp
- cpu_types_riscv_defs.hpp
- cpu_types_riscv_impl.hpp
- cpu_types_scalar.hpp
- cpu_types_vsx.hpp
- cpu_types_vxe.hpp
- cpu_types_x86.hpp
- cpu_wna16.cpp
- dnnl_helper.cpp
- dnnl_helper.h
- dnnl_kernels.cpp
- float_convert.hpp
- generate_cpu_attn_dispatch.py
- layernorm.cpp
- mamba_cpu.cpp
- mamba_kernels.hpp
- mla_decode.cpp
- pos_encoding.cpp
- shm.cpp
- spec_decode_utils.cpp
- torch_bindings.cpp
- utils.cpp
- utils.hpp
- broadcast_load_epilogue_array_c3x.hpp
- broadcast_load_epilogue_c3x.hpp
- cute_utils.cuh
- vllm_custom_types.cuh
- vllm_cutlass_library_extension.py
- vllm_type_utils.cuh
- dcp_direct_a2a_lse_reduce.cu
- dcp_direct_common.cuh
- dcp_direct_kv_gather.cu
- dcp_direct_q_gather.cu
- sm100_mla.hpp
- sm100_fmha_mla_reduction.hpp
- sm100_fmha_mla_tma_warpspecialized.hpp
- sm100_mla_tile_scheduler.hpp
- sm100_cutlass_mla_kernel.cu
- attention_utils.cuh
- merge_attn_states.cu
- math.hpp
- broadcast_load_epilogue_c2x.hpp
- scaled_mm_epilogues_c2x.hpp
- scaled_mm_epilogues_c3x.hpp
- common.cpp
- common.hpp
- torch_utils.hpp
- vllm_collective_builder.cuh
- vllm_numeric_conversion.cuh
- fused_gdn_decode_kernel.cu
- attn_res_kernel.cu
- fused_kda_decode_kernel.cu
- fused_kda_decode_kernel_rocm.cu
- selective_scan.h
- selective_scan_fwd.cu
- static_switch.h
- .gitignore
- generate_kernels.py
- kernel.h
- marlin_template.h
- ops.cu
- dispatch.h
- moe_permute_unpermute_kernel.cu
- moe_permute_unpermute_kernel.h
- moe_permute_unpermute_kernel.inl
- dsv3_router_gemm_bf16_out.cu
- dsv3_router_gemm_entry.cu
- dsv3_router_gemm_float_out.cu
- grouped_topk_kernels.cu
- moe_align_sum_kernels.cu
- moe_ops.h
- moe_permute_unpermute_op.cu
- moe_wna16.cu
- moe_wna16_utils.h
- moeTopKFuncs.cuh
- topk_softmax_kernels.cu
- topk_softplus_sqrt_kernels.cu
- torch_bindings.cpp
- dequantize.cuh
- gemm_kernels.cu
- get_group_starts.cuh
- w4a8_grouped_mm_entry.cu
- w4a8_mm_entry.cu
- w4a8_utils.cu
- w4a8_utils.cuh
- activation_nvfp4_quant_fusion_kernels.cu
- mxfp4_blockwise_moe_kernel.cu
- mxfp4_experts_quant.cu
- nvfp4_blockwise_moe_kernel.cu
- nvfp4_experts_quant.cu
- nvfp4_quant_entry.cu
- nvfp4_quant_kernels.cu
- nvfp4_scaled_mm_entry.cu
- nvfp4_scaled_mm_kernels.cu
- nvfp4_scaled_mm_sm120_kernels.cu
- nvfp4_utils.cuh
- fused_layernorm_dynamic_per_token_quant.cu
- fused_silu_mul_block_quant.cu
- layernorm_utils.cuh
- quant_conversions.cuh
- compat.cuh
- matrix_view.cuh
- q_gemm.cu
- qdq_2.cuh
- qdq_3.cuh
- qdq_4.cuh
- qdq_8.cuh
- qdq_util.cuh
- allspark_qgemm_w8a16.cu
- allspark_repack.cu
- allspark_utils.cuh
- hadamard_transform_cuda.cu
- generate.py
- machete_collective_builder.cuh
- machete_interleaving_utils.cuh
- machete_mainloop.cuh
- machete_mm_kernel.cuh
- machete_mm_launcher.cuh
- machete_prepack_kernel.cuh
- machete_prepack_launcher.cuh
- machete_prepacked_layout.cuh
- machete_pytorch.cu
- Readme.md
- .gitignore
- awq_marlin_repack.cu
- dequant.h
- generate_kernels.py
- gptq_marlin_repack.cu
- kernel.h
- marlin.cu
- marlin.cuh
- marlin_dtypes.cuh
- marlin_int4_fp8_preprocess.cu
- marlin_mma.h
- marlin_template.h
- cutlass_gemm_caller.cuh
- scaled_mm.cuh
- scaled_mm_azp_sm90_int8.cu
- scaled_mm_blockwise_sm100_fp8.cu
- scaled_mm_blockwise_sm100_fp8_dispatch.cuh
- scaled_mm_blockwise_sm120_fp8.cu
- scaled_mm_blockwise_sm120_fp8_dispatch.cuh
- scaled_mm_blockwise_sm90_fp8.cu
- scaled_mm_blockwise_sm90_fp8_dispatch.cuh
- scaled_mm_helper.hpp
- scaled_mm_kernels.hpp
- scaled_mm_sm100_fp8.cu
- scaled_mm_sm100_fp8_dispatch.cuh
- scaled_mm_sm120_fp8.cu
- scaled_mm_sm120_fp8_dispatch.cuh
- scaled_mm_sm90_fp8.cu
- scaled_mm_sm90_fp8_dispatch.cuh
- scaled_mm_sm90_int8.cu
- scaled_mm_sm90_int8_dispatch.cuh
- get_group_starts.cuh
- grouped_mm_c3x.cuh
- grouped_mm_c3x_sm100.cu
- grouped_mm_c3x_sm90.cu
- moe_data.cu
- scaled_mm_c2x.cu
- scaled_mm_c2x.cuh
- scaled_mm_c2x_sm75_dispatch.cuh
- scaled_mm_c2x_sm80_dispatch.cuh
- scaled_mm_c2x_sm89_fp8_dispatch.cuh
- scaled_mm_c2x_sm89_int8_dispatch.cuh
- scaled_mm_c3x_sm100.cu
- scaled_mm_c3x_sm120.cu
- scaled_mm_c3x_sm90.cu
- scaled_mm_entry.cu
- common.cu
- per_token_group_quant.cu
- per_token_group_quant.cu
- scaled_quant.cu
- per_token_group_quant_8bit.h
- activation_kernels.cu
- vectorization.cuh
- vectorization_utils.cuh
- activation_kernels.cu
- async_util.cuh
- cache_kernels.cu
- cache_kernels_fused.cu
- concat_mla_q.cuh
- cooperative_topk.cu
- cooperative_topk.cuh
- cub_helpers.h
- cuda_utils_kernels.cu
- cuda_vec_utils.cuh
- cuda_view.cu
- custom_all_gather_reduce_scatter.cu
- custom_all_gather_reduce_scatter_ops.cpp
- custom_all_reduce.cu
- dispatch_utils.h
- dsv3_fused_a_gemm.cu
- fp32_router_gemm.cu
- fp32_router_gemm_entry.cu
- fused_deepseek_v4_qnorm_rope_kv_insert_kernel.cu
- fused_kimi_k3_mla_key_concat_kv_cache_kernel.cu
- fused_minimax_m3_qknorm_rope_kv_insert_kernel.cu
- fused_qknorm_rope_kernel.cu
- launch_bounds_utils.h
- layernorm_kernels.cu
- layernorm_quant_kernels.cu
- minimax_reduce_rms_kernel.cu
- minimax_reduce_rms_kernel.h
- ngram_embedding_kernels.cu
- nvfp4_kv_cache_kernels.cu
- ops.h
- permute_cols.cu
- persistent_topk.cuh
- pos_encoding_kernels.cu
- sampler.cu
- topk.cu
- topk_histogram_4096.cuh
- torch_bindings.cpp
- torch_utils.h
- type_convert.cuh
- dynamic_4bit_int_moe_cpu.cpp
- Epilogues.md
- quant_utils.cuh
- quant_utils.cuh
- common.cuh
- utils.cuh
- base.h
- quick_reduce.h
- quick_reduce_impl.cuh
- attention.cu
- moe_q_gemm_rdna3.cu
- ops.h
- q_gemm_rdna3.cu
- q_gemm_rdna3_wmma.cu
- qdq_4_rdna3.cuh
- skinny_gemms.cu
- skinny_gemms_int4.cu
- torch_bindings.cpp
- cache.h
- cuda_compat.h
- cuda_utils.h
- cumem_allocator.cpp
- cumem_allocator_compat.h
- custom_all_gather_reduce_scatter.cuh
- custom_all_reduce.cuh
- custom_collective_common.cuh
- custom_quickreduce.cu
- dispatch_utils.h
- flashkda_registration.cpp
- fs_io.cpp
- ops.h
- qutlass_registration.cpp
- spinloop.cpp
- torch_bindings.cpp
- torch_utils.h
- test_vllm_nonroot_entrypoint.sh
- vllm-nonroot-entrypoint.sh
- ci-rocm.hcl
- docker-bake-rocm.hcl
- docker-bake.hcl
- Dockerfile
- Dockerfile.cpu
- Dockerfile.ppc64le
- Dockerfile.rocm
- Dockerfile.rocm_base
- Dockerfile.rocm_base_gfx1250
- Dockerfile.rocm_gfx1250
- Dockerfile.s390x
- Dockerfile.tpu
- Dockerfile.xpu
- versions.json
- .meta.yml
- README.md
- latency-metrics-speculative-decoding-dark.svg
- latency-metrics-speculative-decoding-light.svg
- dockerfile-stages-dependency.png
- load-pattern-examples.png
- vllm_bench_serve_dataset_stats.png
- vllm_bench_serve_timeline.html
- anything-llm-chat-with-doc.png
- anything-llm-chat-without-doc.png
- anything-llm-provider.png
- anything-llm-upload-doc.png
- architecture_helm_deployment.png
- chatbox-chat.png
- chatbox-settings.png
- claude-code-example.png
- dify-chat.png
- dify-create-chatbot.png
- dify-settings.png
- dp_external_lb.png
- dp_internal_lb.png
- hf-inference-endpoints-catalog.png
- hf-inference-endpoints-choose-infra.png
- hf-inference-endpoints-click-deploy-button.png
- hf-inference-endpoints-configure-container.png
- hf-inference-endpoints-create-endpoint.png
- hf-inference-endpoints-locate-deploy-button.png
- hf-inference-endpoints-new-endpoint.png
- hf-inference-endpoints-select-hardware.png
- hf-inference-endpoints-select-model.png
- open_webui.png
- streamlit-chat.png
- entrypoints.excalidraw.png
- llm_engine.excalidraw.png
- v1_process_architecture_tp2_dp4.png
- v1_process_architecture_tp4.png
- current_design.png
- executor_runtime.png
- previous_design.png
- wrapper_flow.png
- design_diagram.png
- dynamic_shapes.png
- tlparse_inductor.png
- fused_experts_blocks.png
- fused_moe_batched.png
- fused_moe_non_batched.png
- prepare_and_finalize_blocks.png
- basic_grouping_example.png
- full_attn.png
- memory_layout.png
- overview.png
- sw_attn.png
- intervals-1.png
- intervals-2.png
- intervals-3.png
- async_no_race_condition.png
- async_race_condition.png
- async_sched.png
- persistent_batch_mrv2.png
- persistent_batch_v1.png
- k_vecs.png
- key.png
- logits_vec.png
- q_vecs.png
- query.png
- v_vec.png
- value.png
- example-time-1.png
- example-time-3.png
- example-time-4.png
- example-time-5.png
- example-time-6.png
- example-time-7.png
- free.png
- overview.png
- most_model_len.png
- hierarchy.png
- disagg_encoder_flow.png
- abstraction.jpg
- high_level_design.png
- overview.jpg
- workflow.png
- speculators-user-flow-dark.svg
- speculators-user-flow-light.svg
- vllm-logo-only-light.ico
- vllm-logo-only-light.png
- vllm-logo-text-dark.png
- vllm-logo-text-light.png
- cheat_sheet.svg
- pooling_types.svg
- score_types.svg
- layerwise.png
- layerwise_bad_loading.png
- layerwise_good_loading.png
- cli.md
- dashboard.md
- README.md
- sweeps.md
- .meta.yml
- .nav.yml
- README.md
- contact_us.md
- meetups.md
- sponsors.md
- conserving_memory.md
- engine_args.md
- env_vars.md
- model_resolution.md
- optimization.md
- README.md
- serve_args.md
- failures.md
- nightly_builds.md
- update_pytorch_version.md
- dockerfile.md
- basic.md
- multimodal.md
- README.md
- registration.md
- tests.md
- transcription.md
- deprecation_policy.md
- editing-agent-instructions.md
- incremental_build.md
- jit_kernel_warmup.md
- labels.md
- profiling.md
- README.md
- vulnerability_management.md
- anyscale.md
- anything-llm.md
- autogen.md
- bentoml.md
- cerebrium.md
- chatbox.md
- crusoe.md
- dify.md
- dstack.md
- haystack.md
- helm.md
- hf_inference_endpoints.md
- litellm.md
- lobe-chat.md
- lws.md
- modal.md
- open-webui.md
- retrieval_augmented_generation.md
- runpod.md
- skypilot.md
- streamlit.md
- triton.md
- aibrix.md
- dynamo.md
- kaito.md
- kserve.md
- kthena.md
- kubeai.md
- kuberay.md
- llamastack.md
- llm-d.md
- llmaz.md
- production-stack.md
- docker.md
- k8s.md
- nginx.md
- arch_overview.md
- attention_backends.md
- cuda_graphs.md
- cuda_graphs_multimodal.md
- custom_op.md
- dbo.md
- debug_vllm_compile.md
- endpoint_plugins.md
- fused_moe_modular_kernel.md
- fusions.md
- huggingface_integration.md
- hybrid_kv_cache_manager.md
- io_processor_plugins.md
- logits_processors.md
- lora_resolver_plugins.md
- metrics.md
- mm_processing.md
- model_runner_v2.md
- moe_kernel_features.md
- multiprocessing.md
- nixl_kv_cache_lease.md
- nixl_kv_push_connector.md
- optimization_levels.md
- paged_attention.md
- plugin_system.md
- prefix_caching.md
- torch_compile.md
- torch_compile_multimodal.md
- vllm_ir.md
- README.md
- fp8.md
- int4.md
- int8_w4a8.md
- int8_w8a8.md
- README.md
- auto_awq.md
- b12x.md
- bnb.md
- fp8_vit_attn.md
- gguf.md
- gptqmodel.md
- inc.md
- modelopt.md
- online.md
- quantized_kvcache.md
- quark.md
- README.md
- torchao.md
- acceptance_metrics.md
- adaptive_verification.md
- draft_model.md
- dynamic_speculative_decoding.md
- eagle.md
- extract_hidden_states.md
- mlp.md
- mtp.md
- n_gram.md
- parallel_draft_model.md
- README.md
- speculators.md
- suffix.md
- automatic_prefix_caching.md
- batch_invariance.md
- context_extension.md
- custom_arguments.md
- custom_logitsprocs.md
- disagg_encoder.md
- disagg_prefill.md
- index_cache.md
- interleaved_thinking.md
- kv_offloading_usage.md
- lora.md
- mooncake_connector_usage.md
- mooncake_store_connector_usage.md
- moriio_connector_usage.md
- multimodal_inputs.md
- nixl_connector_compatibility.md
- nixl_connector_usage.md
- per_request_metrics.md
- prompt_embeds.md
- README.md
- reasoning_outputs.md
- sleep_mode.md
- structured_outputs.md
- tool_calling.md
- .nav.yml
- cpu.apple.inc.md
- cpu.arm.inc.md
- cpu.md
- cpu.s390x.inc.md
- cpu.x86.inc.md
- device.template.md
- gpu.apple.inc.md
- gpu.cuda.inc.md
- gpu.md
- gpu.rocm.inc.md
- gpu.xpu.inc.md
- python_env_setup.inc.md
- README.md
- quickstart.md
- collaboration.md
- committers.md
- process.md
- generate_argparse.py
- generate_attention_backends.py
- generate_examples.py
- generate_metrics.py
- generated_content.py
- autoref_code.py
- remove_announcement.py
- url_schemes.py
- edit_and_feedback.js
- mathjax.js
- reo.js
- run_llm_widget.js
- slack_and_forum.js
- toc-item.html
- main.html
- extra.css
- fastsafetensor.md
- instanttensor.md
- runai_model_streamer.md
- tensorizer.md
- cpu.md
- xpu.md
- classify.md
- embed.md
- README.md
- reward.md
- scoring.md
- specific_models.md
- token_classify.md
- token_embed.md
- generative_models.md
- supported_models.md
- claude_code.md
- codex.md
- langchain.md
- llamaindex.md
- derenderer.md
- generative_scoring.md
- openai_compatible_server.md
- README.md
- renderer.md
- speech_to_text.md
- trace_replay.md
- context_parallel_deployment.md
- data_parallel_deployment.md
- distributed_troubleshooting.md
- expert_parallel_deployment.md
- offline_inference.md
- parallelism_scaling.md
- base.md
- ipc.md
- nccl.md
- README.md
- async_rl.md
- layerwise.md
- rlhf.md
- sampling_mask.md
- trl.md
- faq.md
- metrics.md
- README.md
- reproducibility.md
- security.md
- troubleshooting.md
- usage_stats.md
- v1_guide.md
- .nav.yml
- pre_run_check.sh
- README.md
- client.py
- server.py
- gradio_openai_chatbot_webserver.py
- gradio_webserver.py
- streamlit_openai_chatbot_webserver.py
- retrieval_augmented_generation_with_langchain.py
- retrieval_augmented_generation_with_llamaindex.py
- basic.py
- chat.py
- classify.py
- embed.py
- generate.py
- README.md
- score.py
- openai_chat_completion_client.py
- openai_completion_client.py
- _helpers.tpl
- configmap.yaml
- custom-objects.yaml
- deployment.yaml
- hpa.yaml
- job.yaml
- poddisruptionbudget.yaml
- pvc.yaml
- secrets.yaml
- service.yaml
- deployment_test.yaml
- hpa_test.yaml
- job_test.yaml
- pvc_test.yaml
- service_test.yaml
- .helmignore
- Chart.yaml
- ct.yaml
- lintconf.yaml
- README.md
- values.schema.json
- values.yaml
- async_llm_streaming.py
- llm_engine_example.py
- sagemaker-entrypoint.sh
- disagg_1e1p1d_example.sh
- disagg_1e1pd_example.sh
- disagg_epd_proxy.py
- README.md
- disagg_proxy_demo.py
- disagg_proxy_multiturn.py
- disagg_proxy_pushconnector_demo.py
- kv_events.sh
- moriio_toy_proxy_server.py
- README.md
- ec_both_encoder.sh
- decode_example.py
- prefill_example.py
- README.md
- run.sh
- prefix_caching_flexkv.py
- decode_example.py
- load_recovery_example_connector.py
- prefill_example.py
- README.md
- run.sh
- lmcache-decoder-config.yaml
- lmcache-prefiller-config.yaml
- disagg_example_nixl.sh
- disagg_proxy_server.py
- disagg_vllm_launcher.sh
- cpu_offload_lmcache.py
- cpu_offload_lmcache_mp.sh
- kv_cache_sharing_lmcache_v1.py
- README.md
- mooncake_connector_proxy.py
- run_mooncake_connector.sh
- automatic_prefix_caching_offline.py
- prefix_caching_offline.py
- reproducibility_offline.py
- context_extension_offline.py
- data_parallel_offline.py
- multi_instance_data_parallel.py
- kv_events_subscriber.py
- custom.py
- custom_req.py
- custom_req_init.py
- README.md
- lora_with_quantization_offline.py
- multilora_offline.py
- openai_example_batch.jsonl
- README.md
- data_parallel_pause_resume.py
- pause_resume_offline.py
- run_one_batch_offline.py
- simple_profiling_offline.py
- prompt_embed_inference_with_openai_client.py
- prompt_embed_offline.py
- reset_kv_offline.py
- load_sharded_state_offline.py
- save_sharded_state_offline.py
- extract_hidden_states_offline.py
- mlpspeculator_offline.py
- spec_decode_offline.py
- pyproject.toml
- README.md
- structured_outputs_client.py
- structured_outputs_offline.py
- torchrun_dp_example_offline.py
- torchrun_example_offline.py
- logging_configuration.md
- tensorize_vllm_model.py
- only_thinker.py
- README.md
- only_thinker.py
- audio_language_offline.py
- encoder_decoder_multimodal_offline.py
- mistral-small_offline.py
- openai_chat_completion_client_for_multimodal.py
- vision_language_multi_image_offline.py
- vision_language_offline.py
- batched_chat_completions_online.py
- qwen_1m_offline.py
- trace_replay_offline.py
- performance_statistics.json
- query_statistics.json
- README.md
- performance_statistics.yaml
- query_statistics.yaml
- README.md
- README.md
- offline.py
- dummy_client.py
- README.md
- docker-compose.yaml
- grafana.json
- prometheus.yaml
- README.md
- classification_online.py
- vision_classification_online.py
- client.py
- README.md
- service.sh
- dse_qwen2_vl.jinja
- nemotron_embed_vl.jinja
- vlm2vec_phi3v.jinja
- vlm2vec_qwen2vl.jinja
- embed_jina_embeddings_v3_offline.py
- embed_matryoshka_fy_offline.py
- embedding_requests_base64_online.py
- embedding_requests_bytes_online.py
- openai_embedding_client.py
- openai_embedding_matryoshka_fy_client.py
- vision_embedding_offline.py
- vision_embedding_online.py
- prithvi_geospatial_mae_io_processor.py
- prithvi_geospatial_mae_offline.py
- prithvi_geospatial_mae_online.py
- sequence_reward_offline.py
- sequence_reward_online.py
- token_reward_offline.py
- token_reward_online.py
- bge-reranker-v2-gemma.jinja
- mxbai_rerank_v2.jinja
- nemotron-rerank.jinja
- nemotron-vl-rerank.jinja
- qwen3_reranker.jinja
- qwen3_vl_reranker.jinja
- cohere_rerank_client.py
- colbert_rerank_online.py
- colmodernvbert_rerank_online.py
- colqwen3_5_rerank_online.py
- colqwen3_rerank_online.py
- convert_model_to_seq_cls.py
- qwen3_reranker_offline.py
- qwen3_reranker_online.py
- rerank_api_online.py
- score_api_online.py
- using_template_offline.py
- using_template_online.py
- vision_rerank_api_online.py
- vision_reranker_offline.py
- vision_score_api_online.py
- forced_alignment_offline.py
- forced_alignment_online.py
- ner_offline.py
- ner_online.py
- colqwen3_token_embed_online.py
- jina_embeddings_v4_offline.py
- jina_reranker_v3_offline.py
- jina_reranker_v3_online.py
- multi_vector_retrieval_offline.py
- multi_vector_retrieval_online.py
- bench.sh
- scale.py
- serve_deepseek_v2.sh
- batch_llm_inference.py
- multi-node-serving.sh
- ray_serve_deepseek.py
- run_cluster.sh
- openai_chat_completion_tool_calls_with_reasoning.py
- openai_chat_completion_with_reasoning.py
- openai_chat_completion_with_reasoning_streaming.py
- openai_responses_client.py
- rlhf_async_new_apis.py
- rlhf_http_ipc.py
- rlhf_http_nccl.py
- rlhf_ipc_fsdp_ep.py
- rlhf_nccl_fsdp_ep.py
- rlhf_sparse_nccl.py
- routed_experts_e2e.py
- skip_loading_weights_in_engine_init.py
- __init__.py
- example_mm_serve.py
- token_generation_client.py
- openai_lid_client.py
- openai_transcription_client.py
- openai_translation_client.py
- openai_realtime_client.py
- openai_realtime_microphone_client.py
- chat_with_tools_offline.py
- openai_chat_completion_client_with_tools.py
- openai_chat_completion_client_with_tools_required.py
- openai_chat_completion_client_with_tools_xlam.py
- openai_chat_completion_client_with_tools_xlam_streaming.py
- openai_responses_client_with_mcp_tools.py
- openai_responses_client_with_tools.py
- __init__.py
- template_alpaca.jinja
- template_chatglm.jinja
- template_chatglm2.jinja
- template_chatml.jinja
- template_falcon.jinja
- template_falcon_180b.jinja
- template_inkbot.jinja
- template_teleflm.jinja
- tool_chat_template_apertus.jinja
- tool_chat_template_deepseekr1.jinja
- tool_chat_template_deepseekv3.jinja
- tool_chat_template_deepseekv31.jinja
- tool_chat_template_functiongemma.jinja
- tool_chat_template_gemma3_pythonic.jinja
- tool_chat_template_gemma4.jinja
- tool_chat_template_glm4.jinja
- tool_chat_template_granite.jinja
- tool_chat_template_granite_20b_fc.jinja
- tool_chat_template_hermes.jinja
- tool_chat_template_hunyuan_a13b.jinja
- tool_chat_template_internlm2_tool.jinja
- tool_chat_template_llama3.1_json.jinja
- tool_chat_template_llama3.2_json.jinja
- tool_chat_template_llama3.2_pythonic.jinja
- tool_chat_template_llama4_json.jinja
- tool_chat_template_llama4_pythonic.jinja
- tool_chat_template_mistral.jinja
- tool_chat_template_mistral3.jinja
- tool_chat_template_mistral_parallel.jinja
- tool_chat_template_muse_glimmer.jinja
- tool_chat_template_phi4_mini.jinja
- tool_chat_template_qwen3coder.jinja
- tool_chat_template_toolace.jinja
- tool_chat_template_xlam_llama.jinja
- tool_chat_template_xlam_qwen.jinja
- cpu.txt
- cuda.txt
- rocm.txt
- rust.txt
- tpu.txt
- cpu.txt
- cuda.in
- cuda.txt
- nightly-torch.txt
- rocm.in
- rocm.txt
- xpu.in
- xpu.txt
- common.txt
- cpu.txt
- cuda.txt
- dev.txt
- docs.in
- docs.txt
- kv_connectors.txt
- kv_connectors_rocm.txt
- lint.txt
- rocm.txt
- tpu.txt
- xpu.txt
- nextest.toml
- buf.md
- buf.yaml
- control.proto
- inference.proto
- README.md
- mod.rs
- openai_chat.rs
- openai_completions.rs
- pooling.rs
- streaming.rs
- custom.rs
- hf_dataset.rs
- mod.rs
- multi_turn.rs
- prefix_repetition.rs
- progress.rs
- random.rs
- random_mm.rs
- random_rerank.rs
- sharegpt.rs
- sonnet.rs
- sonnet.txt
- speed_bench.rs
- calculator.rs
- mod.rs
- steady_state.rs
- console.rs
- json.rs
- mod.rs
- benchmark.rs
- cli.rs
- compare.rs
- config.rs
- error.rs
- hub.rs
- lib.rs
- main.rs
- multi_run.rs
- multi_turn.rs
- rate_control.rs
- ready_checker.rs
- sweep.rs
- tiktoken.rs
- tokenizer.rs
- cli_parity.rs
- python_serve_flags.txt
- AGENTS.md
- Cargo.toml
- CLAUDE.md
- README.md
- lib.rs
- Cargo.toml
- external_engine_chat_qwen.rs
- README.md
- hf.rs
- mod.rs
- audio.rs
- expand.rs
- image.rs
- item.rs
- tensor.rs
- video.rs
- mod.rs
- structural_tag.rs
- unified.rs
- mod.rs
- tests.rs
- mod.rs
- structured.rs
- mod.rs
- tests.rs
- mod.rs
- tests.rs
- mod.rs
- unified.rs
- test_input.json
- test_input_developer_tools.json
- test_input_search_w_date.json
- test_input_search_wo_date.json
- test_output_developer_tools.txt
- test_output_search_w_date.txt
- test_output_search_wo_date.txt
- test_output_vllm_parity.txt
- encoding.rs
- mod.rs
- tests.rs
- test_input_1.json
- test_input_2.json
- test_input_developer_tools.json
- test_output_1.txt
- test_output_2.txt
- test_output_developer_tools.txt
- encoding.rs
- mod.rs
- tests.rs
- assistant_history.json
- assistant_history.txt
- developer_tools.json
- developer_tools.txt
- drop_analysis.json
- drop_analysis.txt
- leading_system.json
- leading_system.txt
- request_tools.json
- request_tools.txt
- simple_user.json
- simple_user.txt
- system_instructions_env.txt
- tool_roundtrip.json
- tool_roundtrip.txt
- encoding.rs
- mod.rs
- tests.rs
- error.rs
- format.rs
- mod.rs
- template.rs
- tojson.rs
- text_audio_input.json
- text_audio_output.txt
- text_image_input.json
- text_image_output.txt
- tool_declare_input.json
- tool_declare_output.txt
- tool_round_trip_input.json
- tool_round_trip_output.txt
- mod.rs
- tests.rs
- controls_thinking_off_input.json
- controls_thinking_off_output.txt
- dynamic_only_tool_declare_input.json
- dynamic_only_tool_declare_output.txt
- dynamic_system_tool_declare_input.json
- dynamic_system_tool_declare_output.txt
- history_preserve_and_image_input.json
- history_preserve_and_image_output.txt
- tools_history_and_required_input.json
- tools_history_and_required_output.txt
- encoding.rs
- mod.rs
- tests.rs
- mod.rs
- selection.rs
- test_utils.rs
- error.rs
- event.rs
- lib.rs
- multimodal.rs
- request.rs
- stream.rs
- README.md
- template_alpaca.jinja
- template_chatglm.jinja
- template_chatglm2.jinja
- template_chatml.jinja
- template_falcon.jinja
- template_falcon_180b.jinja
- template_inkbot.jinja
- template_teleflm.jinja
- tool_chat_template_deepseekr1.jinja
- tool_chat_template_deepseekv3.jinja
- tool_chat_template_deepseekv31.jinja
- tool_chat_template_functiongemma.jinja
- tool_chat_template_gemma3_pythonic.jinja
- tool_chat_template_gemma4.jinja
- tool_chat_template_glm4.jinja
- tool_chat_template_granite.jinja
- tool_chat_template_granite_20b_fc.jinja
- tool_chat_template_hermes.jinja
- tool_chat_template_hunyuan_a13b.jinja
- tool_chat_template_internlm2_tool.jinja
- tool_chat_template_llama3.1_json.jinja
- tool_chat_template_llama3.2_json.jinja
- tool_chat_template_llama3.2_pythonic.jinja
- tool_chat_template_llama4_json.jinja
- tool_chat_template_llama4_pythonic.jinja
- tool_chat_template_mistral.jinja
- tool_chat_template_mistral3.jinja
- tool_chat_template_mistral_parallel.jinja
- tool_chat_template_phi4_mini.jinja
- tool_chat_template_qwen3coder.jinja
- tool_chat_template_toolace.jinja
- tool_chat_template_xlam_llama.jinja
- tool_chat_template_xlam_qwen.jinja
- qwen3.jinja
- qwen35.jinja
- chat.rs
- roundtrip.rs
- Cargo.toml
- README.md
- tests.rs
- unsupported.rs
- cli.rs
- main.rs
- Cargo.toml
- external_engine_logprobs.rs
- external_engine_utility_call.rs
- README.md
- imp.rs
- state.rs
- stream.rs
- bootstrap.rs
- external.rs
- handle.rs
- inproc.rs
- mod.rs
- array.rs
- tests.rs
- wire.rs
- dtype.rs
- handshake.rs
- logprobs.rs
- lora.rs
- mod.rs
- multimodal.rs
- output.rs
- request.rs
- sampling.rs
- stats.rs
- structured_outputs.rs
- tensor.rs
- utility.rs
- client.rs
- mod.rs
- python_compat.py
- client.rs
- error.rs
- lib.rs
- metrics.rs
- mock_engine.rs
- runtime.rs
- test_utils.rs
- transport.rs
- Cargo.toml
- external_engine_smoke.rs
- README.md
- error.rs
- inflight.rs
- lib.rs
- log_stats.rs
- output.rs
- request.rs
- request_metrics.rs
- generate.rs
- Cargo.toml
- cli.rs
- lib.rs
- process.rs
- Cargo.toml
- api_server.rs
- lib.rs
- request.rs
- scheduler.rs
- Cargo.toml
- engine.rs
- io.rs
- lib.rs
- main.rs
- tests.rs
- Cargo.toml
- README.md
- adapter.rs
- mod.rs
- deepseek_v3.rs
- deepseek_v31.rs
- deepseek_v32.rs
- gemma4.rs
- glm45_moe.rs
- granite4.rs
- kimi_k2.rs
- llama3_json.rs
- minimax_m2.rs
- qwen3_coder.rs
- qwen3_xml.rs
- lib.rs
- Cargo.toml
- cohere_cmd.rs
- deepseek_r1.rs
- delimited.rs
- hy_v3.rs
- kimi.rs
- minimax_m3.rs
- mod.rs
- qwen3.rs
- seed_oss.rs
- step3p5.rs
- tests.rs
- deepseek_v32.rs
- deepseek_v4.rs
- mod.rs
- deepseek_v3.rs
- deepseek_v31.rs
- mod.rs
- glm45_moe.rs
- glm47_moe.rs
- mod.rs
- structural_tag.rs
- granite4.rs
- hermes.rs
- internlm2.rs
- llama.rs
- mistral.rs
- mod.rs
- phi4mini.rs
- qwen.rs
- error.rs
- hy_v3.rs
- kimi_k2.rs
- minimax_m2.rs
- minimax_m3.rs
- mod.rs
- parameters.rs
- qwen_coder.rs
- seed_oss.rs
- test_utils.rs
- tests.rs
- structural_tag.rs
- combined.rs
- gemma4.rs
- hy_v3.rs
- inkling.rs
- kimi_k3.rs
- mod.rs
- lib.rs
- utils.rs
- Cargo.toml
- external_engine_openai_qwen.rs
- README.md
- control.rs
- convert.rs
- health.rs
- inference.rs
- mod.rs
- tests.rs
- auth.rs
- cors.rs
- load.rs
- metrics.rs
- mod.rs
- offload.rs
- request_id.rs
- convert.rs
- types.rs
- validate.rs
- generate.rs
- mod.rs
- convert.rs
- types.rs
- validate.rs
- convert.rs
- types.rs
- validate.rs
- logprobs.rs
- mod.rs
- structured_outputs.rs
- types.rs
- usage.rs
- validated_json.rs
- chat_completions.rs
- completions.rs
- mod.rs
- models.rs
- types.rs
- abort_requests.rs
- cache.rs
- collective_rpc.rs
- health.rs
- http_client_tests.rs
- load.rs
- lora.rs
- metrics.rs
- pause.rs
- profile.rs
- render.rs
- server_info.rs
- sleep.rs
- tests.rs
- tokenize.rs
- version.rs
- world_size.rs
- config.rs
- error.rs
- lib.rs
- listener.rs
- lora.rs
- render.rs
- routes.rs
- runtime.rs
- server_info.rs
- state.rs
- tls.rs
- tls_tests.rs
- utils.rs
- build.rs
- Cargo.toml
- config.rs
- mod.rs
- model_files.rs
- mod.rs
- logprobs.rs
- sampling.rs
- token_ids.rs
- decoded.rs
- logprobs.rs
- mod.rs
- error.rs
- lib.rs
- lower.rs
- request.rs
- Cargo.toml
- hf.rs
- tiktoken.rs
- added_tokens.rs
- byte_level_decode.rs
- error.rs
- hf.rs
- incremental.rs
- lib.rs
- tekken.rs
- test_utils.rs
- tiktoken.rs
- Cargo.toml
- lib.rs
- Cargo.toml
- .gitattributes
- .gitignore
- AGENTS.md
- Cargo.lock
- Cargo.toml
- CLAUDE.md
- deny.toml
- README.md
- rustfmt.toml
- rustfmt.unstable.toml
- autotune_helion_kernels.py
- benchmark_helion_kernels.py
- __init__.py
- test_basic_correctness.py
- test_cpu_offload.py
- test_mem.py
- test_prefetch_offload.py
- __init__.py
- test_param_sweep.py
- __init__.py
- test_audio_dataset.py
- test_bench_startup.py
- test_bfcl_dataset.py
- test_custom_dataset_chat_template_kwargs.py
- test_custom_dataset_seed.py
- test_custom_image_dataset.py
- test_latency_cli.py
- test_plot_filters.py
- test_random_dataset.py
- test_random_multimodal_dataset_video.py
- test_rust_bench_cli_parity.py
- test_sampling_params.py
- test_serve_cli.py
- test_skip_tokenizer_init.py
- test_throughput_cli.py
- test_txt_slices_dataset.py
- __init__.py
- test_async_tp.py
- test_sequence_parallel.py
- __init__.py
- test_basic_correctness.py
- test_full_cudagraph.py
- test_full_graph.py
- test_multimodal_compile.py
- test_multiple_graphs.py
- test_simple.py
- test_toy_llama.py
- __init__.py
- common.py
- conftest.py
- models.py
- test_tp1_quant.py
- test_tp2_ar_rms.py
- test_tp2_async_tp.py
- __init__.py
- test_startup.py
- __init__.py
- test_async_tp.py
- test_fusion_all_reduce.py
- test_sequence_parallelism.py
- __init__.py
- test_clone_cleanup.py
- test_inplace_functionalization.py
- test_lowering.py
- __init__.py
- test_double_aiter_rms_quant_fusion.py
- test_functionalization.py
- test_fuse_act_padding.py
- test_fuse_mla_dual_rms_norm.py
- test_fusion.py
- test_fusion_attn.py
- test_mla_attn_quant_fusion.py
- test_mla_rope_kvcache_cat_fusion.py
- test_noop_elimination.py
- test_pass_manager.py
- test_qk_norm_rope_fusion.py
- test_rmsnorm_reshape_fusion.py
- test_rocm_aiter_qk_norm_rope_kvcache_fusion.py
- test_rope_kvcache_fusion.py
- test_scatter_split_replace.py
- test_silu_mul_quant_fusion.py
- test_split_coalescing.py
- test_vllm_fusion_pattern_matcher_pass.py
- __init__.py
- backend.py
- conftest.py
- README.md
- silly_attention.py
- test_aot_compile.py
- test_codegen.py
- test_compile_ranges.py
- test_config.py
- test_decorator.py
- test_dynamic_shapes_compilation.py
- test_graph_partition.py
- test_inductor_fallback_allow_list_patch.py
- test_rotary_embedding_compile.py
- test_sequence_parallelism_threshold.py
- test_structured_logging.py
- test_wrapper.py
- base_model_arch_groundtruth.json
- draft_model_arch_groundtruth.json
- test_bailing_mtp_config.py
- test_config.yaml
- test_config_generation.py
- test_config_utils.py
- test_config_with_model.yaml
- test_model_arch_config.py
- test_mp_reducer.py
- test_multimodal_config.py
- test_speculative_draft_hf_overrides.py
- test_speculative_draft_max_position_embeddings.py
- check_device_count_respects_env.py
- check_platform_no_cuda_init.py
- test_cuda_compatibility_path.py
- test_cuda_context.py
- test_platform_no_cuda_init.py
- __init__.py
- test_check_stop_strings.py
- test_disable_detokenization.py
- test_min_tokens.py
- test_stop_reason.py
- test_stop_string_while_stop_model_terminates.py
- test_stop_strings.py
- __init__.py
- conftest.py
- eplb_utils.py
- test_ca_buffer_sharing.py
- test_comm_ops.py
- test_context_parallel.py
- test_custom_all_reduce.py
- test_dcp_a2a.py
- test_dcp_direct_a2a_lse_reduce.py
- test_distributed_oot.py
- test_elastic_ep.py
- test_eplb_algo.py
- test_eplb_events.py
- test_eplb_execute.py
- test_eplb_fused_moe_layer.py
- test_eplb_fused_moe_layer_dep_nvfp4.py
- test_eplb_quant_scale_consistency.py
- test_eplb_spec_decode.py
- test_eplb_utils.py
- test_events.py
- test_expert_parallel.py
- test_expert_placement.py
- test_file_store.py
- test_kimi_linear_context_parallel.py
- test_kv_cache_events.py
- test_kvlayout.py
- test_mnnvl_alltoall.py
- test_mq_connect_ip.py
- test_multi_node_assignment.py
- test_multiproc_executor.py
- test_nccl_symm_mem.py
- test_node_count.py
- test_packed_tensor.py
- test_pipeline_parallel.py
- test_pipeline_partition.py
- test_pp_cudagraph.py
- test_pynccl.py
- test_quick_all_reduce.py
- test_ray_v2_executor.py
- test_ray_v2_executor_e2e.py
- test_rocm_aiter_custom_ar.py
- test_rocm_quick_reduce.py
- test_same_node.py
- test_shm_broadcast.py
- test_shm_buffer.py
- test_shm_storage.py
- test_split_group.py
- test_symm_mem_allreduce.py
- test_torchrun_example.py
- test_torchrun_example_moe.py
- test_utils.py
- test_weight_transfer.py
- __init__.py
- test_arg_utils.py
- test_short_mm_context.py
- __init__.py
- test_anthropic_messages_conversion.py
- test_messages.py
- test_protocol_exports.py
- __init__.py
- test_api_router.py
- test_chat_v2.py
- test_cohere_chat_message.py
- test_protocol.py
- test_registry_and_args.py
- test_serving_conversion.py
- test_serving_streaming.py
- __init__.py
- test_generative_scoring.py
- test_generative_scoring_e2e.py
- __init__.py
- __init__.py
- _api_server_spawn_workers.py
- test_api_server_process_manager.py
- test_multi_api_servers.py
- __init__.py
- test_launch_cli.py
- test_shutdown.py
- test_ssl_cert_refresher.py
- __init__.py
- test_offline_mode.py
- __init__.py
- test_accuracy.py
- test_chat.py
- test_collective_rpc.py
- test_generate.py
- test_gpu_utilization.py
- test_prompt_validation.py
- test_struct_output_generate.py
- __init__.py
- test_chat.py
- test_mm_cache_external_injection.py
- test_mm_cache_stats.py
- test_mm_embeds_only.py
- test_mm_processor_kwargs.py
- __init__.py
- test_audio.py
- test_audio_in_video.py
- test_chat_completion_with_image_embeds.py
- test_chat_completion_with_mixed_audio_embeds.py
- test_chat_completion_with_mixed_image_embeds.py
- test_default_mm_loras.py
- test_video.py
- test_vision.py
- test_vision_embeds.py
- __init__.py
- test_image.py
- __init__.py
- __init__.py
- conftest.py
- __init__.py
- test_batched_chat_completions.py
- test_chat.py
- test_chat_completion.py
- test_chat_completion_with_prompt_embeds.py
- test_chat_echo.py
- test_chat_error.py
- test_chat_logit_bias_validation.py
- test_completion_with_function_calling.py
- test_enable_force_include_usage.py
- test_extra_content_fields.py
- test_include_reasoning.py
- test_logprob_token_ids.py
- test_non_object_body_validation.py
- test_root_path.py
- test_serving_chat.py
- test_thinking_token_budget.py
- test_thinking_token_budget_validation.py
- __init__.py
- test_completion.py
- test_completion_error.py
- test_completion_with_prompt_embeds.py
- test_lora_resolvers.py
- test_prompt_validation.py
- test_tensorizer_entrypoint.py
- test_token_in_token_out.py
- __init__.py
- test_lmeval.py
- __init__.py
- test_models.py
- __init__.py
- test_harmony_render_parity.py
- test_harmony_utils.py
- __init__.py
- conftest.py
- test_basic.py
- test_errors.py
- test_function_call.py
- test_function_call_parsing.py
- test_harmony.py
- test_harmony_utils.py
- test_mcp_tools.py
- test_namespace_tool_separator.py
- test_parsable_context.py
- test_parsable_context_unit.py
- test_protocol.py
- test_response_input_to_harmony.py
- test_responses_utils.py
- test_sampling_params.py
- test_serving_responses.py
- test_simple.py
- test_stateful.py
- test_streaming_events.py
- test_structured_output.py
- __init__.py
- test_async_tokenization.py
- test_chunked_prompt.py
- test_cli_args.py
- test_dp_supervisor.py
- test_openai_schema.py
- test_reasoning_enable_thinking.py
- test_render_parity.py
- test_render_token_offsets.py
- test_return_routed_experts.py
- test_return_token_ids.py
- test_return_tokens_as_ids.py
- test_run_batch.py
- test_session_id.py
- test_stop_token_ids.py
- test_tool_calls_serialization.py
- test_tool_choice_content_none.py
- utils.py
- __init__.py
- test_encode.py
- test_tiling_engine.py
- test_truncation.py
- __init__.py
- test_offline.py
- test_online.py
- test_online_vision.py
- __init__.py
- test_cohere_online.py
- test_cohere_online_vision.py
- test_cohere_openai_parity.py
- test_correctness_mteb.py
- test_io_processor.py
- test_offline.py
- test_online.py
- test_online_dimensions.py
- test_online_long_text.py
- test_online_vision.py
- test_protocol.py
- __init__.py
- test_token_reward_offline.py
- test_token_reward_online.py
- __init__.py
- test_bi_encoder_offline.py
- test_bi_encoder_online.py
- test_cross_encoder_correctness_mteb.py
- test_cross_encoder_offline.py
- test_cross_encoder_online.py
- test_cross_encoder_online_vision.py
- test_jina_ranking_io_processor_unit.py
- test_late_interaction_offline.py
- test_late_interaction_offline_vision.py
- test_late_interaction_online.py
- test_late_interaction_online_vision.py
- test_utils.py
- util.py
- __init__.py
- test_offline.py
- test_online.py
- __init__.py
- test_offline.py
- test_online.py
- __init__.py
- test_factories.py
- test_io_processor.py
- test_utils.py
- __init__.py
- test_derender.py
- test_derender_parity.py
- test_derender_stream.py
- __init__.py
- test_launch_render.py
- test_render.py
- test_render_multimodal.py
- __init__.py
- test_generate_stream.py
- test_mm_serde.py
- test_protocol.py
- test_return_routed_experts.py
- test_serving_multimodal_tokens.py
- test_serving_tokens.py
- test_tokens_logprobs.py
- __init__.py
- __init__.py
- test_pause_resume.py
- conftest.py
- __init__.py
- test_collective_rpc.py
- __init__.py
- test_sleep.py
- __init__.py
- test_error_sanitization.py
- test_http_status_metrics.py
- test_validation_exception_handler.py
- __init__.py
- test_basic.py
- test_metrics.py
- test_orca_metrics.py
- test_uds.py
- __init__.py
- test_lora_adapters.py
- test_serving_models.py
- __init__.py
- test_authentication_middleware.py
- test_optional_middleware.py
- __init__.py
- conftest.py
- test_sagemaker_handler_overrides.py
- test_sagemaker_lora_adapters.py
- test_sagemaker_middleware_integration.py
- test_sagemaker_stateful_sessions.py
- __init__.py
- test_serving_tokenization.py
- test_tokenization.py
- test_tokenization_vlm.py
- test_tokenize_then_chat_vlm.py
- __init__.py
- test_api_utils.py
- test_fingerprint.py
- test_request_logger.py
- __init__.py
- __init__.py
- test_transcription_api_correctness.py
- __init__.py
- test_realtime_validation.py
- __init__.py
- test_chunk_timestamp_offset.py
- test_enable_force_include_usage.py
- test_qwen3_asr_sanitize_prompt.py
- test_transcription_inter_chunk_spacing.py
- test_transcription_validation.py
- test_transcription_validation_whisper.py
- __init__.py
- test_translation_validation.py
- __init__.py
- conftest.py
- test_speech_to_text_cancellation.py
- test_upload_size_limit.py
- __init__.py
- test_granite4_tool_parser.py
- test_hermes_tool_parser.py
- test_openai_tool_parser.py
- __init__.py
- test_chat_utils.py
- test_context.py
- test_grpc_health.py
- test_non_object_body_validation.py
- test_remote_vllm_server.py
- __init__.py
- test_weight_transfer_llm.py
- __init__.py
- conftest.py
- gpt-oss-20b-baseline.yaml
- gpt-oss-20b-flashinfer-mxfp4-bf16-cutlass.yaml
- gpt-oss-20b-flashinfer-mxfp4-bf16-trtllm.yaml
- gpt-oss-20b-flashinfer-mxfp4-mxfp8-cutlass.yaml
- gpt-oss-20b-marlin.yaml
- gpt-oss-20b-rocm-baseline.yaml
- gpt-oss-20b-rocm-quark-mxfp4-bf16-aiter.yaml
- gpt-oss-20b-rocm-quark-mxfp4-bf16-triton.yaml
- gpt-oss-20b-rocm-quark-mxfp4-fp8-triton.yaml
- gpt-oss-20b-sm100-fi-mxfp4-mxfp8-trtllm.yaml
- gpt-oss-20b-sm120.yaml
- gpt-oss-20b-xpu-baseline.yaml
- gpt-oss-20b-xpu-triton-attn.yaml
- models-b200.txt
- models-gfx942.txt
- models-gfx950.txt
- models-h100.txt
- models-spark.txt
- models-xpu.txt
- __init__.py
- conftest.py
- README.md
- test_gpqa_correctness.py
- config-a100-shard-0.txt
- config-a100-shard-1.txt
- config-a100-shard-2.txt
- config-act-fp8.txt
- config-act-int8.txt
- config-h100-shard-0.txt
- config-h100-shard-1.txt
- config-h100-shard-2.txt
- config-int5wc-hadamard.txt
- config.txt
- gpt-oss-20b-humming-act-fp8.yaml
- gpt-oss-20b-humming.yaml
- NVIDIA-Nemotron-3-Nano-30B-A3B-NVFP4-humming.yaml
- Qwen2-1.5B-Instruct-FP8W8-humming-act-fp8.yaml
- Qwen2-1.5B-Instruct-FP8W8-humming.yaml
- Qwen3-0.6B-MXFP8-humming-act-fp8.yaml
- Qwen3-0.6B-MXFP8-humming.yaml
- Qwen3-30B-A3B-AWQ-humming-act-fp8.yaml
- Qwen3-30B-A3B-AWQ-humming-act-int8.yaml
- Qwen3-30B-A3B-AWQ-humming.yaml
- Qwen3-30B-A3B-FP8-block-humming-act-fp8.yaml
- Qwen3-30B-A3B-FP8-block-humming.yaml
- Qwen3-30B-A3B-Fp8-v1-humming-act-fp8.yaml
- Qwen3-30B-A3B-Fp8-v1-humming.yaml
- Qwen3-30B-A3B-GPTQ-Int4-humming-act-fp8.yaml
- Qwen3-30B-A3B-GPTQ-Int4-humming-act-int8.yaml
- Qwen3-30B-A3B-GPTQ-Int4-humming.yaml
- Qwen3-30B-A3B-Instruct-2507-quantized.w8a8-humming-act-int8.yaml
- Qwen3-30B-A3B-Instruct-2507-quantized.w8a8-humming.yaml
- Qwen3-30B-A3B-int5wc-hadamard-humming-act-fp8.yaml
- Qwen3-30B-A3B-int5wc-hadamard-humming-act-int8.yaml
- Qwen3-30B-A3B-int5wc-hadamard-humming.yaml
- Qwen3-30B-A3B-MXFP4A16-humming-act-fp8.yaml
- Qwen3-30B-A3B-MXFP4A16-humming.yaml
- Qwen3-30B-A3B-NVFP4-humming.yaml
- Qwen3-4B-mixed-quant-RTN-humming.yaml
- Qwen3.5-35B-A3B-experts-int8-humming-act-int8.yaml
- Qwen3.5-35B-A3B-experts-int8-humming.yaml
- Qwen3.5-35B-A3B-FP8-humming-act-fp8.yaml
- Qwen3.5-35B-A3B-FP8-humming.yaml
- Qwen3.5-4B-quantized.w4a16-humming-act-fp8.yaml
- Qwen3.5-4B-quantized.w4a16-humming-act-int8.yaml
- Qwen3.5-4B-quantized.w4a16-humming.yaml
- Qwen3.6-35B-A3B-NVFP4-humming.yaml
- config-b200-shard-0.txt
- config-b200-shard-1.txt
- config-b200-shard-2.txt
- config-b200-shard-3.txt
- config-b200.txt
- config-h100.txt
- config-test.txt
- DeepSeek-V4-Flash-deepgemm-mega-moe.yaml
- Llama-4-Scout-BF16-fi-cutlass.yaml
- Llama-4-Scout-BF16-triton.yaml
- Llama-4-Scout-Fp8-CT-vllm-cutlass.yaml
- Llama-4-Scout-Fp8-ModelOpt-fi-cutlass.yaml
- Llama-4-Scout-Fp8-ModelOpt-fi-trtllm.yaml
- Llama-4-Scout-Fp8-ModelOpt-marlin.yaml
- Llama-4-Scout-Fp8-ModelOpt-triton.yaml
- Mixtral-8x7B-BF16-fi-cutlass.yaml
- Mixtral-8x7B-BF16-triton.yaml
- Mixtral-8x7B-Fp8-AutoFp8-fi-cutlass.yaml
- Mixtral-8x7B-Fp8-AutoFp8-triton.yaml
- Nemotron-Nano-30B-Fp8-ModelOpt-fi-trtllm.yaml
- Nemotron-Nano-30B-NvFp4-ModelOpt-fi-cutlass.yaml
- Nemotron-Nano-30B-NvFp4-ModelOpt-vllm-cutlass.yaml
- Qwen3-30B-A3B-BF16-fi-cutlass.yaml
- Qwen3-30B-A3B-BF16-triton.yaml
- Qwen3-30B-A3B-Fp8-AutoFp8-deepgemm.yaml
- Qwen3-30B-A3B-Fp8-AutoFp8-fi-cutlass.yaml
- Qwen3-30B-A3B-Fp8-AutoFp8-fi-trtllm.yaml
- Qwen3-30B-A3B-Fp8-AutoFp8-marlin.yaml
- Qwen3-30B-A3B-Fp8-AutoFp8-triton.yaml
- Qwen3-30B-A3B-Fp8-CT-Block-deepgemm.yaml
- Qwen3-30B-A3B-Fp8-CT-Block-fi-cutlass.yaml
- Qwen3-30B-A3B-Fp8-CT-Block-marlin.yaml
- Qwen3-30B-A3B-Fp8-CT-Block-triton.yaml
- Qwen3-30B-A3B-Fp8-CT-Channel-marlin.yaml
- Qwen3-30B-A3B-Fp8-CT-Channel-vllm-cutlass.yaml
- Qwen3-30B-A3B-NvFp4-CT-fi-cutlass.yaml
- Qwen3-30B-A3B-NvFp4-CT-fi-trtllm.yaml
- Qwen3-30B-A3B-NvFp4-CT-marlin.yaml
- Qwen3-30B-A3B-NvFp4-CT-vllm-cutlass.yaml
- Qwen3-30B-A3B-NvFp4-ModelOpt-fi-cutlass.yaml
- Qwen3-30B-A3B-NvFp4-ModelOpt-fi-trtllm.yaml
- Qwen3-30B-A3B-NvFp4-ModelOpt-marlin.yaml
- Qwen3-30B-A3B-NvFp4-ModelOpt-vllm-cutlass.yaml
- config-b200.txt
- Llama-4-Scout-Fp8-ModelOpt-triton.yaml
- Qwen3-30B-A3B-BF16-triton.yaml
- Qwen3-30B-A3B-Fp8-AutoFp8-deepgemm-deepep-ht.yaml
- Qwen3-30B-A3B-Fp8-AutoFp8-deepgemm-deepep-ll.yaml
- Qwen3-30B-A3B-Fp8-AutoFp8-deepgemm.yaml
- Qwen3-30B-A3B-Fp8-CT-Block-deepgemm-deepep-ht.yaml
- Qwen3-30B-A3B-Fp8-CT-Block-deepgemm-deepep-ll.yaml
- Qwen3-30B-A3B-Fp8-CT-Block-deepgemm.yaml
- Qwen3-30B-A3B-NvFp4-CT-fi-cutedsl-deepep-ll.yaml
- Qwen3-30B-A3B-NvFp4-CT-fi-cutlass.yaml
- Qwen3-30B-A3B-NvFp4-ModelOpt-fi-cutedsl-deepep-ll.yaml
- Qwen3-30B-A3B-NvFp4-ModelOpt-fi-cutlass.yaml
- Qwen3-30B-A3B-NvFp4-ModelOpt-fi-trtllm.yaml
- DeepSeek-R1-DP.yaml
- DeepSeek-R1-DP_MI325.yaml
- DeepSeek-R1-TP.yaml
- DeepSeek-R1-TP_MI325.yaml
- DeepSeek-V2-Lite-Instruct-FP8.yaml
- DeepSeek-V3.2-DP.yaml
- DeepSeek-V3.2-DP_MI325.yaml
- DeepSeek-V3.2-TP.yaml
- DeepSeek-V3.2-TP_MI325.yaml
- DeepSeek-V4-Flash-DSpark-confidence-TP4.yaml
- DeepSeek-V4-Flash-NVFP4.yaml
- DeepSeek-V4-Pro-NVFP4.yaml
- DiffusionGemma-26B-A4B-it-FP8-dynamic.yaml
- gemma-4-E4B-it-qat-mobile-ct.yaml
- GLM-5.2-NVFP4-TP1-PCP4-EP.yaml
- GLM-5.2-NVFP4-TP2-PCP2-EP.yaml
- Laguna-XS.2-NVFP4.yaml
- Llama-3-8B-Instruct-nonuniform-CT.yaml
- Llama-3.2-1B-Instruct-INT8-CT.yaml
- models-blackwell-ep.txt
- models-blackwell.txt
- models-gfx950-large.txt
- models-h200.txt
- models-mi3xx-fp8-and-mixed.txt
- models-mi3xx.txt
- models-pcp.txt
- models-qwen35-blackwell.txt
- models-qwen35-mi355.txt
- models-small-tp.txt
- models-small.txt
- models-spec-decode.txt
- models-turboquant.txt
- Nemotron-3-Super-120B-A12B-BF16.yaml
- Nemotron-3-Super-120B-A12B-FP8.yaml
- Nemotron-3-Super-120B-A12B-NVFP4.yaml
- Qwen1.5-MoE-A2.7B-Chat-INT8.yaml
- Qwen1.5-MoE-W4A16-CT.yaml
- Qwen2.5-VL-3B-Instruct-FP8-dynamic.yaml
- Qwen3-0.6B-FP8.yaml
- Qwen3-1.7B-MXFP4.yaml
- Qwen3-30B-A3B-MXFP4-AITER-TP2-online.yaml
- Qwen3-30B-A3B-MXFP4-AITER-TP2.yaml
- Qwen3-30B-A3B-MXFP4A16.yaml
- Qwen3-30B-A3B-NVFP4.yaml
- Qwen3-30B-A3B-Thinking-2507-FP8.yaml
- Qwen3-30B-A3B-Thinking-2507-PTPC-FP8.yaml
- Qwen3-4B-TQ-k3v4nc.yaml
- Qwen3-4B-TQ-k8v4.yaml
- Qwen3-4B-TQ-t3nc.yaml
- Qwen3-4B-TQ-t4nc.yaml
- Qwen3-Next-80B-A3B-NVFP4-EP2.yaml
- Qwen3-Next-FP8-EP2.yaml
- Qwen3-Next-FP8-EP2_MI355.yaml
- Qwen3.5-35B-A3B-DEP2.yaml
- Qwen3.5-35B-A3B-FP8-DEP2.yaml
- Qwen3.5-35B-A3B-MXFP4-AITER-TP2.yaml
- Qwen3.5-35B-A3B-MXFP4-EMU-TP2.yaml
- Qwen3.5-397B-A17B-NVFP4-DEP2-MTP.yaml
- Qwen3.5-397B-A17B-NVFP4-DEP2.yaml
- __init__.py
- conftest.py
- gsm8k_eval.py
- README.md
- test_gsm8k_correctness.py
- test_gsm8k_offloading.py
- models-small.txt
- Qwen3.5-4B.yaml
- __init__.py
- conftest.py
- mrcr_eval.py
- README.md
- test_mrcr_correctness.py
- __init__.py
- test_quant_activation_contract.py
- ir_test_utils.py
- test_inplace_op.py
- test_op.py
- __init__.py
- conftest.py
- test_hooks.py
- test_hooks_gpu.py
- test_no_runtime_jit.py
- conftest.py
- test_amx_mla.py
- test_attention.py
- test_attention_selector.py
- test_cache.py
- test_cascade_flash_attn.py
- test_cpu_attn.py
- test_cutlass_mla_decode.py
- test_deepgemm_attention.py
- test_flash_attn.py
- test_flashinfer.py
- test_flashinfer_mla_decode.py
- test_flashinfer_trtllm_attention.py
- test_flashmla.py
- test_flashmla_sparse.py
- test_kimi_k3_mla_fused_epilogue.py
- test_kimi_k3_mla_key_concat_kv_cache.py
- test_lightning_attn.py
- test_merge_attn_states.py
- test_mha_attn.py
- test_minimax_m3.py
- test_minimax_m3_msa_cutlass_sparse_decode.py
- test_mixed_causal_attn.py
- test_mla_cross_layer_kernel_equivalence.py
- test_mla_decode_cpu.py
- test_pack_unpack_triton.py
- test_prefix_prefill.py
- test_rocm_aiter_fa.py
- test_rocm_aiter_mla_causal_verify_mask.py
- test_rocm_aiter_mla_decode.py
- test_rocm_aiter_mla_decode_metadata.py
- test_rocm_aiter_mla_fp8_prefill.py
- test_rocm_aiter_mla_fp8_support.py
- test_rocm_aiter_mla_head_padding.py
- test_rocm_aiter_mla_op_registration.py
- test_rocm_aiter_mla_sparse_metadata_sync.py
- test_rocm_aiter_unified_attn.py
- test_rocm_attention_selector.py
- test_rocm_triton_attn_dsv4.py
- test_triton_decode_attention.py
- test_triton_prefill_attention.py
- test_triton_unified_attention.py
- test_triton_unified_attention_diffkv.py
- test_trtllm_kvfp8_dequant.py
- test_use_trtllm_attention.py
- test_xpu_mla_sparse.py
- test_activation.py
- test_apply_rotary_emb.py
- test_batched_weight_rms_norm.py
- test_cpu_activation.py
- test_fused_allreduce_gemma_rms_norm.py
- test_fused_embed_norm.py
- test_fused_q_kv_rmsnorm.py
- test_fused_qk_norm_rope.py
- test_fused_quant_layernorm.py
- test_fused_rms_norm_gated.py
- test_fused_silu_mul_block_quant.py
- test_layernorm.py
- test_minimax_reduce_rms.py
- test_mrope.py
- test_opcheck.py
- test_permute_cols.py
- test_pos_encoding.py
- test_rocm_aiter_ops.py
- test_rotary_embedding.py
- test_rotary_embedding_mla_cache_fused.py
- test_uva.py
- test_vit_bilinear_pos_embed.py
- test_vit_fp8_attn.py
- test_vit_fp8_quant.py
- test_vit_fp8_scaling.py
- helpers.py
- test_autotune.py
- test_benchmark_script.py
- test_case_key.py
- test_config_manager.py
- test_dynamic_per_token_scaled_fp8_quant.py
- test_fused_qk_norm_rope.py
- test_helion_available.py
- test_pattern_matching.py
- test_per_token_group_fp8_quant.py
- test_register.py
- test_rms_norm_dynamic_per_token_quant.py
- test_rms_norm_per_block_quant.py
- test_silu_and_mul_per_block_quant.py
- test_silu_mul_fp8.py
- test_utils.py
- utils.py
- test_ir_ops.py
- test_layernorm.py
- test_cpu_gdn_ops.py
- __init__.py
- test_causal_conv1d.py
- test_cpu_short_conv.py
- test_gdn_forward_core_split.py
- test_gdn_fused_mtp.py
- test_gdn_prefill_cutedsl.py
- test_mamba_mixer2.py
- test_mamba_ssm.py
- test_mamba_ssm_configs.py
- test_mamba_ssm_ssd.py
- test_memcpy_u64_tiled.py
- test_precopy_mamba_align.py
- test_replayssm_prefill_decode_equivalence_mamba2.py
- test_replayssm_standard_decode_mamba2.py
- test_ssu_dispatch.py
- utils.py
- __init__.py
- cli_args.py
- common.py
- make_feature_matrix.py
- mk_objects.py
- parallel_utils.py
- profile_modular_kernel.py
- __init__.py
- conftest.py
- parallel_utils.py
- test_b12x.py
- test_batched_deepgemm.py
- test_batched_moe.py
- test_block_fp8.py
- test_block_int8.py
- test_count_expert_num_tokens.py
- test_cpu_fused_moe.py
- test_cpu_int4_moe.py
- test_cpu_quant_fused_moe.py
- test_cutedsl_moe.py
- test_cutlass_moe.py
- test_deepep_deepgemm_moe.py
- test_deepep_moe.py
- test_deepep_v2_moe.py
- __init__.py
- allclose_default.py
- conftest.py
- __init__.py
- ci_envs.py
- conftest.py
- .clang-format
- .coveragerc
- .dockerignore
- .git-blame-ignore-revs
- .gitignore
- .markdownlint.yaml
- .pre-commit-config.yaml
- .readthedocs.yaml
- .shellcheckrc
- AGENTS.md
- build_rust.sh
- build_vllm_ppc64le.sh
- CLAUDE.md
- CMakeLists.txt
- CODE_OF_CONDUCT.md
- codecov.yml
- CONTRIBUTING.md
- DCO
- LICENSE
- MANIFEST.in
- mkdocs.yaml
- pyproject.toml
- README.md
- RELEASE.md
- rust-toolchain.toml
- SECURITY.md
- setup.py
# Installation Guide
git clone https://github.com/vllm-project/vllm
Downloads the entire project code from GitHub to your computer.
cd vllm
Moves into the project folder you just downloaded.
2. Docker
Easy Recommended- Git Needed to download the project code from GitHub.
- Docker Desktop Needed to build and run containers. Install it and keep it running in the background.
docker build -f docker/Dockerfile -t vllm .
Builds a runnable image based on the Dockerfile.
docker run -p 8080:80 vllm
Runs the built image as an actual container.
3. CMake
Mediummkdir build && cd build
Creates a folder to hold the build output and moves into it.
cmake ..
Analyzes the source code and generates build configuration files (must be run inside the build folder).
make
Compiles the code based on the generated build configuration to produce an executable.
4. Python
Easyuv pip install vllm
Installs the Python libraries listed in requirements.txt (or similar).
Pulled directly from this repo's README.
5. Rust
Medium- Git Needed to download the project code from GitHub.
- Rust (rustup) Installing via rustup also installs cargo.
cd rust
This project's files live in a subfolder, so move into it first.
cargo build --release
Compiles the Rust project.
cargo run
Builds and then immediately runs the program.
