qutlass
QuTLASS: CUTLASS-Powered Quantized BLAS for Deep Learning
File Explorer
Download Latest Version (.zip)- flops_mxfp4_sm100_flashinfer.svg
- flops_mxfp4_sm120_cutlass.svg
- flops_nvfp4_sm100_flashinfer.svg
- flops_nvfp4_sm120_cutlass.svg
- qwen3-14b-end-to-end-prefill-speedup-mxfp4-vs-bf16-on-rtx5090.svg
- qwen3-8b-end-to-end-prefill-speedup-mxfp4-vs-bf16-on-rtx5090.svg
- training.png
- __init__.py
- bench_mxfp4_sm100.py
- bench_mxfp4_sm120.py
- bench_nvfp4_sm100.py
- bench_nvfp4_sm120.py
- mma_mx_sm120.h
- sm100_blockscaled_layout.hpp
- sm100_builder.inl
- collective_builder.hpp
- collective_epilogue.hpp
- sm100_epilogue_tma_warpspecialized.hpp
- callbacks.hpp
- operations.hpp
- sm100_callbacks_tma_warpspecialized.hpp
- sm100_visitor_compute_tma_warpspecialized.hpp
- sm100_visitor_store_tma_warpspecialized.hpp
- sm90_callbacks_tma_warpspecialized.hpp
- linear_combination_quant.h
- default_epilogue_tensor_op_quant.h
- epilogue_quant.h
- default_mx_gemm_configuration.h
- gemm_quant.h
- mx_gemm.h
- default_gemm_quant.h
- default_mx_gemm.h
- gemm_quant.h
- mx_gemm.h
- default_mx_mma.h
- default_mx_mma_core_sm80.h
- mx_mma_base.h
- mx_mma_multistage.h
- default_mma_mx_tensor_op.h
- default_mx_mma_tensor_op_sm80.h
- mma_mx_tensor_op.h
- mma_mx_tensor_op_fast_f32.h
- backward_host.h
- bindings_utils.h
- common.h
- cuda_utils.h
- fused_quantize_host.h
- gemm.h
- registration.h
- bindings.cpp
- fused_quantize_mx.cu
- fused_quantize_mx_mask.cu
- fused_quantize_mx_sm100.cu
- fused_quantize_nv.cu
- fused_quantize_nv_sm100.cu
- gemm.cu
- gemm_ada.cu
- quartet_bwd_sm120.cu
- __init__.py
- utils.py
- __init__.py
- mxfp4_test.py
- mxfp8_test.py
- nvfp4_test.py
- quartet_test.py
- cutlass
- .gitignore
- .gitmodules
- CMakeLists.txt
- LICENSE
- pyproject.toml
- README.md
- requirements.txt
- setup.py
// repository documentation
Was this content helpful?
(0 ratings)
