Download cutlass_extensions/common.hpp from RedHatAI/quantization: direct link, hf CLI and curl.
- Browser
- Download file 1.9 kB
-
https://huggingface.co/RedHatAI/quantization/resolve/refs%2Fpr%2F2/cutlass_extensions/common.hpp
- Command line
-
hf download hf://RedHatAI/quantization@refs/pr/2/cutlass_extensions/common.hpp
-
curl -L -o common.hpp https://huggingface.co/RedHatAI/quantization/resolve/refs%2Fpr%2F2/cutlass_extensions/common.hpp
1.9 kB
| /** | |
| * Helper function for checking CUTLASS errors | |
| */ | |
| inline int get_cuda_max_shared_memory_per_block_opt_in(int const device) { | |
| int max_shared_mem_per_block_opt_in = 0; | |
| cudaDeviceGetAttribute(&max_shared_mem_per_block_opt_in, | |
| cudaDevAttrMaxSharedMemoryPerBlockOptin, device); | |
| return max_shared_mem_per_block_opt_in; | |
| } | |
| int32_t get_sm_version_num(); | |
| /** | |
| * A wrapper for a kernel that is used to guard against compilation on | |
| * architectures that will never use the kernel. The purpose of this is to | |
| * reduce the size of the compiled binary. | |
| * __CUDA_ARCH__ is not defined in host code, so this lets us smuggle the ifdef | |
| * into code that will be executed on the device where it is defined. | |
| */ | |
| template <typename Kernel> | |
| struct enable_sm90_or_later : Kernel { | |
| template <typename... Args> | |
| CUTLASS_DEVICE void operator()(Args&&... args) { | |
| Kernel::operator()(std::forward<Args>(args)...); | |
| } | |
| }; | |
| template <typename Kernel> | |
| struct enable_sm90_only : Kernel { | |
| template <typename... Args> | |
| CUTLASS_DEVICE void operator()(Args&&... args) { | |
| Kernel::operator()(std::forward<Args>(args)...); | |
| } | |
| }; | |
| template <typename Kernel> | |
| struct enable_sm100_only : Kernel { | |
| template <typename... Args> | |
| CUTLASS_DEVICE void operator()(Args&&... args) { | |
| Kernel::operator()(std::forward<Args>(args)...); | |
| } | |
| }; | |