pytorch · YUNQIUGUO · Jan 24, 2025
diff --git a/fbgemm_gpu/experimental/gen_ai/CMakeLists.txt b/fbgemm_gpu/experimental/gen_ai/CMakeLists.txt
@@ -33,7 +33,10 @@ endif()
 # CUDA-specific sources
 file(GLOB_RECURSE experimental_gen_ai_cpp_source_files_cuda
   src/quantize/cutlass_extensions/*.cu
-  src/quantize/cutlass_extensions/**/*.cu)
+  src/quantize/cutlass_extensions/**/*.cu
+  src/quantize/fast_gemv/*.cu
+  src/quantize/fast_gemv/**/*.cu
+  src/quantize/fast_gemv/**/*.cuh)
 
 # HIP-specific sources
 file(GLOB_RECURSE experimental_gen_ai_cpp_source_files_hip

diff --git a/fbgemm_gpu/experimental/gen_ai/bench/quantize_ops.py b/fbgemm_gpu/experimental/gen_ai/bench/quantize_ops.py
@@ -361,6 +361,38 @@ def cuda(self) -> bool:
         return True
 
 
+@register_quantize_op
+class FP16OSSFastGemv(QuantizeOpBase):
+    """
+    FP16 oss fast gemv.
+    """
+
+    def quantize(self, x, w):
+        return x, w
+
+    def compute(self, x, w):
+        out = torch.ops.fbgemm.f16_fast_gemv(x, w)
+        return out
+
+    def quantize_and_compute(self, x, w):
+        # dummy quantize
+        x, w = self.quantize(x, w)
+        return self.compute(x, w)
+
+    @property
+    def name(self) -> str:
+        return "fp16_oss_fast_gemv"
+
+    @property
+    def hip(self) -> bool:
+        # This implementation is specific to cublas.
+        return False
+
+    @property
+    def cuda(self) -> bool:
+        return True
+
+
 @register_quantize_op
 class FP8CublasRowwiseGemm(QuantizeOpBase):
     """

diff --git a/fbgemm_gpu/experimental/gen_ai/src/quantize/fast_gemv/f16_fast_gemv.cu b/fbgemm_gpu/experimental/gen_ai/src/quantize/fast_gemv/f16_fast_gemv.cu
@@ -0,0 +1,60 @@
+/*
+ * Copyright (c) Meta Platforms, Inc. and affiliates.
+ * All rights reserved.
+ *
+ * This source code is licensed under the BSD-style license found in the
+ * LICENSE file in the root directory of this source tree.
+ */
+
+#include <ATen/ATen.h>
+#include <ATen/cuda/CUDAContext.h>
+#include <c10/core/ScalarType.h>
+#include <c10/cuda/CUDAGuard.h>
+
+#include "include/fast_gemv.cuh"
+
+namespace fbgemm_gpu {
+
+#if CUDART_VERSION >= 12000
+
+at::Tensor f16_fast_gemv(at::Tensor X, at::Tensor W) {
+  // note: oss fast gemv implementation accepts vector shape as (size, 1) i.e.
+  // (K, M)
+  // X: K x M
+  // W: N x K
+  auto m = X.size(1);
+  auto n = W.size(0);
+  auto k = W.size(1);
+
+  TORCH_CHECK(X.is_cuda() && X.is_contiguous());
+  TORCH_CHECK(W.is_cuda() && W.is_contiguous());
+
+  auto block_dim_x = k / 8;
+  auto block_dim_y = MAX_THREADS_PER_BLOCK / block_dim_x;
+  dim3 block_dim(block_dim_x, block_dim_y);
+  dim3 grid_dim(1, n / block_dim_y);
+  unsigned int num_per_thread = k / block_dim_x;
+
+  auto stream = at::cuda::getCurrentCUDAStream();
+
+  auto Y = at::empty({n, m}, X.options().dtype(at::kHalf));
+
+  gemv_fp16<<<grid_dim, block_dim, 0, stream>>>(
+      (half*)W.data_ptr(), // mat
+      (half*)X.data_ptr(), // vec
+      (half*)Y.data_ptr(), // res
+      k,
+      num_per_thread);
+
+  return Y;
+}
+
+#else
+
+at::Tensor f16_fast_gemv(at::Tensor X, at::Tensor W) {
+  throw std::runtime_error(
+      "CUDA version is older than 12.0"); // requires CUDA>=12
+}
+#endif
+
+} // namespace fbgemm_gpu