InfiniTensor/include/cuda/cuda_common.h

#pragma once
#include "core/common.h"
#include <cublas_v2.h>
#include <cuda.h>
#include <cuda_profiler_api.h>
#include <cudnn.h>
#include <curand.h>
#include <memory>

#define checkCudaError(call)                                                   \
    if (auto err = call; err != cudaSuccess)                                   \
    throw ::infini::Exception(std::string("[") + __FILE__ + ":" +              \
                              std::to_string(__LINE__) + "] CUDA error (" +    \
                              #call + "): " + cudaGetErrorString(err))

#define checkCUresult(call)                                                    \
    {                                                                          \
        auto err = call;                                                       \
        const char *errName;                                                   \
        if (CUDA_SUCCESS != err) {                                             \
            cuGetErrorString(err, &errName);                                   \
            IT_ASSERT(err == CUDA_SUCCESS,                                     \
                      (string("CU error: ") + string(errName)));               \
        }                                                                      \
    }

#define checkCublasError(call)                                                 \
    {                                                                          \
        auto err = call;                                                       \
        if (CUBLAS_STATUS_SUCCESS != err) {                                    \
            fprintf(stderr, "cuBLAS error in %s:%i : %s.\n", __FILE__,         \
                    __LINE__, cublasGetErrorString(err));                      \
            exit(EXIT_FAILURE);                                                \
        }                                                                      \
    }

#define checkCudnnError(call)                                                  \
    if (auto err = call; err != CUDNN_STATUS_SUCCESS)                          \
    throw ::infini::Exception(std::string("[") + __FILE__ + ":" +              \
                              std::to_string(__LINE__) + "] cuDNN error (" +   \
                              #call + "): " + cudnnGetErrorString(err))

#define checkCurandError(call)                                                 \
    {                                                                          \
        auto err = call;                                                       \
        if (CURAND_STATUS_SUCCESS != err) {                                    \
            fprintf(stderr, "cuRAND error in %s:%i : %s.\n", __FILE__,         \
                    __LINE__, curandGetErrorString(err));                      \
            exit(EXIT_FAILURE);                                                \
        }                                                                      \
    }

namespace infini {

inline const char *cublasGetErrorString(cublasStatus_t error) {
    switch (error) {
    case CUBLAS_STATUS_SUCCESS:
        return "CUBLAS_STATUS_SUCCESS";
    case CUBLAS_STATUS_NOT_INITIALIZED:
        return "CUBLAS_STATUS_NOT_INITIALIZED";
    case CUBLAS_STATUS_ALLOC_FAILED:
        return "CUBLAS_STATUS_ALLOC_FAILED";
    case CUBLAS_STATUS_INVALID_VALUE:
        return "CUBLAS_STATUS_INVALID_VALUE";
    case CUBLAS_STATUS_ARCH_MISMATCH:
        return "CUBLAS_STATUS_ARCH_MISMATCH";
    case CUBLAS_STATUS_MAPPING_ERROR:
        return "CUBLAS_STATUS_MAPPING_ERROR";
    case CUBLAS_STATUS_EXECUTION_FAILED:
        return "CUBLAS_STATUS_EXECUTION_FAILED";
    case CUBLAS_STATUS_INTERNAL_ERROR:
        return "CUBLAS_STATUS_INTERNAL_ERROR";
    case CUBLAS_STATUS_NOT_SUPPORTED:
        return "CUBLAS_STATUS_NOT_SUPPORTED";
    case CUBLAS_STATUS_LICENSE_ERROR:
        return "CUBLAS_STATUS_LICENSE_ERROR";
    }
    return "<unknown>";
}

inline const char *curandGetErrorString(curandStatus_t error) {
    switch (error) {
    case CURAND_STATUS_SUCCESS:
        return "CURAND_STATUS_SUCCESS";
    case CURAND_STATUS_VERSION_MISMATCH:
        return "CURAND_STATUS_VERSION_MISMATCH";
    case CURAND_STATUS_NOT_INITIALIZED:
        return "CURAND_STATUS_NOT_INITIALIZED";
    case CURAND_STATUS_ALLOCATION_FAILED:
        return "CURAND_STATUS_ALLOCATION_FAILED";
    case CURAND_STATUS_TYPE_ERROR:
        return "CURAND_STATUS_TYPE_ERROR";
    case CURAND_STATUS_OUT_OF_RANGE:
        return "CURAND_STATUS_OUT_OF_RANGE";
    case CURAND_STATUS_LENGTH_NOT_MULTIPLE:
        return "CURAND_STATUS_LENGTH_NOT_MULTIPLE";
    case CURAND_STATUS_DOUBLE_PRECISION_REQUIRED:
        return "CURAND_STATUS_DOUBLE_PRECISION_REQUIRED";
    case CURAND_STATUS_LAUNCH_FAILURE:
        return "CURAND_STATUS_LAUNCH_FAILURE";
    case CURAND_STATUS_PREEXISTING_FAILURE:
        return "CURAND_STATUS_PREEXISTING_FAILURE";
    case CURAND_STATUS_INITIALIZATION_FAILED:
        return "CURAND_STATUS_INITIALIZATION_FAILED";
    case CURAND_STATUS_ARCH_MISMATCH:
        return "CURAND_STATUS_ARCH_MISMATCH";
    case CURAND_STATUS_INTERNAL_ERROR:
        return "CURAND_STATUS_INTERNAL_ERROR";
    }
    return "<unknown>";
}

using CudaPtr = void *;

class CUDAStream {
  public:
    CUDAStream(const CUDAStream &) = delete;
    CUDAStream(CUDAStream &&) = delete;
    void operator=(const CUDAStream &) = delete;
    void operator=(CUDAStream &&) = delete;
    static cudaStream_t getCurrentStream() { return _stream; }
    static void Init() { CUDAStream::_stream = 0; };
    static void createStream() { checkCudaError(cudaStreamCreate(&_stream)); }
    static void destroyStream() { checkCudaError(cudaStreamDestroy(_stream)); }

  private:
    CUDAStream(){};
    static cudaStream_t _stream;
};

} // namespace infini
Add CUDA runtime (#6) * Fix: add warm-up and repetition in timing * Add: CUDA runtime and float support * Refactor: Cuda and Cpu runtimes inherit Runtime * Add: environment script for Lotus * Add: Lotus build instructions * Update README.md Co-authored-by: Liyan Zheng <liyan-zheng@outlook.com> 2022-08-22 15:01:03 +08:00			`#pragma once`
			`#include "core/common.h"`
			`#include <cublas_v2.h>`
			`#include <cuda.h>`
Fix CMake USE_CUDA (#36) * Fix: build lib without cuda * Chore: rename GBMM and G2BMM files * Fix: seperate CUDA tests from operator tests * Fix: CMake CMP0104 * Chore: fix typo * Chore: remove unused headers Co-authored-by: Liyan Zheng <liyan-zheng@outlook.com> 2022-09-21 12:28:00 +08:00			`#include <cuda_profiler_api.h>`
Add CUDA runtime (#6) * Fix: add warm-up and repetition in timing * Add: CUDA runtime and float support * Refactor: Cuda and Cpu runtimes inherit Runtime * Add: environment script for Lotus * Add: Lotus build instructions * Update README.md Co-authored-by: Liyan Zheng <liyan-zheng@outlook.com> 2022-08-22 15:01:03 +08:00			`#include <cudnn.h>`
			`#include <curand.h>`
[feature] add cudagraph support (#215) * [feature] add cudagraph support * modify code to pass the cuda_all_reduce test 2024-02-21 14:00:25 +08:00			`#include <memory>`
Add CUDA runtime (#6) * Fix: add warm-up and repetition in timing * Add: CUDA runtime and float support * Refactor: Cuda and Cpu runtimes inherit Runtime * Add: environment script for Lotus * Add: Lotus build instructions * Update README.md Co-authored-by: Liyan Zheng <liyan-zheng@outlook.com> 2022-08-22 15:01:03 +08:00
			`#define checkCudaError(call) \`
tensor parallel for transformer (#125) * add cmake bits about NCCL * move example to examples/NNmodel * impl NCCL communicator * add comm related function to Runtime * export runtime interface * add launch.py * use unique name to distingush the the NCCL ID file * add timeout to communicator init * expose communicator obj from runtime obj, add unit test for nccl communicator * reformat files * Add allReduce operator and cuda nccl allReduce kernel * impl model parallel for resnet * add allGather nccl kernel and operator * Add allreduce allgather operator tests, change allgather kernel to output list of tensor, fix shape infer, handle nullptr output * fix format of onnx.py * use concat following AllGather * get tensor parallel for resnet * fix format of graph_handler.cc * change BUILD_DIST default to OFF * polish code of communicator * update .gitignore * export min/max to python * fix MatMul * modify launch.py to run opt * hack to treat ReduceSum as AllReduceSum * throw exception in cuda error * fix parallel_opt.py * improve the error prompt and cuda error check * fix GatherObj::GatherObj member init * fix size calculation for scalar (rank = 0) tensor * MatMul supports bias * fix add bias for row parallel gemm * add --gen_std to launch.py * fix AllReduceNCCL * update launch.py * less log * update parallel_opt * update launch.py * add __eq__ for Placement sub-classes * less benchmark run * fix placement infer for matmul * fix vacabuary size * fix Exception * Add shard tensor with group to support gpt2 * Add find successor function to find split op at different depth * recover CommunicatorObj * improve error mesasge * optimize parallel_opt.py * optimize launch.py * recover docs for all_reduce and all_gather * Fix API * fix format --------- Co-authored-by: panzezhong <panzezhong@qiyuanlab.com> Co-authored-by: Haojie Wang <haojie0429@gmail.com> 2023-09-14 14:19:45 +08:00			`if (auto err = call; err != cudaSuccess) \`
			`throw ::infini::Exception(std::string("[") + __FILE__ + ":" + \`
			`std::to_string(__LINE__) + "] CUDA error (" + \`
			`#call + "): " + cudaGetErrorString(err))`
Add CUDA runtime (#6) * Fix: add warm-up and repetition in timing * Add: CUDA runtime and float support * Refactor: Cuda and Cpu runtimes inherit Runtime * Add: environment script for Lotus * Add: Lotus build instructions * Update README.md Co-authored-by: Liyan Zheng <liyan-zheng@outlook.com> 2022-08-22 15:01:03 +08:00
Add TVM codegen for MemboundOp (#35) * Add: interface for membound TVM kernel and test * add getAnsorCode * add evaluation, but link failed * add evaluation of kernel, but link failed * Fix: link libcuda and nvrtc * add print * Add: const for source of copy * compile and evaluate the kernel * add compute * fix gen_ansor_op.py * fix membound_TVM * format and fix CMakeLists.txt * fix memory leak Co-authored-by: Liyan Zheng <liyan-zheng@outlook.com> Co-authored-by: huangshuhong <huangsh19@mails.tsinghua.edu.cn> 2022-09-22 18:06:45 +08:00			`#define checkCUresult(call) \`
			`{ \`
			`auto err = call; \`
			`const char *errName; \`
			`if (CUDA_SUCCESS != err) { \`
			`cuGetErrorString(err, &errName); \`
NNET supports TVM backend and kernels (#78) * Add: mutator InfoGAN minimum test * Add: cache and padding (bugs!!) * Add: expression reader as a cmake target * Fix: [Intermediate] NMutator::expressionToGraph To be fix: matmul with implicit broadcast * Add: matmul broadcast * Fix: GraphObj ctor should use cloneTensor * Fix: cuBLAS failure when codegen is enabled * Add: Exception for checkCuError * Fix: graph OpList ctor * Add: expr simplication for TVM * Add: TVM headers and CMake include paths * Add: CMake config * Add: PackedFunc (broken) * Fix: remove cuCtxCreate which makes TVM fails * Fix: membound_tvm * Fix: test_memboundOp * Add: PRelu Expr and AsTVMVisitor * Add: Random generator * Add: support TVM packed function * Fix: specify runtime * Add: CMake support of TVM * Add: detailed output of Matmul * Add: comments for Matmul * Chore: format and comments * Chore: GraphObj::selfCheck without assert control * Fix: CMAKE_CXX_FLAGS in CMakeLists * fix merge bug * update api for mkl batchnorm test * fix lotus env * fig header bug --------- Co-authored-by: Liyan Zheng <liyan-zheng@outlook.com> Co-authored-by: huangshuhong <huangsh19@mails.tsinghua.edu.cn> Co-authored-by: whjthu <haojie0429@gmail.com> 2023-04-18 00:26:36 +08:00			`IT_ASSERT(err == CUDA_SUCCESS, \`
			`(string("CU error: ") + string(errName))); \`
Add TVM codegen for MemboundOp (#35) * Add: interface for membound TVM kernel and test * add getAnsorCode * add evaluation, but link failed * add evaluation of kernel, but link failed * Fix: link libcuda and nvrtc * add print * Add: const for source of copy * compile and evaluate the kernel * add compute * fix gen_ansor_op.py * fix membound_TVM * format and fix CMakeLists.txt * fix memory leak Co-authored-by: Liyan Zheng <liyan-zheng@outlook.com> Co-authored-by: huangshuhong <huangsh19@mails.tsinghua.edu.cn> 2022-09-22 18:06:45 +08:00			`} \`
			`}`

Add CUDA runtime (#6) * Fix: add warm-up and repetition in timing * Add: CUDA runtime and float support * Refactor: Cuda and Cpu runtimes inherit Runtime * Add: environment script for Lotus * Add: Lotus build instructions * Update README.md Co-authored-by: Liyan Zheng <liyan-zheng@outlook.com> 2022-08-22 15:01:03 +08:00			`#define checkCublasError(call) \`
			`{ \`
			`auto err = call; \`
			`if (CUBLAS_STATUS_SUCCESS != err) { \`
			`fprintf(stderr, "cuBLAS error in %s:%i : %s.\n", __FILE__, \`
			`__LINE__, cublasGetErrorString(err)); \`
			`exit(EXIT_FAILURE); \`
			`} \`
			`}`

			`#define checkCudnnError(call) \`
tensor parallel for transformer (#125) * add cmake bits about NCCL * move example to examples/NNmodel * impl NCCL communicator * add comm related function to Runtime * export runtime interface * add launch.py * use unique name to distingush the the NCCL ID file * add timeout to communicator init * expose communicator obj from runtime obj, add unit test for nccl communicator * reformat files * Add allReduce operator and cuda nccl allReduce kernel * impl model parallel for resnet * add allGather nccl kernel and operator * Add allreduce allgather operator tests, change allgather kernel to output list of tensor, fix shape infer, handle nullptr output * fix format of onnx.py * use concat following AllGather * get tensor parallel for resnet * fix format of graph_handler.cc * change BUILD_DIST default to OFF * polish code of communicator * update .gitignore * export min/max to python * fix MatMul * modify launch.py to run opt * hack to treat ReduceSum as AllReduceSum * throw exception in cuda error * fix parallel_opt.py * improve the error prompt and cuda error check * fix GatherObj::GatherObj member init * fix size calculation for scalar (rank = 0) tensor * MatMul supports bias * fix add bias for row parallel gemm * add --gen_std to launch.py * fix AllReduceNCCL * update launch.py * less log * update parallel_opt * update launch.py * add __eq__ for Placement sub-classes * less benchmark run * fix placement infer for matmul * fix vacabuary size * fix Exception * Add shard tensor with group to support gpt2 * Add find successor function to find split op at different depth * recover CommunicatorObj * improve error mesasge * optimize parallel_opt.py * optimize launch.py * recover docs for all_reduce and all_gather * Fix API * fix format --------- Co-authored-by: panzezhong <panzezhong@qiyuanlab.com> Co-authored-by: Haojie Wang <haojie0429@gmail.com> 2023-09-14 14:19:45 +08:00			`if (auto err = call; err != CUDNN_STATUS_SUCCESS) \`
			`throw ::infini::Exception(std::string("[") + __FILE__ + ":" + \`
			`std::to_string(__LINE__) + "] cuDNN error (" + \`
			`#call + "): " + cudnnGetErrorString(err))`
Add CUDA runtime (#6) * Fix: add warm-up and repetition in timing * Add: CUDA runtime and float support * Refactor: Cuda and Cpu runtimes inherit Runtime * Add: environment script for Lotus * Add: Lotus build instructions * Update README.md Co-authored-by: Liyan Zheng <liyan-zheng@outlook.com> 2022-08-22 15:01:03 +08:00
			`#define checkCurandError(call) \`
			`{ \`
			`auto err = call; \`
			`if (CURAND_STATUS_SUCCESS != err) { \`
			`fprintf(stderr, "cuRAND error in %s:%i : %s.\n", __FILE__, \`
			`__LINE__, curandGetErrorString(err)); \`
			`exit(EXIT_FAILURE); \`
			`} \`
			`}`

			`namespace infini {`

			`inline const char *cublasGetErrorString(cublasStatus_t error) {`
			`switch (error) {`
			`case CUBLAS_STATUS_SUCCESS:`
			`return "CUBLAS_STATUS_SUCCESS";`
			`case CUBLAS_STATUS_NOT_INITIALIZED:`
			`return "CUBLAS_STATUS_NOT_INITIALIZED";`
			`case CUBLAS_STATUS_ALLOC_FAILED:`
			`return "CUBLAS_STATUS_ALLOC_FAILED";`
			`case CUBLAS_STATUS_INVALID_VALUE:`
			`return "CUBLAS_STATUS_INVALID_VALUE";`
			`case CUBLAS_STATUS_ARCH_MISMATCH:`
			`return "CUBLAS_STATUS_ARCH_MISMATCH";`
			`case CUBLAS_STATUS_MAPPING_ERROR:`
			`return "CUBLAS_STATUS_MAPPING_ERROR";`
			`case CUBLAS_STATUS_EXECUTION_FAILED:`
			`return "CUBLAS_STATUS_EXECUTION_FAILED";`
			`case CUBLAS_STATUS_INTERNAL_ERROR:`
			`return "CUBLAS_STATUS_INTERNAL_ERROR";`
			`case CUBLAS_STATUS_NOT_SUPPORTED:`
			`return "CUBLAS_STATUS_NOT_SUPPORTED";`
			`case CUBLAS_STATUS_LICENSE_ERROR:`
			`return "CUBLAS_STATUS_LICENSE_ERROR";`
			`}`
			`return "<unknown>";`
			`}`

			`inline const char *curandGetErrorString(curandStatus_t error) {`
			`switch (error) {`
			`case CURAND_STATUS_SUCCESS:`
			`return "CURAND_STATUS_SUCCESS";`
			`case CURAND_STATUS_VERSION_MISMATCH:`
			`return "CURAND_STATUS_VERSION_MISMATCH";`
			`case CURAND_STATUS_NOT_INITIALIZED:`
			`return "CURAND_STATUS_NOT_INITIALIZED";`
			`case CURAND_STATUS_ALLOCATION_FAILED:`
			`return "CURAND_STATUS_ALLOCATION_FAILED";`
			`case CURAND_STATUS_TYPE_ERROR:`
			`return "CURAND_STATUS_TYPE_ERROR";`
			`case CURAND_STATUS_OUT_OF_RANGE:`
			`return "CURAND_STATUS_OUT_OF_RANGE";`
			`case CURAND_STATUS_LENGTH_NOT_MULTIPLE:`
			`return "CURAND_STATUS_LENGTH_NOT_MULTIPLE";`
			`case CURAND_STATUS_DOUBLE_PRECISION_REQUIRED:`
			`return "CURAND_STATUS_DOUBLE_PRECISION_REQUIRED";`
			`case CURAND_STATUS_LAUNCH_FAILURE:`
			`return "CURAND_STATUS_LAUNCH_FAILURE";`
			`case CURAND_STATUS_PREEXISTING_FAILURE:`
			`return "CURAND_STATUS_PREEXISTING_FAILURE";`
			`case CURAND_STATUS_INITIALIZATION_FAILED:`
			`return "CURAND_STATUS_INITIALIZATION_FAILED";`
			`case CURAND_STATUS_ARCH_MISMATCH:`
			`return "CURAND_STATUS_ARCH_MISMATCH";`
			`case CURAND_STATUS_INTERNAL_ERROR:`
			`return "CURAND_STATUS_INTERNAL_ERROR";`
			`}`
			`return "<unknown>";`
			`}`

			`using CudaPtr = void *;`

[feature] add cudagraph support (#215) * [feature] add cudagraph support * modify code to pass the cuda_all_reduce test 2024-02-21 14:00:25 +08:00			`class CUDAStream {`
			`public:`
			`CUDAStream(const CUDAStream &) = delete;`
			`CUDAStream(CUDAStream &&) = delete;`
			`void operator=(const CUDAStream &) = delete;`
			`void operator=(CUDAStream &&) = delete;`
			`static cudaStream_t getCurrentStream() { return _stream; }`
			`static void Init() { CUDAStream::_stream = 0; };`
			`static void createStream() { checkCudaError(cudaStreamCreate(&_stream)); }`
			`static void destroyStream() { checkCudaError(cudaStreamDestroy(_stream)); }`

			`private:`
			`CUDAStream(){};`
			`static cudaStream_t _stream;`
			`};`

ADD: batch norm operator and cuda kernel. (#44) fix numInputs of batchNorm, add new line in file ending. ADD: batch norm operator and cuda kernel. add training remove comments. fix compile error. add batch norm operator and cuda kernel. 2022-10-15 16:29:28 +08:00			`} // namespace infini`