mirror of
https://github.com/NVIDIA/cuda-samples.git
synced 2026-10-11 23:38:25 +08:00
Add and update samples with CUDA 10.1 support
This commit is contained in:
@@ -246,12 +246,6 @@ ifeq ($(TARGET_ARCH),armv7l)
|
||||
SAMPLE_ENABLED := 0
|
||||
endif
|
||||
|
||||
# This sample is not supported on aarch64
|
||||
ifeq ($(TARGET_ARCH),aarch64)
|
||||
$(info >>> WARNING - cudaTensorCoreGemm is not supported on aarch64 - waiving sample <<<)
|
||||
SAMPLE_ENABLED := 0
|
||||
endif
|
||||
|
||||
ALL_LDFLAGS :=
|
||||
ALL_LDFLAGS += $(ALL_CCFLAGS)
|
||||
ALL_LDFLAGS += $(addprefix -Xlinker ,$(LDFLAGS))
|
||||
@@ -264,7 +258,11 @@ LIBRARIES :=
|
||||
################################################################################
|
||||
|
||||
# Gencode arguments
|
||||
ifeq ($(TARGET_ARCH),$(filter $(TARGET_ARCH),armv7l aarch64))
|
||||
SMS ?= 70 72 75
|
||||
else
|
||||
SMS ?= 70 75
|
||||
endif
|
||||
|
||||
ifeq ($(SMS),)
|
||||
$(info >>> WARNING - no SM architectures have been specified - waiving sample <<<)
|
||||
|
||||
@@ -43,12 +43,16 @@ In addition to that, it demonstrates the use of the new CUDA function attribute
|
||||
<scope>1:CUDA Basic Topics</scope>
|
||||
</scopes>
|
||||
<sm-arch>sm70</sm-arch>
|
||||
<sm-arch>sm72</sm-arch>
|
||||
<sm-arch>sm75</sm-arch>
|
||||
<supported_envs>
|
||||
<env>
|
||||
<arch>x86_64</arch>
|
||||
<platform>linux</platform>
|
||||
</env>
|
||||
<env>
|
||||
<arch>aarch64</arch>
|
||||
</env>
|
||||
<env>
|
||||
<platform>windows7</platform>
|
||||
</env>
|
||||
|
||||
@@ -14,7 +14,7 @@ Matrix Multiply, WMMA, Tensor Cores
|
||||
|
||||
## Supported SM Architectures
|
||||
|
||||
[SM 7.0 ](https://developer.nvidia.com/cuda-gpus) [SM 7.5 ](https://developer.nvidia.com/cuda-gpus)
|
||||
[SM 7.0 ](https://developer.nvidia.com/cuda-gpus) [SM 7.2 ](https://developer.nvidia.com/cuda-gpus) [SM 7.5 ](https://developer.nvidia.com/cuda-gpus)
|
||||
|
||||
## Supported OSes
|
||||
|
||||
@@ -22,7 +22,7 @@ Linux, Windows
|
||||
|
||||
## Supported CPU Architecture
|
||||
|
||||
x86_64, ppc64le
|
||||
x86_64, ppc64le, aarch64
|
||||
|
||||
## CUDA APIs involved
|
||||
|
||||
@@ -31,7 +31,7 @@ cudaMallocManaged, cudaDeviceSynchronize, cudaFuncSetAttribute, cudaEventCreate,
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Download and install the [CUDA Toolkit 10.0](https://developer.nvidia.com/cuda-downloads) for your corresponding platform.
|
||||
Download and install the [CUDA Toolkit 10.1](https://developer.nvidia.com/cuda-downloads) for your corresponding platform.
|
||||
|
||||
## Build and Run
|
||||
|
||||
@@ -52,9 +52,9 @@ $ cd <sample_dir>
|
||||
$ make
|
||||
```
|
||||
The samples makefiles can take advantage of certain options:
|
||||
* **TARGET_ARCH=<arch>** - cross-compile targeting a specific architecture. Allowed architectures are x86_64, ppc64le.
|
||||
* **TARGET_ARCH=<arch>** - cross-compile targeting a specific architecture. Allowed architectures are x86_64, ppc64le, aarch64.
|
||||
By default, TARGET_ARCH is set to HOST_ARCH. On a x86_64 machine, not setting TARGET_ARCH is the equivalent of setting TARGET_ARCH=x86_64.<br/>
|
||||
`$ make TARGET_ARCH=x86_64` <br/> `$ make TARGET_ARCH=ppc64le` <br/>
|
||||
`$ make TARGET_ARCH=x86_64` <br/> `$ make TARGET_ARCH=ppc64le` <br/> `$ make TARGET_ARCH=aarch64` <br/>
|
||||
See [here](http://docs.nvidia.com/cuda/cuda-samples/index.html#cross-samples) for more details.
|
||||
* **dbg=1** - build with debug symbols
|
||||
```
|
||||
|
||||
@@ -180,16 +180,16 @@
|
||||
|
||||
using namespace nvcuda;
|
||||
|
||||
__host__ void init_host_matrices(float *a, float *b, float *c) {
|
||||
__host__ void init_host_matrices(half *a, half *b, float *c) {
|
||||
for (int i = 0; i < M_GLOBAL; i++) {
|
||||
for (int j = 0; j < K_GLOBAL; j++) {
|
||||
a[i * K_GLOBAL + j] = static_cast<float>(rand() % 3);
|
||||
a[i * K_GLOBAL + j] = (half)(rand() % 3);
|
||||
}
|
||||
}
|
||||
|
||||
for (int i = 0; i < N_GLOBAL; i++) {
|
||||
for (int j = 0; j < K_GLOBAL; j++) {
|
||||
b[i * K_GLOBAL + j] = static_cast<float>(rand() % 3);
|
||||
b[i * K_GLOBAL + j] = (half)(rand() % 3);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -198,26 +198,6 @@ __host__ void init_host_matrices(float *a, float *b, float *c) {
|
||||
}
|
||||
}
|
||||
|
||||
__global__ void init_device_matrices(const float *A_h, const float *B_h,
|
||||
const float *C_h, half *A, half *B,
|
||||
float *C, float *D) {
|
||||
for (int i = blockDim.x * blockIdx.x + threadIdx.x; i < M_GLOBAL * K_GLOBAL;
|
||||
i += gridDim.x * blockDim.x)
|
||||
A[i] = __float2half(A_h[i]);
|
||||
|
||||
for (int i = blockDim.x * blockIdx.x + threadIdx.x; i < N_GLOBAL * K_GLOBAL;
|
||||
i += gridDim.x * blockDim.x)
|
||||
B[i] = __float2half(B_h[i]);
|
||||
|
||||
for (int i = blockDim.x * blockIdx.x + threadIdx.x; i < M_GLOBAL * N_GLOBAL;
|
||||
i += gridDim.x * blockDim.x)
|
||||
C[i] = C_h[i];
|
||||
|
||||
for (int i = blockDim.x * blockIdx.x + threadIdx.x; i < M_GLOBAL * N_GLOBAL;
|
||||
i += gridDim.x * blockDim.x)
|
||||
D[i] = 0;
|
||||
}
|
||||
|
||||
__global__ void compute_gemm(const half *A, const half *B, const float *C,
|
||||
float *D, float alpha, float beta) {
|
||||
extern __shared__ half shmem[][CHUNK_K * K + SKEW_HALF];
|
||||
@@ -486,7 +466,7 @@ __global__ void simple_wmma_gemm(half *a, half *b, float *c, float *d, int m_ld,
|
||||
}
|
||||
}
|
||||
|
||||
__host__ void matMultiplyOnHost(float *A, float *B, float *C, float alpha,
|
||||
__host__ void matMultiplyOnHost(half *A, half *B, float *C, float alpha,
|
||||
float beta, int numARows, int numAColumns,
|
||||
int numBRows, int numBColumns, int numCRows,
|
||||
int numCColumns) {
|
||||
@@ -495,7 +475,7 @@ __host__ void matMultiplyOnHost(float *A, float *B, float *C, float alpha,
|
||||
float temp = 0.0;
|
||||
|
||||
for (int k = 0; k < numAColumns; k++) {
|
||||
temp += A[i * numAColumns + k] * B[j * numBRows + k];
|
||||
temp += (float)A[i * numAColumns + k] * (float)B[j * numBRows + k];
|
||||
}
|
||||
|
||||
C[i * numCColumns + j] = temp * alpha + beta * C[i * numCColumns + j];
|
||||
@@ -514,7 +494,7 @@ int main(int argc, char **argv) {
|
||||
// Tensor cores require a GPU of Volta (SM7X) architecture or higher.
|
||||
if (deviceProp.major < 7) {
|
||||
printf(
|
||||
"cudaTensorCoreGemm requires requires SM 7.0 or higher to use Tensor "
|
||||
"cudaTensorCoreGemm requires SM 7.0 or higher to use Tensor "
|
||||
"Cores. Exiting...\n");
|
||||
exit(EXIT_WAIVED);
|
||||
}
|
||||
@@ -523,25 +503,20 @@ int main(int argc, char **argv) {
|
||||
printf("N: %d (%d x %d)\n", N_GLOBAL, N, N_TILES);
|
||||
printf("K: %d (%d x %d)\n", K_GLOBAL, K, K_TILES);
|
||||
|
||||
float *A_h = NULL;
|
||||
float *B_h = NULL;
|
||||
half *A_h = NULL;
|
||||
half *B_h = NULL;
|
||||
float *C_h = NULL;
|
||||
#if CPU_DEBUG
|
||||
float *result_hD = NULL;
|
||||
float *result_host = NULL;
|
||||
#endif
|
||||
|
||||
checkCudaErrors(cudaMallocManaged(reinterpret_cast<void **>(&A_h),
|
||||
sizeof(float) * M_GLOBAL * K_GLOBAL));
|
||||
checkCudaErrors(cudaMallocManaged(reinterpret_cast<void **>(&B_h),
|
||||
sizeof(float) * K_GLOBAL * N_GLOBAL));
|
||||
checkCudaErrors(cudaMallocManaged(reinterpret_cast<void **>(&C_h),
|
||||
sizeof(float) * M_GLOBAL * N_GLOBAL));
|
||||
A_h = (half *)malloc(sizeof(half) * M_GLOBAL * K_GLOBAL);
|
||||
B_h = (half *)malloc(sizeof(half) * K_GLOBAL * N_GLOBAL);
|
||||
C_h = (float *)malloc(sizeof(float) * M_GLOBAL * N_GLOBAL);
|
||||
#if CPU_DEBUG
|
||||
checkCudaErrors(cudaMallocManaged((void **)&result_hD,
|
||||
sizeof(float) * M_GLOBAL * N_GLOBAL));
|
||||
checkCudaErrors(cudaMallocManaged((void **)&result_host,
|
||||
sizeof(float) * M_GLOBAL * N_GLOBAL));
|
||||
result_hD = (float *)malloc(sizeof(float) * M_GLOBAL * N_GLOBAL);
|
||||
result_host = (float *)malloc(sizeof(float) * M_GLOBAL * N_GLOBAL);
|
||||
#endif
|
||||
|
||||
half *A = NULL;
|
||||
@@ -567,11 +542,13 @@ int main(int argc, char **argv) {
|
||||
|
||||
printf("Preparing data for GPU...\n");
|
||||
|
||||
checkKernelErrors(
|
||||
(init_device_matrices<<<deviceProp.multiProcessorCount,
|
||||
THREADS_PER_BLOCK>>>(A_h, B_h, C_h, A, B, C, D)));
|
||||
|
||||
checkCudaErrors(cudaDeviceSynchronize());
|
||||
checkCudaErrors(cudaMemcpy(A, A_h, sizeof(half) * M_GLOBAL * K_GLOBAL,
|
||||
cudaMemcpyHostToDevice));
|
||||
checkCudaErrors(cudaMemcpy(B, B_h, sizeof(half) * N_GLOBAL * K_GLOBAL,
|
||||
cudaMemcpyHostToDevice));
|
||||
checkCudaErrors(cudaMemcpy(C, C_h, sizeof(float) * M_GLOBAL * N_GLOBAL,
|
||||
cudaMemcpyHostToDevice));
|
||||
checkCudaErrors(cudaMemset(D, 0, sizeof(float) * M_GLOBAL * N_GLOBAL));
|
||||
|
||||
enum {
|
||||
// Compute the right amount of shared memory to request.
|
||||
@@ -650,6 +627,8 @@ int main(int argc, char **argv) {
|
||||
printf("mismatch i=%d result_hD=%f result_host=%f\n", i, result_hD[i],
|
||||
result_host[i]);
|
||||
}
|
||||
free(result_hD);
|
||||
free(result_host);
|
||||
#endif
|
||||
|
||||
float milliseconds = 0;
|
||||
@@ -662,9 +641,9 @@ int main(int argc, char **argv) {
|
||||
(milliseconds / 1000.)) /
|
||||
1e12);
|
||||
|
||||
checkCudaErrors(cudaFree(reinterpret_cast<void *>(A_h)));
|
||||
checkCudaErrors(cudaFree(reinterpret_cast<void *>(B_h)));
|
||||
checkCudaErrors(cudaFree(reinterpret_cast<void *>(C_h)));
|
||||
free(A_h);
|
||||
free(B_h);
|
||||
free(C_h);
|
||||
checkCudaErrors(cudaFree(reinterpret_cast<void *>(A)));
|
||||
checkCudaErrors(cudaFree(reinterpret_cast<void *>(B)));
|
||||
checkCudaErrors(cudaFree(reinterpret_cast<void *>(C)));
|
||||
|
||||
@@ -33,7 +33,7 @@
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.props" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
@@ -102,6 +102,6 @@
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.targets" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
|
||||
@@ -33,7 +33,7 @@
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.props" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
@@ -102,6 +102,6 @@
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.targets" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
|
||||
@@ -33,7 +33,7 @@
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.props" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
@@ -102,6 +102,6 @@
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.targets" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.props" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
@@ -103,6 +103,6 @@
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.targets" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
|
||||
Reference in New Issue
Block a user