mirror of
https://github.com/NVIDIA/cuda-samples.git
synced 2026-10-11 23:38:25 +08:00
Add and Update samples for CUDA 10.0
This commit is contained in:
@@ -1,31 +1,29 @@
|
||||
################################################################################
|
||||
# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved.
|
||||
#
|
||||
# Copyright 1993-2015 NVIDIA Corporation. All rights reserved.
|
||||
# Redistribution and use in source and binary forms, with or without
|
||||
# modification, are permitted provided that the following conditions
|
||||
# are met:
|
||||
# * Redistributions of source code must retain the above copyright
|
||||
# notice, this list of conditions and the following disclaimer.
|
||||
# * Redistributions in binary form must reproduce the above copyright
|
||||
# notice, this list of conditions and the following disclaimer in the
|
||||
# documentation and/or other materials provided with the distribution.
|
||||
# * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
# contributors may be used to endorse or promote products derived
|
||||
# from this software without specific prior written permission.
|
||||
#
|
||||
# NOTICE TO USER:
|
||||
#
|
||||
# This source code is subject to NVIDIA ownership rights under U.S. and
|
||||
# international Copyright laws.
|
||||
#
|
||||
# NVIDIA MAKES NO REPRESENTATION ABOUT THE SUITABILITY OF THIS SOURCE
|
||||
# CODE FOR ANY PURPOSE. IT IS PROVIDED "AS IS" WITHOUT EXPRESS OR
|
||||
# IMPLIED WARRANTY OF ANY KIND. NVIDIA DISCLAIMS ALL WARRANTIES WITH
|
||||
# REGARD TO THIS SOURCE CODE, INCLUDING ALL IMPLIED WARRANTIES OF
|
||||
# MERCHANTABILITY, NONINFRINGEMENT, AND FITNESS FOR A PARTICULAR PURPOSE.
|
||||
# IN NO EVENT SHALL NVIDIA BE LIABLE FOR ANY SPECIAL, INDIRECT, INCIDENTAL,
|
||||
# OR CONSEQUENTIAL DAMAGES, OR ANY DAMAGES WHATSOEVER RESULTING FROM LOSS
|
||||
# OF USE, DATA OR PROFITS, WHETHER IN AN ACTION OF CONTRACT, NEGLIGENCE
|
||||
# OR OTHER TORTIOUS ACTION, ARISING OUT OF OR IN CONNECTION WITH THE USE
|
||||
# OR PERFORMANCE OF THIS SOURCE CODE.
|
||||
#
|
||||
# U.S. Government End Users. This source code is a "commercial item" as
|
||||
# that term is defined at 48 C.F.R. 2.101 (OCT 1995), consisting of
|
||||
# "commercial computer software" and "commercial computer software
|
||||
# documentation" as such terms are used in 48 C.F.R. 12.212 (SEPT 1995)
|
||||
# and is provided to the U.S. Government only as a commercial end item.
|
||||
# Consistent with 48 C.F.R.12.212 and 48 C.F.R. 227.7202-1 through
|
||||
# 227.7202-4 (JUNE 1995), all U.S. Government End Users acquire the
|
||||
# source code with only those rights set forth herein.
|
||||
# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
# EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
# PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
# CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
# EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
# PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
# PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
# OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#
|
||||
################################################################################
|
||||
#
|
||||
@@ -141,7 +139,7 @@ else ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
export QNX_TARGET
|
||||
HOST_COMPILER ?= $(QNX_HOST)/usr/bin/aarch64-unknown-nto-qnx7.0.0-g++
|
||||
else ifeq ($(TARGET_OS), android)
|
||||
HOST_COMPILER ?= aarch64-linux-android-g++
|
||||
HOST_COMPILER ?= aarch64-linux-android-clang++
|
||||
endif
|
||||
else ifeq ($(TARGET_ARCH),ppc64le)
|
||||
HOST_COMPILER ?= powerpc64le-linux-gnu-g++
|
||||
@@ -266,7 +264,7 @@ LIBRARIES :=
|
||||
################################################################################
|
||||
|
||||
# Gencode arguments
|
||||
SMS ?= 70
|
||||
SMS ?= 70 75
|
||||
|
||||
ifeq ($(SMS),)
|
||||
$(info >>> WARNING - no SM architectures have been specified - waiving sample <<<)
|
||||
|
||||
@@ -43,6 +43,7 @@ In addition to that, it demonstrates the use of the new CUDA function attribute
|
||||
<scope>1:CUDA Basic Topics</scope>
|
||||
</scopes>
|
||||
<sm-arch>sm70</sm-arch>
|
||||
<sm-arch>sm75</sm-arch>
|
||||
<supported_envs>
|
||||
<env>
|
||||
<arch>x86_64</arch>
|
||||
|
||||
@@ -14,7 +14,7 @@ Matrix Multiply, WMMA, Tensor Cores
|
||||
|
||||
## Supported SM Architectures
|
||||
|
||||
[SM 7.0 ](https://developer.nvidia.com/cuda-gpus)
|
||||
[SM 7.0 ](https://developer.nvidia.com/cuda-gpus) [SM 7.5 ](https://developer.nvidia.com/cuda-gpus)
|
||||
|
||||
## Supported OSes
|
||||
|
||||
@@ -31,7 +31,7 @@ cudaMallocManaged, cudaDeviceSynchronize, cudaFuncSetAttribute, cudaEventCreate,
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Download and install the [CUDA Toolkit 9.2](https://developer.nvidia.com/cuda-downloads) for your corresponding platform.
|
||||
Download and install the [CUDA Toolkit 10.0](https://developer.nvidia.com/cuda-downloads) for your corresponding platform.
|
||||
|
||||
## Build and Run
|
||||
|
||||
|
||||
@@ -72,6 +72,21 @@
|
||||
#include <helper_cuda.h>
|
||||
#include <helper_functions.h>
|
||||
|
||||
// Externally configurable parameters.
|
||||
|
||||
#ifndef CPU_DEBUG
|
||||
// Set this to 1 to verify the correctness of the GPU-computed matrix.
|
||||
#define CPU_DEBUG 0
|
||||
#endif
|
||||
|
||||
#ifndef SHARED_MEMORY_LIMIT_64K
|
||||
// Set this to 0 to use more than 64 Kb of shared memory to cache data, to
|
||||
// improve the performance of the computations on GPU.
|
||||
// Note that you need a GPU that can have more than 64 Kb of shared memory
|
||||
// per multiprocessor.
|
||||
#define SHARED_MEMORY_LIMIT_64K 1
|
||||
#endif
|
||||
|
||||
// GPU configuration.
|
||||
|
||||
#define WARP_SIZE 32
|
||||
@@ -82,6 +97,10 @@
|
||||
#define N 16
|
||||
#define K 16
|
||||
|
||||
#define WMMA_M 16
|
||||
#define WMMA_N 16
|
||||
#define WMMA_K 16
|
||||
|
||||
// GEMM configuration.
|
||||
|
||||
#define M_TILES 256
|
||||
@@ -99,7 +118,24 @@
|
||||
#define WARPS_PER_BLOCK 8
|
||||
#define THREADS_PER_BLOCK (WARP_SIZE * WARPS_PER_BLOCK)
|
||||
|
||||
#if SHARED_MEMORY_LIMIT_64K
|
||||
// With only 64 Kb shared memory available, we can fit two 8-tile chunks of
|
||||
// the A and B matrix data, that are 16 * 16 * 8 * 8 * 2 = 32 Kb each
|
||||
// (i.e. two 8x8 arrays of tiles of 16x16 half-typed elements per CTA).
|
||||
// But we cannot account the 8 Kb total skew overhead, without which the
|
||||
// performance would be severely impacted. So we choose to reduce the chunk size
|
||||
// in half, i.e. the amount of A and B matrix data we cache in shared memory.
|
||||
// Accordingly, this doubles the number of outer iterations across the global K
|
||||
// dimension, which only slightly impacts the performance.
|
||||
#define CHUNK_K 4
|
||||
#else
|
||||
#define CHUNK_K 8
|
||||
#endif
|
||||
|
||||
#define CHUNK_LINE_BYTES (CHUNK_K * K * sizeof(half))
|
||||
#define WARP_COPY_BYTES (WARP_SIZE * sizeof(int4))
|
||||
#define CHUNK_COPY_LINES_PER_WARP (WARP_COPY_BYTES / CHUNK_LINE_BYTES)
|
||||
#define CHUNK_COPY_LINE_LANES (WARP_SIZE / CHUNK_COPY_LINES_PER_WARP)
|
||||
|
||||
#define BLOCK_ROW_WARPS 2
|
||||
#define BLOCK_COL_WARPS 4
|
||||
@@ -194,14 +230,14 @@ __global__ void compute_gemm(const half *A, const half *B, const float *C,
|
||||
const size_t shmem_idx_b_off = BLOCK_COL_TILES * M;
|
||||
|
||||
// This pointer is used to access the C and D matrix tiles this warp computes.
|
||||
float *shmem_warp_tile_ptr = reinterpret_cast<float *>(
|
||||
&shmem[0][0] + (warpId / 2) * SHMEM_STRIDE * K * 2 +
|
||||
(warpId % 2) * SHMEM_OFFSET);
|
||||
float *shmem_warp_tile_ptr = (float *)&shmem[0][0] +
|
||||
(warpId / 2) * SHMEM_STRIDE * K * 2 +
|
||||
(warpId % 2) * SHMEM_OFFSET;
|
||||
|
||||
// This pointer is used to stream the C and D matrices block-wide tile to and
|
||||
// from shared memory.
|
||||
float *shmem_warp_stream_ptr =
|
||||
reinterpret_cast<float *>(&shmem[0][0] + warpId * SHMEM_STRIDE * K);
|
||||
(float *)&shmem[0][0] + warpId * SHMEM_STRIDE * K;
|
||||
|
||||
// Adjust the beta scaler, as it'll be multiplied by alpha at the end of
|
||||
// each tile computation. Technically this is not generally correct (may
|
||||
@@ -292,23 +328,24 @@ __global__ void compute_gemm(const half *A, const half *B, const float *C,
|
||||
// First half of the warp copies the first row / column of the matrix,
|
||||
// the second half of the warp copies the next.
|
||||
int4 *lane_ptr = (int4 *)(warp_ptr + tile_k * K +
|
||||
(laneId / (WARP_SIZE / 2)) * K_GLOBAL) +
|
||||
(laneId % (WARP_SIZE / 2));
|
||||
(laneId / CHUNK_COPY_LINE_LANES) * K_GLOBAL) +
|
||||
(laneId % CHUNK_COPY_LINE_LANES);
|
||||
|
||||
// Shift the second half of the warp to the next row / column in the
|
||||
// shared memory.
|
||||
shmem_idx += laneId / (WARP_SIZE / 2);
|
||||
shmem_idx += laneId / CHUNK_COPY_LINE_LANES;
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < (WARP_SIZE / 2); i++) {
|
||||
for (int i = 0; i < ((WARP_SIZE / 2) / CHUNK_COPY_LINES_PER_WARP) * 2;
|
||||
i++) {
|
||||
// Copy 16 bytes at once in each lane.
|
||||
*((int4 *)&shmem[shmem_idx][0] + (laneId % (WARP_SIZE / 2))) =
|
||||
*((int4 *)&shmem[shmem_idx][0] + (laneId % CHUNK_COPY_LINE_LANES)) =
|
||||
*lane_ptr;
|
||||
|
||||
// Advance the global memory pointer and the shared memory index.
|
||||
lane_ptr = reinterpret_cast<int4 *>(
|
||||
reinterpret_cast<half *>(lane_ptr + K_GLOBAL * 2));
|
||||
shmem_idx += 2;
|
||||
lane_ptr =
|
||||
(int4 *)((half *)lane_ptr + K_GLOBAL * CHUNK_COPY_LINES_PER_WARP);
|
||||
shmem_idx += CHUNK_COPY_LINES_PER_WARP;
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
@@ -374,17 +411,98 @@ __global__ void compute_gemm(const half *A, const half *B, const float *C,
|
||||
|
||||
#pragma unroll
|
||||
for (int i = 0; i < K; i++) {
|
||||
*(reinterpret_cast<int4 *>(dst_gmem_warp_stream_ptr +
|
||||
GLOBAL_MEM_STRIDE * i) +
|
||||
laneId) =
|
||||
*(reinterpret_cast<int4 *>(shmem_warp_stream_ptr + SHMEM_STRIDE * i) +
|
||||
laneId);
|
||||
*((int4 *)(dst_gmem_warp_stream_ptr + GLOBAL_MEM_STRIDE * i) + laneId) =
|
||||
*((int4 *)(shmem_warp_stream_ptr + SHMEM_STRIDE * i) + laneId);
|
||||
}
|
||||
|
||||
__syncthreads();
|
||||
}
|
||||
}
|
||||
|
||||
// Performs an MxNxK GEMM (C=alpha*A*B + beta*C) assuming:
|
||||
// 1) Matrices are packed in memory.
|
||||
// 2) M, N and K are multiples of 16.
|
||||
// 3) Neither A nor B are transposed.
|
||||
// Note: This is a less performant version of the compute_gemm kernel. It is
|
||||
// designed for
|
||||
// demonstration purposes only to show the CUDA WMMA API use without
|
||||
// relying on availability of the shared memory.
|
||||
__global__ void simple_wmma_gemm(half *a, half *b, float *c, float *d, int m_ld,
|
||||
int n_ld, int k_ld, float alpha, float beta) {
|
||||
// Leading dimensions. Packed with no transpositions.
|
||||
int lda = m_ld;
|
||||
int ldb = k_ld;
|
||||
int ldc = n_ld;
|
||||
|
||||
// Tile using a 2D grid
|
||||
int warpM = (blockIdx.x * blockDim.x + threadIdx.x) / warpSize;
|
||||
int warpN = (blockIdx.y * blockDim.y + threadIdx.y);
|
||||
|
||||
// Declare the fragments
|
||||
wmma::fragment<wmma::matrix_a, WMMA_M, WMMA_N, WMMA_K, half, wmma::row_major>
|
||||
a_frag;
|
||||
wmma::fragment<wmma::matrix_b, WMMA_M, WMMA_N, WMMA_K, half, wmma::col_major>
|
||||
b_frag;
|
||||
wmma::fragment<wmma::accumulator, WMMA_M, WMMA_N, WMMA_K, float> acc_frag;
|
||||
wmma::fragment<wmma::accumulator, WMMA_M, WMMA_N, WMMA_K, float> c_frag;
|
||||
|
||||
wmma::fill_fragment(acc_frag, 0.0f);
|
||||
|
||||
// Loop over k
|
||||
for (int i = 0; i < k_ld; i += WMMA_K) {
|
||||
int aCol = i;
|
||||
int aRow = warpM * WMMA_M;
|
||||
|
||||
int bCol = i;
|
||||
int bRow = warpN * WMMA_N;
|
||||
|
||||
// Bounds checking
|
||||
if (aRow < m_ld && aCol < k_ld && bRow < k_ld && bCol < n_ld) {
|
||||
// Load the inputs
|
||||
wmma::load_matrix_sync(a_frag, a + aCol + aRow * lda, lda);
|
||||
wmma::load_matrix_sync(b_frag, b + bCol + bRow * ldb, ldb);
|
||||
|
||||
// Perform the matrix multiplication
|
||||
wmma::mma_sync(acc_frag, a_frag, b_frag, acc_frag);
|
||||
}
|
||||
}
|
||||
|
||||
// Load in the current value of c, scale it by beta, and add this our result
|
||||
// scaled by alpha
|
||||
int cCol = warpN * WMMA_N;
|
||||
int cRow = warpM * WMMA_M;
|
||||
|
||||
if (cRow < m_ld && cCol < n_ld) {
|
||||
wmma::load_matrix_sync(c_frag, c + cCol + cRow * ldc, ldc,
|
||||
wmma::mem_row_major);
|
||||
|
||||
for (int i = 0; i < c_frag.num_elements; i++) {
|
||||
c_frag.x[i] = alpha * acc_frag.x[i] + beta * c_frag.x[i];
|
||||
}
|
||||
|
||||
// Store the output
|
||||
wmma::store_matrix_sync(d + cCol + cRow * ldc, c_frag, ldc,
|
||||
wmma::mem_row_major);
|
||||
}
|
||||
}
|
||||
|
||||
__host__ void matMultiplyOnHost(float *A, float *B, float *C, float alpha,
|
||||
float beta, int numARows, int numAColumns,
|
||||
int numBRows, int numBColumns, int numCRows,
|
||||
int numCColumns) {
|
||||
for (int i = 0; i < numCRows; i++) {
|
||||
for (int j = 0; j < numCColumns; j++) {
|
||||
float temp = 0.0;
|
||||
|
||||
for (int k = 0; k < numAColumns; k++) {
|
||||
temp += A[i * numAColumns + k] * B[j * numBRows + k];
|
||||
}
|
||||
|
||||
C[i * numCColumns + j] = temp * alpha + beta * C[i * numCColumns + j];
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
printf("Initializing...\n");
|
||||
|
||||
@@ -408,6 +526,10 @@ int main(int argc, char **argv) {
|
||||
float *A_h = NULL;
|
||||
float *B_h = NULL;
|
||||
float *C_h = NULL;
|
||||
#if CPU_DEBUG
|
||||
float *result_hD = NULL;
|
||||
float *result_host = NULL;
|
||||
#endif
|
||||
|
||||
checkCudaErrors(cudaMallocManaged(reinterpret_cast<void **>(&A_h),
|
||||
sizeof(float) * M_GLOBAL * K_GLOBAL));
|
||||
@@ -415,6 +537,12 @@ int main(int argc, char **argv) {
|
||||
sizeof(float) * K_GLOBAL * N_GLOBAL));
|
||||
checkCudaErrors(cudaMallocManaged(reinterpret_cast<void **>(&C_h),
|
||||
sizeof(float) * M_GLOBAL * N_GLOBAL));
|
||||
#if CPU_DEBUG
|
||||
checkCudaErrors(cudaMallocManaged((void **)&result_hD,
|
||||
sizeof(float) * M_GLOBAL * N_GLOBAL));
|
||||
checkCudaErrors(cudaMallocManaged((void **)&result_host,
|
||||
sizeof(float) * M_GLOBAL * N_GLOBAL));
|
||||
#endif
|
||||
|
||||
half *A = NULL;
|
||||
half *B = NULL;
|
||||
@@ -446,16 +574,22 @@ int main(int argc, char **argv) {
|
||||
checkCudaErrors(cudaDeviceSynchronize());
|
||||
|
||||
enum {
|
||||
SHMEM_SZ =
|
||||
sizeof(half) * (BLOCK_COL_TILES * M) * (CHUNK_K * K + SKEW_HALF) * 2
|
||||
// Compute the right amount of shared memory to request.
|
||||
// We need shared memory to hold per-CTA C and D matrix tiles, and to cache
|
||||
// per-CTA chunks
|
||||
// of the A and B matrices. Therefore, the right amount to request is the
|
||||
// maximum of those
|
||||
// two numbers.
|
||||
SHMEM_SZ = MAX(
|
||||
sizeof(half) * (BLOCK_COL_TILES * M) * (CHUNK_K * K + SKEW_HALF) * 2,
|
||||
M * (BLOCK_ROW_WARPS * WARP_ROW_TILES) * N *
|
||||
(BLOCK_COL_WARPS * WARP_COL_TILES) * sizeof(float))
|
||||
};
|
||||
|
||||
printf("Required shared memory size: %lu Kb\n", SHMEM_SZ / 1024UL);
|
||||
|
||||
checkCudaErrors(cudaFuncSetAttribute(
|
||||
compute_gemm, cudaFuncAttributeMaxDynamicSharedMemorySize, SHMEM_SZ));
|
||||
|
||||
printf("Computing...\n");
|
||||
const float alpha = 1.1f;
|
||||
const float beta = 1.2f;
|
||||
|
||||
cudaEvent_t start, stop;
|
||||
|
||||
@@ -463,16 +597,61 @@ int main(int argc, char **argv) {
|
||||
checkCudaErrors(cudaEventCreate(&stop));
|
||||
checkCudaErrors(cudaEventRecord(start));
|
||||
|
||||
const float alpha = 1.1f;
|
||||
const float beta = 1.2f;
|
||||
// If enough shared memory available on the GPU use high performant kernel
|
||||
if (deviceProp.sharedMemPerMultiprocessor >= SHMEM_SZ) {
|
||||
printf("Computing... using high performance kernel compute_gemm \n");
|
||||
|
||||
checkKernelErrors(
|
||||
(compute_gemm<<<deviceProp.multiProcessorCount, THREADS_PER_BLOCK,
|
||||
SHMEM_SZ>>>(A, B, C, D, alpha, beta)));
|
||||
checkCudaErrors(cudaFuncSetAttribute(
|
||||
compute_gemm, cudaFuncAttributeMaxDynamicSharedMemorySize, SHMEM_SZ));
|
||||
checkKernelErrors(
|
||||
(compute_gemm<<<deviceProp.multiProcessorCount, THREADS_PER_BLOCK,
|
||||
SHMEM_SZ>>>(A, B, C, D, alpha, beta)));
|
||||
#if CPU_DEBUG
|
||||
checkCudaErrors(cudaMemcpy(result_hD, D,
|
||||
sizeof(float) * M_GLOBAL * N_GLOBAL,
|
||||
cudaMemcpyDeviceToHost));
|
||||
#endif
|
||||
} else {
|
||||
dim3 gridDim;
|
||||
dim3 blockDim;
|
||||
|
||||
// blockDim.x must be a multple of warpSize
|
||||
// 128x4 means we have 16 warps and a block computes a 64x64 output tile
|
||||
blockDim.x = 128;
|
||||
blockDim.y = 4;
|
||||
|
||||
gridDim.x = (M_GLOBAL + (WMMA_M * blockDim.x / 32 - 1)) /
|
||||
(WMMA_M * blockDim.x / 32);
|
||||
gridDim.y = (N_GLOBAL + WMMA_N * blockDim.y - 1) / (WMMA_N * blockDim.y);
|
||||
|
||||
printf("Computing... using simple_wmma_gemm kernel\n");
|
||||
simple_wmma_gemm<<<gridDim, blockDim>>>(A, B, C, D, M_GLOBAL, N_GLOBAL,
|
||||
K_GLOBAL, alpha, beta);
|
||||
#if CPU_DEBUG
|
||||
checkCudaErrors(cudaMemcpy(result_hD, D,
|
||||
sizeof(float) * M_GLOBAL * N_GLOBAL,
|
||||
cudaMemcpyDeviceToHost));
|
||||
#endif
|
||||
}
|
||||
|
||||
checkCudaErrors(cudaEventRecord(stop));
|
||||
checkCudaErrors(cudaEventSynchronize(stop));
|
||||
|
||||
#if CPU_DEBUG
|
||||
printf("Verifying correctness of the computations...\n");
|
||||
|
||||
memcpy(result_host, C_h, sizeof(float) * M_GLOBAL * N_GLOBAL);
|
||||
|
||||
matMultiplyOnHost(A_h, B_h, result_host, alpha, beta, M_GLOBAL, K_GLOBAL,
|
||||
K_GLOBAL, N_GLOBAL, M_GLOBAL, N_GLOBAL);
|
||||
|
||||
for (int i = 0; i < N_GLOBAL * M_GLOBAL; i++) {
|
||||
if (fabs(result_hD[i] - result_host[i]) > 0.1f)
|
||||
printf("mismatch i=%d result_hD=%f result_host=%f\n", i, result_hD[i],
|
||||
result_host[i]);
|
||||
}
|
||||
#endif
|
||||
|
||||
float milliseconds = 0;
|
||||
|
||||
checkCudaErrors(cudaEventElapsedTime(&milliseconds, start, stop));
|
||||
|
||||
@@ -1,20 +0,0 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 11.00
|
||||
# Visual Studio 2010
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "cudaTensorCoreGemm", "cudaTensorCoreGemm_vs2010.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
@@ -1,106 +0,0 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>cudaTensorCoreGemm_vs2010</RootNamespace>
|
||||
<ProjectName>cudaTensorCoreGemm</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 9.2.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/cudaTensorCoreGemm.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_70,sm_70;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<CudaCompile Include="cudaTensorCoreGemm.cu" />
|
||||
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 9.2.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
@@ -33,7 +33,7 @@
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 9.2.props" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
@@ -62,7 +62,7 @@
|
||||
<OutputFile>$(OutDir)/cudaTensorCoreGemm.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_70,sm_70;</CodeGeneration>
|
||||
<CodeGeneration>compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
@@ -102,6 +102,6 @@
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 9.2.targets" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
|
||||
@@ -33,7 +33,7 @@
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 9.2.props" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
@@ -62,7 +62,7 @@
|
||||
<OutputFile>$(OutDir)/cudaTensorCoreGemm.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_70,sm_70;</CodeGeneration>
|
||||
<CodeGeneration>compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
@@ -102,6 +102,6 @@
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 9.2.targets" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
|
||||
@@ -33,7 +33,7 @@
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 9.2.props" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
@@ -62,7 +62,7 @@
|
||||
<OutputFile>$(OutDir)/cudaTensorCoreGemm.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_70,sm_70;</CodeGeneration>
|
||||
<CodeGeneration>compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
@@ -102,6 +102,6 @@
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 9.2.targets" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
|
||||
@@ -34,7 +34,7 @@
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 9.2.props" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
@@ -63,7 +63,7 @@
|
||||
<OutputFile>$(OutDir)/cudaTensorCoreGemm.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_70,sm_70;</CodeGeneration>
|
||||
<CodeGeneration>compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
@@ -103,6 +103,6 @@
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 9.2.targets" />
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.0.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
|
||||
Reference in New Issue
Block a user