mirror of
https://github.com/NVIDIA/cuda-samples.git
synced 2026-10-11 23:38:25 +08:00
Add and update samples with CUDA 10.1 support
This commit is contained in:
304
Samples/bandwidthTest/Makefile
Normal file
304
Samples/bandwidthTest/Makefile
Normal file
@@ -0,0 +1,304 @@
|
||||
################################################################################
|
||||
# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved.
|
||||
#
|
||||
# Redistribution and use in source and binary forms, with or without
|
||||
# modification, are permitted provided that the following conditions
|
||||
# are met:
|
||||
# * Redistributions of source code must retain the above copyright
|
||||
# notice, this list of conditions and the following disclaimer.
|
||||
# * Redistributions in binary form must reproduce the above copyright
|
||||
# notice, this list of conditions and the following disclaimer in the
|
||||
# documentation and/or other materials provided with the distribution.
|
||||
# * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
# contributors may be used to endorse or promote products derived
|
||||
# from this software without specific prior written permission.
|
||||
#
|
||||
# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
# EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
# PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
# CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
# EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
# PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
# PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
# OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#
|
||||
################################################################################
|
||||
#
|
||||
# Makefile project only supported on Mac OS X and Linux Platforms)
|
||||
#
|
||||
################################################################################
|
||||
|
||||
# Location of the CUDA Toolkit
|
||||
CUDA_PATH ?= /usr/local/cuda
|
||||
|
||||
##############################
|
||||
# start deprecated interface #
|
||||
##############################
|
||||
ifeq ($(x86_64),1)
|
||||
$(info WARNING - x86_64 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=x86_64 instead)
|
||||
TARGET_ARCH ?= x86_64
|
||||
endif
|
||||
ifeq ($(ARMv7),1)
|
||||
$(info WARNING - ARMv7 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=armv7l instead)
|
||||
TARGET_ARCH ?= armv7l
|
||||
endif
|
||||
ifeq ($(aarch64),1)
|
||||
$(info WARNING - aarch64 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=aarch64 instead)
|
||||
TARGET_ARCH ?= aarch64
|
||||
endif
|
||||
ifeq ($(ppc64le),1)
|
||||
$(info WARNING - ppc64le variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=ppc64le instead)
|
||||
TARGET_ARCH ?= ppc64le
|
||||
endif
|
||||
ifneq ($(GCC),)
|
||||
$(info WARNING - GCC variable has been deprecated)
|
||||
$(info WARNING - please use HOST_COMPILER=$(GCC) instead)
|
||||
HOST_COMPILER ?= $(GCC)
|
||||
endif
|
||||
ifneq ($(abi),)
|
||||
$(error ERROR - abi variable has been removed)
|
||||
endif
|
||||
############################
|
||||
# end deprecated interface #
|
||||
############################
|
||||
|
||||
# architecture
|
||||
HOST_ARCH := $(shell uname -m)
|
||||
TARGET_ARCH ?= $(HOST_ARCH)
|
||||
ifneq (,$(filter $(TARGET_ARCH),x86_64 aarch64 ppc64le armv7l))
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifneq (,$(filter $(TARGET_ARCH),x86_64 aarch64 ppc64le))
|
||||
TARGET_SIZE := 64
|
||||
else ifneq (,$(filter $(TARGET_ARCH),armv7l))
|
||||
TARGET_SIZE := 32
|
||||
endif
|
||||
else
|
||||
TARGET_SIZE := $(shell getconf LONG_BIT)
|
||||
endif
|
||||
else
|
||||
$(error ERROR - unsupported value $(TARGET_ARCH) for TARGET_ARCH!)
|
||||
endif
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq (,$(filter $(HOST_ARCH)-$(TARGET_ARCH),aarch64-armv7l x86_64-armv7l x86_64-aarch64 x86_64-ppc64le))
|
||||
$(error ERROR - cross compiling from $(HOST_ARCH) to $(TARGET_ARCH) is not supported!)
|
||||
endif
|
||||
endif
|
||||
|
||||
# When on native aarch64 system with userspace of 32-bit, change TARGET_ARCH to armv7l
|
||||
ifeq ($(HOST_ARCH)-$(TARGET_ARCH)-$(TARGET_SIZE),aarch64-aarch64-32)
|
||||
TARGET_ARCH = armv7l
|
||||
endif
|
||||
|
||||
# operating system
|
||||
HOST_OS := $(shell uname -s 2>/dev/null | tr "[:upper:]" "[:lower:]")
|
||||
TARGET_OS ?= $(HOST_OS)
|
||||
ifeq (,$(filter $(TARGET_OS),linux darwin qnx android))
|
||||
$(error ERROR - unsupported value $(TARGET_OS) for TARGET_OS!)
|
||||
endif
|
||||
|
||||
# host compiler
|
||||
ifeq ($(TARGET_OS),darwin)
|
||||
ifeq ($(shell expr `xcodebuild -version | grep -i xcode | awk '{print $$2}' | cut -d'.' -f1` \>= 5),1)
|
||||
HOST_COMPILER ?= clang++
|
||||
endif
|
||||
else ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq ($(HOST_ARCH)-$(TARGET_ARCH),x86_64-armv7l)
|
||||
ifeq ($(TARGET_OS),linux)
|
||||
HOST_COMPILER ?= arm-linux-gnueabihf-g++
|
||||
else ifeq ($(TARGET_OS),qnx)
|
||||
ifeq ($(QNX_HOST),)
|
||||
$(error ERROR - QNX_HOST must be passed to the QNX host toolchain)
|
||||
endif
|
||||
ifeq ($(QNX_TARGET),)
|
||||
$(error ERROR - QNX_TARGET must be passed to the QNX target toolchain)
|
||||
endif
|
||||
export QNX_HOST
|
||||
export QNX_TARGET
|
||||
HOST_COMPILER ?= $(QNX_HOST)/usr/bin/arm-unknown-nto-qnx6.6.0eabi-g++
|
||||
else ifeq ($(TARGET_OS),android)
|
||||
HOST_COMPILER ?= arm-linux-androideabi-g++
|
||||
endif
|
||||
else ifeq ($(TARGET_ARCH),aarch64)
|
||||
ifeq ($(TARGET_OS), linux)
|
||||
HOST_COMPILER ?= aarch64-linux-gnu-g++
|
||||
else ifeq ($(TARGET_OS),qnx)
|
||||
ifeq ($(QNX_HOST),)
|
||||
$(error ERROR - QNX_HOST must be passed to the QNX host toolchain)
|
||||
endif
|
||||
ifeq ($(QNX_TARGET),)
|
||||
$(error ERROR - QNX_TARGET must be passed to the QNX target toolchain)
|
||||
endif
|
||||
export QNX_HOST
|
||||
export QNX_TARGET
|
||||
HOST_COMPILER ?= $(QNX_HOST)/usr/bin/aarch64-unknown-nto-qnx7.0.0-g++
|
||||
else ifeq ($(TARGET_OS), android)
|
||||
HOST_COMPILER ?= aarch64-linux-android-clang++
|
||||
endif
|
||||
else ifeq ($(TARGET_ARCH),ppc64le)
|
||||
HOST_COMPILER ?= powerpc64le-linux-gnu-g++
|
||||
endif
|
||||
endif
|
||||
HOST_COMPILER ?= g++
|
||||
NVCC := $(CUDA_PATH)/bin/nvcc -ccbin $(HOST_COMPILER)
|
||||
|
||||
# internal flags
|
||||
NVCCFLAGS := -m${TARGET_SIZE}
|
||||
CCFLAGS :=
|
||||
LDFLAGS :=
|
||||
|
||||
# build flags
|
||||
ifeq ($(TARGET_OS),darwin)
|
||||
LDFLAGS += -rpath $(CUDA_PATH)/lib
|
||||
CCFLAGS += -arch $(HOST_ARCH)
|
||||
else ifeq ($(HOST_ARCH)-$(TARGET_ARCH)-$(TARGET_OS),x86_64-armv7l-linux)
|
||||
LDFLAGS += --dynamic-linker=/lib/ld-linux-armhf.so.3
|
||||
CCFLAGS += -mfloat-abi=hard
|
||||
else ifeq ($(TARGET_OS),android)
|
||||
LDFLAGS += -pie
|
||||
CCFLAGS += -fpie -fpic -fexceptions
|
||||
endif
|
||||
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-linux)
|
||||
ifneq ($(TARGET_FS),)
|
||||
GCCVERSIONLTEQ46 := $(shell expr `$(HOST_COMPILER) -dumpversion` \<= 4.6)
|
||||
ifeq ($(GCCVERSIONLTEQ46),1)
|
||||
CCFLAGS += --sysroot=$(TARGET_FS)
|
||||
endif
|
||||
LDFLAGS += --sysroot=$(TARGET_FS)
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib/arm-linux-gnueabihf
|
||||
endif
|
||||
endif
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-linux)
|
||||
ifneq ($(TARGET_FS),)
|
||||
GCCVERSIONLTEQ46 := $(shell expr `$(HOST_COMPILER) -dumpversion` \<= 4.6)
|
||||
ifeq ($(GCCVERSIONLTEQ46),1)
|
||||
CCFLAGS += --sysroot=$(TARGET_FS)
|
||||
endif
|
||||
LDFLAGS += --sysroot=$(TARGET_FS)
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/lib -L $(TARGET_FS)/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib -L $(TARGET_FS)/usr/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib/aarch64-linux-gnu -L $(TARGET_FS)/usr/lib/aarch64-linux-gnu
|
||||
LDFLAGS += --unresolved-symbols=ignore-in-shared-libs
|
||||
CCFLAGS += -isystem=$(TARGET_FS)/usr/include
|
||||
CCFLAGS += -isystem=$(TARGET_FS)/usr/include/aarch64-linux-gnu
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(TARGET_OS),qnx)
|
||||
CCFLAGS += -DWIN_INTERFACE_CUSTOM
|
||||
LDFLAGS += -lsocket
|
||||
endif
|
||||
|
||||
# Install directory of different arch
|
||||
CUDA_INSTALL_TARGET_DIR :=
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-linux)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/armv7-linux-gnueabihf/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-linux)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-linux/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-android)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/armv7-linux-androideabi/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-android)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-linux-androideabi/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-qnx)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/ARMv7-linux-QNX/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-qnx)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-qnx/
|
||||
else ifeq ($(TARGET_ARCH),ppc64le)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/ppc64le-linux/
|
||||
endif
|
||||
|
||||
# Debug build flags
|
||||
ifeq ($(dbg),1)
|
||||
NVCCFLAGS += -g -G
|
||||
BUILD_TYPE := debug
|
||||
else
|
||||
BUILD_TYPE := release
|
||||
endif
|
||||
|
||||
ALL_CCFLAGS :=
|
||||
ALL_CCFLAGS += $(NVCCFLAGS)
|
||||
ALL_CCFLAGS += $(EXTRA_NVCCFLAGS)
|
||||
ALL_CCFLAGS += $(addprefix -Xcompiler ,$(CCFLAGS))
|
||||
ALL_CCFLAGS += $(addprefix -Xcompiler ,$(EXTRA_CCFLAGS))
|
||||
|
||||
SAMPLE_ENABLED := 1
|
||||
|
||||
ALL_LDFLAGS :=
|
||||
ALL_LDFLAGS += $(ALL_CCFLAGS)
|
||||
ALL_LDFLAGS += $(addprefix -Xlinker ,$(LDFLAGS))
|
||||
ALL_LDFLAGS += $(addprefix -Xlinker ,$(EXTRA_LDFLAGS))
|
||||
|
||||
# Common includes and paths for CUDA
|
||||
INCLUDES := -I../../Common
|
||||
LIBRARIES :=
|
||||
|
||||
################################################################################
|
||||
|
||||
# Gencode arguments
|
||||
ifeq ($(TARGET_ARCH),$(filter $(TARGET_ARCH),armv7l aarch64))
|
||||
SMS ?= 30 35 37 50 52 60 61 70 72 75
|
||||
else
|
||||
SMS ?= 30 35 37 50 52 60 61 70 75
|
||||
endif
|
||||
|
||||
ifeq ($(SMS),)
|
||||
$(info >>> WARNING - no SM architectures have been specified - waiving sample <<<)
|
||||
SAMPLE_ENABLED := 0
|
||||
endif
|
||||
|
||||
ifeq ($(GENCODE_FLAGS),)
|
||||
# Generate SASS code for each SM architecture listed in $(SMS)
|
||||
$(foreach sm,$(SMS),$(eval GENCODE_FLAGS += -gencode arch=compute_$(sm),code=sm_$(sm)))
|
||||
|
||||
# Generate PTX code from the highest SM architecture in $(SMS) to guarantee forward-compatibility
|
||||
HIGHEST_SM := $(lastword $(sort $(SMS)))
|
||||
ifneq ($(HIGHEST_SM),)
|
||||
GENCODE_FLAGS += -gencode arch=compute_$(HIGHEST_SM),code=compute_$(HIGHEST_SM)
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(SAMPLE_ENABLED),0)
|
||||
EXEC ?= @echo "[@]"
|
||||
endif
|
||||
|
||||
################################################################################
|
||||
|
||||
# Target rules
|
||||
all: build
|
||||
|
||||
build: bandwidthTest
|
||||
|
||||
check.deps:
|
||||
ifeq ($(SAMPLE_ENABLED),0)
|
||||
@echo "Sample will be waived due to the above missing dependencies"
|
||||
else
|
||||
@echo "Sample is ready - all dependencies have been met"
|
||||
endif
|
||||
|
||||
bandwidthTest.o:bandwidthTest.cu
|
||||
$(EXEC) $(NVCC) $(INCLUDES) $(ALL_CCFLAGS) $(GENCODE_FLAGS) -o $@ -c $<
|
||||
|
||||
bandwidthTest: bandwidthTest.o
|
||||
$(EXEC) $(NVCC) $(ALL_LDFLAGS) $(GENCODE_FLAGS) -o $@ $+ $(LIBRARIES)
|
||||
$(EXEC) mkdir -p ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)
|
||||
$(EXEC) cp $@ ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)
|
||||
|
||||
run: build
|
||||
$(EXEC) ./bandwidthTest
|
||||
|
||||
clean:
|
||||
rm -f bandwidthTest bandwidthTest.o
|
||||
rm -rf ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)/bandwidthTest
|
||||
|
||||
clobber: clean
|
||||
79
Samples/bandwidthTest/NsightEclipse.xml
Normal file
79
Samples/bandwidthTest/NsightEclipse.xml
Normal file
@@ -0,0 +1,79 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE entry SYSTEM "SamplesInfo.dtd">
|
||||
<entry>
|
||||
<name>bandwidthTest</name>
|
||||
<cuda_api_list>
|
||||
<toolkit>cudaSetDevice</toolkit>
|
||||
<toolkit>cudaHostAlloc</toolkit>
|
||||
<toolkit>cudaFree</toolkit>
|
||||
<toolkit>cudaMallocHost</toolkit>
|
||||
<toolkit>cudaFreeHost</toolkit>
|
||||
<toolkit>cudaMemcpy</toolkit>
|
||||
<toolkit>cudaMemcpyAsync</toolkit>
|
||||
<toolkit>cudaEventCreate</toolkit>
|
||||
<toolkit>cudaEventRecord</toolkit>
|
||||
<toolkit>cudaEventDestroy</toolkit>
|
||||
<toolkit>cudaDeviceSynchronize</toolkit>
|
||||
<toolkit>cudaEventElapsedTime</toolkit>
|
||||
</cuda_api_list>
|
||||
<description><![CDATA[This is a simple test program to measure the memcopy bandwidth of the GPU and memcpy bandwidth across PCI-e. This test application is capable of measuring device to device copy bandwidth, host to device copy bandwidth for pageable and page-locked memory, and device to host copy bandwidth for pageable and page-locked memory.]]></description>
|
||||
<devicecompilation>whole</devicecompilation>
|
||||
<includepaths>
|
||||
<path>./</path>
|
||||
<path>../</path>
|
||||
<path>../../common/inc</path>
|
||||
</includepaths>
|
||||
<keyconcepts>
|
||||
<concept level="basic">CUDA Streams and Events</concept>
|
||||
<concept level="basic">Performance Strategies</concept>
|
||||
</keyconcepts>
|
||||
<keywords>
|
||||
<keyword>GPGPU</keyword>
|
||||
<keyword>bandwidth</keyword>
|
||||
</keywords>
|
||||
<libraries>
|
||||
</libraries>
|
||||
<librarypaths>
|
||||
</librarypaths>
|
||||
<nsight_eclipse>true</nsight_eclipse>
|
||||
<primary_file>bandwidthTest.cu</primary_file>
|
||||
<scopes>
|
||||
<scope>1:CUDA Basic Topics</scope>
|
||||
<scope>1:Performance Strategies</scope>
|
||||
</scopes>
|
||||
<sm-arch>sm30</sm-arch>
|
||||
<sm-arch>sm35</sm-arch>
|
||||
<sm-arch>sm37</sm-arch>
|
||||
<sm-arch>sm50</sm-arch>
|
||||
<sm-arch>sm52</sm-arch>
|
||||
<sm-arch>sm60</sm-arch>
|
||||
<sm-arch>sm61</sm-arch>
|
||||
<sm-arch>sm70</sm-arch>
|
||||
<sm-arch>sm72</sm-arch>
|
||||
<sm-arch>sm75</sm-arch>
|
||||
<supported_envs>
|
||||
<env>
|
||||
<arch>x86_64</arch>
|
||||
<platform>linux</platform>
|
||||
</env>
|
||||
<env>
|
||||
<platform>windows7</platform>
|
||||
</env>
|
||||
<env>
|
||||
<arch>x86_64</arch>
|
||||
<platform>macosx</platform>
|
||||
</env>
|
||||
<env>
|
||||
<arch>arm</arch>
|
||||
</env>
|
||||
<env>
|
||||
<arch>ppc64le</arch>
|
||||
<platform>linux</platform>
|
||||
</env>
|
||||
</supported_envs>
|
||||
<supported_sm_architectures>
|
||||
<include>all</include>
|
||||
</supported_sm_architectures>
|
||||
<title>Bandwidth Test</title>
|
||||
<type>exe</type>
|
||||
</entry>
|
||||
94
Samples/bandwidthTest/README.md
Normal file
94
Samples/bandwidthTest/README.md
Normal file
@@ -0,0 +1,94 @@
|
||||
# bandwidthTest - Bandwidth Test
|
||||
|
||||
## Description
|
||||
|
||||
This is a simple test program to measure the memcopy bandwidth of the GPU and memcpy bandwidth across PCI-e. This test application is capable of measuring device to device copy bandwidth, host to device copy bandwidth for pageable and page-locked memory, and device to host copy bandwidth for pageable and page-locked memory.
|
||||
|
||||
## Key Concepts
|
||||
|
||||
CUDA Streams and Events, Performance Strategies
|
||||
|
||||
## Supported SM Architectures
|
||||
|
||||
[SM 3.0 ](https://developer.nvidia.com/cuda-gpus) [SM 3.5 ](https://developer.nvidia.com/cuda-gpus) [SM 3.7 ](https://developer.nvidia.com/cuda-gpus) [SM 5.0 ](https://developer.nvidia.com/cuda-gpus) [SM 5.2 ](https://developer.nvidia.com/cuda-gpus) [SM 6.0 ](https://developer.nvidia.com/cuda-gpus) [SM 6.1 ](https://developer.nvidia.com/cuda-gpus) [SM 7.0 ](https://developer.nvidia.com/cuda-gpus) [SM 7.2 ](https://developer.nvidia.com/cuda-gpus) [SM 7.5 ](https://developer.nvidia.com/cuda-gpus)
|
||||
|
||||
## Supported OSes
|
||||
|
||||
Linux, Windows, MacOSX
|
||||
|
||||
## Supported CPU Architecture
|
||||
|
||||
x86_64, ppc64le, armv7l
|
||||
|
||||
## CUDA APIs involved
|
||||
|
||||
### [CUDA Runtime API](http://docs.nvidia.com/cuda/cuda-runtime-api/index.html)
|
||||
cudaSetDevice, cudaHostAlloc, cudaFree, cudaMallocHost, cudaFreeHost, cudaMemcpy, cudaMemcpyAsync, cudaEventCreate, cudaEventRecord, cudaEventDestroy, cudaDeviceSynchronize, cudaEventElapsedTime
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Download and install the [CUDA Toolkit 10.1](https://developer.nvidia.com/cuda-downloads) for your corresponding platform.
|
||||
|
||||
## Build and Run
|
||||
|
||||
### Windows
|
||||
The Windows samples are built using the Visual Studio IDE. Solution files (.sln) are provided for each supported version of Visual Studio, using the format:
|
||||
```
|
||||
*_vs<version>.sln - for Visual Studio <version>
|
||||
```
|
||||
Each individual sample has its own set of solution files in its directory:
|
||||
|
||||
To build/examine all the samples at once, the complete solution files should be used. To build/examine a single sample, the individual sample solution files should be used.
|
||||
> **Note:** Some samples require that the Microsoft DirectX SDK (June 2010 or newer) be installed and that the VC++ directory paths are properly set up (**Tools > Options...**). Check DirectX Dependencies section for details."
|
||||
|
||||
### Linux
|
||||
The Linux samples are built using makefiles. To use the makefiles, change the current directory to the sample directory you wish to build, and run make:
|
||||
```
|
||||
$ cd <sample_dir>
|
||||
$ make
|
||||
```
|
||||
The samples makefiles can take advantage of certain options:
|
||||
* **TARGET_ARCH=<arch>** - cross-compile targeting a specific architecture. Allowed architectures are x86_64, ppc64le, armv7l.
|
||||
By default, TARGET_ARCH is set to HOST_ARCH. On a x86_64 machine, not setting TARGET_ARCH is the equivalent of setting TARGET_ARCH=x86_64.<br/>
|
||||
`$ make TARGET_ARCH=x86_64` <br/> `$ make TARGET_ARCH=ppc64le` <br/> `$ make TARGET_ARCH=armv7l` <br/>
|
||||
See [here](http://docs.nvidia.com/cuda/cuda-samples/index.html#cross-samples) for more details.
|
||||
* **dbg=1** - build with debug symbols
|
||||
```
|
||||
$ make dbg=1
|
||||
```
|
||||
* **SMS="A B ..."** - override the SM architectures for which the sample will be built, where `"A B ..."` is a space-delimited list of SM architectures. For example, to generate SASS for SM 50 and SM 60, use `SMS="50 60"`.
|
||||
```
|
||||
$ make SMS="50 60"
|
||||
```
|
||||
|
||||
* **HOST_COMPILER=<host_compiler>** - override the default g++ host compiler. See the [Linux Installation Guide](http://docs.nvidia.com/cuda/cuda-installation-guide-linux/index.html#system-requirements) for a list of supported host compilers.
|
||||
```
|
||||
$ make HOST_COMPILER=g++
|
||||
```
|
||||
|
||||
### Mac
|
||||
The Mac samples are built using makefiles. To use the makefiles, change directory into the sample directory you wish to build, and run make:
|
||||
```
|
||||
$ cd <sample_dir>
|
||||
$ make
|
||||
```
|
||||
|
||||
The samples makefiles can take advantage of certain options:
|
||||
|
||||
* **dbg=1** - build with debug symbols
|
||||
```
|
||||
$ make dbg=1
|
||||
```
|
||||
|
||||
* **SMS="A B ..."** - override the SM architectures for which the sample will be built, where "A B ..." is a space-delimited list of SM architectures. For example, to generate SASS for SM 50 and SM 60, use SMS="50 60".
|
||||
```
|
||||
$ make SMS="A B ..."
|
||||
```
|
||||
|
||||
* **HOST_COMPILER=<host_compiler>** - override the default clang host compiler. See the [Mac Installation Guide](http://docs.nvidia.com/cuda/cuda-installation-guide-mac-os-x/index.html#system-requirements) for a list of supported host compilers.
|
||||
```
|
||||
$ make HOST_COMPILER=clang
|
||||
```
|
||||
|
||||
## References (for more details)
|
||||
|
||||
969
Samples/bandwidthTest/bandwidthTest.cu
Normal file
969
Samples/bandwidthTest/bandwidthTest.cu
Normal file
@@ -0,0 +1,969 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
/*
|
||||
* This is a simple test program to measure the memcopy bandwidth of the GPU.
|
||||
* It can measure device to device copy bandwidth, host to device copy bandwidth
|
||||
* for pageable and pinned memory, and device to host copy bandwidth for
|
||||
* pageable and pinned memory.
|
||||
*
|
||||
* Usage:
|
||||
* ./bandwidthTest [option]...
|
||||
*/
|
||||
|
||||
// CUDA runtime
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
// includes
|
||||
#include <helper_cuda.h> // helper functions for CUDA error checking and initialization
|
||||
#include <helper_functions.h> // helper for shared functions common to CUDA Samples
|
||||
|
||||
#include <cuda.h>
|
||||
|
||||
#include <cassert>
|
||||
#include <iostream>
|
||||
#include <memory>
|
||||
|
||||
static const char *sSDKsample = "CUDA Bandwidth Test";
|
||||
|
||||
// defines, project
|
||||
#define MEMCOPY_ITERATIONS 100
|
||||
#define DEFAULT_SIZE (32 * (1e6)) // 32 M
|
||||
#define DEFAULT_INCREMENT (4 * (1e6)) // 4 M
|
||||
#define CACHE_CLEAR_SIZE (16 * (1e6)) // 16 M
|
||||
|
||||
// shmoo mode defines
|
||||
#define SHMOO_MEMSIZE_MAX (64 * (1e6)) // 64 M
|
||||
#define SHMOO_MEMSIZE_START (1e3) // 1 KB
|
||||
#define SHMOO_INCREMENT_1KB (1e3) // 1 KB
|
||||
#define SHMOO_INCREMENT_2KB (2 * 1e3) // 2 KB
|
||||
#define SHMOO_INCREMENT_10KB (10 * (1e3)) // 10KB
|
||||
#define SHMOO_INCREMENT_100KB (100 * (1e3)) // 100 KB
|
||||
#define SHMOO_INCREMENT_1MB (1e6) // 1 MB
|
||||
#define SHMOO_INCREMENT_2MB (2 * 1e6) // 2 MB
|
||||
#define SHMOO_INCREMENT_4MB (4 * 1e6) // 4 MB
|
||||
#define SHMOO_LIMIT_20KB (20 * (1e3)) // 20 KB
|
||||
#define SHMOO_LIMIT_50KB (50 * (1e3)) // 50 KB
|
||||
#define SHMOO_LIMIT_100KB (100 * (1e3)) // 100 KB
|
||||
#define SHMOO_LIMIT_1MB (1e6) // 1 MB
|
||||
#define SHMOO_LIMIT_16MB (16 * 1e6) // 16 MB
|
||||
#define SHMOO_LIMIT_32MB (32 * 1e6) // 32 MB
|
||||
|
||||
// CPU cache flush
|
||||
#define FLUSH_SIZE (256 * 1024 * 1024)
|
||||
char *flush_buf;
|
||||
|
||||
// enums, project
|
||||
enum testMode { QUICK_MODE, RANGE_MODE, SHMOO_MODE };
|
||||
enum memcpyKind { DEVICE_TO_HOST, HOST_TO_DEVICE, DEVICE_TO_DEVICE };
|
||||
enum printMode { USER_READABLE, CSV };
|
||||
enum memoryMode { PINNED, PAGEABLE };
|
||||
|
||||
const char *sMemoryCopyKind[] = {"Device to Host", "Host to Device",
|
||||
"Device to Device", NULL};
|
||||
|
||||
const char *sMemoryMode[] = {"PINNED", "PAGEABLE", NULL};
|
||||
|
||||
// if true, use CPU based timing for everything
|
||||
static bool bDontUseGPUTiming;
|
||||
|
||||
int *pArgc = NULL;
|
||||
char **pArgv = NULL;
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
// declaration, forward
|
||||
int runTest(const int argc, const char **argv);
|
||||
void testBandwidth(unsigned int start, unsigned int end, unsigned int increment,
|
||||
testMode mode, memcpyKind kind, printMode printmode,
|
||||
memoryMode memMode, int startDevice, int endDevice, bool wc);
|
||||
void testBandwidthQuick(unsigned int size, memcpyKind kind, printMode printmode,
|
||||
memoryMode memMode, int startDevice, int endDevice,
|
||||
bool wc);
|
||||
void testBandwidthRange(unsigned int start, unsigned int end,
|
||||
unsigned int increment, memcpyKind kind,
|
||||
printMode printmode, memoryMode memMode,
|
||||
int startDevice, int endDevice, bool wc);
|
||||
void testBandwidthShmoo(memcpyKind kind, printMode printmode,
|
||||
memoryMode memMode, int startDevice, int endDevice,
|
||||
bool wc);
|
||||
float testDeviceToHostTransfer(unsigned int memSize, memoryMode memMode,
|
||||
bool wc);
|
||||
float testHostToDeviceTransfer(unsigned int memSize, memoryMode memMode,
|
||||
bool wc);
|
||||
float testDeviceToDeviceTransfer(unsigned int memSize);
|
||||
void printResultsReadable(unsigned int *memSizes, double *bandwidths,
|
||||
unsigned int count, memcpyKind kind,
|
||||
memoryMode memMode, int iNumDevs, bool wc);
|
||||
void printResultsCSV(unsigned int *memSizes, double *bandwidths,
|
||||
unsigned int count, memcpyKind kind, memoryMode memMode,
|
||||
int iNumDevs, bool wc);
|
||||
void printHelp(void);
|
||||
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
// Program main
|
||||
////////////////////////////////////////////////////////////////////////////////
|
||||
int main(int argc, char **argv) {
|
||||
pArgc = &argc;
|
||||
pArgv = argv;
|
||||
|
||||
flush_buf = (char *)malloc(FLUSH_SIZE);
|
||||
|
||||
// set logfile name and start logs
|
||||
printf("[%s] - Starting...\n", sSDKsample);
|
||||
|
||||
int iRetVal = runTest(argc, (const char **)argv);
|
||||
|
||||
if (iRetVal < 0) {
|
||||
checkCudaErrors(cudaSetDevice(0));
|
||||
}
|
||||
|
||||
// finish
|
||||
printf("%s\n", (iRetVal == 0) ? "Result = PASS" : "Result = FAIL");
|
||||
|
||||
printf(
|
||||
"\nNOTE: The CUDA Samples are not meant for performance measurements. "
|
||||
"Results may vary when GPU Boost is enabled.\n");
|
||||
|
||||
free(flush_buf);
|
||||
|
||||
exit((iRetVal == 0) ? EXIT_SUCCESS : EXIT_FAILURE);
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// Parse args, run the appropriate tests
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
int runTest(const int argc, const char **argv) {
|
||||
int start = DEFAULT_SIZE;
|
||||
int end = DEFAULT_SIZE;
|
||||
int startDevice = 0;
|
||||
int endDevice = 0;
|
||||
int increment = DEFAULT_INCREMENT;
|
||||
testMode mode = QUICK_MODE;
|
||||
bool htod = false;
|
||||
bool dtoh = false;
|
||||
bool dtod = false;
|
||||
bool wc = false;
|
||||
char *modeStr;
|
||||
char *device = NULL;
|
||||
printMode printmode = USER_READABLE;
|
||||
char *memModeStr = NULL;
|
||||
memoryMode memMode = PINNED;
|
||||
|
||||
// process command line args
|
||||
if (checkCmdLineFlag(argc, argv, "help")) {
|
||||
printHelp();
|
||||
return 0;
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, argv, "csv")) {
|
||||
printmode = CSV;
|
||||
}
|
||||
|
||||
if (getCmdLineArgumentString(argc, argv, "memory", &memModeStr)) {
|
||||
if (strcmp(memModeStr, "pageable") == 0) {
|
||||
memMode = PAGEABLE;
|
||||
} else if (strcmp(memModeStr, "pinned") == 0) {
|
||||
memMode = PINNED;
|
||||
} else {
|
||||
printf("Invalid memory mode - valid modes are pageable or pinned\n");
|
||||
printf("See --help for more information\n");
|
||||
return -1000;
|
||||
}
|
||||
} else {
|
||||
// default - pinned memory
|
||||
memMode = PINNED;
|
||||
}
|
||||
|
||||
if (getCmdLineArgumentString(argc, argv, "device", &device)) {
|
||||
int deviceCount;
|
||||
cudaError_t error_id = cudaGetDeviceCount(&deviceCount);
|
||||
|
||||
if (error_id != cudaSuccess) {
|
||||
printf("cudaGetDeviceCount returned %d\n-> %s\n", (int)error_id,
|
||||
cudaGetErrorString(error_id));
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
if (deviceCount == 0) {
|
||||
printf("!!!!!No devices found!!!!!\n");
|
||||
return -2000;
|
||||
}
|
||||
|
||||
if (strcmp(device, "all") == 0) {
|
||||
printf(
|
||||
"\n!!!!!Cumulative Bandwidth to be computed from all the devices "
|
||||
"!!!!!!\n\n");
|
||||
startDevice = 0;
|
||||
endDevice = deviceCount - 1;
|
||||
} else {
|
||||
startDevice = endDevice = atoi(device);
|
||||
|
||||
if (startDevice >= deviceCount || startDevice < 0) {
|
||||
printf(
|
||||
"\n!!!!!Invalid GPU number %d given hence default gpu %d will be "
|
||||
"used !!!!!\n",
|
||||
startDevice, 0);
|
||||
startDevice = endDevice = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
printf("Running on...\n\n");
|
||||
|
||||
for (int currentDevice = startDevice; currentDevice <= endDevice;
|
||||
currentDevice++) {
|
||||
cudaDeviceProp deviceProp;
|
||||
cudaError_t error_id = cudaGetDeviceProperties(&deviceProp, currentDevice);
|
||||
|
||||
if (error_id == cudaSuccess) {
|
||||
printf(" Device %d: %s\n", currentDevice, deviceProp.name);
|
||||
|
||||
if (deviceProp.computeMode == cudaComputeModeProhibited) {
|
||||
fprintf(stderr,
|
||||
"Error: device is running in <Compute Mode Prohibited>, no "
|
||||
"threads can use ::cudaSetDevice().\n");
|
||||
checkCudaErrors(cudaSetDevice(currentDevice));
|
||||
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
} else {
|
||||
printf("cudaGetDeviceProperties returned %d\n-> %s\n", (int)error_id,
|
||||
cudaGetErrorString(error_id));
|
||||
checkCudaErrors(cudaSetDevice(currentDevice));
|
||||
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
}
|
||||
|
||||
if (getCmdLineArgumentString(argc, argv, "mode", &modeStr)) {
|
||||
// figure out the mode
|
||||
if (strcmp(modeStr, "quick") == 0) {
|
||||
printf(" Quick Mode\n\n");
|
||||
mode = QUICK_MODE;
|
||||
} else if (strcmp(modeStr, "shmoo") == 0) {
|
||||
printf(" Shmoo Mode\n\n");
|
||||
mode = SHMOO_MODE;
|
||||
} else if (strcmp(modeStr, "range") == 0) {
|
||||
printf(" Range Mode\n\n");
|
||||
mode = RANGE_MODE;
|
||||
} else {
|
||||
printf("Invalid mode - valid modes are quick, range, or shmoo\n");
|
||||
printf("See --help for more information\n");
|
||||
return -3000;
|
||||
}
|
||||
} else {
|
||||
// default mode - quick
|
||||
printf(" Quick Mode\n\n");
|
||||
mode = QUICK_MODE;
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, argv, "htod")) {
|
||||
htod = true;
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, argv, "dtoh")) {
|
||||
dtoh = true;
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, argv, "dtod")) {
|
||||
dtod = true;
|
||||
}
|
||||
|
||||
#if CUDART_VERSION >= 2020
|
||||
|
||||
if (checkCmdLineFlag(argc, argv, "wc")) {
|
||||
wc = true;
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
if (checkCmdLineFlag(argc, argv, "cputiming")) {
|
||||
bDontUseGPUTiming = true;
|
||||
}
|
||||
|
||||
if (!htod && !dtoh && !dtod) {
|
||||
// default: All
|
||||
htod = true;
|
||||
dtoh = true;
|
||||
dtod = true;
|
||||
}
|
||||
|
||||
if (RANGE_MODE == mode) {
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "start")) {
|
||||
start = getCmdLineArgumentInt(argc, argv, "start");
|
||||
|
||||
if (start <= 0) {
|
||||
printf("Illegal argument - start must be greater than zero\n");
|
||||
return -4000;
|
||||
}
|
||||
} else {
|
||||
printf("Must specify a starting size in range mode\n");
|
||||
printf("See --help for more information\n");
|
||||
return -5000;
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "end")) {
|
||||
end = getCmdLineArgumentInt(argc, argv, "end");
|
||||
|
||||
if (end <= 0) {
|
||||
printf("Illegal argument - end must be greater than zero\n");
|
||||
return -6000;
|
||||
}
|
||||
|
||||
if (start > end) {
|
||||
printf("Illegal argument - start is greater than end\n");
|
||||
return -7000;
|
||||
}
|
||||
} else {
|
||||
printf("Must specify an end size in range mode.\n");
|
||||
printf("See --help for more information\n");
|
||||
return -8000;
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, argv, "increment")) {
|
||||
increment = getCmdLineArgumentInt(argc, argv, "increment");
|
||||
|
||||
if (increment <= 0) {
|
||||
printf("Illegal argument - increment must be greater than zero\n");
|
||||
return -9000;
|
||||
}
|
||||
} else {
|
||||
printf("Must specify an increment in user mode\n");
|
||||
printf("See --help for more information\n");
|
||||
return -10000;
|
||||
}
|
||||
}
|
||||
|
||||
if (htod) {
|
||||
testBandwidth((unsigned int)start, (unsigned int)end,
|
||||
(unsigned int)increment, mode, HOST_TO_DEVICE, printmode,
|
||||
memMode, startDevice, endDevice, wc);
|
||||
}
|
||||
|
||||
if (dtoh) {
|
||||
testBandwidth((unsigned int)start, (unsigned int)end,
|
||||
(unsigned int)increment, mode, DEVICE_TO_HOST, printmode,
|
||||
memMode, startDevice, endDevice, wc);
|
||||
}
|
||||
|
||||
if (dtod) {
|
||||
testBandwidth((unsigned int)start, (unsigned int)end,
|
||||
(unsigned int)increment, mode, DEVICE_TO_DEVICE, printmode,
|
||||
memMode, startDevice, endDevice, wc);
|
||||
}
|
||||
|
||||
// Ensure that we reset all CUDA Devices in question
|
||||
for (int nDevice = startDevice; nDevice <= endDevice; nDevice++) {
|
||||
cudaSetDevice(nDevice);
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// Run a bandwidth test
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
void testBandwidth(unsigned int start, unsigned int end, unsigned int increment,
|
||||
testMode mode, memcpyKind kind, printMode printmode,
|
||||
memoryMode memMode, int startDevice, int endDevice,
|
||||
bool wc) {
|
||||
switch (mode) {
|
||||
case QUICK_MODE:
|
||||
testBandwidthQuick(DEFAULT_SIZE, kind, printmode, memMode, startDevice,
|
||||
endDevice, wc);
|
||||
break;
|
||||
|
||||
case RANGE_MODE:
|
||||
testBandwidthRange(start, end, increment, kind, printmode, memMode,
|
||||
startDevice, endDevice, wc);
|
||||
break;
|
||||
|
||||
case SHMOO_MODE:
|
||||
testBandwidthShmoo(kind, printmode, memMode, startDevice, endDevice, wc);
|
||||
break;
|
||||
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////////////////////
|
||||
// Run a quick mode bandwidth test
|
||||
//////////////////////////////////////////////////////////////////////
|
||||
void testBandwidthQuick(unsigned int size, memcpyKind kind, printMode printmode,
|
||||
memoryMode memMode, int startDevice, int endDevice,
|
||||
bool wc) {
|
||||
testBandwidthRange(size, size, DEFAULT_INCREMENT, kind, printmode, memMode,
|
||||
startDevice, endDevice, wc);
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////
|
||||
// Run a range mode bandwidth test
|
||||
//////////////////////////////////////////////////////////////////////
|
||||
void testBandwidthRange(unsigned int start, unsigned int end,
|
||||
unsigned int increment, memcpyKind kind,
|
||||
printMode printmode, memoryMode memMode,
|
||||
int startDevice, int endDevice, bool wc) {
|
||||
// count the number of copies we're going to run
|
||||
unsigned int count = 1 + ((end - start) / increment);
|
||||
|
||||
unsigned int *memSizes = (unsigned int *)malloc(count * sizeof(unsigned int));
|
||||
double *bandwidths = (double *)malloc(count * sizeof(double));
|
||||
|
||||
// Before calculating the cumulative bandwidth, initialize bandwidths array to
|
||||
// NULL
|
||||
for (unsigned int i = 0; i < count; i++) {
|
||||
bandwidths[i] = 0.0;
|
||||
}
|
||||
|
||||
// Use the device asked by the user
|
||||
for (int currentDevice = startDevice; currentDevice <= endDevice;
|
||||
currentDevice++) {
|
||||
cudaSetDevice(currentDevice);
|
||||
|
||||
// run each of the copies
|
||||
for (unsigned int i = 0; i < count; i++) {
|
||||
memSizes[i] = start + i * increment;
|
||||
|
||||
switch (kind) {
|
||||
case DEVICE_TO_HOST:
|
||||
bandwidths[i] += testDeviceToHostTransfer(memSizes[i], memMode, wc);
|
||||
break;
|
||||
|
||||
case HOST_TO_DEVICE:
|
||||
bandwidths[i] += testHostToDeviceTransfer(memSizes[i], memMode, wc);
|
||||
break;
|
||||
|
||||
case DEVICE_TO_DEVICE:
|
||||
bandwidths[i] += testDeviceToDeviceTransfer(memSizes[i]);
|
||||
break;
|
||||
}
|
||||
}
|
||||
} // Complete the bandwidth computation on all the devices
|
||||
|
||||
// print results
|
||||
if (printmode == CSV) {
|
||||
printResultsCSV(memSizes, bandwidths, count, kind, memMode,
|
||||
(1 + endDevice - startDevice), wc);
|
||||
} else {
|
||||
printResultsReadable(memSizes, bandwidths, count, kind, memMode,
|
||||
(1 + endDevice - startDevice), wc);
|
||||
}
|
||||
|
||||
// clean up
|
||||
free(memSizes);
|
||||
free(bandwidths);
|
||||
}
|
||||
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
// Intense shmoo mode - covers a large range of values with varying increments
|
||||
//////////////////////////////////////////////////////////////////////////////
|
||||
void testBandwidthShmoo(memcpyKind kind, printMode printmode,
|
||||
memoryMode memMode, int startDevice, int endDevice,
|
||||
bool wc) {
|
||||
// count the number of copies to make
|
||||
unsigned int count =
|
||||
1 + (SHMOO_LIMIT_20KB / SHMOO_INCREMENT_1KB) +
|
||||
((SHMOO_LIMIT_50KB - SHMOO_LIMIT_20KB) / SHMOO_INCREMENT_2KB) +
|
||||
((SHMOO_LIMIT_100KB - SHMOO_LIMIT_50KB) / SHMOO_INCREMENT_10KB) +
|
||||
((SHMOO_LIMIT_1MB - SHMOO_LIMIT_100KB) / SHMOO_INCREMENT_100KB) +
|
||||
((SHMOO_LIMIT_16MB - SHMOO_LIMIT_1MB) / SHMOO_INCREMENT_1MB) +
|
||||
((SHMOO_LIMIT_32MB - SHMOO_LIMIT_16MB) / SHMOO_INCREMENT_2MB) +
|
||||
((SHMOO_MEMSIZE_MAX - SHMOO_LIMIT_32MB) / SHMOO_INCREMENT_4MB);
|
||||
|
||||
unsigned int *memSizes = (unsigned int *)malloc(count * sizeof(unsigned int));
|
||||
double *bandwidths = (double *)malloc(count * sizeof(double));
|
||||
|
||||
// Before calculating the cumulative bandwidth, initialize bandwidths array to
|
||||
// NULL
|
||||
for (unsigned int i = 0; i < count; i++) {
|
||||
bandwidths[i] = 0.0;
|
||||
}
|
||||
|
||||
// Use the device asked by the user
|
||||
for (int currentDevice = startDevice; currentDevice <= endDevice;
|
||||
currentDevice++) {
|
||||
cudaSetDevice(currentDevice);
|
||||
// Run the shmoo
|
||||
int iteration = 0;
|
||||
unsigned int memSize = 0;
|
||||
|
||||
while (memSize <= SHMOO_MEMSIZE_MAX) {
|
||||
if (memSize < SHMOO_LIMIT_20KB) {
|
||||
memSize += SHMOO_INCREMENT_1KB;
|
||||
} else if (memSize < SHMOO_LIMIT_50KB) {
|
||||
memSize += SHMOO_INCREMENT_2KB;
|
||||
} else if (memSize < SHMOO_LIMIT_100KB) {
|
||||
memSize += SHMOO_INCREMENT_10KB;
|
||||
} else if (memSize < SHMOO_LIMIT_1MB) {
|
||||
memSize += SHMOO_INCREMENT_100KB;
|
||||
} else if (memSize < SHMOO_LIMIT_16MB) {
|
||||
memSize += SHMOO_INCREMENT_1MB;
|
||||
} else if (memSize < SHMOO_LIMIT_32MB) {
|
||||
memSize += SHMOO_INCREMENT_2MB;
|
||||
} else {
|
||||
memSize += SHMOO_INCREMENT_4MB;
|
||||
}
|
||||
|
||||
memSizes[iteration] = memSize;
|
||||
|
||||
switch (kind) {
|
||||
case DEVICE_TO_HOST:
|
||||
bandwidths[iteration] +=
|
||||
testDeviceToHostTransfer(memSizes[iteration], memMode, wc);
|
||||
break;
|
||||
|
||||
case HOST_TO_DEVICE:
|
||||
bandwidths[iteration] +=
|
||||
testHostToDeviceTransfer(memSizes[iteration], memMode, wc);
|
||||
break;
|
||||
|
||||
case DEVICE_TO_DEVICE:
|
||||
bandwidths[iteration] +=
|
||||
testDeviceToDeviceTransfer(memSizes[iteration]);
|
||||
break;
|
||||
}
|
||||
|
||||
iteration++;
|
||||
printf(".");
|
||||
fflush(0);
|
||||
}
|
||||
} // Complete the bandwidth computation on all the devices
|
||||
|
||||
// print results
|
||||
printf("\n");
|
||||
|
||||
if (CSV == printmode) {
|
||||
printResultsCSV(memSizes, bandwidths, count, kind, memMode,
|
||||
(1 + endDevice - startDevice), wc);
|
||||
} else {
|
||||
printResultsReadable(memSizes, bandwidths, count, kind, memMode,
|
||||
(1 + endDevice - startDevice), wc);
|
||||
}
|
||||
|
||||
// clean up
|
||||
free(memSizes);
|
||||
free(bandwidths);
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
// test the bandwidth of a device to host memcopy of a specific size
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
float testDeviceToHostTransfer(unsigned int memSize, memoryMode memMode,
|
||||
bool wc) {
|
||||
StopWatchInterface *timer = NULL;
|
||||
float elapsedTimeInMs = 0.0f;
|
||||
float bandwidthInGBs = 0.0f;
|
||||
unsigned char *h_idata = NULL;
|
||||
unsigned char *h_odata = NULL;
|
||||
cudaEvent_t start, stop;
|
||||
|
||||
sdkCreateTimer(&timer);
|
||||
checkCudaErrors(cudaEventCreate(&start));
|
||||
checkCudaErrors(cudaEventCreate(&stop));
|
||||
|
||||
// allocate host memory
|
||||
if (PINNED == memMode) {
|
||||
// pinned memory mode - use special function to get OS-pinned memory
|
||||
#if CUDART_VERSION >= 2020
|
||||
checkCudaErrors(cudaHostAlloc((void **)&h_idata, memSize,
|
||||
(wc) ? cudaHostAllocWriteCombined : 0));
|
||||
checkCudaErrors(cudaHostAlloc((void **)&h_odata, memSize,
|
||||
(wc) ? cudaHostAllocWriteCombined : 0));
|
||||
#else
|
||||
checkCudaErrors(cudaMallocHost((void **)&h_idata, memSize));
|
||||
checkCudaErrors(cudaMallocHost((void **)&h_odata, memSize));
|
||||
#endif
|
||||
} else {
|
||||
// pageable memory mode - use malloc
|
||||
h_idata = (unsigned char *)malloc(memSize);
|
||||
h_odata = (unsigned char *)malloc(memSize);
|
||||
|
||||
if (h_idata == 0 || h_odata == 0) {
|
||||
fprintf(stderr, "Not enough memory avaialable on host to run test!\n");
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
}
|
||||
|
||||
// initialize the memory
|
||||
for (unsigned int i = 0; i < memSize / sizeof(unsigned char); i++) {
|
||||
h_idata[i] = (unsigned char)(i & 0xff);
|
||||
}
|
||||
|
||||
// allocate device memory
|
||||
unsigned char *d_idata;
|
||||
checkCudaErrors(cudaMalloc((void **)&d_idata, memSize));
|
||||
|
||||
// initialize the device memory
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(d_idata, h_idata, memSize, cudaMemcpyHostToDevice));
|
||||
|
||||
// copy data from GPU to Host
|
||||
if (PINNED == memMode) {
|
||||
if (bDontUseGPUTiming) sdkStartTimer(&timer);
|
||||
checkCudaErrors(cudaEventRecord(start, 0));
|
||||
for (unsigned int i = 0; i < MEMCOPY_ITERATIONS; i++) {
|
||||
checkCudaErrors(cudaMemcpyAsync(h_odata, d_idata, memSize,
|
||||
cudaMemcpyDeviceToHost, 0));
|
||||
}
|
||||
checkCudaErrors(cudaEventRecord(stop, 0));
|
||||
checkCudaErrors(cudaDeviceSynchronize());
|
||||
checkCudaErrors(cudaEventElapsedTime(&elapsedTimeInMs, start, stop));
|
||||
if (bDontUseGPUTiming) {
|
||||
sdkStopTimer(&timer);
|
||||
elapsedTimeInMs = sdkGetTimerValue(&timer);
|
||||
sdkResetTimer(&timer);
|
||||
}
|
||||
} else {
|
||||
elapsedTimeInMs = 0;
|
||||
for (unsigned int i = 0; i < MEMCOPY_ITERATIONS; i++) {
|
||||
sdkStartTimer(&timer);
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(h_odata, d_idata, memSize, cudaMemcpyDeviceToHost));
|
||||
sdkStopTimer(&timer);
|
||||
elapsedTimeInMs += sdkGetTimerValue(&timer);
|
||||
sdkResetTimer(&timer);
|
||||
memset(flush_buf, i, FLUSH_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
// calculate bandwidth in GB/s
|
||||
double time_s = elapsedTimeInMs / 1e3;
|
||||
bandwidthInGBs = (memSize * (float)MEMCOPY_ITERATIONS) / (double)1e9;
|
||||
bandwidthInGBs = bandwidthInGBs / time_s;
|
||||
// clean up memory
|
||||
checkCudaErrors(cudaEventDestroy(stop));
|
||||
checkCudaErrors(cudaEventDestroy(start));
|
||||
sdkDeleteTimer(&timer);
|
||||
|
||||
if (PINNED == memMode) {
|
||||
checkCudaErrors(cudaFreeHost(h_idata));
|
||||
checkCudaErrors(cudaFreeHost(h_odata));
|
||||
} else {
|
||||
free(h_idata);
|
||||
free(h_odata);
|
||||
}
|
||||
|
||||
checkCudaErrors(cudaFree(d_idata));
|
||||
|
||||
return bandwidthInGBs;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
//! test the bandwidth of a host to device memcopy of a specific size
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
float testHostToDeviceTransfer(unsigned int memSize, memoryMode memMode,
|
||||
bool wc) {
|
||||
StopWatchInterface *timer = NULL;
|
||||
float elapsedTimeInMs = 0.0f;
|
||||
float bandwidthInGBs = 0.0f;
|
||||
cudaEvent_t start, stop;
|
||||
sdkCreateTimer(&timer);
|
||||
checkCudaErrors(cudaEventCreate(&start));
|
||||
checkCudaErrors(cudaEventCreate(&stop));
|
||||
|
||||
// allocate host memory
|
||||
unsigned char *h_odata = NULL;
|
||||
|
||||
if (PINNED == memMode) {
|
||||
#if CUDART_VERSION >= 2020
|
||||
// pinned memory mode - use special function to get OS-pinned memory
|
||||
checkCudaErrors(cudaHostAlloc((void **)&h_odata, memSize,
|
||||
(wc) ? cudaHostAllocWriteCombined : 0));
|
||||
#else
|
||||
// pinned memory mode - use special function to get OS-pinned memory
|
||||
checkCudaErrors(cudaMallocHost((void **)&h_odata, memSize));
|
||||
#endif
|
||||
} else {
|
||||
// pageable memory mode - use malloc
|
||||
h_odata = (unsigned char *)malloc(memSize);
|
||||
|
||||
if (h_odata == 0) {
|
||||
fprintf(stderr, "Not enough memory available on host to run test!\n");
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
}
|
||||
|
||||
unsigned char *h_cacheClear1 = (unsigned char *)malloc(CACHE_CLEAR_SIZE);
|
||||
unsigned char *h_cacheClear2 = (unsigned char *)malloc(CACHE_CLEAR_SIZE);
|
||||
|
||||
if (h_cacheClear1 == 0 || h_cacheClear2 == 0) {
|
||||
fprintf(stderr, "Not enough memory available on host to run test!\n");
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
// initialize the memory
|
||||
for (unsigned int i = 0; i < memSize / sizeof(unsigned char); i++) {
|
||||
h_odata[i] = (unsigned char)(i & 0xff);
|
||||
}
|
||||
|
||||
for (unsigned int i = 0; i < CACHE_CLEAR_SIZE / sizeof(unsigned char); i++) {
|
||||
h_cacheClear1[i] = (unsigned char)(i & 0xff);
|
||||
h_cacheClear2[i] = (unsigned char)(0xff - (i & 0xff));
|
||||
}
|
||||
|
||||
// allocate device memory
|
||||
unsigned char *d_idata;
|
||||
checkCudaErrors(cudaMalloc((void **)&d_idata, memSize));
|
||||
|
||||
// copy host memory to device memory
|
||||
if (PINNED == memMode) {
|
||||
if (bDontUseGPUTiming) sdkStartTimer(&timer);
|
||||
checkCudaErrors(cudaEventRecord(start, 0));
|
||||
for (unsigned int i = 0; i < MEMCOPY_ITERATIONS; i++) {
|
||||
checkCudaErrors(cudaMemcpyAsync(d_idata, h_odata, memSize,
|
||||
cudaMemcpyHostToDevice, 0));
|
||||
}
|
||||
checkCudaErrors(cudaEventRecord(stop, 0));
|
||||
checkCudaErrors(cudaDeviceSynchronize());
|
||||
checkCudaErrors(cudaEventElapsedTime(&elapsedTimeInMs, start, stop));
|
||||
if (bDontUseGPUTiming) {
|
||||
sdkStopTimer(&timer);
|
||||
elapsedTimeInMs = sdkGetTimerValue(&timer);
|
||||
sdkResetTimer(&timer);
|
||||
}
|
||||
} else {
|
||||
elapsedTimeInMs = 0;
|
||||
for (unsigned int i = 0; i < MEMCOPY_ITERATIONS; i++) {
|
||||
sdkStartTimer(&timer);
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(d_idata, h_odata, memSize, cudaMemcpyHostToDevice));
|
||||
sdkStopTimer(&timer);
|
||||
elapsedTimeInMs += sdkGetTimerValue(&timer);
|
||||
sdkResetTimer(&timer);
|
||||
memset(flush_buf, i, FLUSH_SIZE);
|
||||
}
|
||||
}
|
||||
|
||||
// calculate bandwidth in GB/s
|
||||
double time_s = elapsedTimeInMs / 1e3;
|
||||
bandwidthInGBs = (memSize * (float)MEMCOPY_ITERATIONS) / (double)1e9;
|
||||
bandwidthInGBs = bandwidthInGBs / time_s;
|
||||
// clean up memory
|
||||
checkCudaErrors(cudaEventDestroy(stop));
|
||||
checkCudaErrors(cudaEventDestroy(start));
|
||||
sdkDeleteTimer(&timer);
|
||||
|
||||
if (PINNED == memMode) {
|
||||
checkCudaErrors(cudaFreeHost(h_odata));
|
||||
} else {
|
||||
free(h_odata);
|
||||
}
|
||||
|
||||
free(h_cacheClear1);
|
||||
free(h_cacheClear2);
|
||||
checkCudaErrors(cudaFree(d_idata));
|
||||
|
||||
return bandwidthInGBs;
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
//! test the bandwidth of a device to device memcopy of a specific size
|
||||
///////////////////////////////////////////////////////////////////////////////
|
||||
float testDeviceToDeviceTransfer(unsigned int memSize) {
|
||||
StopWatchInterface *timer = NULL;
|
||||
float elapsedTimeInMs = 0.0f;
|
||||
float bandwidthInGBs = 0.0f;
|
||||
cudaEvent_t start, stop;
|
||||
|
||||
sdkCreateTimer(&timer);
|
||||
checkCudaErrors(cudaEventCreate(&start));
|
||||
checkCudaErrors(cudaEventCreate(&stop));
|
||||
|
||||
// allocate host memory
|
||||
unsigned char *h_idata = (unsigned char *)malloc(memSize);
|
||||
|
||||
if (h_idata == 0) {
|
||||
fprintf(stderr, "Not enough memory avaialable on host to run test!\n");
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
// initialize the host memory
|
||||
for (unsigned int i = 0; i < memSize / sizeof(unsigned char); i++) {
|
||||
h_idata[i] = (unsigned char)(i & 0xff);
|
||||
}
|
||||
|
||||
// allocate device memory
|
||||
unsigned char *d_idata;
|
||||
checkCudaErrors(cudaMalloc((void **)&d_idata, memSize));
|
||||
unsigned char *d_odata;
|
||||
checkCudaErrors(cudaMalloc((void **)&d_odata, memSize));
|
||||
|
||||
// initialize memory
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(d_idata, h_idata, memSize, cudaMemcpyHostToDevice));
|
||||
|
||||
// run the memcopy
|
||||
sdkStartTimer(&timer);
|
||||
checkCudaErrors(cudaEventRecord(start, 0));
|
||||
|
||||
for (unsigned int i = 0; i < MEMCOPY_ITERATIONS; i++) {
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(d_odata, d_idata, memSize, cudaMemcpyDeviceToDevice));
|
||||
}
|
||||
|
||||
checkCudaErrors(cudaEventRecord(stop, 0));
|
||||
|
||||
// Since device to device memory copies are non-blocking,
|
||||
// cudaDeviceSynchronize() is required in order to get
|
||||
// proper timing.
|
||||
checkCudaErrors(cudaDeviceSynchronize());
|
||||
|
||||
// get the total elapsed time in ms
|
||||
sdkStopTimer(&timer);
|
||||
checkCudaErrors(cudaEventElapsedTime(&elapsedTimeInMs, start, stop));
|
||||
|
||||
if (bDontUseGPUTiming) {
|
||||
elapsedTimeInMs = sdkGetTimerValue(&timer);
|
||||
}
|
||||
|
||||
// calculate bandwidth in GB/s
|
||||
double time_s = elapsedTimeInMs / 1e3;
|
||||
bandwidthInGBs = (2.0f * memSize * (float)MEMCOPY_ITERATIONS) / (double)1e9;
|
||||
bandwidthInGBs = bandwidthInGBs / time_s;
|
||||
|
||||
// clean up memory
|
||||
sdkDeleteTimer(&timer);
|
||||
free(h_idata);
|
||||
checkCudaErrors(cudaEventDestroy(stop));
|
||||
checkCudaErrors(cudaEventDestroy(start));
|
||||
checkCudaErrors(cudaFree(d_idata));
|
||||
checkCudaErrors(cudaFree(d_odata));
|
||||
|
||||
return bandwidthInGBs;
|
||||
}
|
||||
|
||||
/////////////////////////////////////////////////////////
|
||||
// print results in an easily read format
|
||||
////////////////////////////////////////////////////////
|
||||
void printResultsReadable(unsigned int *memSizes, double *bandwidths,
|
||||
unsigned int count, memcpyKind kind,
|
||||
memoryMode memMode, int iNumDevs, bool wc) {
|
||||
printf(" %s Bandwidth, %i Device(s)\n", sMemoryCopyKind[kind], iNumDevs);
|
||||
printf(" %s Memory Transfers\n", sMemoryMode[memMode]);
|
||||
|
||||
if (wc) {
|
||||
printf(" Write-Combined Memory Writes are Enabled");
|
||||
}
|
||||
|
||||
printf(" Transfer Size (Bytes)\tBandwidth(GB/s)\n");
|
||||
unsigned int i;
|
||||
|
||||
for (i = 0; i < (count - 1); i++) {
|
||||
printf(" %u\t\t\t%s%.1f\n", memSizes[i],
|
||||
(memSizes[i] < 10000) ? "\t" : "", bandwidths[i]);
|
||||
}
|
||||
|
||||
printf(" %u\t\t\t%s%.1f\n\n", memSizes[i],
|
||||
(memSizes[i] < 10000) ? "\t" : "", bandwidths[i]);
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// print results in a database format
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
void printResultsCSV(unsigned int *memSizes, double *bandwidths,
|
||||
unsigned int count, memcpyKind kind, memoryMode memMode,
|
||||
int iNumDevs, bool wc) {
|
||||
std::string sConfig;
|
||||
|
||||
// log config information
|
||||
if (kind == DEVICE_TO_DEVICE) {
|
||||
sConfig += "D2D";
|
||||
} else {
|
||||
if (kind == DEVICE_TO_HOST) {
|
||||
sConfig += "D2H";
|
||||
} else if (kind == HOST_TO_DEVICE) {
|
||||
sConfig += "H2D";
|
||||
}
|
||||
|
||||
if (memMode == PAGEABLE) {
|
||||
sConfig += "-Paged";
|
||||
} else if (memMode == PINNED) {
|
||||
sConfig += "-Pinned";
|
||||
|
||||
if (wc) {
|
||||
sConfig += "-WriteCombined";
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
unsigned int i;
|
||||
double dSeconds = 0.0;
|
||||
|
||||
for (i = 0; i < count; i++) {
|
||||
dSeconds = (double)memSizes[i] / (bandwidths[i] * (double)(1 << 20));
|
||||
printf(
|
||||
"bandwidthTest-%s, Bandwidth = %.1f GB/s, Time = %.5f s, Size = %u "
|
||||
"bytes, NumDevsUsed = %d\n",
|
||||
sConfig.c_str(), bandwidths[i], dSeconds, memSizes[i], iNumDevs);
|
||||
}
|
||||
}
|
||||
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
// Print help screen
|
||||
///////////////////////////////////////////////////////////////////////////
|
||||
void printHelp(void) {
|
||||
printf("Usage: bandwidthTest [OPTION]...\n");
|
||||
printf(
|
||||
"Test the bandwidth for device to host, host to device, and device to "
|
||||
"device transfers\n");
|
||||
printf("\n");
|
||||
printf(
|
||||
"Example: measure the bandwidth of device to host pinned memory copies "
|
||||
"in the range 1024 Bytes to 102400 Bytes in 1024 Byte increments\n");
|
||||
printf(
|
||||
"./bandwidthTest --memory=pinned --mode=range --start=1024 --end=102400 "
|
||||
"--increment=1024 --dtoh\n");
|
||||
|
||||
printf("\n");
|
||||
printf("Options:\n");
|
||||
printf("--help\tDisplay this help menu\n");
|
||||
printf("--csv\tPrint results as a CSV\n");
|
||||
printf("--device=[deviceno]\tSpecify the device device to be used\n");
|
||||
printf(" all - compute cumulative bandwidth on all the devices\n");
|
||||
printf(" 0,1,2,...,n - Specify any particular device to be used\n");
|
||||
printf("--memory=[MEMMODE]\tSpecify which memory mode to use\n");
|
||||
printf(" pageable - pageable memory\n");
|
||||
printf(" pinned - non-pageable system memory\n");
|
||||
printf("--mode=[MODE]\tSpecify the mode to use\n");
|
||||
printf(" quick - performs a quick measurement\n");
|
||||
printf(" range - measures a user-specified range of values\n");
|
||||
printf(" shmoo - performs an intense shmoo of a large range of values\n");
|
||||
|
||||
printf("--htod\tMeasure host to device transfers\n");
|
||||
printf("--dtoh\tMeasure device to host transfers\n");
|
||||
printf("--dtod\tMeasure device to device transfers\n");
|
||||
#if CUDART_VERSION >= 2020
|
||||
printf("--wc\tAllocate pinned memory as write-combined\n");
|
||||
#endif
|
||||
printf("--cputiming\tForce CPU-based timing always\n");
|
||||
|
||||
printf("Range mode options\n");
|
||||
printf("--start=[SIZE]\tStarting transfer size in bytes\n");
|
||||
printf("--end=[SIZE]\tEnding transfer size in bytes\n");
|
||||
printf("--increment=[SIZE]\tIncrement size in bytes\n");
|
||||
}
|
||||
20
Samples/bandwidthTest/bandwidthTest_vs2012.sln
Normal file
20
Samples/bandwidthTest/bandwidthTest_vs2012.sln
Normal file
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 12.00
|
||||
# Visual Studio 2012
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "bandwidthTest", "bandwidthTest_vs2012.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
107
Samples/bandwidthTest/bandwidthTest_vs2012.vcxproj
Normal file
107
Samples/bandwidthTest/bandwidthTest_vs2012.vcxproj
Normal file
@@ -0,0 +1,107 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>bandwidthTest_vs2012</RootNamespace>
|
||||
<ProjectName>bandwidthTest</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v110</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/bandwidthTest.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_30,sm_30;compute_35,sm_35;compute_37,sm_37;compute_50,sm_50;compute_52,sm_52;compute_60,sm_60;compute_61,sm_61;compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<CudaCompile Include="bandwidthTest.cu" />
|
||||
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
20
Samples/bandwidthTest/bandwidthTest_vs2013.sln
Normal file
20
Samples/bandwidthTest/bandwidthTest_vs2013.sln
Normal file
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 13.00
|
||||
# Visual Studio 2013
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "bandwidthTest", "bandwidthTest_vs2013.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
107
Samples/bandwidthTest/bandwidthTest_vs2013.vcxproj
Normal file
107
Samples/bandwidthTest/bandwidthTest_vs2013.vcxproj
Normal file
@@ -0,0 +1,107 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>bandwidthTest_vs2013</RootNamespace>
|
||||
<ProjectName>bandwidthTest</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v120</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/bandwidthTest.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_30,sm_30;compute_35,sm_35;compute_37,sm_37;compute_50,sm_50;compute_52,sm_52;compute_60,sm_60;compute_61,sm_61;compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<CudaCompile Include="bandwidthTest.cu" />
|
||||
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
20
Samples/bandwidthTest/bandwidthTest_vs2015.sln
Normal file
20
Samples/bandwidthTest/bandwidthTest_vs2015.sln
Normal file
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 14.00
|
||||
# Visual Studio 2015
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "bandwidthTest", "bandwidthTest_vs2015.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
107
Samples/bandwidthTest/bandwidthTest_vs2015.vcxproj
Normal file
107
Samples/bandwidthTest/bandwidthTest_vs2015.vcxproj
Normal file
@@ -0,0 +1,107 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>bandwidthTest_vs2015</RootNamespace>
|
||||
<ProjectName>bandwidthTest</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v140</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/bandwidthTest.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_30,sm_30;compute_35,sm_35;compute_37,sm_37;compute_50,sm_50;compute_52,sm_52;compute_60,sm_60;compute_61,sm_61;compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<CudaCompile Include="bandwidthTest.cu" />
|
||||
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
20
Samples/bandwidthTest/bandwidthTest_vs2017.sln
Normal file
20
Samples/bandwidthTest/bandwidthTest_vs2017.sln
Normal file
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 12.00
|
||||
# Visual Studio 2017
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "bandwidthTest", "bandwidthTest_vs2017.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
108
Samples/bandwidthTest/bandwidthTest_vs2017.vcxproj
Normal file
108
Samples/bandwidthTest/bandwidthTest_vs2017.vcxproj
Normal file
@@ -0,0 +1,108 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>bandwidthTest_vs2017</RootNamespace>
|
||||
<ProjectName>bandwidthTest</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v141</PlatformToolset>
|
||||
<WindowsTargetPlatformVersion>10.0.15063.0</WindowsTargetPlatformVersion>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/bandwidthTest.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_30,sm_30;compute_35,sm_35;compute_37,sm_37;compute_50,sm_50;compute_52,sm_52;compute_60,sm_60;compute_61,sm_61;compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<CudaCompile Include="bandwidthTest.cu" />
|
||||
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
Reference in New Issue
Block a user