mirror of
https://github.com/NVIDIA/cuda-samples.git
synced 2026-10-11 23:38:25 +08:00
Add and update samples with CUDA 10.1 Update 1 support
This commit is contained in:
322
Samples/NV12toBGRandResize/Makefile
Normal file
322
Samples/NV12toBGRandResize/Makefile
Normal file
@@ -0,0 +1,322 @@
|
||||
################################################################################
|
||||
# Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
#
|
||||
# Redistribution and use in source and binary forms, with or without
|
||||
# modification, are permitted provided that the following conditions
|
||||
# are met:
|
||||
# * Redistributions of source code must retain the above copyright
|
||||
# notice, this list of conditions and the following disclaimer.
|
||||
# * Redistributions in binary form must reproduce the above copyright
|
||||
# notice, this list of conditions and the following disclaimer in the
|
||||
# documentation and/or other materials provided with the distribution.
|
||||
# * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
# contributors may be used to endorse or promote products derived
|
||||
# from this software without specific prior written permission.
|
||||
#
|
||||
# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
# EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
# PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
# CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
# EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
# PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
# PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
# OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#
|
||||
################################################################################
|
||||
#
|
||||
# Makefile project only supported on Mac OS X and Linux Platforms)
|
||||
#
|
||||
################################################################################
|
||||
|
||||
# Location of the CUDA Toolkit
|
||||
CUDA_PATH ?= /usr/local/cuda
|
||||
|
||||
##############################
|
||||
# start deprecated interface #
|
||||
##############################
|
||||
ifeq ($(x86_64),1)
|
||||
$(info WARNING - x86_64 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=x86_64 instead)
|
||||
TARGET_ARCH ?= x86_64
|
||||
endif
|
||||
ifeq ($(ARMv7),1)
|
||||
$(info WARNING - ARMv7 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=armv7l instead)
|
||||
TARGET_ARCH ?= armv7l
|
||||
endif
|
||||
ifeq ($(aarch64),1)
|
||||
$(info WARNING - aarch64 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=aarch64 instead)
|
||||
TARGET_ARCH ?= aarch64
|
||||
endif
|
||||
ifeq ($(ppc64le),1)
|
||||
$(info WARNING - ppc64le variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=ppc64le instead)
|
||||
TARGET_ARCH ?= ppc64le
|
||||
endif
|
||||
ifneq ($(GCC),)
|
||||
$(info WARNING - GCC variable has been deprecated)
|
||||
$(info WARNING - please use HOST_COMPILER=$(GCC) instead)
|
||||
HOST_COMPILER ?= $(GCC)
|
||||
endif
|
||||
ifneq ($(abi),)
|
||||
$(error ERROR - abi variable has been removed)
|
||||
endif
|
||||
############################
|
||||
# end deprecated interface #
|
||||
############################
|
||||
|
||||
# architecture
|
||||
HOST_ARCH := $(shell uname -m)
|
||||
TARGET_ARCH ?= $(HOST_ARCH)
|
||||
ifneq (,$(filter $(TARGET_ARCH),x86_64 aarch64 ppc64le armv7l))
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifneq (,$(filter $(TARGET_ARCH),x86_64 aarch64 ppc64le))
|
||||
TARGET_SIZE := 64
|
||||
else ifneq (,$(filter $(TARGET_ARCH),armv7l))
|
||||
TARGET_SIZE := 32
|
||||
endif
|
||||
else
|
||||
TARGET_SIZE := $(shell getconf LONG_BIT)
|
||||
endif
|
||||
else
|
||||
$(error ERROR - unsupported value $(TARGET_ARCH) for TARGET_ARCH!)
|
||||
endif
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq (,$(filter $(HOST_ARCH)-$(TARGET_ARCH),aarch64-armv7l x86_64-armv7l x86_64-aarch64 x86_64-ppc64le))
|
||||
$(error ERROR - cross compiling from $(HOST_ARCH) to $(TARGET_ARCH) is not supported!)
|
||||
endif
|
||||
endif
|
||||
|
||||
# When on native aarch64 system with userspace of 32-bit, change TARGET_ARCH to armv7l
|
||||
ifeq ($(HOST_ARCH)-$(TARGET_ARCH)-$(TARGET_SIZE),aarch64-aarch64-32)
|
||||
TARGET_ARCH = armv7l
|
||||
endif
|
||||
|
||||
# operating system
|
||||
HOST_OS := $(shell uname -s 2>/dev/null | tr "[:upper:]" "[:lower:]")
|
||||
TARGET_OS ?= $(HOST_OS)
|
||||
ifeq (,$(filter $(TARGET_OS),linux darwin qnx android))
|
||||
$(error ERROR - unsupported value $(TARGET_OS) for TARGET_OS!)
|
||||
endif
|
||||
|
||||
# host compiler
|
||||
ifeq ($(TARGET_OS),darwin)
|
||||
ifeq ($(shell expr `xcodebuild -version | grep -i xcode | awk '{print $$2}' | cut -d'.' -f1` \>= 5),1)
|
||||
HOST_COMPILER ?= clang++
|
||||
endif
|
||||
else ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq ($(HOST_ARCH)-$(TARGET_ARCH),x86_64-armv7l)
|
||||
ifeq ($(TARGET_OS),linux)
|
||||
HOST_COMPILER ?= arm-linux-gnueabihf-g++
|
||||
else ifeq ($(TARGET_OS),qnx)
|
||||
ifeq ($(QNX_HOST),)
|
||||
$(error ERROR - QNX_HOST must be passed to the QNX host toolchain)
|
||||
endif
|
||||
ifeq ($(QNX_TARGET),)
|
||||
$(error ERROR - QNX_TARGET must be passed to the QNX target toolchain)
|
||||
endif
|
||||
export QNX_HOST
|
||||
export QNX_TARGET
|
||||
HOST_COMPILER ?= $(QNX_HOST)/usr/bin/arm-unknown-nto-qnx6.6.0eabi-g++
|
||||
else ifeq ($(TARGET_OS),android)
|
||||
HOST_COMPILER ?= arm-linux-androideabi-g++
|
||||
endif
|
||||
else ifeq ($(TARGET_ARCH),aarch64)
|
||||
ifeq ($(TARGET_OS), linux)
|
||||
HOST_COMPILER ?= aarch64-linux-gnu-g++
|
||||
else ifeq ($(TARGET_OS),qnx)
|
||||
ifeq ($(QNX_HOST),)
|
||||
$(error ERROR - QNX_HOST must be passed to the QNX host toolchain)
|
||||
endif
|
||||
ifeq ($(QNX_TARGET),)
|
||||
$(error ERROR - QNX_TARGET must be passed to the QNX target toolchain)
|
||||
endif
|
||||
export QNX_HOST
|
||||
export QNX_TARGET
|
||||
HOST_COMPILER ?= $(QNX_HOST)/usr/bin/aarch64-unknown-nto-qnx7.0.0-g++
|
||||
else ifeq ($(TARGET_OS), android)
|
||||
HOST_COMPILER ?= aarch64-linux-android-clang++
|
||||
endif
|
||||
else ifeq ($(TARGET_ARCH),ppc64le)
|
||||
HOST_COMPILER ?= powerpc64le-linux-gnu-g++
|
||||
endif
|
||||
endif
|
||||
HOST_COMPILER ?= g++
|
||||
NVCC := $(CUDA_PATH)/bin/nvcc -ccbin $(HOST_COMPILER)
|
||||
|
||||
# internal flags
|
||||
NVCCFLAGS := -m${TARGET_SIZE}
|
||||
CCFLAGS :=
|
||||
LDFLAGS :=
|
||||
|
||||
# build flags
|
||||
ifeq ($(TARGET_OS),darwin)
|
||||
LDFLAGS += -rpath $(CUDA_PATH)/lib
|
||||
CCFLAGS += -arch $(HOST_ARCH)
|
||||
else ifeq ($(HOST_ARCH)-$(TARGET_ARCH)-$(TARGET_OS),x86_64-armv7l-linux)
|
||||
LDFLAGS += --dynamic-linker=/lib/ld-linux-armhf.so.3
|
||||
CCFLAGS += -mfloat-abi=hard
|
||||
else ifeq ($(TARGET_OS),android)
|
||||
LDFLAGS += -pie
|
||||
CCFLAGS += -fpie -fpic -fexceptions
|
||||
endif
|
||||
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-linux)
|
||||
ifneq ($(TARGET_FS),)
|
||||
GCCVERSIONLTEQ46 := $(shell expr `$(HOST_COMPILER) -dumpversion` \<= 4.6)
|
||||
ifeq ($(GCCVERSIONLTEQ46),1)
|
||||
CCFLAGS += --sysroot=$(TARGET_FS)
|
||||
endif
|
||||
LDFLAGS += --sysroot=$(TARGET_FS)
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib/arm-linux-gnueabihf
|
||||
endif
|
||||
endif
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-linux)
|
||||
ifneq ($(TARGET_FS),)
|
||||
GCCVERSIONLTEQ46 := $(shell expr `$(HOST_COMPILER) -dumpversion` \<= 4.6)
|
||||
ifeq ($(GCCVERSIONLTEQ46),1)
|
||||
CCFLAGS += --sysroot=$(TARGET_FS)
|
||||
endif
|
||||
LDFLAGS += --sysroot=$(TARGET_FS)
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/lib -L $(TARGET_FS)/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib -L $(TARGET_FS)/usr/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib/aarch64-linux-gnu -L $(TARGET_FS)/usr/lib/aarch64-linux-gnu
|
||||
LDFLAGS += --unresolved-symbols=ignore-in-shared-libs
|
||||
CCFLAGS += -isystem=$(TARGET_FS)/usr/include
|
||||
CCFLAGS += -isystem=$(TARGET_FS)/usr/include/aarch64-linux-gnu
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(TARGET_OS),qnx)
|
||||
CCFLAGS += -DWIN_INTERFACE_CUSTOM
|
||||
LDFLAGS += -lsocket
|
||||
endif
|
||||
|
||||
# Install directory of different arch
|
||||
CUDA_INSTALL_TARGET_DIR :=
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-linux)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/armv7-linux-gnueabihf/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-linux)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-linux/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-android)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/armv7-linux-androideabi/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-android)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-linux-androideabi/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-qnx)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/ARMv7-linux-QNX/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-qnx)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-qnx/
|
||||
else ifeq ($(TARGET_ARCH),ppc64le)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/ppc64le-linux/
|
||||
endif
|
||||
|
||||
# Debug build flags
|
||||
ifeq ($(dbg),1)
|
||||
NVCCFLAGS += -g -G
|
||||
BUILD_TYPE := debug
|
||||
else
|
||||
BUILD_TYPE := release
|
||||
endif
|
||||
|
||||
ALL_CCFLAGS :=
|
||||
ALL_CCFLAGS += $(NVCCFLAGS)
|
||||
ALL_CCFLAGS += $(EXTRA_NVCCFLAGS)
|
||||
ALL_CCFLAGS += $(addprefix -Xcompiler ,$(CCFLAGS))
|
||||
ALL_CCFLAGS += $(addprefix -Xcompiler ,$(EXTRA_CCFLAGS))
|
||||
|
||||
SAMPLE_ENABLED := 1
|
||||
|
||||
# This sample is not supported on ARMv7
|
||||
ifeq ($(TARGET_ARCH),armv7l)
|
||||
$(info >>> WARNING - NV12toBGRandResize is not supported on ARMv7 - waiving sample <<<)
|
||||
SAMPLE_ENABLED := 0
|
||||
endif
|
||||
|
||||
ALL_LDFLAGS :=
|
||||
ALL_LDFLAGS += $(ALL_CCFLAGS)
|
||||
ALL_LDFLAGS += $(addprefix -Xlinker ,$(LDFLAGS))
|
||||
ALL_LDFLAGS += $(addprefix -Xlinker ,$(EXTRA_LDFLAGS))
|
||||
|
||||
# Common includes and paths for CUDA
|
||||
INCLUDES := -I../../Common
|
||||
LIBRARIES :=
|
||||
|
||||
################################################################################
|
||||
|
||||
# Gencode arguments
|
||||
ifeq ($(TARGET_ARCH),$(filter $(TARGET_ARCH),armv7l aarch64))
|
||||
SMS ?= 30 35 37 50 52 60 61 70 72 75
|
||||
else
|
||||
SMS ?= 30 35 37 50 52 60 61 70 75
|
||||
endif
|
||||
|
||||
ifeq ($(SMS),)
|
||||
$(info >>> WARNING - no SM architectures have been specified - waiving sample <<<)
|
||||
SAMPLE_ENABLED := 0
|
||||
endif
|
||||
|
||||
ifeq ($(GENCODE_FLAGS),)
|
||||
# Generate SASS code for each SM architecture listed in $(SMS)
|
||||
$(foreach sm,$(SMS),$(eval GENCODE_FLAGS += -gencode arch=compute_$(sm),code=sm_$(sm)))
|
||||
|
||||
# Generate PTX code from the highest SM architecture in $(SMS) to guarantee forward-compatibility
|
||||
HIGHEST_SM := $(lastword $(sort $(SMS)))
|
||||
ifneq ($(HIGHEST_SM),)
|
||||
GENCODE_FLAGS += -gencode arch=compute_$(HIGHEST_SM),code=compute_$(HIGHEST_SM)
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(SAMPLE_ENABLED),0)
|
||||
EXEC ?= @echo "[@]"
|
||||
endif
|
||||
|
||||
################################################################################
|
||||
|
||||
# Target rules
|
||||
all: build
|
||||
|
||||
build: NV12toBGRandResize
|
||||
|
||||
check.deps:
|
||||
ifeq ($(SAMPLE_ENABLED),0)
|
||||
@echo "Sample will be waived due to the above missing dependencies"
|
||||
else
|
||||
@echo "Sample is ready - all dependencies have been met"
|
||||
endif
|
||||
|
||||
bgr_resize.o:bgr_resize.cu
|
||||
$(EXEC) $(NVCC) $(INCLUDES) $(ALL_CCFLAGS) $(GENCODE_FLAGS) -o $@ -c $<
|
||||
|
||||
nv12_resize.o:nv12_resize.cu
|
||||
$(EXEC) $(NVCC) $(INCLUDES) $(ALL_CCFLAGS) $(GENCODE_FLAGS) -o $@ -c $<
|
||||
|
||||
nv12_to_bgr_planar.o:nv12_to_bgr_planar.cu
|
||||
$(EXEC) $(NVCC) $(INCLUDES) $(ALL_CCFLAGS) $(GENCODE_FLAGS) -o $@ -c $<
|
||||
|
||||
resize_convert_main.o:resize_convert_main.cpp
|
||||
$(EXEC) $(NVCC) $(INCLUDES) $(ALL_CCFLAGS) $(GENCODE_FLAGS) -o $@ -c $<
|
||||
|
||||
utils.o:utils.cu
|
||||
$(EXEC) $(NVCC) $(INCLUDES) $(ALL_CCFLAGS) $(GENCODE_FLAGS) -o $@ -c $<
|
||||
|
||||
NV12toBGRandResize: bgr_resize.o nv12_resize.o nv12_to_bgr_planar.o resize_convert_main.o utils.o
|
||||
$(EXEC) $(NVCC) $(ALL_LDFLAGS) $(GENCODE_FLAGS) -o $@ $+ $(LIBRARIES)
|
||||
$(EXEC) mkdir -p ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)
|
||||
$(EXEC) cp $@ ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)
|
||||
|
||||
run: build
|
||||
$(EXEC) ./NV12toBGRandResize
|
||||
|
||||
clean:
|
||||
rm -f NV12toBGRandResize bgr_resize.o nv12_resize.o nv12_to_bgr_planar.o resize_convert_main.o utils.o
|
||||
rm -rf ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)/NV12toBGRandResize
|
||||
|
||||
clobber: clean
|
||||
20
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2012.sln
Normal file
20
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2012.sln
Normal file
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 12.00
|
||||
# Visual Studio 2012
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "NV12toBGRandResize", "NV12toBGRandResize_vs2012.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
112
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2012.vcxproj
Normal file
112
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2012.vcxproj
Normal file
@@ -0,0 +1,112 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>NV12toBGRandResize_vs2012</RootNamespace>
|
||||
<ProjectName>NV12toBGRandResize</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v110</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/NV12toBGRandResize.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_30,sm_30;compute_35,sm_35;compute_37,sm_37;compute_50,sm_50;compute_52,sm_52;compute_60,sm_60;compute_61,sm_61;compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<CudaCompile Include="bgr_resize.cu" />
|
||||
<CudaCompile Include="nv12_resize.cu" />
|
||||
<CudaCompile Include="nv12_to_bgr_planar.cu" />
|
||||
<ClCompile Include="resize_convert_main.cpp" />
|
||||
<CudaCompile Include="utils.cu" />
|
||||
<ClInclude Include="resize_convert.h" />
|
||||
<ClInclude Include="utils.h" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
20
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2013.sln
Normal file
20
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2013.sln
Normal file
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 13.00
|
||||
# Visual Studio 2013
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "NV12toBGRandResize", "NV12toBGRandResize_vs2013.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
112
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2013.vcxproj
Normal file
112
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2013.vcxproj
Normal file
@@ -0,0 +1,112 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>NV12toBGRandResize_vs2013</RootNamespace>
|
||||
<ProjectName>NV12toBGRandResize</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v120</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/NV12toBGRandResize.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_30,sm_30;compute_35,sm_35;compute_37,sm_37;compute_50,sm_50;compute_52,sm_52;compute_60,sm_60;compute_61,sm_61;compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<CudaCompile Include="bgr_resize.cu" />
|
||||
<CudaCompile Include="nv12_resize.cu" />
|
||||
<CudaCompile Include="nv12_to_bgr_planar.cu" />
|
||||
<ClCompile Include="resize_convert_main.cpp" />
|
||||
<CudaCompile Include="utils.cu" />
|
||||
<ClInclude Include="resize_convert.h" />
|
||||
<ClInclude Include="utils.h" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
20
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2015.sln
Normal file
20
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2015.sln
Normal file
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 14.00
|
||||
# Visual Studio 2015
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "NV12toBGRandResize", "NV12toBGRandResize_vs2015.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
112
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2015.vcxproj
Normal file
112
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2015.vcxproj
Normal file
@@ -0,0 +1,112 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>NV12toBGRandResize_vs2015</RootNamespace>
|
||||
<ProjectName>NV12toBGRandResize</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v140</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/NV12toBGRandResize.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_30,sm_30;compute_35,sm_35;compute_37,sm_37;compute_50,sm_50;compute_52,sm_52;compute_60,sm_60;compute_61,sm_61;compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<CudaCompile Include="bgr_resize.cu" />
|
||||
<CudaCompile Include="nv12_resize.cu" />
|
||||
<CudaCompile Include="nv12_to_bgr_planar.cu" />
|
||||
<ClCompile Include="resize_convert_main.cpp" />
|
||||
<CudaCompile Include="utils.cu" />
|
||||
<ClInclude Include="resize_convert.h" />
|
||||
<ClInclude Include="utils.h" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
20
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2017.sln
Normal file
20
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2017.sln
Normal file
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 12.00
|
||||
# Visual Studio 2017
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "NV12toBGRandResize", "NV12toBGRandResize_vs2017.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
117
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2017.vcxproj
Normal file
117
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2017.vcxproj
Normal file
@@ -0,0 +1,117 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>NV12toBGRandResize_vs2017</RootNamespace>
|
||||
<ProjectName>NV12toBGRandResize</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(WindowsTargetPlatformVersion)'==''">
|
||||
<LatestTargetPlatformVersion>$([Microsoft.Build.Utilities.ToolLocationHelper]::GetLatestSDKTargetPlatformVersion('Windows', '10.0'))</LatestTargetPlatformVersion>
|
||||
<WindowsTargetPlatformVersion Condition="'$(WindowsTargetPlatformVersion)' == ''">$(LatestTargetPlatformVersion)</WindowsTargetPlatformVersion>
|
||||
<TargetPlatformVersion>$(WindowsTargetPlatformVersion)</TargetPlatformVersion>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v141</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/NV12toBGRandResize.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_30,sm_30;compute_35,sm_35;compute_37,sm_37;compute_50,sm_50;compute_52,sm_52;compute_60,sm_60;compute_61,sm_61;compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<CudaCompile Include="bgr_resize.cu" />
|
||||
<CudaCompile Include="nv12_resize.cu" />
|
||||
<CudaCompile Include="nv12_to_bgr_planar.cu" />
|
||||
<ClCompile Include="resize_convert_main.cpp" />
|
||||
<CudaCompile Include="utils.cu" />
|
||||
<ClInclude Include="resize_convert.h" />
|
||||
<ClInclude Include="utils.h" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
20
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2019.sln
Normal file
20
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2019.sln
Normal file
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 12.00
|
||||
# Visual Studio 2019
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "NV12toBGRandResize", "NV12toBGRandResize_vs2019.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
113
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2019.vcxproj
Normal file
113
Samples/NV12toBGRandResize/NV12toBGRandResize_vs2019.vcxproj
Normal file
@@ -0,0 +1,113 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>NV12toBGRandResize_vs2019</RootNamespace>
|
||||
<ProjectName>NV12toBGRandResize</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v142</PlatformToolset>
|
||||
<WindowsTargetPlatformVersion>10.0</WindowsTargetPlatformVersion>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/NV12toBGRandResize.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration>compute_30,sm_30;compute_35,sm_35;compute_37,sm_37;compute_50,sm_50;compute_52,sm_52;compute_60,sm_60;compute_61,sm_61;compute_70,sm_70;compute_75,sm_75;</CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<CudaCompile Include="bgr_resize.cu" />
|
||||
<CudaCompile Include="nv12_resize.cu" />
|
||||
<CudaCompile Include="nv12_to_bgr_planar.cu" />
|
||||
<ClCompile Include="resize_convert_main.cpp" />
|
||||
<CudaCompile Include="utils.cu" />
|
||||
<ClInclude Include="resize_convert.h" />
|
||||
<ClInclude Include="utils.h" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
70
Samples/NV12toBGRandResize/NsightEclipse.xml
Normal file
70
Samples/NV12toBGRandResize/NsightEclipse.xml
Normal file
@@ -0,0 +1,70 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE entry SYSTEM "SamplesInfo.dtd">
|
||||
<entry>
|
||||
<name>NV12toBGRandResize</name>
|
||||
<cuda_api_list>
|
||||
<toolkit>cudaMemcpy2D</toolkit>
|
||||
<toolkit>cudaMallocManaged</toolkit>
|
||||
</cuda_api_list>
|
||||
<description><![CDATA[This code shows two ways to convert and resize NV12 frames to BGR 3 planars frames using CUDA in batch. Way-1, Convert NV12 Input to BGR @ Input Resolution-1, then Resize to Resolution#2. Way-2, resize NV12 Input to Resolution#2 then convert it to BGR Output. NVIDIA HW Decoder, both dGPU and Tegra, normally outputs NV12 pitch format frames. For the inference using TensorRT, the input frame needs to be BGR planar format with possibly different size. So, conversion and resizing from NV12 to BGR planar is usually required for the inference following decoding. This CUDA code provides a reference implementation for conversion and resizing.]]></description>
|
||||
<devicecompilation>whole</devicecompilation>
|
||||
<includepaths>
|
||||
<path>./</path>
|
||||
<path>../</path>
|
||||
<path>../../common/inc</path>
|
||||
</includepaths>
|
||||
<keyconcepts>
|
||||
<concept level="basic">Graphics Interop</concept>
|
||||
<concept level="basic">Image Processing</concept>
|
||||
<concept level="basic">Video Processing</concept>
|
||||
</keyconcepts>
|
||||
<keywords>
|
||||
<keyword>GPGPU</keyword>
|
||||
</keywords>
|
||||
<libraries>
|
||||
</libraries>
|
||||
<librarypaths>
|
||||
</librarypaths>
|
||||
<nsight_eclipse>true</nsight_eclipse>
|
||||
<primary_file>resize_convert_main.cpp</primary_file>
|
||||
<scopes>
|
||||
<scope>1:CUDA Basic Topics</scope>
|
||||
<scope>2:Image Processing</scope>
|
||||
<scope>2:Computer Vision</scope>
|
||||
</scopes>
|
||||
<sm-arch>sm30</sm-arch>
|
||||
<sm-arch>sm35</sm-arch>
|
||||
<sm-arch>sm37</sm-arch>
|
||||
<sm-arch>sm50</sm-arch>
|
||||
<sm-arch>sm52</sm-arch>
|
||||
<sm-arch>sm60</sm-arch>
|
||||
<sm-arch>sm61</sm-arch>
|
||||
<sm-arch>sm70</sm-arch>
|
||||
<sm-arch>sm72</sm-arch>
|
||||
<sm-arch>sm75</sm-arch>
|
||||
<supported_envs>
|
||||
<env>
|
||||
<arch>x86_64</arch>
|
||||
<platform>linux</platform>
|
||||
</env>
|
||||
<env>
|
||||
<platform>windows7</platform>
|
||||
</env>
|
||||
<env>
|
||||
<arch>x86_64</arch>
|
||||
<platform>macosx</platform>
|
||||
</env>
|
||||
<env>
|
||||
<arch>aarch64</arch>
|
||||
</env>
|
||||
<env>
|
||||
<arch>ppc64le</arch>
|
||||
<platform>linux</platform>
|
||||
</env>
|
||||
</supported_envs>
|
||||
<supported_sm_architectures>
|
||||
<include>all</include>
|
||||
</supported_sm_architectures>
|
||||
<title>NV12toBGRandResize</title>
|
||||
<type>exe</type>
|
||||
</entry>
|
||||
94
Samples/NV12toBGRandResize/README.md
Normal file
94
Samples/NV12toBGRandResize/README.md
Normal file
@@ -0,0 +1,94 @@
|
||||
# NV12toBGRandResize - NV12toBGRandResize
|
||||
|
||||
## Description
|
||||
|
||||
This code shows two ways to convert and resize NV12 frames to BGR 3 planars frames using CUDA in batch. Way-1, Convert NV12 Input to BGR @ Input Resolution-1, then Resize to Resolution#2. Way-2, resize NV12 Input to Resolution#2 then convert it to BGR Output. NVIDIA HW Decoder, both dGPU and Tegra, normally outputs NV12 pitch format frames. For the inference using TensorRT, the input frame needs to be BGR planar format with possibly different size. So, conversion and resizing from NV12 to BGR planar is usually required for the inference following decoding. This CUDA code provides a reference implementation for conversion and resizing.
|
||||
|
||||
## Key Concepts
|
||||
|
||||
Graphics Interop, Image Processing, Video Processing
|
||||
|
||||
## Supported SM Architectures
|
||||
|
||||
[SM 3.0 ](https://developer.nvidia.com/cuda-gpus) [SM 3.5 ](https://developer.nvidia.com/cuda-gpus) [SM 3.7 ](https://developer.nvidia.com/cuda-gpus) [SM 5.0 ](https://developer.nvidia.com/cuda-gpus) [SM 5.2 ](https://developer.nvidia.com/cuda-gpus) [SM 6.0 ](https://developer.nvidia.com/cuda-gpus) [SM 6.1 ](https://developer.nvidia.com/cuda-gpus) [SM 7.0 ](https://developer.nvidia.com/cuda-gpus) [SM 7.2 ](https://developer.nvidia.com/cuda-gpus) [SM 7.5 ](https://developer.nvidia.com/cuda-gpus)
|
||||
|
||||
## Supported OSes
|
||||
|
||||
Linux, Windows, MacOSX
|
||||
|
||||
## Supported CPU Architecture
|
||||
|
||||
x86_64, ppc64le, aarch64
|
||||
|
||||
## CUDA APIs involved
|
||||
|
||||
### [CUDA Runtime API](http://docs.nvidia.com/cuda/cuda-runtime-api/index.html)
|
||||
cudaMemcpy2D, cudaMallocManaged
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Download and install the [CUDA Toolkit 10.1](https://developer.nvidia.com/cuda-downloads) for your corresponding platform.
|
||||
|
||||
## Build and Run
|
||||
|
||||
### Windows
|
||||
The Windows samples are built using the Visual Studio IDE. Solution files (.sln) are provided for each supported version of Visual Studio, using the format:
|
||||
```
|
||||
*_vs<version>.sln - for Visual Studio <version>
|
||||
```
|
||||
Each individual sample has its own set of solution files in its directory:
|
||||
|
||||
To build/examine all the samples at once, the complete solution files should be used. To build/examine a single sample, the individual sample solution files should be used.
|
||||
> **Note:** Some samples require that the Microsoft DirectX SDK (June 2010 or newer) be installed and that the VC++ directory paths are properly set up (**Tools > Options...**). Check DirectX Dependencies section for details."
|
||||
|
||||
### Linux
|
||||
The Linux samples are built using makefiles. To use the makefiles, change the current directory to the sample directory you wish to build, and run make:
|
||||
```
|
||||
$ cd <sample_dir>
|
||||
$ make
|
||||
```
|
||||
The samples makefiles can take advantage of certain options:
|
||||
* **TARGET_ARCH=<arch>** - cross-compile targeting a specific architecture. Allowed architectures are x86_64, ppc64le, aarch64.
|
||||
By default, TARGET_ARCH is set to HOST_ARCH. On a x86_64 machine, not setting TARGET_ARCH is the equivalent of setting TARGET_ARCH=x86_64.<br/>
|
||||
`$ make TARGET_ARCH=x86_64` <br/> `$ make TARGET_ARCH=ppc64le` <br/> `$ make TARGET_ARCH=aarch64` <br/>
|
||||
See [here](http://docs.nvidia.com/cuda/cuda-samples/index.html#cross-samples) for more details.
|
||||
* **dbg=1** - build with debug symbols
|
||||
```
|
||||
$ make dbg=1
|
||||
```
|
||||
* **SMS="A B ..."** - override the SM architectures for which the sample will be built, where `"A B ..."` is a space-delimited list of SM architectures. For example, to generate SASS for SM 50 and SM 60, use `SMS="50 60"`.
|
||||
```
|
||||
$ make SMS="50 60"
|
||||
```
|
||||
|
||||
* **HOST_COMPILER=<host_compiler>** - override the default g++ host compiler. See the [Linux Installation Guide](http://docs.nvidia.com/cuda/cuda-installation-guide-linux/index.html#system-requirements) for a list of supported host compilers.
|
||||
```
|
||||
$ make HOST_COMPILER=g++
|
||||
```
|
||||
|
||||
### Mac
|
||||
The Mac samples are built using makefiles. To use the makefiles, change directory into the sample directory you wish to build, and run make:
|
||||
```
|
||||
$ cd <sample_dir>
|
||||
$ make
|
||||
```
|
||||
|
||||
The samples makefiles can take advantage of certain options:
|
||||
|
||||
* **dbg=1** - build with debug symbols
|
||||
```
|
||||
$ make dbg=1
|
||||
```
|
||||
|
||||
* **SMS="A B ..."** - override the SM architectures for which the sample will be built, where "A B ..." is a space-delimited list of SM architectures. For example, to generate SASS for SM 50 and SM 60, use SMS="50 60".
|
||||
```
|
||||
$ make SMS="A B ..."
|
||||
```
|
||||
|
||||
* **HOST_COMPILER=<host_compiler>** - override the default clang host compiler. See the [Mac Installation Guide](http://docs.nvidia.com/cuda/cuda-installation-guide-mac-os-x/index.html#system-requirements) for a list of supported host compilers.
|
||||
```
|
||||
$ make HOST_COMPILER=clang
|
||||
```
|
||||
|
||||
## References (for more details)
|
||||
|
||||
134
Samples/NV12toBGRandResize/bgr_resize.cu
Normal file
134
Samples/NV12toBGRandResize/bgr_resize.cu
Normal file
@@ -0,0 +1,134 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
|
||||
// Implements BGR 3 progressive planars frames batch resize
|
||||
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
#include "resize_convert.h"
|
||||
|
||||
__global__ void resizeBGRplanarBatchKernel(cudaTextureObject_t texSrc,
|
||||
float *pDst, int nDstPitch, int nDstHeight, int nSrcHeight,
|
||||
int batch, float scaleX, float scaleY,
|
||||
int cropX, int cropY, int cropW, int cropH) {
|
||||
int x = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
int y = threadIdx.y + blockIdx.y * blockDim.y;
|
||||
|
||||
if (x >= (int)(cropW/scaleX) || y >= (int)(cropH/scaleY))
|
||||
return;
|
||||
|
||||
int frameSize = nDstPitch*nDstHeight;
|
||||
float *p = NULL;
|
||||
for (int i = blockIdx.z; i < batch; i += gridDim.z) {
|
||||
#pragma unroll
|
||||
for (int channel=0; channel < 3; channel++){
|
||||
p = pDst + i * 3 * frameSize + y * nDstPitch + x + channel * frameSize;
|
||||
*p = tex2D<float>(texSrc, x * scaleX + cropX,
|
||||
((3 * i + channel) * nSrcHeight + y * scaleY + cropY));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
static void resizeBGRplanarBatchCore(
|
||||
float *dpSrc, int nSrcPitch, int nSrcWidth, int nSrcHeight,
|
||||
float *dpDst, int nDstPitch, int nDstWidth, int nDstHeight,
|
||||
int nBatchSize, cudaStream_t stream, bool whSameResizeRatio,
|
||||
int cropX, int cropY, int cropW, int cropH) {
|
||||
cudaTextureObject_t texSrc[2];
|
||||
int nTiles = 1, h, iTile;
|
||||
|
||||
h = nSrcHeight * 3 * nBatchSize;
|
||||
while ((h + nTiles - 1) / nTiles > 65536)
|
||||
nTiles++;
|
||||
|
||||
if (nTiles > 2)
|
||||
return;
|
||||
|
||||
int batchTile = nBatchSize / nTiles;
|
||||
int batchTileLast = nBatchSize - batchTile * (nTiles-1);
|
||||
|
||||
for (iTile = 0; iTile < nTiles; ++iTile) {
|
||||
int bs = (iTile == nTiles - 1) ? batchTileLast : batchTile;
|
||||
float *dpSrcNew = dpSrc +
|
||||
iTile * (batchTile * 3 * nSrcHeight * nSrcPitch);
|
||||
|
||||
cudaResourceDesc resDesc = {};
|
||||
resDesc.resType = cudaResourceTypePitch2D;
|
||||
resDesc.res.pitch2D.devPtr = dpSrcNew;
|
||||
resDesc.res.pitch2D.desc = cudaCreateChannelDesc<float>();
|
||||
resDesc.res.pitch2D.width = nSrcWidth;
|
||||
resDesc.res.pitch2D.height = bs * 3 * nSrcHeight;
|
||||
resDesc.res.pitch2D.pitchInBytes = nSrcPitch * sizeof(float);
|
||||
cudaTextureDesc texDesc = {};
|
||||
texDesc.filterMode = cudaFilterModeLinear;
|
||||
texDesc.readMode = cudaReadModeElementType;
|
||||
|
||||
checkCudaErrors(cudaCreateTextureObject(&texSrc[iTile], &resDesc, &texDesc, NULL));
|
||||
float *dpDstNew = dpDst +
|
||||
iTile * (batchTile * 3 * nDstHeight * nDstPitch);
|
||||
|
||||
if(cropW == 0 || cropH == 0) {
|
||||
cropX = 0;
|
||||
cropY = 0;
|
||||
cropW = nSrcWidth;
|
||||
cropH = nSrcHeight;
|
||||
}
|
||||
|
||||
float scaleX = (cropW*1.0f / nDstWidth);
|
||||
float scaleY = (cropH*1.0f / nDstHeight);
|
||||
|
||||
if(whSameResizeRatio == true)
|
||||
scaleX = scaleY = scaleX > scaleY ? scaleX : scaleY;
|
||||
dim3 block(32, 32, 1);
|
||||
|
||||
size_t blockDimZ = bs;
|
||||
// Restricting blocks in Z-dim till 32 to not launch too many blocks
|
||||
blockDimZ = (blockDimZ > 32) ? 32 : blockDimZ;
|
||||
dim3 grid((cropW*1.0f/scaleX + block.x - 1) / block.x,
|
||||
(cropH*1.0f/scaleY + block.y - 1) / block.y, blockDimZ);
|
||||
|
||||
resizeBGRplanarBatchKernel<<<grid, block, 0, stream>>>
|
||||
(texSrc[iTile], dpDstNew, nDstPitch, nDstHeight, nSrcHeight,
|
||||
bs, scaleX, scaleY, cropX, cropY, cropW, cropH);
|
||||
|
||||
}
|
||||
|
||||
for (iTile = 0; iTile < nTiles; ++iTile)
|
||||
checkCudaErrors(cudaDestroyTextureObject(texSrc[iTile]));
|
||||
}
|
||||
|
||||
void resizeBGRplanarBatch(
|
||||
float *dpSrc, int nSrcPitch, int nSrcWidth, int nSrcHeight,
|
||||
float *dpDst, int nDstPitch, int nDstWidth, int nDstHeight,
|
||||
int nBatchSize, cudaStream_t stream,
|
||||
int cropX, int cropY, int cropW, int cropH, bool whSameResizeRatio) {
|
||||
resizeBGRplanarBatchCore(dpSrc, nSrcPitch, nSrcWidth, nSrcHeight,
|
||||
dpDst, nDstPitch, nDstWidth, nDstHeight, nBatchSize, stream,
|
||||
whSameResizeRatio, cropX, cropY, cropW, cropH);
|
||||
}
|
||||
30
Samples/NV12toBGRandResize/data/test1280x720.nv12
Normal file
30
Samples/NV12toBGRandResize/data/test1280x720.nv12
Normal file
File diff suppressed because one or more lines are too long
72
Samples/NV12toBGRandResize/data/test1920x1080.nv12
Normal file
72
Samples/NV12toBGRandResize/data/test1920x1080.nv12
Normal file
File diff suppressed because one or more lines are too long
1
Samples/NV12toBGRandResize/data/test640x480.nv12
Normal file
1
Samples/NV12toBGRandResize/data/test640x480.nv12
Normal file
File diff suppressed because one or more lines are too long
112
Samples/NV12toBGRandResize/nv12_resize.cu
Normal file
112
Samples/NV12toBGRandResize/nv12_resize.cu
Normal file
@@ -0,0 +1,112 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
// Implements interlace NV12 frames batch resize
|
||||
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
#include "resize_convert.h"
|
||||
|
||||
__global__ static void resizeNV12BatchKernel(cudaTextureObject_t texSrcLuma,
|
||||
cudaTextureObject_t texSrcChroma,
|
||||
uint8_t *pDstNv12, int nSrcWidth,
|
||||
int nSrcHeight, int nDstPitch,
|
||||
int nDstWidth, int nDstHeight,
|
||||
int nBatchSize) {
|
||||
int x = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
int y = threadIdx.y + blockIdx.y * blockDim.y;
|
||||
|
||||
int px = x * 2, py = y * 2;
|
||||
|
||||
if ((px + 1) >= nDstWidth || (py + 1) >= nDstHeight) return;
|
||||
|
||||
float fxScale = 1.0f * nSrcWidth / nDstWidth;
|
||||
float fyScale = 1.0f * nSrcHeight / nDstHeight;
|
||||
|
||||
uint8_t *p = pDstNv12 + px + py * nDstPitch;
|
||||
int hh = nDstHeight * 3 / 2;
|
||||
int nByte = nDstPitch * hh;
|
||||
int px_fxScale = px * fxScale;
|
||||
int px_fxScale_1 = (px + 1) * fxScale;
|
||||
int py_fyScale = py * fyScale;
|
||||
int py_fyScale_1 = (py + 1) * fyScale;
|
||||
|
||||
for (int i = blockIdx.z; i < nBatchSize; i+=gridDim.z) {
|
||||
*(uchar2 *)p = make_uchar2(tex2D<uint8_t>(texSrcLuma, px_fxScale, py_fyScale),
|
||||
tex2D<uint8_t>(texSrcLuma, px_fxScale_1, py_fyScale));
|
||||
*(uchar2 *)(p + nDstPitch) =
|
||||
make_uchar2(tex2D<uint8_t>(texSrcLuma, px_fxScale, py_fyScale_1),
|
||||
tex2D<uint8_t>(texSrcLuma, px_fxScale_1, py_fyScale_1));
|
||||
*(uchar2 *)(p + (nDstHeight - y) * nDstPitch) = tex2D<uchar2>(
|
||||
texSrcChroma, x * fxScale, (hh * i + nDstHeight + y) * fyScale);
|
||||
p += nByte;
|
||||
py += hh;
|
||||
}
|
||||
}
|
||||
|
||||
void resizeNV12Batch(uint8_t *dpSrc, int nSrcPitch, int nSrcWidth,
|
||||
int nSrcHeight, uint8_t *dpDst, int nDstPitch,
|
||||
int nDstWidth, int nDstHeight, int nBatchSize,
|
||||
cudaStream_t stream) {
|
||||
int hhSrc = ceilf(nSrcHeight * 3.0f / 2.0f);
|
||||
cudaResourceDesc resDesc = {};
|
||||
resDesc.resType = cudaResourceTypePitch2D;
|
||||
resDesc.res.pitch2D.devPtr = dpSrc;
|
||||
resDesc.res.pitch2D.desc = cudaCreateChannelDesc<uint8_t>();
|
||||
resDesc.res.pitch2D.width = nSrcWidth;
|
||||
resDesc.res.pitch2D.height = hhSrc * nBatchSize;
|
||||
resDesc.res.pitch2D.pitchInBytes = nSrcPitch;
|
||||
|
||||
cudaTextureDesc texDesc = {};
|
||||
texDesc.filterMode = cudaFilterModePoint;
|
||||
texDesc.readMode = cudaReadModeElementType;
|
||||
|
||||
cudaTextureObject_t texLuma = 0;
|
||||
checkCudaErrors(cudaCreateTextureObject(&texLuma, &resDesc, &texDesc, NULL));
|
||||
|
||||
resDesc.res.pitch2D.desc = cudaCreateChannelDesc<uchar2>();
|
||||
resDesc.res.pitch2D.width /= 2;
|
||||
|
||||
cudaTextureObject_t texChroma = 0;
|
||||
checkCudaErrors(cudaCreateTextureObject(&texChroma, &resDesc, &texDesc, NULL));
|
||||
|
||||
dim3 block(32, 32, 1);
|
||||
|
||||
size_t blockDimZ = nBatchSize;
|
||||
|
||||
// Restricting blocks in Z-dim till 32 to not launch too many blocks
|
||||
blockDimZ = (blockDimZ > 32) ? 32 : blockDimZ;
|
||||
|
||||
dim3 grid((nDstWidth / 2 + block.x) / block.x,
|
||||
(nDstHeight / 2 + block.y) / block.y, blockDimZ);
|
||||
resizeNV12BatchKernel<<<grid, block, 0, stream>>>(
|
||||
texLuma, texChroma, dpDst, nSrcWidth, nSrcHeight, nDstPitch, nDstWidth,
|
||||
nDstHeight, nBatchSize);
|
||||
|
||||
checkCudaErrors(cudaDestroyTextureObject(texLuma));
|
||||
checkCudaErrors(cudaDestroyTextureObject(texChroma));
|
||||
}
|
||||
154
Samples/NV12toBGRandResize/nv12_to_bgr_planar.cu
Normal file
154
Samples/NV12toBGRandResize/nv12_to_bgr_planar.cu
Normal file
@@ -0,0 +1,154 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
|
||||
// Implements NV12 to BGR batch conversion
|
||||
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#include "resize_convert.h"
|
||||
|
||||
#define CONV_THREADS_X 64
|
||||
#define CONV_THREADS_Y 10
|
||||
|
||||
__forceinline__ __device__ static float clampF(float x, float lower,
|
||||
float upper) {
|
||||
return x < lower ? lower : (x > upper ? upper : x);
|
||||
}
|
||||
|
||||
__global__ static void nv12ToBGRplanarBatchKernel(const uint8_t *pNv12,
|
||||
int nNv12Pitch, float *pBgr,
|
||||
int nRgbPitch, int nWidth,
|
||||
int nHeight, int nBatchSize) {
|
||||
int x = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
int y = threadIdx.y + blockIdx.y * blockDim.y;
|
||||
|
||||
if ((x << 2) + 1 > nWidth || (y << 1) + 1 > nHeight) return;
|
||||
|
||||
const uint8_t *__restrict__ pSrc = pNv12;
|
||||
|
||||
for (int i = blockIdx.z; i < nBatchSize; i += gridDim.z) {
|
||||
pSrc = pNv12 + i * ((nHeight * nNv12Pitch * 3) >> 1) + (x << 2) +
|
||||
(y << 1) * nNv12Pitch;
|
||||
uchar4 luma2x01, luma2x23, uv2;
|
||||
*(uint32_t *)&luma2x01 = *(uint32_t *)pSrc;
|
||||
*(uint32_t *)&luma2x23 = *(uint32_t *)(pSrc + nNv12Pitch);
|
||||
*(uint32_t *)&uv2 = *(uint32_t *)(pSrc + (nHeight - y) * nNv12Pitch);
|
||||
|
||||
float *pDstBlock = (pBgr + i * ((nHeight * nRgbPitch * 3) >> 2) +
|
||||
((blockIdx.x * blockDim.x) << 2) +
|
||||
((blockIdx.y * blockDim.y) << 1) * (nRgbPitch >> 2));
|
||||
|
||||
float2 add1;
|
||||
float2 add2;
|
||||
float2 add3;
|
||||
float2 add00, add01, add02, add03;
|
||||
float2 d, e;
|
||||
|
||||
add00.x = 1.1644f * luma2x01.x;
|
||||
add01.x = 1.1644f * luma2x01.y;
|
||||
add00.y = 1.1644f * luma2x01.z;
|
||||
add01.y = 1.1644f * luma2x01.w;
|
||||
|
||||
add02.x = 1.1644f * luma2x23.x;
|
||||
add03.x = 1.1644f * luma2x23.y;
|
||||
add02.y = 1.1644f * luma2x23.z;
|
||||
add03.y = 1.1644f * luma2x23.w;
|
||||
|
||||
d.x = uv2.x - 128.0f;
|
||||
e.x = uv2.y - 128.0f;
|
||||
d.y = uv2.z - 128.0f;
|
||||
e.y = uv2.w - 128.0f;
|
||||
|
||||
add1.x = 2.0172f * d.x;
|
||||
add1.y = 2.0172f * d.y;
|
||||
|
||||
add2.x = (-0.3918f) * d.x + (-0.8130f) * e.x;
|
||||
add2.y = (-0.3918f) * d.y + (-0.8130f) * e.y;
|
||||
|
||||
add3.x = 1.5960f * e.x;
|
||||
add3.y = 1.5960f * e.y;
|
||||
|
||||
int rowStride = (threadIdx.y << 1) * (nRgbPitch >> 2);
|
||||
int nextRowStride = ((threadIdx.y << 1) + 1) * (nRgbPitch >> 2);
|
||||
// B
|
||||
*((float4 *)&pDstBlock[rowStride + (threadIdx.x << 2)]) =
|
||||
make_float4(clampF(add00.x + add1.x, 0.0f, 255.0f),
|
||||
clampF(add01.x + add1.x, 0.0f, 255.0f),
|
||||
clampF(add00.y + add1.y, 0.0f, 255.0f),
|
||||
clampF(add01.y + add1.y, 0.0f, 255.0f));
|
||||
*((float4 *)&pDstBlock[nextRowStride + (threadIdx.x << 2)]) =
|
||||
make_float4(clampF(add02.x + add1.x, 0.0f, 255.0f),
|
||||
clampF(add03.x + add1.x, 0.0f, 255.0f),
|
||||
clampF(add02.y + add1.y, 0.0f, 255.0f),
|
||||
clampF(add03.y + add1.y, 0.0f, 255.0f));
|
||||
|
||||
int planeStride = nHeight * nRgbPitch >> 2;
|
||||
// G
|
||||
*((float4 *)&pDstBlock[planeStride + rowStride + (threadIdx.x << 2)]) =
|
||||
make_float4(clampF(add00.x + add2.x, 0.0f, 255.0f),
|
||||
clampF(add01.x + add2.x, 0.0f, 255.0f),
|
||||
clampF(add00.y + add2.y, 0.0f, 255.0f),
|
||||
clampF(add01.y + add2.y, 0.0f, 255.0f));
|
||||
*((float4 *)&pDstBlock[planeStride + nextRowStride + (threadIdx.x << 2)]) =
|
||||
make_float4(clampF(add02.x + add2.x, 0.0f, 255.0f),
|
||||
clampF(add03.x + add2.x, 0.0f, 255.0f),
|
||||
clampF(add02.y + add2.y, 0.0f, 255.0f),
|
||||
clampF(add03.y + add2.y, 0.0f, 255.0f));
|
||||
|
||||
// R
|
||||
*((float4
|
||||
*)&pDstBlock[(planeStride << 1) + rowStride + (threadIdx.x << 2)]) =
|
||||
make_float4(clampF(add00.x + add3.x, 0.0f, 255.0f),
|
||||
clampF(add01.x + add3.x, 0.0f, 255.0f),
|
||||
clampF(add00.y + add3.y, 0.0f, 255.0f),
|
||||
clampF(add01.y + add3.y, 0.0f, 255.0f));
|
||||
*((float4 *)&pDstBlock[(planeStride << 1) + nextRowStride +
|
||||
(threadIdx.x << 2)]) =
|
||||
make_float4(clampF(add02.x + add3.x, 0.0f, 255.0f),
|
||||
clampF(add03.x + add3.x, 0.0f, 255.0f),
|
||||
clampF(add02.y + add3.y, 0.0f, 255.0f),
|
||||
clampF(add03.y + add3.y, 0.0f, 255.0f));
|
||||
}
|
||||
}
|
||||
|
||||
void nv12ToBGRplanarBatch(uint8_t *pNv12, int nNv12Pitch, float *pBgr,
|
||||
int nRgbPitch, int nWidth, int nHeight,
|
||||
int nBatchSize, cudaStream_t stream) {
|
||||
dim3 threads(CONV_THREADS_X, CONV_THREADS_Y);
|
||||
|
||||
size_t blockDimZ = nBatchSize;
|
||||
|
||||
// Restricting blocks in Z-dim till 32 to not launch too many blocks
|
||||
blockDimZ = (blockDimZ > 32) ? 32 : blockDimZ;
|
||||
|
||||
dim3 blocks((nWidth / 4 - 1) / threads.x + 1,
|
||||
(nHeight / 2 - 1) / threads.y + 1, blockDimZ);
|
||||
nv12ToBGRplanarBatchKernel<<<blocks, threads, 0, stream>>>(
|
||||
pNv12, nNv12Pitch, pBgr, nRgbPitch, nWidth, nHeight, nBatchSize);
|
||||
}
|
||||
56
Samples/NV12toBGRandResize/resize_convert.h
Normal file
56
Samples/NV12toBGRandResize/resize_convert.h
Normal file
@@ -0,0 +1,56 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
|
||||
#ifndef __H_RESIZE_CONVERT__
|
||||
#define __H_RESIZE_CONVERT__
|
||||
|
||||
#include <iostream>
|
||||
#include <helper_cuda.h>
|
||||
|
||||
// nv12 resize
|
||||
extern "C"
|
||||
void resizeNV12Batch(
|
||||
uint8_t *dpSrc, int nSrcPitch, int nSrcWidth, int nSrcHeight,
|
||||
uint8_t *dpDst, int nDstPitch, int nDstWidth, int nDstHeight,
|
||||
int nBatchSize, cudaStream_t stream = 0);
|
||||
|
||||
// bgr resize
|
||||
extern "C"
|
||||
void resizeBGRplanarBatch(
|
||||
float *dpSrc, int nSrcPitch, int nSrcWidth, int nSrcHeight,
|
||||
float *dpDst, int nDstPitch, int nDstWidth, int nDstHeight,
|
||||
int nBatchSize, cudaStream_t stream = 0,
|
||||
int cropX = 0, int cropY = 0, int cropW = 0, int cropH = 0,
|
||||
bool whSameResizeRatio = false);
|
||||
|
||||
//NV12 to bgr planar
|
||||
extern "C"
|
||||
void nv12ToBGRplanarBatch(uint8_t *pNv12, int nNv12Pitch,
|
||||
float *pRgb, int nRgbPitch, int nWidth, int nHeight,
|
||||
int nBatchSize, cudaStream_t stream=0);
|
||||
#endif
|
||||
448
Samples/NV12toBGRandResize/resize_convert_main.cpp
Normal file
448
Samples/NV12toBGRandResize/resize_convert_main.cpp
Normal file
@@ -0,0 +1,448 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
|
||||
/*
|
||||
NVIDIA HW Decoder, both dGPU and Tegra, normally outputs NV12 pitch format
|
||||
frames. For the inference using TensorRT, the input frame needs to be BGR planar
|
||||
format with possibly different size. So, conversion and resizing from NV12 to
|
||||
BGR planar is usually required for the inference following decoding.
|
||||
This CUDA code is to provide a reference implementation for conversion and
|
||||
resizing.
|
||||
|
||||
Limitaion
|
||||
=========
|
||||
NV12resize needs the height to be a even value.
|
||||
|
||||
Note
|
||||
====
|
||||
Resize function needs the pitch of image buffer to be 32 alignment.
|
||||
|
||||
Run
|
||||
====
|
||||
./NV12toBGRandResize
|
||||
OR
|
||||
./NV12toBGRandResize -input=data/test1920x1080.nv12 -width=1920 -height=1080 \
|
||||
-dst_width=640 -dst_height=480 -batch=40 -device=0
|
||||
|
||||
*/
|
||||
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#include <math.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
#include <cassert>
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <memory>
|
||||
|
||||
#include "resize_convert.h"
|
||||
#include "utils.h"
|
||||
|
||||
#define TEST_LOOP 20
|
||||
|
||||
typedef struct _nv12_to_bgr24_context_t {
|
||||
int width;
|
||||
int height;
|
||||
int pitch;
|
||||
|
||||
int dst_width;
|
||||
int dst_height;
|
||||
int dst_pitch;
|
||||
|
||||
int batch;
|
||||
int device; // cuda device ID
|
||||
|
||||
char *input_nv12_file;
|
||||
|
||||
int ctx_pitch; // the value will be suitable for Texture memroy.
|
||||
int ctx_heights; // the value will be even.
|
||||
|
||||
} nv12_to_bgr24_context;
|
||||
|
||||
nv12_to_bgr24_context g_ctx;
|
||||
|
||||
static void printHelp(const char *app_name) {
|
||||
std::cout << "Usage:" << app_name << " [options]\n\n";
|
||||
std::cout << "OPTIONS:\n";
|
||||
std::cout << "\t-h,--help\n\n";
|
||||
std::cout << "\t-input=nv12file nv12 input file\n";
|
||||
std::cout
|
||||
<< "\t-width=width input nv12 image width, <1 -- 4096>\n";
|
||||
std::cout
|
||||
<< "\t-height=height input nv12 image height, <1 -- 4096>\n";
|
||||
std::cout
|
||||
<< "\t-pitch=pitch(optional) input nv12 image pitch, <0 -- 4096>\n";
|
||||
std::cout
|
||||
<< "\t-dst_width=width output BGR image width, <1 -- 4096>\n";
|
||||
std::cout
|
||||
<< "\t-dst_height=height output BGR image height, <1 -- 4096>\n";
|
||||
std::cout
|
||||
<< "\t-dst_pitch=pitch(optional) output BGR image pitch, <0 -- 4096>\n";
|
||||
std::cout
|
||||
<< "\t-batch=batch process frames count, <1 -- 4096>\n\n";
|
||||
std::cout
|
||||
<< "\t-device=device_num(optional) cuda device number, <0 -- 4096>\n\n";
|
||||
|
||||
return;
|
||||
}
|
||||
|
||||
int parseCmdLine(int argc, char *argv[]) {
|
||||
char **argp = (char **)argv;
|
||||
char *arg = (char *)argv[0];
|
||||
|
||||
memset(&g_ctx, 0, sizeof(g_ctx));
|
||||
|
||||
if ((arg && (!strcmp(arg, "-h") || !strcmp(arg, "--help")))) {
|
||||
printHelp(argv[0]);
|
||||
return -1;
|
||||
}
|
||||
|
||||
if (argc == 1) {
|
||||
// Run using default arguments
|
||||
|
||||
g_ctx.input_nv12_file = sdkFindFilePath("test1920x1080.nv12", argv[0]);
|
||||
if (g_ctx.input_nv12_file == NULL) {
|
||||
printf("Cannot find input file test1920x1080.nv12\n Exiting\n");
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
g_ctx.width = 1920;
|
||||
g_ctx.height = 1080;
|
||||
g_ctx.dst_width = 640;
|
||||
g_ctx.dst_height = 480;
|
||||
g_ctx.batch = 24;
|
||||
} else if (argc > 1) {
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "width")) {
|
||||
g_ctx.width = getCmdLineArgumentInt(argc, (const char **)argv, "width");
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "height")) {
|
||||
g_ctx.height = getCmdLineArgumentInt(argc, (const char **)argv, "height");
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "pitch")) {
|
||||
g_ctx.pitch = getCmdLineArgumentInt(argc, (const char **)argv, "pitch");
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "input")) {
|
||||
getCmdLineArgumentString(argc, (const char **)argv, "input",
|
||||
(char **)&g_ctx.input_nv12_file);
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "dst_width")) {
|
||||
g_ctx.dst_width =
|
||||
getCmdLineArgumentInt(argc, (const char **)argv, "dst_width");
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "dst_height")) {
|
||||
g_ctx.dst_height =
|
||||
getCmdLineArgumentInt(argc, (const char **)argv, "dst_height");
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "dst_pitch")) {
|
||||
g_ctx.dst_pitch =
|
||||
getCmdLineArgumentInt(argc, (const char **)argv, "dst_pitch");
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "batch")) {
|
||||
g_ctx.batch = getCmdLineArgumentInt(argc, (const char **)argv, "batch");
|
||||
}
|
||||
}
|
||||
|
||||
g_ctx.device = findCudaDevice(argc, (const char **)argv);
|
||||
|
||||
if ((g_ctx.width == 0) || (g_ctx.height == 0) || (g_ctx.dst_width == 0) ||
|
||||
(g_ctx.dst_height == 0) || !g_ctx.input_nv12_file) {
|
||||
printHelp(argv[0]);
|
||||
return -1;
|
||||
}
|
||||
|
||||
if (g_ctx.pitch == 0) g_ctx.pitch = g_ctx.width;
|
||||
if (g_ctx.dst_pitch == 0) g_ctx.dst_pitch = g_ctx.dst_width;
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
/*
|
||||
load nv12 yuvfile data into GPU device memory with batch of copy
|
||||
*/
|
||||
static int loadNV12Frame(unsigned char *d_inputNV12) {
|
||||
unsigned char *pNV12FrameData;
|
||||
unsigned char *d_nv12;
|
||||
int frameSize;
|
||||
std::ifstream nv12File(g_ctx.input_nv12_file, std::ifstream::in | std::ios::binary);
|
||||
|
||||
if (!nv12File.is_open()) {
|
||||
std::cerr << "Can't open files\n";
|
||||
return -1;
|
||||
}
|
||||
|
||||
frameSize = g_ctx.pitch * g_ctx.ctx_heights;
|
||||
|
||||
#if USE_UVM_MEM
|
||||
pNV12FrameData = d_inputNV12;
|
||||
#else
|
||||
pNV12FrameData = (unsigned char *)malloc(frameSize);
|
||||
if (pNV12FrameData == NULL) {
|
||||
std::cerr << "Failed to malloc pNV12FrameData\n";
|
||||
return -1;
|
||||
}
|
||||
#endif
|
||||
|
||||
nv12File.read((char *)pNV12FrameData, frameSize);
|
||||
|
||||
if (nv12File.gcount() < frameSize) {
|
||||
std::cerr << "can't get one frame!\n";
|
||||
return -1;
|
||||
}
|
||||
|
||||
#if USE_UVM_MEM
|
||||
// Prefetch to GPU for following GPU operation
|
||||
cudaStreamAttachMemAsync(NULL, pNV12FrameData, 0, cudaMemAttachGlobal);
|
||||
#endif
|
||||
|
||||
// expand one frame to multi frames for batch processing
|
||||
d_nv12 = d_inputNV12;
|
||||
for (int i = 0; i < g_ctx.batch; i++) {
|
||||
checkCudaErrors(cudaMemcpy2D((void *)d_nv12, g_ctx.ctx_pitch,
|
||||
pNV12FrameData, g_ctx.width, g_ctx.width,
|
||||
g_ctx.ctx_heights, cudaMemcpyHostToDevice));
|
||||
|
||||
d_nv12 += g_ctx.ctx_pitch * g_ctx.ctx_heights;
|
||||
}
|
||||
|
||||
#if (USE_UVM_MEM == 0)
|
||||
free(pNV12FrameData);
|
||||
#endif
|
||||
nv12File.close();
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
/*
|
||||
1. resize interlace nv12 to target size
|
||||
2. convert nv12 to bgr 3 progressive planars
|
||||
*/
|
||||
void nv12ResizeAndNV12ToBGR(unsigned char *d_inputNV12) {
|
||||
unsigned char *d_resizedNV12;
|
||||
float *d_outputBGR;
|
||||
int size;
|
||||
char filename[40];
|
||||
|
||||
/* allocate device memory for resized nv12 output */
|
||||
size = g_ctx.dst_width * ceil(g_ctx.dst_height * 3.0f / 2.0f) * g_ctx.batch *
|
||||
sizeof(unsigned char);
|
||||
checkCudaErrors(cudaMalloc((void **)&d_resizedNV12, size));
|
||||
|
||||
/* allocate device memory for bgr output */
|
||||
size = g_ctx.dst_pitch * g_ctx.dst_height * 3 * g_ctx.batch * sizeof(float);
|
||||
checkCudaErrors(cudaMalloc((void **)&d_outputBGR, size));
|
||||
|
||||
cudaStream_t stream;
|
||||
checkCudaErrors(cudaStreamCreate(&stream));
|
||||
/* create cuda event handles */
|
||||
cudaEvent_t start, stop;
|
||||
checkCudaErrors(cudaEventCreate(&start));
|
||||
checkCudaErrors(cudaEventCreate(&stop));
|
||||
float elapsedTime = 0.0f;
|
||||
|
||||
/* resize interlace nv12 */
|
||||
|
||||
cudaEventRecord(start, 0);
|
||||
for (int i = 0; i < TEST_LOOP; i++) {
|
||||
resizeNV12Batch(d_inputNV12, g_ctx.ctx_pitch, g_ctx.width, g_ctx.height,
|
||||
d_resizedNV12, g_ctx.dst_width, g_ctx.dst_width,
|
||||
g_ctx.dst_height, g_ctx.batch);
|
||||
}
|
||||
cudaEventRecord(stop, 0);
|
||||
cudaEventSynchronize(stop);
|
||||
|
||||
cudaEventElapsedTime(&elapsedTime, start, stop);
|
||||
printf(
|
||||
" CUDA resize nv12(%dx%d --> %dx%d), batch: %d,"
|
||||
" average time: %.3f ms ==> %.3f ms/frame\n",
|
||||
g_ctx.width, g_ctx.height, g_ctx.dst_width, g_ctx.dst_height, g_ctx.batch,
|
||||
(elapsedTime / (TEST_LOOP * 1.0f)),
|
||||
(elapsedTime / (TEST_LOOP * 1.0f)) / g_ctx.batch);
|
||||
|
||||
sprintf(filename, "resized_nv12_%dx%d", g_ctx.dst_width, g_ctx.dst_height);
|
||||
|
||||
/* convert nv12 to bgr 3 progressive planars */
|
||||
cudaEventRecord(start, 0);
|
||||
for (int i = 0; i < TEST_LOOP; i++) {
|
||||
nv12ToBGRplanarBatch(d_resizedNV12, g_ctx.dst_pitch, // intput
|
||||
d_outputBGR,
|
||||
g_ctx.dst_pitch * sizeof(float), // output
|
||||
g_ctx.dst_width, g_ctx.dst_height, // output
|
||||
g_ctx.batch, 0);
|
||||
}
|
||||
cudaEventRecord(stop, 0);
|
||||
cudaEventSynchronize(stop);
|
||||
|
||||
cudaEventElapsedTime(&elapsedTime, start, stop);
|
||||
|
||||
printf(
|
||||
" CUDA convert nv12(%dx%d) to bgr(%dx%d), batch: %d,"
|
||||
" average time: %.3f ms ==> %.3f ms/frame\n",
|
||||
g_ctx.dst_width, g_ctx.dst_height, g_ctx.dst_width, g_ctx.dst_height,
|
||||
g_ctx.batch, (elapsedTime / (TEST_LOOP * 1.0f)),
|
||||
(elapsedTime / (TEST_LOOP * 1.0f)) / g_ctx.batch);
|
||||
|
||||
sprintf(filename, "converted_bgr_%dx%d", g_ctx.dst_width, g_ctx.dst_height);
|
||||
dumpBGR(d_outputBGR, g_ctx.dst_pitch, g_ctx.dst_width, g_ctx.dst_height,
|
||||
g_ctx.batch, (char *)"t1", filename);
|
||||
|
||||
/* release resources */
|
||||
checkCudaErrors(cudaEventDestroy(start));
|
||||
checkCudaErrors(cudaEventDestroy(stop));
|
||||
checkCudaErrors(cudaStreamDestroy(stream));
|
||||
checkCudaErrors(cudaFree(d_resizedNV12));
|
||||
checkCudaErrors(cudaFree(d_outputBGR));
|
||||
}
|
||||
|
||||
/*
|
||||
1. convert nv12 to bgr 3 progressive planars
|
||||
2. resize bgr 3 planars to target size
|
||||
*/
|
||||
void nv12ToBGRandBGRresize(unsigned char *d_inputNV12) {
|
||||
float *d_bgr;
|
||||
float *d_resizedBGR;
|
||||
int size;
|
||||
char filename[40];
|
||||
|
||||
/* allocate device memory for bgr output */
|
||||
size = g_ctx.ctx_pitch * g_ctx.height * 3 * g_ctx.batch * sizeof(float);
|
||||
checkCudaErrors(cudaMalloc((void **)&d_bgr, size));
|
||||
|
||||
/* allocate device memory for resized bgr output */
|
||||
size = g_ctx.dst_width * g_ctx.dst_height * 3 * g_ctx.batch * sizeof(float);
|
||||
checkCudaErrors(cudaMalloc((void **)&d_resizedBGR, size));
|
||||
|
||||
cudaStream_t stream;
|
||||
checkCudaErrors(cudaStreamCreate(&stream));
|
||||
/* create cuda event handles */
|
||||
cudaEvent_t start, stop;
|
||||
checkCudaErrors(cudaEventCreate(&start));
|
||||
checkCudaErrors(cudaEventCreate(&stop));
|
||||
float elapsedTime = 0.0f;
|
||||
|
||||
/* convert interlace nv12 to bgr 3 progressive planars */
|
||||
cudaEventRecord(start, 0);
|
||||
cudaDeviceSynchronize();
|
||||
for (int i = 0; i < TEST_LOOP; i++) {
|
||||
nv12ToBGRplanarBatch(d_inputNV12, g_ctx.ctx_pitch, d_bgr,
|
||||
g_ctx.ctx_pitch * sizeof(float), g_ctx.width,
|
||||
g_ctx.height, g_ctx.batch, 0);
|
||||
}
|
||||
cudaEventRecord(stop, 0);
|
||||
cudaEventSynchronize(stop);
|
||||
|
||||
cudaEventElapsedTime(&elapsedTime, start, stop);
|
||||
printf(
|
||||
" CUDA convert nv12(%dx%d) to bgr(%dx%d), batch: %d,"
|
||||
" average time: %.3f ms ==> %.3f ms/frame\n",
|
||||
g_ctx.width, g_ctx.height, g_ctx.width, g_ctx.height, g_ctx.batch,
|
||||
(elapsedTime / (TEST_LOOP * 1.0f)),
|
||||
(elapsedTime / (TEST_LOOP * 1.0f)) / g_ctx.batch);
|
||||
|
||||
sprintf(filename, "converted_bgr_%dx%d", g_ctx.width, g_ctx.height);
|
||||
|
||||
/* resize bgr 3 progressive planars */
|
||||
cudaEventRecord(start, 0);
|
||||
for (int i = 0; i < TEST_LOOP; i++) {
|
||||
resizeBGRplanarBatch(d_bgr, g_ctx.ctx_pitch, g_ctx.width, g_ctx.height,
|
||||
d_resizedBGR, g_ctx.dst_width, g_ctx.dst_width,
|
||||
g_ctx.dst_height, g_ctx.batch);
|
||||
}
|
||||
cudaEventRecord(stop, 0);
|
||||
cudaEventSynchronize(stop);
|
||||
|
||||
cudaEventElapsedTime(&elapsedTime, start, stop);
|
||||
printf(
|
||||
" CUDA resize bgr(%dx%d --> %dx%d), batch: %d,"
|
||||
" average time: %.3f ms ==> %.3f ms/frame\n",
|
||||
g_ctx.width, g_ctx.height, g_ctx.dst_width, g_ctx.dst_height, g_ctx.batch,
|
||||
(elapsedTime / (TEST_LOOP * 1.0f)),
|
||||
(elapsedTime / (TEST_LOOP * 1.0f)) / g_ctx.batch);
|
||||
|
||||
memset(filename, 0, sizeof(filename));
|
||||
sprintf(filename, "resized_bgr_%dx%d", g_ctx.dst_width, g_ctx.dst_height);
|
||||
dumpBGR(d_resizedBGR, g_ctx.dst_pitch, g_ctx.dst_width, g_ctx.dst_height,
|
||||
g_ctx.batch, (char *)"t2", filename);
|
||||
|
||||
/* release resources */
|
||||
checkCudaErrors(cudaEventDestroy(start));
|
||||
checkCudaErrors(cudaEventDestroy(stop));
|
||||
checkCudaErrors(cudaStreamDestroy(stream));
|
||||
checkCudaErrors(cudaFree(d_bgr));
|
||||
checkCudaErrors(cudaFree(d_resizedBGR));
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
unsigned char *d_inputNV12;
|
||||
|
||||
if (parseCmdLine(argc, argv) < 0) return EXIT_FAILURE;
|
||||
|
||||
g_ctx.ctx_pitch = g_ctx.width;
|
||||
int ctx_alignment = 32;
|
||||
g_ctx.ctx_pitch += (g_ctx.ctx_pitch % ctx_alignment != 0)
|
||||
? (ctx_alignment - g_ctx.ctx_pitch % ctx_alignment)
|
||||
: 0;
|
||||
|
||||
g_ctx.ctx_heights = ceil(g_ctx.height * 3.0f / 2.0f);
|
||||
|
||||
/* load nv12 yuv data into d_inputNV12 with batch of copies */
|
||||
#if USE_UVM_MEM
|
||||
checkCudaErrors(cudaMallocManaged(
|
||||
(void **)&d_inputNV12,
|
||||
(g_ctx.ctx_pitch * g_ctx.ctx_heights * g_ctx.batch), cudaMemAttachHost));
|
||||
printf("\nUSE_UVM_MEM\n");
|
||||
#else
|
||||
checkCudaErrors(
|
||||
cudaMalloc((void **)&d_inputNV12,
|
||||
(g_ctx.ctx_pitch * g_ctx.ctx_heights * g_ctx.batch)));
|
||||
#endif
|
||||
if (loadNV12Frame(d_inputNV12)) {
|
||||
std::cerr << "failed to load batch data!\n";
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
/* firstly resize nv12, then convert nv12 to bgr */
|
||||
printf("\nTEST#1:\n");
|
||||
nv12ResizeAndNV12ToBGR(d_inputNV12);
|
||||
|
||||
/* first convert nv12 to bgr, then resize bgr */
|
||||
printf("\nTEST#2:\n");
|
||||
nv12ToBGRandBGRresize(d_inputNV12);
|
||||
|
||||
checkCudaErrors(cudaFree(d_inputNV12));
|
||||
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
152
Samples/NV12toBGRandResize/utils.cu
Normal file
152
Samples/NV12toBGRandResize/utils.cu
Normal file
@@ -0,0 +1,152 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
#include <stdlib.h>
|
||||
#include <sys/stat.h>
|
||||
#include <sys/types.h>
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
|
||||
#include <cuda.h>
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#include "resize_convert.h"
|
||||
#include "utils.h"
|
||||
|
||||
__global__ void floatToChar(float *src, unsigned char *dst, int height,
|
||||
int width, int batchSize) {
|
||||
int x = threadIdx.x + blockIdx.x * blockDim.x;
|
||||
|
||||
if (x >= height * width) return;
|
||||
|
||||
int offset = height * width * 3;
|
||||
|
||||
for (int j = 0; j < batchSize; j++) {
|
||||
// b
|
||||
*(dst + j * offset + x * 3 + 0) =
|
||||
(unsigned char)*(src + j * offset + height * width * 0 + x);
|
||||
// g
|
||||
*(dst + j * offset + x * 3 + 1) =
|
||||
(unsigned char)*(src + j * offset + height * width * 1 + x);
|
||||
// r
|
||||
*(dst + j * offset + x * 3 + 2) =
|
||||
(unsigned char)*(src + j * offset + height * width * 2 + x);
|
||||
}
|
||||
}
|
||||
|
||||
void floatPlanarToChar(float *src, unsigned char *dst, int height, int width,
|
||||
int batchSize) {
|
||||
floatToChar<<<(height * width - 1) / 1024 + 1, 1024, 0, NULL>>>(
|
||||
src, dst, height, width, batchSize);
|
||||
}
|
||||
|
||||
void dumpRawBGR(float *d_srcBGR, int pitch, int width, int height,
|
||||
int batchSize, char *folder, char *tag) {
|
||||
float *bgr, *d_bgr;
|
||||
int frameSize;
|
||||
char directory[120];
|
||||
char mkdir_cmd[256];
|
||||
#if !defined(_WIN32)
|
||||
sprintf(directory, "output/%s", folder);
|
||||
sprintf(mkdir_cmd, "mkdir -p %s 2> /dev/null", directory);
|
||||
#else
|
||||
sprintf(directory, "output\\%s", folder);
|
||||
sprintf(mkdir_cmd, "mkdir %s 2> nul", directory);
|
||||
#endif
|
||||
|
||||
int ret = system(mkdir_cmd);
|
||||
|
||||
frameSize = width * height * 3 * sizeof(float);
|
||||
bgr = (float *)malloc(frameSize);
|
||||
if (bgr == NULL) {
|
||||
std::cerr << "Failed malloc for bgr\n";
|
||||
return;
|
||||
}
|
||||
|
||||
d_bgr = d_srcBGR;
|
||||
for (int i = 0; i < batchSize; i++) {
|
||||
char filename[120];
|
||||
std::ofstream *outputFile;
|
||||
|
||||
checkCudaErrors(cudaMemcpy((void *)bgr, (void *)d_bgr, frameSize,
|
||||
cudaMemcpyDeviceToHost));
|
||||
sprintf(filename, "%s/%s_%d.raw", directory, tag, (i + 1));
|
||||
|
||||
outputFile = new std::ofstream(filename);
|
||||
if (outputFile) {
|
||||
outputFile->write((char *)bgr, frameSize);
|
||||
delete outputFile;
|
||||
}
|
||||
|
||||
d_bgr += pitch * height * 3;
|
||||
}
|
||||
|
||||
free(bgr);
|
||||
}
|
||||
|
||||
void dumpBGR(float *d_srcBGR, int pitch, int width, int height, int batchSize,
|
||||
char *folder, char *tag) {
|
||||
dumpRawBGR(d_srcBGR, pitch, width, height, batchSize, folder, tag);
|
||||
}
|
||||
|
||||
void dumpYUV(unsigned char *d_nv12, int size, char *folder, char *tag) {
|
||||
unsigned char *nv12Data;
|
||||
std::ofstream *nv12File;
|
||||
char filename[120];
|
||||
char directory[60];
|
||||
char mkdir_cmd[256];
|
||||
#if !defined(_WIN32)
|
||||
sprintf(directory, "output/%s", folder);
|
||||
sprintf(mkdir_cmd, "mkdir -p %s 2> /dev/null", directory);
|
||||
#else
|
||||
sprintf(directory, "output\\%s", folder);
|
||||
sprintf(mkdir_cmd, "mkdir %s 2> nul", directory);
|
||||
#endif
|
||||
|
||||
int ret = system(mkdir_cmd);
|
||||
|
||||
sprintf(filename, "%s/%s.nv12", directory, tag);
|
||||
|
||||
nv12File = new std::ofstream(filename);
|
||||
if (nv12File == NULL) {
|
||||
std::cerr << "Failed to new " << filename;
|
||||
return;
|
||||
}
|
||||
|
||||
nv12Data = (unsigned char *)malloc(size * (sizeof(char)));
|
||||
if (nv12Data == NULL) {
|
||||
std::cerr << "Failed to allcoate memory\n";
|
||||
return;
|
||||
}
|
||||
|
||||
cudaMemcpy((void *)nv12Data, (void *)d_nv12, size, cudaMemcpyDeviceToHost);
|
||||
|
||||
nv12File->write((const char *)nv12Data, size);
|
||||
|
||||
free(nv12Data);
|
||||
delete nv12File;
|
||||
}
|
||||
37
Samples/NV12toBGRandResize/utils.h
Normal file
37
Samples/NV12toBGRandResize/utils.h
Normal file
@@ -0,0 +1,37 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
|
||||
#ifndef __H_UTIL_
|
||||
#define __H_UTIL_
|
||||
|
||||
extern "C"
|
||||
void dumpBGR(float *d_srcBGR, int pitch, int width, int height,
|
||||
int batchSize, char *folder, char *tag);
|
||||
extern "C"
|
||||
void dumpYUV(unsigned char *d_nv12, int size, char *folder, char *tag);
|
||||
#endif
|
||||
Reference in New Issue
Block a user