Add and update samples with CUDA 10.1 support
301
Samples/nvJPEG/Makefile
Normal file
@@ -0,0 +1,301 @@
|
||||
################################################################################
|
||||
# Copyright (c) 2018, NVIDIA CORPORATION. All rights reserved.
|
||||
#
|
||||
# Redistribution and use in source and binary forms, with or without
|
||||
# modification, are permitted provided that the following conditions
|
||||
# are met:
|
||||
# * Redistributions of source code must retain the above copyright
|
||||
# notice, this list of conditions and the following disclaimer.
|
||||
# * Redistributions in binary form must reproduce the above copyright
|
||||
# notice, this list of conditions and the following disclaimer in the
|
||||
# documentation and/or other materials provided with the distribution.
|
||||
# * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
# contributors may be used to endorse or promote products derived
|
||||
# from this software without specific prior written permission.
|
||||
#
|
||||
# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
# EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
# PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
# CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
# EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
# PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
# PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
# OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#
|
||||
################################################################################
|
||||
#
|
||||
# Makefile project only supported on Mac OS X and Linux Platforms)
|
||||
#
|
||||
################################################################################
|
||||
|
||||
# Location of the CUDA Toolkit
|
||||
CUDA_PATH ?= /usr/local/cuda
|
||||
|
||||
##############################
|
||||
# start deprecated interface #
|
||||
##############################
|
||||
ifeq ($(x86_64),1)
|
||||
$(info WARNING - x86_64 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=x86_64 instead)
|
||||
TARGET_ARCH ?= x86_64
|
||||
endif
|
||||
ifeq ($(ARMv7),1)
|
||||
$(info WARNING - ARMv7 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=armv7l instead)
|
||||
TARGET_ARCH ?= armv7l
|
||||
endif
|
||||
ifeq ($(aarch64),1)
|
||||
$(info WARNING - aarch64 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=aarch64 instead)
|
||||
TARGET_ARCH ?= aarch64
|
||||
endif
|
||||
ifeq ($(ppc64le),1)
|
||||
$(info WARNING - ppc64le variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=ppc64le instead)
|
||||
TARGET_ARCH ?= ppc64le
|
||||
endif
|
||||
ifneq ($(GCC),)
|
||||
$(info WARNING - GCC variable has been deprecated)
|
||||
$(info WARNING - please use HOST_COMPILER=$(GCC) instead)
|
||||
HOST_COMPILER ?= $(GCC)
|
||||
endif
|
||||
ifneq ($(abi),)
|
||||
$(error ERROR - abi variable has been removed)
|
||||
endif
|
||||
############################
|
||||
# end deprecated interface #
|
||||
############################
|
||||
|
||||
# architecture
|
||||
HOST_ARCH := $(shell uname -m)
|
||||
TARGET_ARCH ?= $(HOST_ARCH)
|
||||
ifneq (,$(filter $(TARGET_ARCH),x86_64 aarch64 ppc64le armv7l))
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifneq (,$(filter $(TARGET_ARCH),x86_64 aarch64 ppc64le))
|
||||
TARGET_SIZE := 64
|
||||
else ifneq (,$(filter $(TARGET_ARCH),armv7l))
|
||||
TARGET_SIZE := 32
|
||||
endif
|
||||
else
|
||||
TARGET_SIZE := $(shell getconf LONG_BIT)
|
||||
endif
|
||||
else
|
||||
$(error ERROR - unsupported value $(TARGET_ARCH) for TARGET_ARCH!)
|
||||
endif
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq (,$(filter $(HOST_ARCH)-$(TARGET_ARCH),aarch64-armv7l x86_64-armv7l x86_64-aarch64 x86_64-ppc64le))
|
||||
$(error ERROR - cross compiling from $(HOST_ARCH) to $(TARGET_ARCH) is not supported!)
|
||||
endif
|
||||
endif
|
||||
|
||||
# When on native aarch64 system with userspace of 32-bit, change TARGET_ARCH to armv7l
|
||||
ifeq ($(HOST_ARCH)-$(TARGET_ARCH)-$(TARGET_SIZE),aarch64-aarch64-32)
|
||||
TARGET_ARCH = armv7l
|
||||
endif
|
||||
|
||||
# operating system
|
||||
HOST_OS := $(shell uname -s 2>/dev/null | tr "[:upper:]" "[:lower:]")
|
||||
TARGET_OS ?= $(HOST_OS)
|
||||
ifeq (,$(filter $(TARGET_OS),linux darwin qnx android))
|
||||
$(error ERROR - unsupported value $(TARGET_OS) for TARGET_OS!)
|
||||
endif
|
||||
|
||||
# host compiler
|
||||
ifeq ($(TARGET_OS),darwin)
|
||||
ifeq ($(shell expr `xcodebuild -version | grep -i xcode | awk '{print $$2}' | cut -d'.' -f1` \>= 5),1)
|
||||
HOST_COMPILER ?= clang++
|
||||
endif
|
||||
else ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq ($(HOST_ARCH)-$(TARGET_ARCH),x86_64-armv7l)
|
||||
ifeq ($(TARGET_OS),linux)
|
||||
HOST_COMPILER ?= arm-linux-gnueabihf-g++
|
||||
else ifeq ($(TARGET_OS),qnx)
|
||||
ifeq ($(QNX_HOST),)
|
||||
$(error ERROR - QNX_HOST must be passed to the QNX host toolchain)
|
||||
endif
|
||||
ifeq ($(QNX_TARGET),)
|
||||
$(error ERROR - QNX_TARGET must be passed to the QNX target toolchain)
|
||||
endif
|
||||
export QNX_HOST
|
||||
export QNX_TARGET
|
||||
HOST_COMPILER ?= $(QNX_HOST)/usr/bin/arm-unknown-nto-qnx6.6.0eabi-g++
|
||||
else ifeq ($(TARGET_OS),android)
|
||||
HOST_COMPILER ?= arm-linux-androideabi-g++
|
||||
endif
|
||||
else ifeq ($(TARGET_ARCH),aarch64)
|
||||
ifeq ($(TARGET_OS), linux)
|
||||
HOST_COMPILER ?= aarch64-linux-gnu-g++
|
||||
else ifeq ($(TARGET_OS),qnx)
|
||||
ifeq ($(QNX_HOST),)
|
||||
$(error ERROR - QNX_HOST must be passed to the QNX host toolchain)
|
||||
endif
|
||||
ifeq ($(QNX_TARGET),)
|
||||
$(error ERROR - QNX_TARGET must be passed to the QNX target toolchain)
|
||||
endif
|
||||
export QNX_HOST
|
||||
export QNX_TARGET
|
||||
HOST_COMPILER ?= $(QNX_HOST)/usr/bin/aarch64-unknown-nto-qnx7.0.0-g++
|
||||
else ifeq ($(TARGET_OS), android)
|
||||
HOST_COMPILER ?= aarch64-linux-android-clang++
|
||||
endif
|
||||
else ifeq ($(TARGET_ARCH),ppc64le)
|
||||
HOST_COMPILER ?= powerpc64le-linux-gnu-g++
|
||||
endif
|
||||
endif
|
||||
HOST_COMPILER ?= g++
|
||||
NVCC := $(CUDA_PATH)/bin/nvcc -ccbin $(HOST_COMPILER)
|
||||
|
||||
# internal flags
|
||||
NVCCFLAGS := -m${TARGET_SIZE}
|
||||
CCFLAGS :=
|
||||
LDFLAGS :=
|
||||
|
||||
# build flags
|
||||
ifeq ($(TARGET_OS),darwin)
|
||||
LDFLAGS += -rpath $(CUDA_PATH)/lib
|
||||
CCFLAGS += -arch $(HOST_ARCH)
|
||||
else ifeq ($(HOST_ARCH)-$(TARGET_ARCH)-$(TARGET_OS),x86_64-armv7l-linux)
|
||||
LDFLAGS += --dynamic-linker=/lib/ld-linux-armhf.so.3
|
||||
CCFLAGS += -mfloat-abi=hard
|
||||
else ifeq ($(TARGET_OS),android)
|
||||
LDFLAGS += -pie
|
||||
CCFLAGS += -fpie -fpic -fexceptions
|
||||
endif
|
||||
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-linux)
|
||||
ifneq ($(TARGET_FS),)
|
||||
GCCVERSIONLTEQ46 := $(shell expr `$(HOST_COMPILER) -dumpversion` \<= 4.6)
|
||||
ifeq ($(GCCVERSIONLTEQ46),1)
|
||||
CCFLAGS += --sysroot=$(TARGET_FS)
|
||||
endif
|
||||
LDFLAGS += --sysroot=$(TARGET_FS)
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib/arm-linux-gnueabihf
|
||||
endif
|
||||
endif
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-linux)
|
||||
ifneq ($(TARGET_FS),)
|
||||
GCCVERSIONLTEQ46 := $(shell expr `$(HOST_COMPILER) -dumpversion` \<= 4.6)
|
||||
ifeq ($(GCCVERSIONLTEQ46),1)
|
||||
CCFLAGS += --sysroot=$(TARGET_FS)
|
||||
endif
|
||||
LDFLAGS += --sysroot=$(TARGET_FS)
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/lib -L $(TARGET_FS)/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib -L $(TARGET_FS)/usr/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib/aarch64-linux-gnu -L $(TARGET_FS)/usr/lib/aarch64-linux-gnu
|
||||
LDFLAGS += --unresolved-symbols=ignore-in-shared-libs
|
||||
CCFLAGS += -isystem=$(TARGET_FS)/usr/include
|
||||
CCFLAGS += -isystem=$(TARGET_FS)/usr/include/aarch64-linux-gnu
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(TARGET_OS),qnx)
|
||||
CCFLAGS += -DWIN_INTERFACE_CUSTOM
|
||||
LDFLAGS += -lsocket
|
||||
endif
|
||||
|
||||
# Install directory of different arch
|
||||
CUDA_INSTALL_TARGET_DIR :=
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-linux)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/armv7-linux-gnueabihf/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-linux)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-linux/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-android)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/armv7-linux-androideabi/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-android)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-linux-androideabi/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-qnx)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/ARMv7-linux-QNX/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-qnx)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-qnx/
|
||||
else ifeq ($(TARGET_ARCH),ppc64le)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/ppc64le-linux/
|
||||
endif
|
||||
|
||||
# Debug build flags
|
||||
ifeq ($(dbg),1)
|
||||
NVCCFLAGS += -g -G
|
||||
BUILD_TYPE := debug
|
||||
else
|
||||
BUILD_TYPE := release
|
||||
endif
|
||||
|
||||
ALL_CCFLAGS :=
|
||||
ALL_CCFLAGS += $(NVCCFLAGS)
|
||||
ALL_CCFLAGS += $(EXTRA_NVCCFLAGS)
|
||||
ALL_CCFLAGS += $(addprefix -Xcompiler ,$(CCFLAGS))
|
||||
ALL_CCFLAGS += $(addprefix -Xcompiler ,$(EXTRA_CCFLAGS))
|
||||
|
||||
SAMPLE_ENABLED := 1
|
||||
|
||||
# This sample is not supported on Mac OSX
|
||||
ifeq ($(TARGET_OS),darwin)
|
||||
$(info >>> WARNING - nvJPEG is not supported on Mac OSX - waiving sample <<<)
|
||||
SAMPLE_ENABLED := 0
|
||||
endif
|
||||
|
||||
# This sample is not supported on ARMv7
|
||||
ifeq ($(TARGET_ARCH),armv7l)
|
||||
$(info >>> WARNING - nvJPEG is not supported on ARMv7 - waiving sample <<<)
|
||||
SAMPLE_ENABLED := 0
|
||||
endif
|
||||
|
||||
# This sample is not supported on aarch64
|
||||
ifeq ($(TARGET_ARCH),aarch64)
|
||||
$(info >>> WARNING - nvJPEG is not supported on aarch64 - waiving sample <<<)
|
||||
SAMPLE_ENABLED := 0
|
||||
endif
|
||||
|
||||
ALL_LDFLAGS :=
|
||||
ALL_LDFLAGS += $(ALL_CCFLAGS)
|
||||
ALL_LDFLAGS += $(addprefix -Xlinker ,$(LDFLAGS))
|
||||
ALL_LDFLAGS += $(addprefix -Xlinker ,$(EXTRA_LDFLAGS))
|
||||
|
||||
# Common includes and paths for CUDA
|
||||
INCLUDES := -I../../Common
|
||||
LIBRARIES :=
|
||||
|
||||
################################################################################
|
||||
|
||||
LIBRARIES += -lnvjpeg
|
||||
|
||||
ifeq ($(SAMPLE_ENABLED),0)
|
||||
EXEC ?= @echo "[@]"
|
||||
endif
|
||||
|
||||
################################################################################
|
||||
|
||||
# Target rules
|
||||
all: build
|
||||
|
||||
build: nvJPEG
|
||||
|
||||
check.deps:
|
||||
ifeq ($(SAMPLE_ENABLED),0)
|
||||
@echo "Sample will be waived due to the above missing dependencies"
|
||||
else
|
||||
@echo "Sample is ready - all dependencies have been met"
|
||||
endif
|
||||
|
||||
nvJPEG.o:nvJPEG.cpp
|
||||
$(EXEC) $(NVCC) $(INCLUDES) $(ALL_CCFLAGS) $(GENCODE_FLAGS) -o $@ -c $<
|
||||
|
||||
nvJPEG: nvJPEG.o
|
||||
$(EXEC) $(NVCC) $(ALL_LDFLAGS) $(GENCODE_FLAGS) -o $@ $+ $(LIBRARIES)
|
||||
$(EXEC) mkdir -p ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)
|
||||
$(EXEC) cp $@ ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)
|
||||
|
||||
run: build
|
||||
$(EXEC) ./nvJPEG
|
||||
|
||||
clean:
|
||||
rm -f nvJPEG nvJPEG.o
|
||||
rm -rf ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)/nvJPEG
|
||||
|
||||
clobber: clean
|
||||
58
Samples/nvJPEG/NsightEclipse.xml
Normal file
@@ -0,0 +1,58 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE entry SYSTEM "SamplesInfo.dtd">
|
||||
<entry>
|
||||
<name>nvJPEG</name>
|
||||
<description><![CDATA[A CUDA Sample that demonstrates single and batched decoding of jpeg images using NVJPEG Library.]]></description>
|
||||
<devicecompilation>whole</devicecompilation>
|
||||
<includepaths>
|
||||
<path>./</path>
|
||||
<path>../</path>
|
||||
<path>../../common/inc</path>
|
||||
</includepaths>
|
||||
<keyconcepts>
|
||||
<concept level="basic">Image Decoding</concept>
|
||||
<concept level="basic">NVJPEG Library</concept>
|
||||
</keyconcepts>
|
||||
<keywords>
|
||||
<keyword>NVJPEG</keyword>
|
||||
<keyword>JPEG Decoding</keyword>
|
||||
</keywords>
|
||||
<libraries>
|
||||
<library>nvjpeg</library>
|
||||
</libraries>
|
||||
<librarypaths>
|
||||
</librarypaths>
|
||||
<nsight_eclipse>true</nsight_eclipse>
|
||||
<primary_file>nvJPEG.cpp</primary_file>
|
||||
<qatests>
|
||||
<qatest>-i ../../../../Samples/nvJPEG/images/</qatest>
|
||||
</qatests>
|
||||
<required_dependencies>
|
||||
<dependency>NVJPEG</dependency>
|
||||
</required_dependencies>
|
||||
<scopes>
|
||||
<scope>1:CUDA Basic Topics</scope>
|
||||
<scope>3:JPEG Decoding</scope>
|
||||
</scopes>
|
||||
<sm-arch>sm30</sm-arch>
|
||||
<sm-arch>sm35</sm-arch>
|
||||
<sm-arch>sm37</sm-arch>
|
||||
<sm-arch>sm50</sm-arch>
|
||||
<sm-arch>sm52</sm-arch>
|
||||
<sm-arch>sm60</sm-arch>
|
||||
<sm-arch>sm61</sm-arch>
|
||||
<sm-arch>sm70</sm-arch>
|
||||
<sm-arch>sm72</sm-arch>
|
||||
<sm-arch>sm75</sm-arch>
|
||||
<supported_envs>
|
||||
<env>
|
||||
<arch>x86_64</arch>
|
||||
<platform>linux</platform>
|
||||
</env>
|
||||
</supported_envs>
|
||||
<supported_sm_architectures>
|
||||
<from>3.0</from>
|
||||
</supported_sm_architectures>
|
||||
<title>NVJPEG simple</title>
|
||||
<type>exe</type>
|
||||
</entry>
|
||||
61
Samples/nvJPEG/README.md
Normal file
@@ -0,0 +1,61 @@
|
||||
# nvJPEG - NVJPEG simple
|
||||
|
||||
## Description
|
||||
|
||||
A CUDA Sample that demonstrates single and batched decoding of jpeg images using NVJPEG Library.
|
||||
|
||||
## Key Concepts
|
||||
|
||||
Image Decoding, NVJPEG Library
|
||||
|
||||
## Supported SM Architectures
|
||||
|
||||
[SM 3.0 ](https://developer.nvidia.com/cuda-gpus) [SM 3.5 ](https://developer.nvidia.com/cuda-gpus) [SM 3.7 ](https://developer.nvidia.com/cuda-gpus) [SM 5.0 ](https://developer.nvidia.com/cuda-gpus) [SM 5.2 ](https://developer.nvidia.com/cuda-gpus) [SM 6.0 ](https://developer.nvidia.com/cuda-gpus) [SM 6.1 ](https://developer.nvidia.com/cuda-gpus) [SM 7.0 ](https://developer.nvidia.com/cuda-gpus) [SM 7.2 ](https://developer.nvidia.com/cuda-gpus) [SM 7.5 ](https://developer.nvidia.com/cuda-gpus)
|
||||
|
||||
## Supported OSes
|
||||
|
||||
Linux
|
||||
|
||||
## Supported CPU Architecture
|
||||
|
||||
x86_64
|
||||
|
||||
## CUDA APIs involved
|
||||
|
||||
## Dependencies needed to build/run
|
||||
[NVJPEG](../../README.md#nvjpeg)
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Download and install the [CUDA Toolkit 10.1](https://developer.nvidia.com/cuda-downloads) for your corresponding platform.
|
||||
Make sure the dependencies mentioned in [Dependencies]() section above are installed.
|
||||
|
||||
## Build and Run
|
||||
|
||||
### Linux
|
||||
The Linux samples are built using makefiles. To use the makefiles, change the current directory to the sample directory you wish to build, and run make:
|
||||
```
|
||||
$ cd <sample_dir>
|
||||
$ make
|
||||
```
|
||||
The samples makefiles can take advantage of certain options:
|
||||
* **TARGET_ARCH=<arch>** - cross-compile targeting a specific architecture. Allowed architectures are x86_64.
|
||||
By default, TARGET_ARCH is set to HOST_ARCH. On a x86_64 machine, not setting TARGET_ARCH is the equivalent of setting TARGET_ARCH=x86_64.<br/>
|
||||
`$ make TARGET_ARCH=x86_64` <br/>
|
||||
See [here](http://docs.nvidia.com/cuda/cuda-samples/index.html#cross-samples) for more details.
|
||||
* **dbg=1** - build with debug symbols
|
||||
```
|
||||
$ make dbg=1
|
||||
```
|
||||
* **SMS="A B ..."** - override the SM architectures for which the sample will be built, where `"A B ..."` is a space-delimited list of SM architectures. For example, to generate SASS for SM 50 and SM 60, use `SMS="50 60"`.
|
||||
```
|
||||
$ make SMS="50 60"
|
||||
```
|
||||
|
||||
* **HOST_COMPILER=<host_compiler>** - override the default g++ host compiler. See the [Linux Installation Guide](http://docs.nvidia.com/cuda/cuda-installation-guide-linux/index.html#system-requirements) for a list of supported host compilers.
|
||||
```
|
||||
$ make HOST_COMPILER=g++
|
||||
```
|
||||
|
||||
## References (for more details)
|
||||
|
||||
BIN
Samples/nvJPEG/images/img1.jpg
Normal file
|
After Width: | Height: | Size: 66 KiB |
BIN
Samples/nvJPEG/images/img2.jpg
Normal file
|
After Width: | Height: | Size: 50 KiB |
BIN
Samples/nvJPEG/images/img3.jpg
Normal file
|
After Width: | Height: | Size: 34 KiB |
BIN
Samples/nvJPEG/images/img4.jpg
Normal file
|
After Width: | Height: | Size: 30 KiB |
BIN
Samples/nvJPEG/images/img5.jpg
Normal file
|
After Width: | Height: | Size: 80 KiB |
BIN
Samples/nvJPEG/images/img6.jpg
Normal file
|
After Width: | Height: | Size: 63 KiB |
BIN
Samples/nvJPEG/images/img7.jpg
Normal file
|
After Width: | Height: | Size: 92 KiB |
BIN
Samples/nvJPEG/images/img8.jpg
Normal file
|
After Width: | Height: | Size: 52 KiB |
559
Samples/nvJPEG/nvJPEG.cpp
Normal file
@@ -0,0 +1,559 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
// This sample needs at least CUDA 10.0. It demonstrates usages of the nvJPEG
|
||||
// library nvJPEG supports single and multiple image(batched) decode. Multiple
|
||||
// images can be decoded using the API for batch mode
|
||||
|
||||
#include <cuda_runtime_api.h>
|
||||
#include "nvJPEG_helper.hxx"
|
||||
|
||||
int dev_malloc(void **p, size_t s) { return (int)cudaMalloc(p, s); }
|
||||
|
||||
int dev_free(void *p) { return (int)cudaFree(p); }
|
||||
|
||||
typedef std::vector<std::string> FileNames;
|
||||
typedef std::vector<std::vector<char> > FileData;
|
||||
|
||||
struct decode_params_t {
|
||||
std::string input_dir;
|
||||
int batch_size;
|
||||
int total_images;
|
||||
int dev;
|
||||
int warmup;
|
||||
|
||||
nvjpegJpegState_t nvjpeg_state;
|
||||
nvjpegHandle_t nvjpeg_handle;
|
||||
cudaStream_t stream;
|
||||
|
||||
nvjpegOutputFormat_t fmt;
|
||||
bool write_decoded;
|
||||
std::string output_dir;
|
||||
|
||||
bool pipelined;
|
||||
bool batched;
|
||||
};
|
||||
|
||||
int read_next_batch(FileNames &image_names, int batch_size,
|
||||
FileNames::iterator &cur_iter, FileData &raw_data,
|
||||
std::vector<size_t> &raw_len, FileNames ¤t_names) {
|
||||
int counter = 0;
|
||||
|
||||
while (counter < batch_size) {
|
||||
if (cur_iter == image_names.end()) {
|
||||
std::cerr << "Image list is too short to fill the batch, adding files "
|
||||
"from the beginning of the image list"
|
||||
<< std::endl;
|
||||
cur_iter = image_names.begin();
|
||||
}
|
||||
|
||||
if (image_names.size() == 0) {
|
||||
std::cerr << "No valid images left in the input list, exit" << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
// Read an image from disk.
|
||||
std::ifstream input(cur_iter->c_str(),
|
||||
std::ios::in | std::ios::binary | std::ios::ate);
|
||||
if (!(input.is_open())) {
|
||||
std::cerr << "Cannot open image: " << *cur_iter
|
||||
<< ", removing it from image list" << std::endl;
|
||||
image_names.erase(cur_iter);
|
||||
continue;
|
||||
}
|
||||
|
||||
// Get the size
|
||||
std::streamsize file_size = input.tellg();
|
||||
input.seekg(0, std::ios::beg);
|
||||
// resize if buffer is too small
|
||||
if (raw_data[counter].size() < file_size) {
|
||||
raw_data[counter].resize(file_size);
|
||||
}
|
||||
if (!input.read(raw_data[counter].data(), file_size)) {
|
||||
std::cerr << "Cannot read from file: " << *cur_iter
|
||||
<< ", removing it from image list" << std::endl;
|
||||
image_names.erase(cur_iter);
|
||||
continue;
|
||||
}
|
||||
raw_len[counter] = file_size;
|
||||
|
||||
current_names[counter] = *cur_iter;
|
||||
|
||||
counter++;
|
||||
cur_iter++;
|
||||
}
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
|
||||
// prepare buffers for RGBi output format
|
||||
int prepare_buffers(FileData &file_data, std::vector<size_t> &file_len,
|
||||
std::vector<int> &img_width, std::vector<int> &img_height,
|
||||
std::vector<nvjpegImage_t> &ibuf,
|
||||
std::vector<nvjpegImage_t> &isz, FileNames ¤t_names,
|
||||
decode_params_t ¶ms) {
|
||||
int widths[NVJPEG_MAX_COMPONENT];
|
||||
int heights[NVJPEG_MAX_COMPONENT];
|
||||
int channels;
|
||||
nvjpegChromaSubsampling_t subsampling;
|
||||
|
||||
for (int i = 0; i < file_data.size(); i++) {
|
||||
checkCudaErrors(nvjpegGetImageInfo(
|
||||
params.nvjpeg_handle, (unsigned char *)file_data[i].data(), file_len[i],
|
||||
&channels, &subsampling, widths, heights));
|
||||
|
||||
img_width[i] = widths[0];
|
||||
img_height[i] = heights[0];
|
||||
|
||||
std::cout << "Processing: " << current_names[i] << std::endl;
|
||||
std::cout << "Image is " << channels << " channels." << std::endl;
|
||||
for (int c = 0; c < channels; c++) {
|
||||
std::cout << "Channel #" << c << " size: " << widths[c] << " x "
|
||||
<< heights[c] << std::endl;
|
||||
}
|
||||
|
||||
switch (subsampling) {
|
||||
case NVJPEG_CSS_444:
|
||||
std::cout << "YUV 4:4:4 chroma subsampling" << std::endl;
|
||||
break;
|
||||
case NVJPEG_CSS_440:
|
||||
std::cout << "YUV 4:4:0 chroma subsampling" << std::endl;
|
||||
break;
|
||||
case NVJPEG_CSS_422:
|
||||
std::cout << "YUV 4:2:2 chroma subsampling" << std::endl;
|
||||
break;
|
||||
case NVJPEG_CSS_420:
|
||||
std::cout << "YUV 4:2:0 chroma subsampling" << std::endl;
|
||||
break;
|
||||
case NVJPEG_CSS_411:
|
||||
std::cout << "YUV 4:1:1 chroma subsampling" << std::endl;
|
||||
break;
|
||||
case NVJPEG_CSS_410:
|
||||
std::cout << "YUV 4:1:0 chroma subsampling" << std::endl;
|
||||
break;
|
||||
case NVJPEG_CSS_GRAY:
|
||||
std::cout << "Grayscale JPEG " << std::endl;
|
||||
break;
|
||||
case NVJPEG_CSS_UNKNOWN:
|
||||
std::cout << "Unknown chroma subsampling" << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
int mul = 1;
|
||||
// in the case of interleaved RGB output, write only to single channel, but
|
||||
// 3 samples at once
|
||||
if (params.fmt == NVJPEG_OUTPUT_RGBI || params.fmt == NVJPEG_OUTPUT_BGRI) {
|
||||
channels = 1;
|
||||
mul = 3;
|
||||
}
|
||||
// in the case of rgb create 3 buffers with sizes of original image
|
||||
else if (params.fmt == NVJPEG_OUTPUT_RGB ||
|
||||
params.fmt == NVJPEG_OUTPUT_BGR) {
|
||||
channels = 3;
|
||||
widths[1] = widths[2] = widths[0];
|
||||
heights[1] = heights[2] = heights[0];
|
||||
}
|
||||
|
||||
// realloc output buffer if required
|
||||
for (int c = 0; c < channels; c++) {
|
||||
int aw = mul * widths[c];
|
||||
int ah = heights[c];
|
||||
int sz = aw * ah;
|
||||
ibuf[i].pitch[c] = aw;
|
||||
if (sz > isz[i].pitch[c]) {
|
||||
if (ibuf[i].channel[c]) {
|
||||
checkCudaErrors(cudaFree(ibuf[i].channel[c]));
|
||||
}
|
||||
checkCudaErrors(cudaMalloc(&ibuf[i].channel[c], sz));
|
||||
isz[i].pitch[c] = sz;
|
||||
}
|
||||
}
|
||||
}
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
|
||||
void release_buffers(std::vector<nvjpegImage_t> &ibuf) {
|
||||
for (int i = 0; i < ibuf.size(); i++) {
|
||||
for (int c = 0; c < NVJPEG_MAX_COMPONENT; c++)
|
||||
if (ibuf[i].channel[c]) checkCudaErrors(cudaFree(ibuf[i].channel[c]));
|
||||
}
|
||||
}
|
||||
|
||||
int decode_images(const FileData &img_data, const std::vector<size_t> &img_len,
|
||||
std::vector<nvjpegImage_t> &out, decode_params_t ¶ms,
|
||||
double &time) {
|
||||
checkCudaErrors(cudaStreamSynchronize(params.stream));
|
||||
nvjpegStatus_t err;
|
||||
StopWatchInterface *timer = NULL;
|
||||
sdkCreateTimer(&timer);
|
||||
|
||||
if (!params.batched) {
|
||||
if (!params.pipelined) // decode one image at a time
|
||||
{
|
||||
int thread_idx = 0;
|
||||
sdkStartTimer(&timer);
|
||||
for (int i = 0; i < params.batch_size; i++) {
|
||||
checkCudaErrors(nvjpegDecode(params.nvjpeg_handle, params.nvjpeg_state,
|
||||
(const unsigned char *)img_data[i].data(),
|
||||
img_len[i], params.fmt, &out[i],
|
||||
params.stream));
|
||||
checkCudaErrors(cudaStreamSynchronize(params.stream));
|
||||
}
|
||||
} else {
|
||||
int thread_idx = 0;
|
||||
sdkStartTimer(&timer);
|
||||
for (int i = 0; i < params.batch_size; i++) {
|
||||
checkCudaErrors(
|
||||
nvjpegDecodePhaseOne(params.nvjpeg_handle, params.nvjpeg_state,
|
||||
(const unsigned char *)img_data[i].data(),
|
||||
img_len[i], params.fmt, params.stream));
|
||||
checkCudaErrors(cudaStreamSynchronize(params.stream));
|
||||
checkCudaErrors(nvjpegDecodePhaseTwo(
|
||||
params.nvjpeg_handle, params.nvjpeg_state, params.stream));
|
||||
checkCudaErrors(nvjpegDecodePhaseThree(
|
||||
params.nvjpeg_handle, params.nvjpeg_state, &out[i], params.stream));
|
||||
}
|
||||
checkCudaErrors(cudaStreamSynchronize(params.stream));
|
||||
}
|
||||
} else {
|
||||
std::vector<const unsigned char *> raw_inputs;
|
||||
for (int i = 0; i < params.batch_size; i++) {
|
||||
raw_inputs.push_back((const unsigned char *)img_data[i].data());
|
||||
}
|
||||
|
||||
if (!params.pipelined) // decode multiple images in a single batch
|
||||
{
|
||||
sdkStartTimer(&timer);
|
||||
checkCudaErrors(nvjpegDecodeBatched(
|
||||
params.nvjpeg_handle, params.nvjpeg_state, raw_inputs.data(),
|
||||
img_len.data(), out.data(), params.stream));
|
||||
checkCudaErrors(cudaStreamSynchronize(params.stream));
|
||||
} else {
|
||||
int thread_idx = 0;
|
||||
for (int i = 0; i < params.batch_size; i++) {
|
||||
checkCudaErrors(nvjpegDecodeBatchedPhaseOne(
|
||||
params.nvjpeg_handle, params.nvjpeg_state, raw_inputs[i],
|
||||
img_len[i], i, thread_idx, params.stream));
|
||||
}
|
||||
checkCudaErrors(nvjpegDecodeBatchedPhaseTwo(
|
||||
params.nvjpeg_handle, params.nvjpeg_state, params.stream));
|
||||
checkCudaErrors(nvjpegDecodeBatchedPhaseThree(params.nvjpeg_handle,
|
||||
params.nvjpeg_state,
|
||||
out.data(), params.stream));
|
||||
checkCudaErrors(cudaStreamSynchronize(params.stream));
|
||||
}
|
||||
}
|
||||
sdkStopTimer(&timer);
|
||||
time = sdkGetAverageTimerValue(&timer)/1000.0f;
|
||||
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
|
||||
int write_images(std::vector<nvjpegImage_t> &iout, std::vector<int> &widths,
|
||||
std::vector<int> &heights, decode_params_t ¶ms,
|
||||
FileNames &filenames) {
|
||||
for (int i = 0; i < params.batch_size; i++) {
|
||||
// Get the file name, without extension.
|
||||
// This will be used to rename the output file.
|
||||
size_t position = filenames[i].rfind("/");
|
||||
std::string sFileName =
|
||||
(std::string::npos == position)
|
||||
? filenames[i]
|
||||
: filenames[i].substr(position + 1, filenames[i].size());
|
||||
position = sFileName.rfind(".");
|
||||
sFileName = (std::string::npos == position) ? sFileName
|
||||
: sFileName.substr(0, position);
|
||||
std::string fname(params.output_dir + "/" + sFileName + ".bmp");
|
||||
|
||||
int err;
|
||||
if (params.fmt == NVJPEG_OUTPUT_RGB || params.fmt == NVJPEG_OUTPUT_BGR) {
|
||||
err = writeBMP(fname.c_str(), iout[i].channel[0], iout[i].pitch[0],
|
||||
iout[i].channel[1], iout[i].pitch[1], iout[i].channel[2],
|
||||
iout[i].pitch[2], widths[i], heights[i]);
|
||||
} else if (params.fmt == NVJPEG_OUTPUT_RGBI ||
|
||||
params.fmt == NVJPEG_OUTPUT_BGRI) {
|
||||
// Write BMP from interleaved data
|
||||
err = writeBMPi(fname.c_str(), iout[i].channel[0], iout[i].pitch[0],
|
||||
widths[i], heights[i]);
|
||||
}
|
||||
if (err) {
|
||||
std::cout << "Cannot write output file: " << fname << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
std::cout << "Done writing decoded image to file: " << fname << std::endl;
|
||||
}
|
||||
}
|
||||
|
||||
double process_images(FileNames &image_names, decode_params_t ¶ms,
|
||||
double &total) {
|
||||
// vector for storing raw files and file lengths
|
||||
FileData file_data(params.batch_size);
|
||||
std::vector<size_t> file_len(params.batch_size);
|
||||
FileNames current_names(params.batch_size);
|
||||
std::vector<int> widths(params.batch_size);
|
||||
std::vector<int> heights(params.batch_size);
|
||||
// we wrap over image files to process total_images of files
|
||||
FileNames::iterator file_iter = image_names.begin();
|
||||
|
||||
// stream for decoding
|
||||
checkCudaErrors(
|
||||
cudaStreamCreateWithFlags(¶ms.stream, cudaStreamNonBlocking));
|
||||
|
||||
int total_processed = 0;
|
||||
|
||||
// output buffers
|
||||
std::vector<nvjpegImage_t> iout(params.batch_size);
|
||||
// output buffer sizes, for convenience
|
||||
std::vector<nvjpegImage_t> isz(params.batch_size);
|
||||
|
||||
for (int i = 0; i < iout.size(); i++) {
|
||||
for (int c = 0; c < NVJPEG_MAX_COMPONENT; c++) {
|
||||
iout[i].channel[c] = NULL;
|
||||
iout[i].pitch[c] = 0;
|
||||
isz[i].pitch[c] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
double test_time = 0;
|
||||
int warmup = 0;
|
||||
while (total_processed < params.total_images) {
|
||||
if (read_next_batch(image_names, params.batch_size, file_iter, file_data,
|
||||
file_len, current_names))
|
||||
return EXIT_FAILURE;
|
||||
|
||||
if (prepare_buffers(file_data, file_len, widths, heights, iout, isz,
|
||||
current_names, params))
|
||||
return EXIT_FAILURE;
|
||||
|
||||
double time;
|
||||
if (decode_images(file_data, file_len, iout, params, time))
|
||||
return EXIT_FAILURE;
|
||||
if (warmup < params.warmup) {
|
||||
warmup++;
|
||||
} else {
|
||||
total_processed += params.batch_size;
|
||||
test_time += time;
|
||||
}
|
||||
|
||||
if (params.write_decoded)
|
||||
write_images(iout, widths, heights, params, current_names);
|
||||
}
|
||||
total = test_time;
|
||||
|
||||
release_buffers(iout);
|
||||
|
||||
checkCudaErrors(cudaStreamDestroy(params.stream));
|
||||
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
|
||||
// parse parameters
|
||||
int findParamIndex(const char **argv, int argc, const char *parm) {
|
||||
int count = 0;
|
||||
int index = -1;
|
||||
|
||||
for (int i = 0; i < argc; i++) {
|
||||
if (strncmp(argv[i], parm, 100) == 0) {
|
||||
index = i;
|
||||
count++;
|
||||
}
|
||||
}
|
||||
|
||||
if (count == 0 || count == 1) {
|
||||
return index;
|
||||
} else {
|
||||
std::cout << "Error, parameter " << parm
|
||||
<< " has been specified more than once, exiting\n"
|
||||
<< std::endl;
|
||||
return -1;
|
||||
}
|
||||
|
||||
return -1;
|
||||
}
|
||||
|
||||
int main(int argc, const char *argv[]) {
|
||||
int pidx;
|
||||
|
||||
if ((pidx = findParamIndex(argv, argc, "-h")) != -1 ||
|
||||
(pidx = findParamIndex(argv, argc, "--help")) != -1) {
|
||||
std::cout << "Usage: " << argv[0]
|
||||
<< " -i images_dir [-b batch_size] [-t total_images] [-device= "
|
||||
"device_id] [-w warmup_iterations] [-o output_dir] "
|
||||
"[-pipelined] [-batched] [-fmt output_format]\n";
|
||||
std::cout << "Parameters: " << std::endl;
|
||||
std::cout << "\timages_dir\t:\tPath to single image or directory of images"
|
||||
<< std::endl;
|
||||
std::cout << "\tbatch_size\t:\tDecode images from input by batches of "
|
||||
"specified size"
|
||||
<< std::endl;
|
||||
std::cout << "\ttotal_images\t:\tDecode this much images, if there are "
|
||||
"less images \n"
|
||||
<< "\t\t\t\t\tin the input than total images, decoder will loop "
|
||||
"over the input"
|
||||
<< std::endl;
|
||||
std::cout << "\tdevice_id\t:\tWhich device to use for decoding"
|
||||
<< std::endl;
|
||||
std::cout << "\twarmup_iterations\t:\tRun this amount of batches first "
|
||||
"without measuring performance"
|
||||
<< std::endl;
|
||||
std::cout
|
||||
<< "\toutput_dir\t:\tWrite decoded images as BMPs to this directory"
|
||||
<< std::endl;
|
||||
std::cout << "\tpipelined\t:\tUse decoding in phases" << std::endl;
|
||||
std::cout << "\tbatched\t\t:\tUse batched interface" << std::endl;
|
||||
std::cout << "\toutput_format\t:\tnvJPEG output format for decoding. One "
|
||||
"of [rgb, rgbi, bgr, bgri, yuv, y, unchanged]"
|
||||
<< std::endl;
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
|
||||
decode_params_t params;
|
||||
|
||||
params.input_dir = "./";
|
||||
if ((pidx = findParamIndex(argv, argc, "-i")) != -1) {
|
||||
params.input_dir = argv[pidx + 1];
|
||||
} else {
|
||||
std::cerr << "Please specify input directory with encoded images"
|
||||
<< std::endl;
|
||||
return EXIT_WAIVED;
|
||||
}
|
||||
|
||||
params.batch_size = 1;
|
||||
if ((pidx = findParamIndex(argv, argc, "-b")) != -1) {
|
||||
params.batch_size = std::atoi(argv[pidx + 1]);
|
||||
}
|
||||
|
||||
params.total_images = -1;
|
||||
if ((pidx = findParamIndex(argv, argc, "-t")) != -1) {
|
||||
params.total_images = std::atoi(argv[pidx + 1]);
|
||||
}
|
||||
|
||||
params.dev = 0;
|
||||
params.dev = findCudaDevice(argc, argv);
|
||||
|
||||
params.warmup = 0;
|
||||
if ((pidx = findParamIndex(argv, argc, "-w")) != -1) {
|
||||
params.warmup = std::atoi(argv[pidx + 1]);
|
||||
}
|
||||
|
||||
params.batched = false;
|
||||
if ((pidx = findParamIndex(argv, argc, "-batched")) != -1) {
|
||||
params.batched = true;
|
||||
}
|
||||
|
||||
params.pipelined = false;
|
||||
if ((pidx = findParamIndex(argv, argc, "-pipelined")) != -1) {
|
||||
params.pipelined = true;
|
||||
}
|
||||
|
||||
params.fmt = NVJPEG_OUTPUT_RGB;
|
||||
if ((pidx = findParamIndex(argv, argc, "-fmt")) != -1) {
|
||||
std::string sfmt = argv[pidx + 1];
|
||||
if (sfmt == "rgb")
|
||||
params.fmt = NVJPEG_OUTPUT_RGB;
|
||||
else if (sfmt == "bgr")
|
||||
params.fmt = NVJPEG_OUTPUT_BGR;
|
||||
else if (sfmt == "rgbi")
|
||||
params.fmt = NVJPEG_OUTPUT_RGBI;
|
||||
else if (sfmt == "bgri")
|
||||
params.fmt = NVJPEG_OUTPUT_BGRI;
|
||||
else if (sfmt == "yuv")
|
||||
params.fmt = NVJPEG_OUTPUT_YUV;
|
||||
else if (sfmt == "y")
|
||||
params.fmt = NVJPEG_OUTPUT_Y;
|
||||
else if (sfmt == "unchanged")
|
||||
params.fmt = NVJPEG_OUTPUT_UNCHANGED;
|
||||
else {
|
||||
std::cout << "Unknown format: " << sfmt << std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
}
|
||||
|
||||
params.write_decoded = false;
|
||||
if ((pidx = findParamIndex(argv, argc, "-o")) != -1) {
|
||||
params.output_dir = argv[pidx + 1];
|
||||
if (params.fmt != NVJPEG_OUTPUT_RGB && params.fmt != NVJPEG_OUTPUT_BGR &&
|
||||
params.fmt != NVJPEG_OUTPUT_RGBI && params.fmt != NVJPEG_OUTPUT_BGRI) {
|
||||
std::cout << "We can write ony BMPs, which require output format be "
|
||||
"either RGB/BGR or RGBi/BGRi"
|
||||
<< std::endl;
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
params.write_decoded = true;
|
||||
}
|
||||
|
||||
cudaDeviceProp props;
|
||||
checkCudaErrors(cudaGetDeviceProperties(&props, params.dev));
|
||||
|
||||
printf("Using GPU %d (%s, %d SMs, %d th/SM max, CC %d.%d, ECC %s)\n",
|
||||
params.dev, props.name, props.multiProcessorCount,
|
||||
props.maxThreadsPerMultiProcessor, props.major, props.minor,
|
||||
props.ECCEnabled ? "on" : "off");
|
||||
|
||||
nvjpegDevAllocator_t dev_allocator = {&dev_malloc, &dev_free};
|
||||
checkCudaErrors(nvjpegCreate(NVJPEG_BACKEND_DEFAULT, &dev_allocator,
|
||||
¶ms.nvjpeg_handle));
|
||||
checkCudaErrors(
|
||||
nvjpegJpegStateCreate(params.nvjpeg_handle, ¶ms.nvjpeg_state));
|
||||
checkCudaErrors(
|
||||
nvjpegDecodeBatchedInitialize(params.nvjpeg_handle, params.nvjpeg_state,
|
||||
params.batch_size, 1, params.fmt));
|
||||
|
||||
// read source images
|
||||
FileNames image_names;
|
||||
readInput(params.input_dir, image_names);
|
||||
|
||||
if (params.total_images == -1) {
|
||||
params.total_images = image_names.size();
|
||||
} else if (params.total_images % params.batch_size) {
|
||||
params.total_images =
|
||||
((params.total_images) / params.batch_size) * params.batch_size;
|
||||
std::cout << "Changing total_images number to " << params.total_images
|
||||
<< " to be multiple of batch_size - " << params.batch_size
|
||||
<< std::endl;
|
||||
}
|
||||
|
||||
std::cout << "Decoding images in directory: " << params.input_dir
|
||||
<< ", total " << params.total_images << ", batchsize "
|
||||
<< params.batch_size << std::endl;
|
||||
|
||||
double total;
|
||||
if (process_images(image_names, params, total)) return EXIT_FAILURE;
|
||||
std::cout << "Total decoding time: " << total << std::endl;
|
||||
std::cout << "Avg decoding time per image: " << total / params.total_images
|
||||
<< std::endl;
|
||||
std::cout << "Avg images per sec: " << params.total_images / total
|
||||
<< std::endl;
|
||||
std::cout << "Avg decoding time per batch: "
|
||||
<< total / ((params.total_images + params.batch_size - 1) /
|
||||
params.batch_size)
|
||||
<< std::endl;
|
||||
|
||||
checkCudaErrors(nvjpegJpegStateDestroy(params.nvjpeg_state));
|
||||
checkCudaErrors(nvjpegDestroy(params.nvjpeg_handle));
|
||||
|
||||
return EXIT_SUCCESS;
|
||||
}
|
||||
338
Samples/nvJPEG/nvJPEG_helper.hxx
Normal file
@@ -0,0 +1,338 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
// This sample needs at least CUDA 10.0.
|
||||
// It demonstrates usages of the nvJPEG library
|
||||
|
||||
#ifndef NV_JPEG_EXAMPLE
|
||||
#define NV_JPEG_EXAMPLE
|
||||
|
||||
#include "cuda_runtime.h"
|
||||
#include "nvjpeg.h"
|
||||
#include "helper_cuda.h"
|
||||
#include "helper_timer.h"
|
||||
|
||||
#include <cstdlib>
|
||||
#include <fstream>
|
||||
#include <iostream>
|
||||
#include <sstream>
|
||||
#include <string>
|
||||
#include <vector>
|
||||
|
||||
#include <string.h> // strcmpi
|
||||
#include <sys/time.h> // timings
|
||||
|
||||
#include <dirent.h> // linux dir traverse
|
||||
#include <sys/stat.h>
|
||||
#include <sys/types.h>
|
||||
#include <unistd.h>
|
||||
|
||||
// write bmp, input - RGB, device
|
||||
int writeBMP(const char *filename, const unsigned char *d_chanR, int pitchR,
|
||||
const unsigned char *d_chanG, int pitchG,
|
||||
const unsigned char *d_chanB, int pitchB, int width, int height) {
|
||||
unsigned int headers[13];
|
||||
FILE *outfile;
|
||||
int extrabytes;
|
||||
int paddedsize;
|
||||
int x;
|
||||
int y;
|
||||
int n;
|
||||
int red, green, blue;
|
||||
|
||||
std::vector<unsigned char> vchanR(height * width);
|
||||
std::vector<unsigned char> vchanG(height * width);
|
||||
std::vector<unsigned char> vchanB(height * width);
|
||||
unsigned char *chanR = vchanR.data();
|
||||
unsigned char *chanG = vchanG.data();
|
||||
unsigned char *chanB = vchanB.data();
|
||||
checkCudaErrors(cudaMemcpy2D(chanR, (size_t)width, d_chanR, (size_t)pitchR,
|
||||
width, height, cudaMemcpyDeviceToHost));
|
||||
checkCudaErrors(cudaMemcpy2D(chanG, (size_t)width, d_chanG, (size_t)pitchR,
|
||||
width, height, cudaMemcpyDeviceToHost));
|
||||
checkCudaErrors(cudaMemcpy2D(chanB, (size_t)width, d_chanB, (size_t)pitchR,
|
||||
width, height, cudaMemcpyDeviceToHost));
|
||||
|
||||
extrabytes =
|
||||
4 - ((width * 3) % 4); // How many bytes of padding to add to each
|
||||
// horizontal line - the size of which must
|
||||
// be a multiple of 4 bytes.
|
||||
if (extrabytes == 4) extrabytes = 0;
|
||||
|
||||
paddedsize = ((width * 3) + extrabytes) * height;
|
||||
|
||||
// Headers...
|
||||
// Note that the "BM" identifier in bytes 0 and 1 is NOT included in these
|
||||
// "headers".
|
||||
|
||||
headers[0] = paddedsize + 54; // bfSize (whole file size)
|
||||
headers[1] = 0; // bfReserved (both)
|
||||
headers[2] = 54; // bfOffbits
|
||||
headers[3] = 40; // biSize
|
||||
headers[4] = width; // biWidth
|
||||
headers[5] = height; // biHeight
|
||||
|
||||
// Would have biPlanes and biBitCount in position 6, but they're shorts.
|
||||
// It's easier to write them out separately (see below) than pretend
|
||||
// they're a single int, especially with endian issues...
|
||||
|
||||
headers[7] = 0; // biCompression
|
||||
headers[8] = paddedsize; // biSizeImage
|
||||
headers[9] = 0; // biXPelsPerMeter
|
||||
headers[10] = 0; // biYPelsPerMeter
|
||||
headers[11] = 0; // biClrUsed
|
||||
headers[12] = 0; // biClrImportant
|
||||
|
||||
if (!(outfile = fopen(filename, "wb"))) {
|
||||
std::cerr << "Cannot open file: " << filename << std::endl;
|
||||
return 1;
|
||||
}
|
||||
|
||||
//
|
||||
// Headers begin...
|
||||
// When printing ints and shorts, we write out 1 character at a time to avoid
|
||||
// endian issues.
|
||||
//
|
||||
fprintf(outfile, "BM");
|
||||
|
||||
for (n = 0; n <= 5; n++) {
|
||||
fprintf(outfile, "%c", headers[n] & 0x000000FF);
|
||||
fprintf(outfile, "%c", (headers[n] & 0x0000FF00) >> 8);
|
||||
fprintf(outfile, "%c", (headers[n] & 0x00FF0000) >> 16);
|
||||
fprintf(outfile, "%c", (headers[n] & (unsigned int)0xFF000000) >> 24);
|
||||
}
|
||||
|
||||
// These next 4 characters are for the biPlanes and biBitCount fields.
|
||||
|
||||
fprintf(outfile, "%c", 1);
|
||||
fprintf(outfile, "%c", 0);
|
||||
fprintf(outfile, "%c", 24);
|
||||
fprintf(outfile, "%c", 0);
|
||||
|
||||
for (n = 7; n <= 12; n++) {
|
||||
fprintf(outfile, "%c", headers[n] & 0x000000FF);
|
||||
fprintf(outfile, "%c", (headers[n] & 0x0000FF00) >> 8);
|
||||
fprintf(outfile, "%c", (headers[n] & 0x00FF0000) >> 16);
|
||||
fprintf(outfile, "%c", (headers[n] & (unsigned int)0xFF000000) >> 24);
|
||||
}
|
||||
|
||||
//
|
||||
// Headers done, now write the data...
|
||||
//
|
||||
|
||||
for (y = height - 1; y >= 0;
|
||||
y--) // BMP image format is written from bottom to top...
|
||||
{
|
||||
for (x = 0; x <= width - 1; x++) {
|
||||
red = chanR[y * width + x];
|
||||
green = chanG[y * width + x];
|
||||
blue = chanB[y * width + x];
|
||||
|
||||
if (red > 255) red = 255;
|
||||
if (red < 0) red = 0;
|
||||
if (green > 255) green = 255;
|
||||
if (green < 0) green = 0;
|
||||
if (blue > 255) blue = 255;
|
||||
if (blue < 0) blue = 0;
|
||||
// Also, it's written in (b,g,r) format...
|
||||
|
||||
fprintf(outfile, "%c", blue);
|
||||
fprintf(outfile, "%c", green);
|
||||
fprintf(outfile, "%c", red);
|
||||
}
|
||||
if (extrabytes) // See above - BMP lines must be of lengths divisible by 4.
|
||||
{
|
||||
for (n = 1; n <= extrabytes; n++) {
|
||||
fprintf(outfile, "%c", 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fclose(outfile);
|
||||
return 0;
|
||||
}
|
||||
|
||||
// write bmp, input - RGB, device
|
||||
int writeBMPi(const char *filename, const unsigned char *d_RGB, int pitch,
|
||||
int width, int height) {
|
||||
unsigned int headers[13];
|
||||
FILE *outfile;
|
||||
int extrabytes;
|
||||
int paddedsize;
|
||||
int x;
|
||||
int y;
|
||||
int n;
|
||||
int red, green, blue;
|
||||
|
||||
std::vector<unsigned char> vchanRGB(height * width * 3);
|
||||
unsigned char *chanRGB = vchanRGB.data();
|
||||
checkCudaErrors(cudaMemcpy2D(chanRGB, (size_t)width * 3, d_RGB, (size_t)pitch,
|
||||
width * 3, height, cudaMemcpyDeviceToHost));
|
||||
|
||||
extrabytes =
|
||||
4 - ((width * 3) % 4); // How many bytes of padding to add to each
|
||||
// horizontal line - the size of which must
|
||||
// be a multiple of 4 bytes.
|
||||
if (extrabytes == 4) extrabytes = 0;
|
||||
|
||||
paddedsize = ((width * 3) + extrabytes) * height;
|
||||
|
||||
// Headers...
|
||||
// Note that the "BM" identifier in bytes 0 and 1 is NOT included in these
|
||||
// "headers".
|
||||
headers[0] = paddedsize + 54; // bfSize (whole file size)
|
||||
headers[1] = 0; // bfReserved (both)
|
||||
headers[2] = 54; // bfOffbits
|
||||
headers[3] = 40; // biSize
|
||||
headers[4] = width; // biWidth
|
||||
headers[5] = height; // biHeight
|
||||
|
||||
// Would have biPlanes and biBitCount in position 6, but they're shorts.
|
||||
// It's easier to write them out separately (see below) than pretend
|
||||
// they're a single int, especially with endian issues...
|
||||
|
||||
headers[7] = 0; // biCompression
|
||||
headers[8] = paddedsize; // biSizeImage
|
||||
headers[9] = 0; // biXPelsPerMeter
|
||||
headers[10] = 0; // biYPelsPerMeter
|
||||
headers[11] = 0; // biClrUsed
|
||||
headers[12] = 0; // biClrImportant
|
||||
|
||||
if (!(outfile = fopen(filename, "wb"))) {
|
||||
std::cerr << "Cannot open file: " << filename << std::endl;
|
||||
return 1;
|
||||
}
|
||||
|
||||
//
|
||||
// Headers begin...
|
||||
// When printing ints and shorts, we write out 1 character at a time to avoid
|
||||
// endian issues.
|
||||
//
|
||||
|
||||
fprintf(outfile, "BM");
|
||||
|
||||
for (n = 0; n <= 5; n++) {
|
||||
fprintf(outfile, "%c", headers[n] & 0x000000FF);
|
||||
fprintf(outfile, "%c", (headers[n] & 0x0000FF00) >> 8);
|
||||
fprintf(outfile, "%c", (headers[n] & 0x00FF0000) >> 16);
|
||||
fprintf(outfile, "%c", (headers[n] & (unsigned int)0xFF000000) >> 24);
|
||||
}
|
||||
|
||||
// These next 4 characters are for the biPlanes and biBitCount fields.
|
||||
|
||||
fprintf(outfile, "%c", 1);
|
||||
fprintf(outfile, "%c", 0);
|
||||
fprintf(outfile, "%c", 24);
|
||||
fprintf(outfile, "%c", 0);
|
||||
|
||||
for (n = 7; n <= 12; n++) {
|
||||
fprintf(outfile, "%c", headers[n] & 0x000000FF);
|
||||
fprintf(outfile, "%c", (headers[n] & 0x0000FF00) >> 8);
|
||||
fprintf(outfile, "%c", (headers[n] & 0x00FF0000) >> 16);
|
||||
fprintf(outfile, "%c", (headers[n] & (unsigned int)0xFF000000) >> 24);
|
||||
}
|
||||
|
||||
//
|
||||
// Headers done, now write the data...
|
||||
//
|
||||
for (y = height - 1; y >= 0;
|
||||
y--) // BMP image format is written from bottom to top...
|
||||
{
|
||||
for (x = 0; x <= width - 1; x++) {
|
||||
red = chanRGB[(y * width + x) * 3];
|
||||
green = chanRGB[(y * width + x) * 3 + 1];
|
||||
blue = chanRGB[(y * width + x) * 3 + 2];
|
||||
|
||||
if (red > 255) red = 255;
|
||||
if (red < 0) red = 0;
|
||||
if (green > 255) green = 255;
|
||||
if (green < 0) green = 0;
|
||||
if (blue > 255) blue = 255;
|
||||
if (blue < 0) blue = 0;
|
||||
// Also, it's written in (b,g,r) format...
|
||||
|
||||
fprintf(outfile, "%c", blue);
|
||||
fprintf(outfile, "%c", green);
|
||||
fprintf(outfile, "%c", red);
|
||||
}
|
||||
if (extrabytes) // See above - BMP lines must be of lengths divisible by 4.
|
||||
{
|
||||
for (n = 1; n <= extrabytes; n++) {
|
||||
fprintf(outfile, "%c", 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fclose(outfile);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int readInput(const std::string &sInputPath,
|
||||
std::vector<std::string> &filelist) {
|
||||
int error_code = 1;
|
||||
struct stat s;
|
||||
|
||||
if (stat(sInputPath.c_str(), &s) == 0) {
|
||||
if (s.st_mode & S_IFREG) {
|
||||
filelist.push_back(sInputPath);
|
||||
} else if (s.st_mode & S_IFDIR) {
|
||||
// processing each file in directory
|
||||
DIR *dir_handle;
|
||||
struct dirent *dir;
|
||||
dir_handle = opendir(sInputPath.c_str());
|
||||
std::vector<std::string> filenames;
|
||||
if (dir_handle) {
|
||||
error_code = 0;
|
||||
while ((dir = readdir(dir_handle)) != NULL) {
|
||||
if (dir->d_type == DT_REG) {
|
||||
std::string sFileName = sInputPath + dir->d_name;
|
||||
filelist.push_back(sFileName);
|
||||
} else if (dir->d_type == DT_DIR) {
|
||||
std::string sname = dir->d_name;
|
||||
if (sname != "." && sname != "..") {
|
||||
readInput(sInputPath + sname + "/", filelist);
|
||||
}
|
||||
}
|
||||
}
|
||||
closedir(dir_handle);
|
||||
} else {
|
||||
std::cout << "Cannot open input directory: " << sInputPath << std::endl;
|
||||
return error_code;
|
||||
}
|
||||
} else {
|
||||
std::cout << "Cannot open input: " << sInputPath << std::endl;
|
||||
return error_code;
|
||||
}
|
||||
} else {
|
||||
std::cout << "Cannot find input path " << sInputPath << std::endl;
|
||||
return error_code;
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
#endif
|
||||