mirror of
https://github.com/NVIDIA/cuda-samples.git
synced 2026-10-11 23:38:25 +08:00
Add and update samples with CUDA 10.1 Update 1 support
This commit is contained in:
299
Samples/cuSolverDn_LinearSolver/Makefile
Normal file
299
Samples/cuSolverDn_LinearSolver/Makefile
Normal file
@@ -0,0 +1,299 @@
|
||||
################################################################################
|
||||
# Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
#
|
||||
# Redistribution and use in source and binary forms, with or without
|
||||
# modification, are permitted provided that the following conditions
|
||||
# are met:
|
||||
# * Redistributions of source code must retain the above copyright
|
||||
# notice, this list of conditions and the following disclaimer.
|
||||
# * Redistributions in binary form must reproduce the above copyright
|
||||
# notice, this list of conditions and the following disclaimer in the
|
||||
# documentation and/or other materials provided with the distribution.
|
||||
# * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
# contributors may be used to endorse or promote products derived
|
||||
# from this software without specific prior written permission.
|
||||
#
|
||||
# THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
# EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
# IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
# PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
# CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
# EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
# PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
# PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
# OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
# (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
# OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
#
|
||||
################################################################################
|
||||
#
|
||||
# Makefile project only supported on Mac OS X and Linux Platforms)
|
||||
#
|
||||
################################################################################
|
||||
|
||||
# Location of the CUDA Toolkit
|
||||
CUDA_PATH ?= /usr/local/cuda
|
||||
|
||||
##############################
|
||||
# start deprecated interface #
|
||||
##############################
|
||||
ifeq ($(x86_64),1)
|
||||
$(info WARNING - x86_64 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=x86_64 instead)
|
||||
TARGET_ARCH ?= x86_64
|
||||
endif
|
||||
ifeq ($(ARMv7),1)
|
||||
$(info WARNING - ARMv7 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=armv7l instead)
|
||||
TARGET_ARCH ?= armv7l
|
||||
endif
|
||||
ifeq ($(aarch64),1)
|
||||
$(info WARNING - aarch64 variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=aarch64 instead)
|
||||
TARGET_ARCH ?= aarch64
|
||||
endif
|
||||
ifeq ($(ppc64le),1)
|
||||
$(info WARNING - ppc64le variable has been deprecated)
|
||||
$(info WARNING - please use TARGET_ARCH=ppc64le instead)
|
||||
TARGET_ARCH ?= ppc64le
|
||||
endif
|
||||
ifneq ($(GCC),)
|
||||
$(info WARNING - GCC variable has been deprecated)
|
||||
$(info WARNING - please use HOST_COMPILER=$(GCC) instead)
|
||||
HOST_COMPILER ?= $(GCC)
|
||||
endif
|
||||
ifneq ($(abi),)
|
||||
$(error ERROR - abi variable has been removed)
|
||||
endif
|
||||
############################
|
||||
# end deprecated interface #
|
||||
############################
|
||||
|
||||
# architecture
|
||||
HOST_ARCH := $(shell uname -m)
|
||||
TARGET_ARCH ?= $(HOST_ARCH)
|
||||
ifneq (,$(filter $(TARGET_ARCH),x86_64 aarch64 ppc64le armv7l))
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifneq (,$(filter $(TARGET_ARCH),x86_64 aarch64 ppc64le))
|
||||
TARGET_SIZE := 64
|
||||
else ifneq (,$(filter $(TARGET_ARCH),armv7l))
|
||||
TARGET_SIZE := 32
|
||||
endif
|
||||
else
|
||||
TARGET_SIZE := $(shell getconf LONG_BIT)
|
||||
endif
|
||||
else
|
||||
$(error ERROR - unsupported value $(TARGET_ARCH) for TARGET_ARCH!)
|
||||
endif
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq (,$(filter $(HOST_ARCH)-$(TARGET_ARCH),aarch64-armv7l x86_64-armv7l x86_64-aarch64 x86_64-ppc64le))
|
||||
$(error ERROR - cross compiling from $(HOST_ARCH) to $(TARGET_ARCH) is not supported!)
|
||||
endif
|
||||
endif
|
||||
|
||||
# When on native aarch64 system with userspace of 32-bit, change TARGET_ARCH to armv7l
|
||||
ifeq ($(HOST_ARCH)-$(TARGET_ARCH)-$(TARGET_SIZE),aarch64-aarch64-32)
|
||||
TARGET_ARCH = armv7l
|
||||
endif
|
||||
|
||||
# operating system
|
||||
HOST_OS := $(shell uname -s 2>/dev/null | tr "[:upper:]" "[:lower:]")
|
||||
TARGET_OS ?= $(HOST_OS)
|
||||
ifeq (,$(filter $(TARGET_OS),linux darwin qnx android))
|
||||
$(error ERROR - unsupported value $(TARGET_OS) for TARGET_OS!)
|
||||
endif
|
||||
|
||||
# host compiler
|
||||
ifeq ($(TARGET_OS),darwin)
|
||||
ifeq ($(shell expr `xcodebuild -version | grep -i xcode | awk '{print $$2}' | cut -d'.' -f1` \>= 5),1)
|
||||
HOST_COMPILER ?= clang++
|
||||
endif
|
||||
else ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq ($(HOST_ARCH)-$(TARGET_ARCH),x86_64-armv7l)
|
||||
ifeq ($(TARGET_OS),linux)
|
||||
HOST_COMPILER ?= arm-linux-gnueabihf-g++
|
||||
else ifeq ($(TARGET_OS),qnx)
|
||||
ifeq ($(QNX_HOST),)
|
||||
$(error ERROR - QNX_HOST must be passed to the QNX host toolchain)
|
||||
endif
|
||||
ifeq ($(QNX_TARGET),)
|
||||
$(error ERROR - QNX_TARGET must be passed to the QNX target toolchain)
|
||||
endif
|
||||
export QNX_HOST
|
||||
export QNX_TARGET
|
||||
HOST_COMPILER ?= $(QNX_HOST)/usr/bin/arm-unknown-nto-qnx6.6.0eabi-g++
|
||||
else ifeq ($(TARGET_OS),android)
|
||||
HOST_COMPILER ?= arm-linux-androideabi-g++
|
||||
endif
|
||||
else ifeq ($(TARGET_ARCH),aarch64)
|
||||
ifeq ($(TARGET_OS), linux)
|
||||
HOST_COMPILER ?= aarch64-linux-gnu-g++
|
||||
else ifeq ($(TARGET_OS),qnx)
|
||||
ifeq ($(QNX_HOST),)
|
||||
$(error ERROR - QNX_HOST must be passed to the QNX host toolchain)
|
||||
endif
|
||||
ifeq ($(QNX_TARGET),)
|
||||
$(error ERROR - QNX_TARGET must be passed to the QNX target toolchain)
|
||||
endif
|
||||
export QNX_HOST
|
||||
export QNX_TARGET
|
||||
HOST_COMPILER ?= $(QNX_HOST)/usr/bin/aarch64-unknown-nto-qnx7.0.0-g++
|
||||
else ifeq ($(TARGET_OS), android)
|
||||
HOST_COMPILER ?= aarch64-linux-android-clang++
|
||||
endif
|
||||
else ifeq ($(TARGET_ARCH),ppc64le)
|
||||
HOST_COMPILER ?= powerpc64le-linux-gnu-g++
|
||||
endif
|
||||
endif
|
||||
HOST_COMPILER ?= g++
|
||||
NVCC := $(CUDA_PATH)/bin/nvcc -ccbin $(HOST_COMPILER)
|
||||
|
||||
# internal flags
|
||||
NVCCFLAGS := -m${TARGET_SIZE}
|
||||
CCFLAGS :=
|
||||
LDFLAGS :=
|
||||
|
||||
# build flags
|
||||
ifeq ($(TARGET_OS),darwin)
|
||||
LDFLAGS += -rpath $(CUDA_PATH)/lib
|
||||
CCFLAGS += -arch $(HOST_ARCH)
|
||||
else ifeq ($(HOST_ARCH)-$(TARGET_ARCH)-$(TARGET_OS),x86_64-armv7l-linux)
|
||||
LDFLAGS += --dynamic-linker=/lib/ld-linux-armhf.so.3
|
||||
CCFLAGS += -mfloat-abi=hard
|
||||
else ifeq ($(TARGET_OS),android)
|
||||
LDFLAGS += -pie
|
||||
CCFLAGS += -fpie -fpic -fexceptions
|
||||
endif
|
||||
|
||||
ifneq ($(TARGET_ARCH),$(HOST_ARCH))
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-linux)
|
||||
ifneq ($(TARGET_FS),)
|
||||
GCCVERSIONLTEQ46 := $(shell expr `$(HOST_COMPILER) -dumpversion` \<= 4.6)
|
||||
ifeq ($(GCCVERSIONLTEQ46),1)
|
||||
CCFLAGS += --sysroot=$(TARGET_FS)
|
||||
endif
|
||||
LDFLAGS += --sysroot=$(TARGET_FS)
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib/arm-linux-gnueabihf
|
||||
endif
|
||||
endif
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-linux)
|
||||
ifneq ($(TARGET_FS),)
|
||||
GCCVERSIONLTEQ46 := $(shell expr `$(HOST_COMPILER) -dumpversion` \<= 4.6)
|
||||
ifeq ($(GCCVERSIONLTEQ46),1)
|
||||
CCFLAGS += --sysroot=$(TARGET_FS)
|
||||
endif
|
||||
LDFLAGS += --sysroot=$(TARGET_FS)
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/lib -L $(TARGET_FS)/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib -L $(TARGET_FS)/usr/lib
|
||||
LDFLAGS += -rpath-link=$(TARGET_FS)/usr/lib/aarch64-linux-gnu -L $(TARGET_FS)/usr/lib/aarch64-linux-gnu
|
||||
LDFLAGS += --unresolved-symbols=ignore-in-shared-libs
|
||||
CCFLAGS += -isystem=$(TARGET_FS)/usr/include
|
||||
CCFLAGS += -isystem=$(TARGET_FS)/usr/include/aarch64-linux-gnu
|
||||
endif
|
||||
endif
|
||||
endif
|
||||
|
||||
ifeq ($(TARGET_OS),qnx)
|
||||
CCFLAGS += -DWIN_INTERFACE_CUSTOM
|
||||
LDFLAGS += -lsocket
|
||||
endif
|
||||
|
||||
# Install directory of different arch
|
||||
CUDA_INSTALL_TARGET_DIR :=
|
||||
ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-linux)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/armv7-linux-gnueabihf/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-linux)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-linux/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-android)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/armv7-linux-androideabi/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-android)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-linux-androideabi/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),armv7l-qnx)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/ARMv7-linux-QNX/
|
||||
else ifeq ($(TARGET_ARCH)-$(TARGET_OS),aarch64-qnx)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/aarch64-qnx/
|
||||
else ifeq ($(TARGET_ARCH),ppc64le)
|
||||
CUDA_INSTALL_TARGET_DIR = targets/ppc64le-linux/
|
||||
endif
|
||||
|
||||
# Debug build flags
|
||||
ifeq ($(dbg),1)
|
||||
NVCCFLAGS += -g -G
|
||||
BUILD_TYPE := debug
|
||||
else
|
||||
BUILD_TYPE := release
|
||||
endif
|
||||
|
||||
ALL_CCFLAGS :=
|
||||
ALL_CCFLAGS += $(NVCCFLAGS)
|
||||
ALL_CCFLAGS += $(EXTRA_NVCCFLAGS)
|
||||
ALL_CCFLAGS += $(addprefix -Xcompiler ,$(CCFLAGS))
|
||||
ALL_CCFLAGS += $(addprefix -Xcompiler ,$(EXTRA_CCFLAGS))
|
||||
|
||||
SAMPLE_ENABLED := 1
|
||||
|
||||
# This sample is not supported on ARMv7
|
||||
ifeq ($(TARGET_ARCH),armv7l)
|
||||
$(info >>> WARNING - cuSolverDn_LinearSolver is not supported on ARMv7 - waiving sample <<<)
|
||||
SAMPLE_ENABLED := 0
|
||||
endif
|
||||
|
||||
ifeq ($(TARGET_OS),linux)
|
||||
ALL_CCFLAGS += -Xcompiler \"-Wl,--no-as-needed\"
|
||||
endif
|
||||
|
||||
ALL_LDFLAGS :=
|
||||
ALL_LDFLAGS += $(ALL_CCFLAGS)
|
||||
ALL_LDFLAGS += $(addprefix -Xlinker ,$(LDFLAGS))
|
||||
ALL_LDFLAGS += $(addprefix -Xlinker ,$(EXTRA_LDFLAGS))
|
||||
|
||||
# Common includes and paths for CUDA
|
||||
INCLUDES := -I../../Common
|
||||
LIBRARIES :=
|
||||
|
||||
################################################################################
|
||||
|
||||
LIBRARIES += -lcusolver -lcublas -lcusparse
|
||||
|
||||
ifeq ($(SAMPLE_ENABLED),0)
|
||||
EXEC ?= @echo "[@]"
|
||||
endif
|
||||
|
||||
################################################################################
|
||||
|
||||
# Target rules
|
||||
all: build
|
||||
|
||||
build: cuSolverDn_LinearSolver
|
||||
|
||||
check.deps:
|
||||
ifeq ($(SAMPLE_ENABLED),0)
|
||||
@echo "Sample will be waived due to the above missing dependencies"
|
||||
else
|
||||
@echo "Sample is ready - all dependencies have been met"
|
||||
endif
|
||||
|
||||
cuSolverDn_LinearSolver.o:cuSolverDn_LinearSolver.cpp
|
||||
$(EXEC) $(NVCC) $(INCLUDES) $(ALL_CCFLAGS) $(GENCODE_FLAGS) -o $@ -c $<
|
||||
|
||||
mmio.c.o:mmio.c
|
||||
$(EXEC) $(NVCC) $(INCLUDES) $(ALL_CCFLAGS) $(GENCODE_FLAGS) -o $@ -c $<
|
||||
|
||||
mmio_wrapper.o:mmio_wrapper.cpp
|
||||
$(EXEC) $(NVCC) $(INCLUDES) $(ALL_CCFLAGS) $(GENCODE_FLAGS) -o $@ -c $<
|
||||
|
||||
cuSolverDn_LinearSolver: cuSolverDn_LinearSolver.o mmio.c.o mmio_wrapper.o
|
||||
$(EXEC) $(NVCC) $(ALL_LDFLAGS) $(GENCODE_FLAGS) -o $@ $+ $(LIBRARIES)
|
||||
$(EXEC) mkdir -p ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)
|
||||
$(EXEC) cp $@ ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)
|
||||
|
||||
run: build
|
||||
$(EXEC) ./cuSolverDn_LinearSolver
|
||||
|
||||
clean:
|
||||
rm -f cuSolverDn_LinearSolver cuSolverDn_LinearSolver.o mmio.c.o mmio_wrapper.o
|
||||
rm -rf ../../bin/$(TARGET_ARCH)/$(TARGET_OS)/$(BUILD_TYPE)/cuSolverDn_LinearSolver
|
||||
|
||||
clobber: clean
|
||||
95
Samples/cuSolverDn_LinearSolver/README.md
Normal file
95
Samples/cuSolverDn_LinearSolver/README.md
Normal file
@@ -0,0 +1,95 @@
|
||||
# cuSolverDn_LinearSolver - cuSolverDn Linear Solver
|
||||
|
||||
## Description
|
||||
|
||||
A CUDA Sample that demonstrates cuSolverDN's LU, QR and Cholesky factorization.
|
||||
|
||||
## Key Concepts
|
||||
|
||||
Linear Algebra, CUSOLVER Library
|
||||
|
||||
## Supported SM Architectures
|
||||
|
||||
[SM 3.0 ](https://developer.nvidia.com/cuda-gpus) [SM 3.5 ](https://developer.nvidia.com/cuda-gpus) [SM 3.7 ](https://developer.nvidia.com/cuda-gpus) [SM 5.0 ](https://developer.nvidia.com/cuda-gpus) [SM 5.2 ](https://developer.nvidia.com/cuda-gpus) [SM 6.0 ](https://developer.nvidia.com/cuda-gpus) [SM 6.1 ](https://developer.nvidia.com/cuda-gpus) [SM 7.0 ](https://developer.nvidia.com/cuda-gpus) [SM 7.2 ](https://developer.nvidia.com/cuda-gpus) [SM 7.5 ](https://developer.nvidia.com/cuda-gpus)
|
||||
|
||||
## Supported OSes
|
||||
|
||||
Linux, Windows, MacOSX
|
||||
|
||||
## Supported CPU Architecture
|
||||
|
||||
x86_64, ppc64le, aarch64
|
||||
|
||||
## CUDA APIs involved
|
||||
|
||||
## Dependencies needed to build/run
|
||||
[CUSOLVER](../../README.md#cusolver), [CUBLAS](../../README.md#cublas), [CUSPARSE](../../README.md#cusparse)
|
||||
|
||||
## Prerequisites
|
||||
|
||||
Download and install the [CUDA Toolkit 10.1](https://developer.nvidia.com/cuda-downloads) for your corresponding platform.
|
||||
Make sure the dependencies mentioned in [Dependencies]() section above are installed.
|
||||
|
||||
## Build and Run
|
||||
|
||||
### Windows
|
||||
The Windows samples are built using the Visual Studio IDE. Solution files (.sln) are provided for each supported version of Visual Studio, using the format:
|
||||
```
|
||||
*_vs<version>.sln - for Visual Studio <version>
|
||||
```
|
||||
Each individual sample has its own set of solution files in its directory:
|
||||
|
||||
To build/examine all the samples at once, the complete solution files should be used. To build/examine a single sample, the individual sample solution files should be used.
|
||||
> **Note:** Some samples require that the Microsoft DirectX SDK (June 2010 or newer) be installed and that the VC++ directory paths are properly set up (**Tools > Options...**). Check DirectX Dependencies section for details."
|
||||
|
||||
### Linux
|
||||
The Linux samples are built using makefiles. To use the makefiles, change the current directory to the sample directory you wish to build, and run make:
|
||||
```
|
||||
$ cd <sample_dir>
|
||||
$ make
|
||||
```
|
||||
The samples makefiles can take advantage of certain options:
|
||||
* **TARGET_ARCH=<arch>** - cross-compile targeting a specific architecture. Allowed architectures are x86_64, ppc64le, aarch64.
|
||||
By default, TARGET_ARCH is set to HOST_ARCH. On a x86_64 machine, not setting TARGET_ARCH is the equivalent of setting TARGET_ARCH=x86_64.<br/>
|
||||
`$ make TARGET_ARCH=x86_64` <br/> `$ make TARGET_ARCH=ppc64le` <br/> `$ make TARGET_ARCH=aarch64` <br/>
|
||||
See [here](http://docs.nvidia.com/cuda/cuda-samples/index.html#cross-samples) for more details.
|
||||
* **dbg=1** - build with debug symbols
|
||||
```
|
||||
$ make dbg=1
|
||||
```
|
||||
* **SMS="A B ..."** - override the SM architectures for which the sample will be built, where `"A B ..."` is a space-delimited list of SM architectures. For example, to generate SASS for SM 50 and SM 60, use `SMS="50 60"`.
|
||||
```
|
||||
$ make SMS="50 60"
|
||||
```
|
||||
|
||||
* **HOST_COMPILER=<host_compiler>** - override the default g++ host compiler. See the [Linux Installation Guide](http://docs.nvidia.com/cuda/cuda-installation-guide-linux/index.html#system-requirements) for a list of supported host compilers.
|
||||
```
|
||||
$ make HOST_COMPILER=g++
|
||||
```
|
||||
|
||||
### Mac
|
||||
The Mac samples are built using makefiles. To use the makefiles, change directory into the sample directory you wish to build, and run make:
|
||||
```
|
||||
$ cd <sample_dir>
|
||||
$ make
|
||||
```
|
||||
|
||||
The samples makefiles can take advantage of certain options:
|
||||
|
||||
* **dbg=1** - build with debug symbols
|
||||
```
|
||||
$ make dbg=1
|
||||
```
|
||||
|
||||
* **SMS="A B ..."** - override the SM architectures for which the sample will be built, where "A B ..." is a space-delimited list of SM architectures. For example, to generate SASS for SM 50 and SM 60, use SMS="50 60".
|
||||
```
|
||||
$ make SMS="A B ..."
|
||||
```
|
||||
|
||||
* **HOST_COMPILER=<host_compiler>** - override the default clang host compiler. See the [Mac Installation Guide](http://docs.nvidia.com/cuda/cuda-installation-guide-mac-os-x/index.html#system-requirements) for a list of supported host compilers.
|
||||
```
|
||||
$ make HOST_COMPILER=clang
|
||||
```
|
||||
|
||||
## References (for more details)
|
||||
|
||||
584
Samples/cuSolverDn_LinearSolver/cuSolverDn_LinearSolver.cpp
Normal file
584
Samples/cuSolverDn_LinearSolver/cuSolverDn_LinearSolver.cpp
Normal file
@@ -0,0 +1,584 @@
|
||||
/* Copyright (c) 2019, NVIDIA CORPORATION. All rights reserved.
|
||||
*
|
||||
* Redistribution and use in source and binary forms, with or without
|
||||
* modification, are permitted provided that the following conditions
|
||||
* are met:
|
||||
* * Redistributions of source code must retain the above copyright
|
||||
* notice, this list of conditions and the following disclaimer.
|
||||
* * Redistributions in binary form must reproduce the above copyright
|
||||
* notice, this list of conditions and the following disclaimer in the
|
||||
* documentation and/or other materials provided with the distribution.
|
||||
* * Neither the name of NVIDIA CORPORATION nor the names of its
|
||||
* contributors may be used to endorse or promote products derived
|
||||
* from this software without specific prior written permission.
|
||||
*
|
||||
* THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS ``AS IS'' AND ANY
|
||||
* EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT LIMITED TO, THE
|
||||
* IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A PARTICULAR
|
||||
* PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT OWNER OR
|
||||
* CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
|
||||
* EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
|
||||
* PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR
|
||||
* PROFITS; OR BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY
|
||||
* OF LIABILITY, WHETHER IN CONTRACT, STRICT LIABILITY, OR TORT
|
||||
* (INCLUDING NEGLIGENCE OR OTHERWISE) ARISING IN ANY WAY OUT OF THE USE
|
||||
* OF THIS SOFTWARE, EVEN IF ADVISED OF THE POSSIBILITY OF SUCH DAMAGE.
|
||||
*/
|
||||
|
||||
/*
|
||||
* Test three linear solvers, including Cholesky, LU and QR.
|
||||
* The user has to prepare a sparse matrix of "matrix market format" (with
|
||||
* extension .mtx). For example, the user can download matrices in Florida
|
||||
* Sparse Matrix Collection.
|
||||
* (http://www.cise.ufl.edu/research/sparse/matrices/)
|
||||
*
|
||||
* The user needs to choose a solver by switch -R<solver> and
|
||||
* to provide the path of the matrix by switch -F<file>, then
|
||||
* the program solves
|
||||
* A*x = b where b = ones(m,1)
|
||||
* and reports relative error
|
||||
* |b-A*x|/(|A|*|x|)
|
||||
*
|
||||
* The elapsed time is also reported so the user can compare efficiency of
|
||||
* different solvers.
|
||||
*
|
||||
* How to use
|
||||
* ./cuSolverDn_LinearSolver // Default: cholesky
|
||||
* ./cuSolverDn_LinearSolver -R=chol -filefile> // cholesky factorization
|
||||
* ./cuSolverDn_LinearSolver -R=lu -file<file> // LU with partial
|
||||
* pivoting
|
||||
* ./cuSolverDn_LinearSolver -R=qr -file<file> // QR factorization
|
||||
*
|
||||
* Remark: the absolute error on solution x is meaningless without knowing
|
||||
* condition number of A. The relative error on residual should be close to
|
||||
* machine zero, i.e. 1.e-15.
|
||||
*/
|
||||
|
||||
#include <assert.h>
|
||||
#include <ctype.h>
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
#include <cuda_runtime.h>
|
||||
|
||||
#include "cublas_v2.h"
|
||||
#include "cusolverDn.h"
|
||||
#include "helper_cuda.h"
|
||||
|
||||
#include "helper_cusolver.h"
|
||||
|
||||
template <typename T_ELEM>
|
||||
int loadMMSparseMatrix(char *filename, char elem_type, bool csrFormat, int *m,
|
||||
int *n, int *nnz, T_ELEM **aVal, int **aRowInd,
|
||||
int **aColInd, int extendSymMatrix);
|
||||
|
||||
void UsageDN(void) {
|
||||
printf("<options>\n");
|
||||
printf("-h : display this help\n");
|
||||
printf("-R=<name> : choose a linear solver\n");
|
||||
printf(" chol (cholesky factorization), this is default\n");
|
||||
printf(" qr (QR factorization)\n");
|
||||
printf(" lu (LU factorization)\n");
|
||||
printf("-lda=<int> : leading dimension of A , m by default\n");
|
||||
printf("-file=<filename>: filename containing a matrix in MM format\n");
|
||||
printf("-device=<device_id> : <device_id> if want to run on specific GPU\n");
|
||||
|
||||
exit(0);
|
||||
}
|
||||
|
||||
/*
|
||||
* solve A*x = b by Cholesky factorization
|
||||
*
|
||||
*/
|
||||
int linearSolverCHOL(cusolverDnHandle_t handle, int n, const double *Acopy,
|
||||
int lda, const double *b, double *x) {
|
||||
int bufferSize = 0;
|
||||
int *info = NULL;
|
||||
double *buffer = NULL;
|
||||
double *A = NULL;
|
||||
int h_info = 0;
|
||||
double start, stop;
|
||||
double time_solve;
|
||||
cublasFillMode_t uplo = CUBLAS_FILL_MODE_LOWER;
|
||||
|
||||
checkCudaErrors(cusolverDnDpotrf_bufferSize(handle, uplo, n, (double *)Acopy,
|
||||
lda, &bufferSize));
|
||||
|
||||
checkCudaErrors(cudaMalloc(&info, sizeof(int)));
|
||||
checkCudaErrors(cudaMalloc(&buffer, sizeof(double) * bufferSize));
|
||||
checkCudaErrors(cudaMalloc(&A, sizeof(double) * lda * n));
|
||||
|
||||
// prepare a copy of A because potrf will overwrite A with L
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(A, Acopy, sizeof(double) * lda * n, cudaMemcpyDeviceToDevice));
|
||||
checkCudaErrors(cudaMemset(info, 0, sizeof(int)));
|
||||
|
||||
start = second();
|
||||
start = second();
|
||||
|
||||
checkCudaErrors(
|
||||
cusolverDnDpotrf(handle, uplo, n, A, lda, buffer, bufferSize, info));
|
||||
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(&h_info, info, sizeof(int), cudaMemcpyDeviceToHost));
|
||||
|
||||
if (0 != h_info) {
|
||||
fprintf(stderr, "Error: Cholesky factorization failed\n");
|
||||
}
|
||||
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(x, b, sizeof(double) * n, cudaMemcpyDeviceToDevice));
|
||||
|
||||
checkCudaErrors(cusolverDnDpotrs(handle, uplo, n, 1, A, lda, x, n, info));
|
||||
|
||||
checkCudaErrors(cudaDeviceSynchronize());
|
||||
stop = second();
|
||||
|
||||
time_solve = stop - start;
|
||||
fprintf(stdout, "timing: cholesky = %10.6f sec\n", time_solve);
|
||||
|
||||
if (info) {
|
||||
checkCudaErrors(cudaFree(info));
|
||||
}
|
||||
if (buffer) {
|
||||
checkCudaErrors(cudaFree(buffer));
|
||||
}
|
||||
if (A) {
|
||||
checkCudaErrors(cudaFree(A));
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
/*
|
||||
* solve A*x = b by LU with partial pivoting
|
||||
*
|
||||
*/
|
||||
int linearSolverLU(cusolverDnHandle_t handle, int n, const double *Acopy,
|
||||
int lda, const double *b, double *x) {
|
||||
int bufferSize = 0;
|
||||
int *info = NULL;
|
||||
double *buffer = NULL;
|
||||
double *A = NULL;
|
||||
int *ipiv = NULL; // pivoting sequence
|
||||
int h_info = 0;
|
||||
double start, stop;
|
||||
double time_solve;
|
||||
|
||||
checkCudaErrors(cusolverDnDgetrf_bufferSize(handle, n, n, (double *)Acopy,
|
||||
lda, &bufferSize));
|
||||
|
||||
checkCudaErrors(cudaMalloc(&info, sizeof(int)));
|
||||
checkCudaErrors(cudaMalloc(&buffer, sizeof(double) * bufferSize));
|
||||
checkCudaErrors(cudaMalloc(&A, sizeof(double) * lda * n));
|
||||
checkCudaErrors(cudaMalloc(&ipiv, sizeof(int) * n));
|
||||
|
||||
// prepare a copy of A because getrf will overwrite A with L
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(A, Acopy, sizeof(double) * lda * n, cudaMemcpyDeviceToDevice));
|
||||
checkCudaErrors(cudaMemset(info, 0, sizeof(int)));
|
||||
|
||||
start = second();
|
||||
start = second();
|
||||
|
||||
checkCudaErrors(cusolverDnDgetrf(handle, n, n, A, lda, buffer, ipiv, info));
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(&h_info, info, sizeof(int), cudaMemcpyDeviceToHost));
|
||||
|
||||
if (0 != h_info) {
|
||||
fprintf(stderr, "Error: LU factorization failed\n");
|
||||
}
|
||||
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(x, b, sizeof(double) * n, cudaMemcpyDeviceToDevice));
|
||||
checkCudaErrors(
|
||||
cusolverDnDgetrs(handle, CUBLAS_OP_N, n, 1, A, lda, ipiv, x, n, info));
|
||||
checkCudaErrors(cudaDeviceSynchronize());
|
||||
stop = second();
|
||||
|
||||
time_solve = stop - start;
|
||||
fprintf(stdout, "timing: LU = %10.6f sec\n", time_solve);
|
||||
|
||||
if (info) {
|
||||
checkCudaErrors(cudaFree(info));
|
||||
}
|
||||
if (buffer) {
|
||||
checkCudaErrors(cudaFree(buffer));
|
||||
}
|
||||
if (A) {
|
||||
checkCudaErrors(cudaFree(A));
|
||||
}
|
||||
if (ipiv) {
|
||||
checkCudaErrors(cudaFree(ipiv));
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
/*
|
||||
* solve A*x = b by QR
|
||||
*
|
||||
*/
|
||||
int linearSolverQR(cusolverDnHandle_t handle, int n, const double *Acopy,
|
||||
int lda, const double *b, double *x) {
|
||||
cublasHandle_t cublasHandle = NULL; // used in residual evaluation
|
||||
int bufferSize = 0;
|
||||
int bufferSize_geqrf = 0;
|
||||
int bufferSize_ormqr = 0;
|
||||
int *info = NULL;
|
||||
double *buffer = NULL;
|
||||
double *A = NULL;
|
||||
double *tau = NULL;
|
||||
int h_info = 0;
|
||||
double start, stop;
|
||||
double time_solve;
|
||||
const double one = 1.0;
|
||||
|
||||
checkCudaErrors(cublasCreate(&cublasHandle));
|
||||
|
||||
checkCudaErrors(cusolverDnDgeqrf_bufferSize(handle, n, n, (double *)Acopy,
|
||||
lda, &bufferSize_geqrf));
|
||||
checkCudaErrors(cusolverDnDormqr_bufferSize(handle, CUBLAS_SIDE_LEFT,
|
||||
CUBLAS_OP_T, n, 1, n, A, lda,
|
||||
NULL, x, n, &bufferSize_ormqr));
|
||||
|
||||
printf("buffer_geqrf = %d, buffer_ormqr = %d \n", bufferSize_geqrf,
|
||||
bufferSize_ormqr);
|
||||
|
||||
bufferSize = (bufferSize_geqrf > bufferSize_ormqr) ? bufferSize_geqrf
|
||||
: bufferSize_ormqr;
|
||||
|
||||
checkCudaErrors(cudaMalloc(&info, sizeof(int)));
|
||||
checkCudaErrors(cudaMalloc(&buffer, sizeof(double) * bufferSize));
|
||||
checkCudaErrors(cudaMalloc(&A, sizeof(double) * lda * n));
|
||||
checkCudaErrors(cudaMalloc((void **)&tau, sizeof(double) * n));
|
||||
|
||||
// prepare a copy of A because getrf will overwrite A with L
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(A, Acopy, sizeof(double) * lda * n, cudaMemcpyDeviceToDevice));
|
||||
|
||||
checkCudaErrors(cudaMemset(info, 0, sizeof(int)));
|
||||
|
||||
start = second();
|
||||
start = second();
|
||||
|
||||
// compute QR factorization
|
||||
checkCudaErrors(
|
||||
cusolverDnDgeqrf(handle, n, n, A, lda, tau, buffer, bufferSize, info));
|
||||
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(&h_info, info, sizeof(int), cudaMemcpyDeviceToHost));
|
||||
|
||||
if (0 != h_info) {
|
||||
fprintf(stderr, "Error: LU factorization failed\n");
|
||||
}
|
||||
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(x, b, sizeof(double) * n, cudaMemcpyDeviceToDevice));
|
||||
|
||||
// compute Q^T*b
|
||||
checkCudaErrors(cusolverDnDormqr(handle, CUBLAS_SIDE_LEFT, CUBLAS_OP_T, n, 1,
|
||||
n, A, lda, tau, x, n, buffer, bufferSize,
|
||||
info));
|
||||
|
||||
// x = R \ Q^T*b
|
||||
checkCudaErrors(cublasDtrsm(cublasHandle, CUBLAS_SIDE_LEFT,
|
||||
CUBLAS_FILL_MODE_UPPER, CUBLAS_OP_N,
|
||||
CUBLAS_DIAG_NON_UNIT, n, 1, &one, A, lda, x, n));
|
||||
checkCudaErrors(cudaDeviceSynchronize());
|
||||
stop = second();
|
||||
|
||||
time_solve = stop - start;
|
||||
fprintf(stdout, "timing: QR = %10.6f sec\n", time_solve);
|
||||
|
||||
if (cublasHandle) {
|
||||
checkCudaErrors(cublasDestroy(cublasHandle));
|
||||
}
|
||||
if (info) {
|
||||
checkCudaErrors(cudaFree(info));
|
||||
}
|
||||
if (buffer) {
|
||||
checkCudaErrors(cudaFree(buffer));
|
||||
}
|
||||
if (A) {
|
||||
checkCudaErrors(cudaFree(A));
|
||||
}
|
||||
if (tau) {
|
||||
checkCudaErrors(cudaFree(tau));
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
void parseCommandLineArguments(int argc, char *argv[], struct testOpts &opts) {
|
||||
memset(&opts, 0, sizeof(opts));
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "-h")) {
|
||||
UsageDN();
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "R")) {
|
||||
char *solverType = NULL;
|
||||
getCmdLineArgumentString(argc, (const char **)argv, "R", &solverType);
|
||||
|
||||
if (solverType) {
|
||||
if ((STRCASECMP(solverType, "chol") != 0) &&
|
||||
(STRCASECMP(solverType, "lu") != 0) &&
|
||||
(STRCASECMP(solverType, "qr") != 0)) {
|
||||
printf("\nIncorrect argument passed to -R option\n");
|
||||
UsageDN();
|
||||
} else {
|
||||
opts.testFunc = solverType;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "file")) {
|
||||
char *fileName = 0;
|
||||
getCmdLineArgumentString(argc, (const char **)argv, "file", &fileName);
|
||||
|
||||
if (fileName) {
|
||||
opts.sparse_mat_filename = fileName;
|
||||
} else {
|
||||
printf("\nIncorrect filename passed to -file \n ");
|
||||
UsageDN();
|
||||
}
|
||||
}
|
||||
|
||||
if (checkCmdLineFlag(argc, (const char **)argv, "lda")) {
|
||||
opts.lda = getCmdLineArgumentInt(argc, (const char **)argv, "lda");
|
||||
}
|
||||
}
|
||||
|
||||
int main(int argc, char *argv[]) {
|
||||
struct testOpts opts;
|
||||
cusolverDnHandle_t handle = NULL;
|
||||
cublasHandle_t cublasHandle = NULL; // used in residual evaluation
|
||||
cudaStream_t stream = NULL;
|
||||
|
||||
int rowsA = 0; // number of rows of A
|
||||
int colsA = 0; // number of columns of A
|
||||
int nnzA = 0; // number of nonzeros of A
|
||||
int baseA = 0; // base index in CSR format
|
||||
int lda = 0; // leading dimension in dense matrix
|
||||
|
||||
// CSR(A) from I/O
|
||||
int *h_csrRowPtrA = NULL;
|
||||
int *h_csrColIndA = NULL;
|
||||
double *h_csrValA = NULL;
|
||||
|
||||
double *h_A = NULL; // dense matrix from CSR(A)
|
||||
double *h_x = NULL; // a copy of d_x
|
||||
double *h_b = NULL; // b = ones(m,1)
|
||||
double *h_r = NULL; // r = b - A*x, a copy of d_r
|
||||
|
||||
double *d_A = NULL; // a copy of h_A
|
||||
double *d_x = NULL; // x = A \ b
|
||||
double *d_b = NULL; // a copy of h_b
|
||||
double *d_r = NULL; // r = b - A*x
|
||||
|
||||
// the constants are used in residual evaluation, r = b - A*x
|
||||
const double minus_one = -1.0;
|
||||
const double one = 1.0;
|
||||
|
||||
double x_inf = 0.0;
|
||||
double r_inf = 0.0;
|
||||
double A_inf = 0.0;
|
||||
int errors = 0;
|
||||
|
||||
parseCommandLineArguments(argc, argv, opts);
|
||||
|
||||
if (NULL == opts.testFunc) {
|
||||
opts.testFunc = "chol"; // By default running Cholesky as NO solver
|
||||
// selected with -R option.
|
||||
}
|
||||
|
||||
findCudaDevice(argc, (const char **)argv);
|
||||
|
||||
printf("step 1: read matrix market format\n");
|
||||
|
||||
if (opts.sparse_mat_filename == NULL) {
|
||||
opts.sparse_mat_filename = sdkFindFilePath("gr_900_900_crg.mtx", argv[0]);
|
||||
if (opts.sparse_mat_filename != NULL)
|
||||
printf("Using default input file [%s]\n", opts.sparse_mat_filename);
|
||||
else
|
||||
printf("Could not find gr_900_900_crg.mtx\n");
|
||||
} else {
|
||||
printf("Using input file [%s]\n", opts.sparse_mat_filename);
|
||||
}
|
||||
|
||||
if (opts.sparse_mat_filename == NULL) {
|
||||
fprintf(stderr, "Error: input matrix is not provided\n");
|
||||
return EXIT_FAILURE;
|
||||
}
|
||||
|
||||
if (loadMMSparseMatrix<double>(opts.sparse_mat_filename, 'd', true, &rowsA,
|
||||
&colsA, &nnzA, &h_csrValA, &h_csrRowPtrA,
|
||||
&h_csrColIndA, true)) {
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
baseA = h_csrRowPtrA[0]; // baseA = {0,1}
|
||||
|
||||
printf("sparse matrix A is %d x %d with %d nonzeros, base=%d\n", rowsA, colsA,
|
||||
nnzA, baseA);
|
||||
|
||||
if (rowsA != colsA) {
|
||||
fprintf(stderr, "Error: only support square matrix\n");
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
printf("step 2: convert CSR(A) to dense matrix\n");
|
||||
|
||||
lda = opts.lda ? opts.lda : rowsA;
|
||||
if (lda < rowsA) {
|
||||
fprintf(stderr, "Error: lda must be greater or equal to dimension of A\n");
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
|
||||
h_A = (double *)malloc(sizeof(double) * lda * colsA);
|
||||
h_x = (double *)malloc(sizeof(double) * colsA);
|
||||
h_b = (double *)malloc(sizeof(double) * rowsA);
|
||||
h_r = (double *)malloc(sizeof(double) * rowsA);
|
||||
assert(NULL != h_A);
|
||||
assert(NULL != h_x);
|
||||
assert(NULL != h_b);
|
||||
assert(NULL != h_r);
|
||||
|
||||
memset(h_A, 0, sizeof(double) * lda * colsA);
|
||||
|
||||
for (int row = 0; row < rowsA; row++) {
|
||||
const int start = h_csrRowPtrA[row] - baseA;
|
||||
const int end = h_csrRowPtrA[row + 1] - baseA;
|
||||
for (int colidx = start; colidx < end; colidx++) {
|
||||
const int col = h_csrColIndA[colidx] - baseA;
|
||||
const double Areg = h_csrValA[colidx];
|
||||
h_A[row + col * lda] = Areg;
|
||||
}
|
||||
}
|
||||
|
||||
printf("step 3: set right hand side vector (b) to 1\n");
|
||||
for (int row = 0; row < rowsA; row++) {
|
||||
h_b[row] = 1.0;
|
||||
}
|
||||
|
||||
// verify if A is symmetric or not.
|
||||
if (0 == strcmp(opts.testFunc, "chol")) {
|
||||
int issym = 1;
|
||||
for (int j = 0; j < colsA; j++) {
|
||||
for (int i = j; i < rowsA; i++) {
|
||||
double Aij = h_A[i + j * lda];
|
||||
double Aji = h_A[j + i * lda];
|
||||
if (Aij != Aji) {
|
||||
issym = 0;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (!issym) {
|
||||
printf("Error: A has no symmetric pattern, please use LU or QR \n");
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
}
|
||||
|
||||
checkCudaErrors(cusolverDnCreate(&handle));
|
||||
checkCudaErrors(cublasCreate(&cublasHandle));
|
||||
checkCudaErrors(cudaStreamCreate(&stream));
|
||||
|
||||
checkCudaErrors(cusolverDnSetStream(handle, stream));
|
||||
checkCudaErrors(cublasSetStream(cublasHandle, stream));
|
||||
|
||||
checkCudaErrors(cudaMalloc((void **)&d_A, sizeof(double) * lda * colsA));
|
||||
checkCudaErrors(cudaMalloc((void **)&d_x, sizeof(double) * colsA));
|
||||
checkCudaErrors(cudaMalloc((void **)&d_b, sizeof(double) * rowsA));
|
||||
checkCudaErrors(cudaMalloc((void **)&d_r, sizeof(double) * rowsA));
|
||||
|
||||
printf("step 4: prepare data on device\n");
|
||||
checkCudaErrors(cudaMemcpy(d_A, h_A, sizeof(double) * lda * colsA,
|
||||
cudaMemcpyHostToDevice));
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(d_b, h_b, sizeof(double) * rowsA, cudaMemcpyHostToDevice));
|
||||
|
||||
printf("step 5: solve A*x = b \n");
|
||||
// d_A and d_b are read-only
|
||||
if (0 == strcmp(opts.testFunc, "chol")) {
|
||||
linearSolverCHOL(handle, rowsA, d_A, lda, d_b, d_x);
|
||||
} else if (0 == strcmp(opts.testFunc, "lu")) {
|
||||
linearSolverLU(handle, rowsA, d_A, lda, d_b, d_x);
|
||||
} else if (0 == strcmp(opts.testFunc, "qr")) {
|
||||
linearSolverQR(handle, rowsA, d_A, lda, d_b, d_x);
|
||||
} else {
|
||||
fprintf(stderr, "Error: %s is unknown function\n", opts.testFunc);
|
||||
exit(EXIT_FAILURE);
|
||||
}
|
||||
printf("step 6: evaluate residual\n");
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(d_r, d_b, sizeof(double) * rowsA, cudaMemcpyDeviceToDevice));
|
||||
|
||||
// r = b - A*x
|
||||
checkCudaErrors(cublasDgemm_v2(cublasHandle, CUBLAS_OP_N, CUBLAS_OP_N, rowsA,
|
||||
1, colsA, &minus_one, d_A, lda, d_x, rowsA,
|
||||
&one, d_r, rowsA));
|
||||
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(h_x, d_x, sizeof(double) * colsA, cudaMemcpyDeviceToHost));
|
||||
checkCudaErrors(
|
||||
cudaMemcpy(h_r, d_r, sizeof(double) * rowsA, cudaMemcpyDeviceToHost));
|
||||
|
||||
x_inf = vec_norminf(colsA, h_x);
|
||||
r_inf = vec_norminf(rowsA, h_r);
|
||||
A_inf = mat_norminf(rowsA, colsA, h_A, lda);
|
||||
|
||||
printf("|b - A*x| = %E \n", r_inf);
|
||||
printf("|A| = %E \n", A_inf);
|
||||
printf("|x| = %E \n", x_inf);
|
||||
printf("|b - A*x|/(|A|*|x|) = %E \n", r_inf / (A_inf * x_inf));
|
||||
|
||||
if (handle) {
|
||||
checkCudaErrors(cusolverDnDestroy(handle));
|
||||
}
|
||||
if (cublasHandle) {
|
||||
checkCudaErrors(cublasDestroy(cublasHandle));
|
||||
}
|
||||
if (stream) {
|
||||
checkCudaErrors(cudaStreamDestroy(stream));
|
||||
}
|
||||
|
||||
if (h_csrValA) {
|
||||
free(h_csrValA);
|
||||
}
|
||||
if (h_csrRowPtrA) {
|
||||
free(h_csrRowPtrA);
|
||||
}
|
||||
if (h_csrColIndA) {
|
||||
free(h_csrColIndA);
|
||||
}
|
||||
|
||||
if (h_A) {
|
||||
free(h_A);
|
||||
}
|
||||
if (h_x) {
|
||||
free(h_x);
|
||||
}
|
||||
if (h_b) {
|
||||
free(h_b);
|
||||
}
|
||||
if (h_r) {
|
||||
free(h_r);
|
||||
}
|
||||
|
||||
if (d_A) {
|
||||
checkCudaErrors(cudaFree(d_A));
|
||||
}
|
||||
if (d_x) {
|
||||
checkCudaErrors(cudaFree(d_x));
|
||||
}
|
||||
if (d_b) {
|
||||
checkCudaErrors(cudaFree(d_b));
|
||||
}
|
||||
if (d_r) {
|
||||
checkCudaErrors(cudaFree(d_r));
|
||||
}
|
||||
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 12.00
|
||||
# Visual Studio 2012
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "cuSolverDn_LinearSolver", "cuSolverDn_LinearSolver_vs2012.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
@@ -0,0 +1,109 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>cuSolverDn_LinearSolver_vs2012</RootNamespace>
|
||||
<ProjectName>cuSolverDn_LinearSolver</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v110</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;$(CudaToolkitIncludeDir);</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cusolver.lib;cublas.lib;cusparse.lib;cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/cuSolverDn_LinearSolver.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration></CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<ClCompile Include="cuSolverDn_LinearSolver.cpp" />
|
||||
<ClCompile Include="mmio.c" />
|
||||
<ClCompile Include="mmio_wrapper.cpp" />
|
||||
<ClInclude Include="mmio.h" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 13.00
|
||||
# Visual Studio 2013
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "cuSolverDn_LinearSolver", "cuSolverDn_LinearSolver_vs2013.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
@@ -0,0 +1,109 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>cuSolverDn_LinearSolver_vs2013</RootNamespace>
|
||||
<ProjectName>cuSolverDn_LinearSolver</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v120</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;$(CudaToolkitIncludeDir);</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cusolver.lib;cublas.lib;cusparse.lib;cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/cuSolverDn_LinearSolver.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration></CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<ClCompile Include="cuSolverDn_LinearSolver.cpp" />
|
||||
<ClCompile Include="mmio.c" />
|
||||
<ClCompile Include="mmio_wrapper.cpp" />
|
||||
<ClInclude Include="mmio.h" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 14.00
|
||||
# Visual Studio 2015
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "cuSolverDn_LinearSolver", "cuSolverDn_LinearSolver_vs2015.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
@@ -0,0 +1,109 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>cuSolverDn_LinearSolver_vs2015</RootNamespace>
|
||||
<ProjectName>cuSolverDn_LinearSolver</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v140</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;$(CudaToolkitIncludeDir);</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cusolver.lib;cublas.lib;cusparse.lib;cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/cuSolverDn_LinearSolver.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration></CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<ClCompile Include="cuSolverDn_LinearSolver.cpp" />
|
||||
<ClCompile Include="mmio.c" />
|
||||
<ClCompile Include="mmio_wrapper.cpp" />
|
||||
<ClInclude Include="mmio.h" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 12.00
|
||||
# Visual Studio 2017
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "cuSolverDn_LinearSolver", "cuSolverDn_LinearSolver_vs2017.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
@@ -0,0 +1,114 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>cuSolverDn_LinearSolver_vs2017</RootNamespace>
|
||||
<ProjectName>cuSolverDn_LinearSolver</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(WindowsTargetPlatformVersion)'==''">
|
||||
<LatestTargetPlatformVersion>$([Microsoft.Build.Utilities.ToolLocationHelper]::GetLatestSDKTargetPlatformVersion('Windows', '10.0'))</LatestTargetPlatformVersion>
|
||||
<WindowsTargetPlatformVersion Condition="'$(WindowsTargetPlatformVersion)' == ''">$(LatestTargetPlatformVersion)</WindowsTargetPlatformVersion>
|
||||
<TargetPlatformVersion>$(WindowsTargetPlatformVersion)</TargetPlatformVersion>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v141</PlatformToolset>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;$(CudaToolkitIncludeDir);</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cusolver.lib;cublas.lib;cusparse.lib;cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/cuSolverDn_LinearSolver.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration></CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<ClCompile Include="cuSolverDn_LinearSolver.cpp" />
|
||||
<ClCompile Include="mmio.c" />
|
||||
<ClCompile Include="mmio_wrapper.cpp" />
|
||||
<ClInclude Include="mmio.h" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
@@ -0,0 +1,20 @@
|
||||
|
||||
Microsoft Visual Studio Solution File, Format Version 12.00
|
||||
# Visual Studio 2019
|
||||
Project("{8BC9CEB8-8B4A-11D0-8D11-00A0C91BC942}") = "cuSolverDn_LinearSolver", "cuSolverDn_LinearSolver_vs2019.vcxproj", "{997E0757-EA74-4A4E-A0FC-47D8C8831A15}"
|
||||
EndProject
|
||||
Global
|
||||
GlobalSection(SolutionConfigurationPlatforms) = preSolution
|
||||
Debug|x64 = Debug|x64
|
||||
Release|x64 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(ProjectConfigurationPlatforms) = postSolution
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.ActiveCfg = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Debug|x64.Build.0 = Debug|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.ActiveCfg = Release|x64
|
||||
{997E0757-EA74-4A4E-A0FC-47D8C8831A15}.Release|x64.Build.0 = Release|x64
|
||||
EndGlobalSection
|
||||
GlobalSection(SolutionProperties) = preSolution
|
||||
HideSolutionNode = FALSE
|
||||
EndGlobalSection
|
||||
EndGlobal
|
||||
@@ -0,0 +1,110 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<Project DefaultTargets="Build" ToolsVersion="4.0" xmlns="http://schemas.microsoft.com/developer/msbuild/2003">
|
||||
<PropertyGroup>
|
||||
<CUDAPropsPath Condition="'$(CUDAPropsPath)'==''">$(VCTargetsPath)\BuildCustomizations</CUDAPropsPath>
|
||||
</PropertyGroup>
|
||||
<ItemGroup Label="ProjectConfigurations">
|
||||
<ProjectConfiguration Include="Debug|x64">
|
||||
<Configuration>Debug</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
<ProjectConfiguration Include="Release|x64">
|
||||
<Configuration>Release</Configuration>
|
||||
<Platform>x64</Platform>
|
||||
</ProjectConfiguration>
|
||||
</ItemGroup>
|
||||
<PropertyGroup Label="Globals">
|
||||
<ProjectGuid>{997E0757-EA74-4A4E-A0FC-47D8C8831A15}</ProjectGuid>
|
||||
<RootNamespace>cuSolverDn_LinearSolver_vs2019</RootNamespace>
|
||||
<ProjectName>cuSolverDn_LinearSolver</ProjectName>
|
||||
<CudaToolkitCustomDir />
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.Default.props" />
|
||||
<PropertyGroup>
|
||||
<ConfigurationType>Application</ConfigurationType>
|
||||
<CharacterSet>MultiByte</CharacterSet>
|
||||
<PlatformToolset>v142</PlatformToolset>
|
||||
<WindowsTargetPlatformVersion>10.0</WindowsTargetPlatformVersion>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<UseDebugLibraries>true</UseDebugLibraries>
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Configuration)'=='Release'">
|
||||
<WholeProgramOptimization>true</WholeProgramOptimization>
|
||||
</PropertyGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.props" />
|
||||
<ImportGroup Label="ExtensionSettings">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.props" />
|
||||
</ImportGroup>
|
||||
<ImportGroup Label="PropertySheets">
|
||||
<Import Condition="exists('$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props')" Label="LocalAppDataPlatform" Project="$(UserRootDir)\Microsoft.Cpp.$(Platform).user.props" />
|
||||
</ImportGroup>
|
||||
<PropertyGroup Label="UserMacros" />
|
||||
<PropertyGroup>
|
||||
<IntDir>$(Platform)/$(Configuration)/</IntDir>
|
||||
<IncludePath>$(IncludePath)</IncludePath>
|
||||
<CodeAnalysisRuleSet>AllRules.ruleset</CodeAnalysisRuleSet>
|
||||
<CodeAnalysisRules />
|
||||
<CodeAnalysisRuleAssemblies />
|
||||
</PropertyGroup>
|
||||
<PropertyGroup Condition="'$(Platform)'=='x64'">
|
||||
<OutDir>../../bin/win64/$(Configuration)/</OutDir>
|
||||
</PropertyGroup>
|
||||
<ItemDefinitionGroup>
|
||||
<ClCompile>
|
||||
<WarningLevel>Level3</WarningLevel>
|
||||
<PreprocessorDefinitions>WIN32;_MBCS;%(PreprocessorDefinitions)</PreprocessorDefinitions>
|
||||
<AdditionalIncludeDirectories>./;$(CudaToolkitDir)/include;../../Common;$(CudaToolkitIncludeDir);</AdditionalIncludeDirectories>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<SubSystem>Console</SubSystem>
|
||||
<AdditionalDependencies>cusolver.lib;cublas.lib;cusparse.lib;cudart_static.lib;kernel32.lib;user32.lib;gdi32.lib;winspool.lib;comdlg32.lib;advapi32.lib;shell32.lib;ole32.lib;oleaut32.lib;uuid.lib;odbc32.lib;odbccp32.lib;%(AdditionalDependencies)</AdditionalDependencies>
|
||||
<AdditionalLibraryDirectories>$(CudaToolkitLibDir);</AdditionalLibraryDirectories>
|
||||
<OutputFile>$(OutDir)/cuSolverDn_LinearSolver.exe</OutputFile>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<CodeGeneration></CodeGeneration>
|
||||
<AdditionalOptions>-Xcompiler "/wd 4819" %(AdditionalOptions)</AdditionalOptions>
|
||||
<Include>./;../../Common</Include>
|
||||
<Defines>WIN32</Defines>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Debug'">
|
||||
<ClCompile>
|
||||
<Optimization>Disabled</Optimization>
|
||||
<RuntimeLibrary>MultiThreadedDebug</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>true</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>Default</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MTd</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemDefinitionGroup Condition="'$(Configuration)'=='Release'">
|
||||
<ClCompile>
|
||||
<Optimization>MaxSpeed</Optimization>
|
||||
<RuntimeLibrary>MultiThreaded</RuntimeLibrary>
|
||||
</ClCompile>
|
||||
<Link>
|
||||
<GenerateDebugInformation>false</GenerateDebugInformation>
|
||||
<LinkTimeCodeGeneration>UseLinkTimeCodeGeneration</LinkTimeCodeGeneration>
|
||||
</Link>
|
||||
<CudaCompile>
|
||||
<Runtime>MT</Runtime>
|
||||
<TargetMachinePlatform>64</TargetMachinePlatform>
|
||||
</CudaCompile>
|
||||
</ItemDefinitionGroup>
|
||||
<ItemGroup>
|
||||
<ClCompile Include="cuSolverDn_LinearSolver.cpp" />
|
||||
<ClCompile Include="mmio.c" />
|
||||
<ClCompile Include="mmio_wrapper.cpp" />
|
||||
<ClInclude Include="mmio.h" />
|
||||
</ItemGroup>
|
||||
<Import Project="$(VCTargetsPath)\Microsoft.Cpp.targets" />
|
||||
<ImportGroup Label="ExtensionTargets">
|
||||
<Import Project="$(CUDAPropsPath)\CUDA 10.1.targets" />
|
||||
</ImportGroup>
|
||||
</Project>
|
||||
4330
Samples/cuSolverDn_LinearSolver/gr_900_900_crg.mtx
Normal file
4330
Samples/cuSolverDn_LinearSolver/gr_900_900_crg.mtx
Normal file
File diff suppressed because it is too large
Load Diff
30803
Samples/cuSolverDn_LinearSolver/lap3D_7pt_n20.mtx
Normal file
30803
Samples/cuSolverDn_LinearSolver/lap3D_7pt_n20.mtx
Normal file
File diff suppressed because it is too large
Load Diff
521
Samples/cuSolverDn_LinearSolver/mmio.c
Normal file
521
Samples/cuSolverDn_LinearSolver/mmio.c
Normal file
@@ -0,0 +1,521 @@
|
||||
/*
|
||||
* Matrix Market I/O library for ANSI C
|
||||
*
|
||||
* See http://math.nist.gov/MatrixMarket for details.
|
||||
*
|
||||
*
|
||||
*/
|
||||
|
||||
/* avoid Windows warnings (for example: strcpy, fscanf, etc.) */
|
||||
#if defined(_WIN32)
|
||||
#define _CRT_SECURE_NO_WARNINGS
|
||||
#endif
|
||||
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
#include <stdlib.h>
|
||||
#include <ctype.h>
|
||||
|
||||
#include "mmio.h"
|
||||
|
||||
int mm_read_unsymmetric_sparse(const char *fname, int *M_, int *N_, int *nz_,
|
||||
double **val_, int **I_, int **J_)
|
||||
{
|
||||
FILE *f;
|
||||
MM_typecode matcode;
|
||||
int M, N, nz;
|
||||
int i;
|
||||
double *val;
|
||||
int *I, *J;
|
||||
|
||||
if ((f = fopen(fname, "r")) == NULL)
|
||||
return -1;
|
||||
|
||||
|
||||
if (mm_read_banner(f, &matcode) != 0)
|
||||
{
|
||||
printf("mm_read_unsymetric: Could not process Matrix Market banner ");
|
||||
printf(" in file [%s]\n", fname);
|
||||
return -1;
|
||||
}
|
||||
|
||||
|
||||
|
||||
if ( !(mm_is_real(matcode) && mm_is_matrix(matcode) &&
|
||||
mm_is_sparse(matcode)))
|
||||
{
|
||||
fprintf(stderr, "Sorry, this application does not support ");
|
||||
fprintf(stderr, "Market Market type: [%s]\n",
|
||||
mm_typecode_to_str(matcode));
|
||||
return -1;
|
||||
}
|
||||
|
||||
/* find out size of sparse matrix: M, N, nz .... */
|
||||
|
||||
if (mm_read_mtx_crd_size(f, &M, &N, &nz) !=0)
|
||||
{
|
||||
fprintf(stderr, "read_unsymmetric_sparse(): could not parse matrix size.\n");
|
||||
return -1;
|
||||
}
|
||||
|
||||
*M_ = M;
|
||||
*N_ = N;
|
||||
*nz_ = nz;
|
||||
|
||||
/* reseve memory for matrices */
|
||||
|
||||
I = (int *) malloc(nz * sizeof(int));
|
||||
J = (int *) malloc(nz * sizeof(int));
|
||||
val = (double *) malloc(nz * sizeof(double));
|
||||
|
||||
*val_ = val;
|
||||
*I_ = I;
|
||||
*J_ = J;
|
||||
|
||||
/* NOTE: when reading in doubles, ANSI C requires the use of the "l" */
|
||||
/* specifier as in "%lg", "%lf", "%le", otherwise errors will occur */
|
||||
/* (ANSI C X3.159-1989, Sec. 4.9.6.2, p. 136 lines 13-15) */
|
||||
|
||||
for (i=0; i<nz; i++)
|
||||
{
|
||||
if (fscanf(f, "%d %d %lg\n", &I[i], &J[i], &val[i]) != 3) {
|
||||
return -1;
|
||||
}
|
||||
I[i]--; /* adjust from 1-based to 0-based */
|
||||
J[i]--;
|
||||
}
|
||||
fclose(f);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int mm_is_valid(MM_typecode matcode)
|
||||
{
|
||||
if (!mm_is_matrix(matcode)) return 0;
|
||||
if (mm_is_dense(matcode) && mm_is_pattern(matcode)) return 0;
|
||||
if (mm_is_real(matcode) && mm_is_hermitian(matcode)) return 0;
|
||||
if (mm_is_pattern(matcode) && (mm_is_hermitian(matcode) ||
|
||||
mm_is_skew(matcode))) return 0;
|
||||
return 1;
|
||||
}
|
||||
|
||||
int mm_read_banner(FILE *f, MM_typecode *matcode)
|
||||
{
|
||||
char line[MM_MAX_LINE_LENGTH];
|
||||
char banner[MM_MAX_TOKEN_LENGTH];
|
||||
char mtx[MM_MAX_TOKEN_LENGTH];
|
||||
char crd[MM_MAX_TOKEN_LENGTH];
|
||||
char data_type[MM_MAX_TOKEN_LENGTH];
|
||||
char storage_scheme[MM_MAX_TOKEN_LENGTH];
|
||||
char *p;
|
||||
|
||||
|
||||
mm_clear_typecode(matcode);
|
||||
|
||||
if (fgets(line, MM_MAX_LINE_LENGTH, f) == NULL)
|
||||
return MM_PREMATURE_EOF;
|
||||
|
||||
if (sscanf(line, "%s %s %s %s %s", banner, mtx, crd, data_type,
|
||||
storage_scheme) != 5)
|
||||
return MM_PREMATURE_EOF;
|
||||
|
||||
for (p=mtx; *p!='\0'; *p=tolower(*p),p++); /* convert to lower case */
|
||||
for (p=crd; *p!='\0'; *p=tolower(*p),p++);
|
||||
for (p=data_type; *p!='\0'; *p=tolower(*p),p++);
|
||||
for (p=storage_scheme; *p!='\0'; *p=tolower(*p),p++);
|
||||
|
||||
/* check for banner */
|
||||
if (strncmp(banner, MatrixMarketBanner, strlen(MatrixMarketBanner)) != 0)
|
||||
return MM_NO_HEADER;
|
||||
|
||||
/* first field should be "mtx" */
|
||||
if (strcmp(mtx, MM_MTX_STR) != 0)
|
||||
return MM_UNSUPPORTED_TYPE;
|
||||
mm_set_matrix(matcode);
|
||||
|
||||
|
||||
/* second field describes whether this is a sparse matrix (in coordinate
|
||||
storgae) or a dense array */
|
||||
|
||||
|
||||
if (strcmp(crd, MM_SPARSE_STR) == 0)
|
||||
mm_set_sparse(matcode);
|
||||
else
|
||||
if (strcmp(crd, MM_DENSE_STR) == 0)
|
||||
mm_set_dense(matcode);
|
||||
else
|
||||
return MM_UNSUPPORTED_TYPE;
|
||||
|
||||
|
||||
/* third field */
|
||||
|
||||
if (strcmp(data_type, MM_REAL_STR) == 0)
|
||||
mm_set_real(matcode);
|
||||
else
|
||||
if (strcmp(data_type, MM_COMPLEX_STR) == 0)
|
||||
mm_set_complex(matcode);
|
||||
else
|
||||
if (strcmp(data_type, MM_PATTERN_STR) == 0)
|
||||
mm_set_pattern(matcode);
|
||||
else
|
||||
if (strcmp(data_type, MM_INT_STR) == 0)
|
||||
mm_set_integer(matcode);
|
||||
else
|
||||
return MM_UNSUPPORTED_TYPE;
|
||||
|
||||
|
||||
/* fourth field */
|
||||
|
||||
if (strcmp(storage_scheme, MM_GENERAL_STR) == 0)
|
||||
mm_set_general(matcode);
|
||||
else
|
||||
if (strcmp(storage_scheme, MM_SYMM_STR) == 0)
|
||||
mm_set_symmetric(matcode);
|
||||
else
|
||||
if (strcmp(storage_scheme, MM_HERM_STR) == 0)
|
||||
mm_set_hermitian(matcode);
|
||||
else
|
||||
if (strcmp(storage_scheme, MM_SKEW_STR) == 0)
|
||||
mm_set_skew(matcode);
|
||||
else
|
||||
return MM_UNSUPPORTED_TYPE;
|
||||
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int mm_write_mtx_crd_size(FILE *f, int M, int N, int nz)
|
||||
{
|
||||
if (fprintf(f, "%d %d %d\n", M, N, nz) != 3)
|
||||
return MM_COULD_NOT_WRITE_FILE;
|
||||
else
|
||||
return 0;
|
||||
}
|
||||
|
||||
int mm_read_mtx_crd_size(FILE *f, int *M, int *N, int *nz )
|
||||
{
|
||||
char line[MM_MAX_LINE_LENGTH];
|
||||
int num_items_read;
|
||||
|
||||
/* set return null parameter values, in case we exit with errors */
|
||||
*M = *N = *nz = 0;
|
||||
|
||||
/* now continue scanning until you reach the end-of-comments */
|
||||
do
|
||||
{
|
||||
if (fgets(line,MM_MAX_LINE_LENGTH,f) == NULL)
|
||||
return MM_PREMATURE_EOF;
|
||||
}while (line[0] == '%');
|
||||
|
||||
/* line[] is either blank or has M,N, nz */
|
||||
if (sscanf(line, "%d %d %d", M, N, nz) == 3)
|
||||
return 0;
|
||||
|
||||
else
|
||||
do
|
||||
{
|
||||
num_items_read = fscanf(f, "%d %d %d", M, N, nz);
|
||||
if (num_items_read == EOF) return MM_PREMATURE_EOF;
|
||||
}
|
||||
while (num_items_read != 3);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
int mm_read_mtx_array_size(FILE *f, int *M, int *N)
|
||||
{
|
||||
char line[MM_MAX_LINE_LENGTH];
|
||||
int num_items_read;
|
||||
/* set return null parameter values, in case we exit with errors */
|
||||
*M = *N = 0;
|
||||
|
||||
/* now continue scanning until you reach the end-of-comments */
|
||||
do
|
||||
{
|
||||
if (fgets(line,MM_MAX_LINE_LENGTH,f) == NULL)
|
||||
return MM_PREMATURE_EOF;
|
||||
}while (line[0] == '%');
|
||||
|
||||
/* line[] is either blank or has M,N, nz */
|
||||
if (sscanf(line, "%d %d", M, N) == 2)
|
||||
return 0;
|
||||
|
||||
else /* we have a blank line */
|
||||
do
|
||||
{
|
||||
num_items_read = fscanf(f, "%d %d", M, N);
|
||||
if (num_items_read == EOF) return MM_PREMATURE_EOF;
|
||||
}
|
||||
while (num_items_read != 2);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
int mm_write_mtx_array_size(FILE *f, int M, int N)
|
||||
{
|
||||
if (fprintf(f, "%d %d\n", M, N) != 2)
|
||||
return MM_COULD_NOT_WRITE_FILE;
|
||||
else
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
|
||||
/*-------------------------------------------------------------------------*/
|
||||
|
||||
/******************************************************************/
|
||||
/* use when I[], J[], and val[]J, and val[] are already allocated */
|
||||
/******************************************************************/
|
||||
|
||||
int mm_read_mtx_crd_data(FILE *f, int M, int N, int nz, int I[], int J[],
|
||||
double val[], MM_typecode matcode)
|
||||
{
|
||||
int i;
|
||||
if (mm_is_complex(matcode))
|
||||
{
|
||||
for (i=0; i<nz; i++)
|
||||
if (fscanf(f, "%d %d %lg %lg", &I[i], &J[i], &val[2*i], &val[2*i+1])
|
||||
!= 4) return MM_PREMATURE_EOF;
|
||||
}
|
||||
else if (mm_is_real(matcode) || mm_is_integer(matcode))
|
||||
{
|
||||
for (i=0; i<nz; i++)
|
||||
{
|
||||
if (fscanf(f, "%d %d %lg\n", &I[i], &J[i], &val[i])
|
||||
!= 3) return MM_PREMATURE_EOF;
|
||||
|
||||
}
|
||||
}
|
||||
|
||||
else if (mm_is_pattern(matcode))
|
||||
{
|
||||
for (i=0; i<nz; i++)
|
||||
if (fscanf(f, "%d %d", &I[i], &J[i])
|
||||
!= 2) return MM_PREMATURE_EOF;
|
||||
}
|
||||
else
|
||||
return MM_UNSUPPORTED_TYPE;
|
||||
|
||||
return 0;
|
||||
|
||||
}
|
||||
|
||||
int mm_read_mtx_crd_entry(FILE *f, int *I, int *J,
|
||||
double *real, double *imag, MM_typecode matcode)
|
||||
{
|
||||
if (mm_is_complex(matcode))
|
||||
{
|
||||
if (fscanf(f, "%d %d %lg %lg", I, J, real, imag)
|
||||
!= 4) return MM_PREMATURE_EOF;
|
||||
}
|
||||
else if (mm_is_real(matcode) || mm_is_integer(matcode))
|
||||
{
|
||||
if (fscanf(f, "%d %d %lg\n", I, J, real)
|
||||
!= 3) return MM_PREMATURE_EOF;
|
||||
|
||||
}
|
||||
|
||||
else if (mm_is_pattern(matcode))
|
||||
{
|
||||
if (fscanf(f, "%d %d", I, J) != 2) return MM_PREMATURE_EOF;
|
||||
}
|
||||
else
|
||||
return MM_UNSUPPORTED_TYPE;
|
||||
|
||||
return 0;
|
||||
|
||||
}
|
||||
|
||||
|
||||
/************************************************************************
|
||||
mm_read_mtx_crd() fills M, N, nz, array of values, and return
|
||||
type code, e.g. 'MCRS'
|
||||
|
||||
if matrix is complex, values[] is of size 2*nz,
|
||||
(nz pairs of real/imaginary values)
|
||||
************************************************************************/
|
||||
|
||||
int mm_read_mtx_crd(char *fname, int *M, int *N, int *nz, int **I, int **J,
|
||||
double **val, MM_typecode *matcode)
|
||||
{
|
||||
int ret_code;
|
||||
FILE *f;
|
||||
|
||||
if (strcmp(fname, "stdin") == 0) f=stdin;
|
||||
else
|
||||
if ((f = fopen(fname, "r")) == NULL)
|
||||
return MM_COULD_NOT_READ_FILE;
|
||||
|
||||
|
||||
if ((ret_code = mm_read_banner(f, matcode)) != 0)
|
||||
return ret_code;
|
||||
|
||||
if (!(mm_is_valid(*matcode) && mm_is_sparse(*matcode) &&
|
||||
mm_is_matrix(*matcode)))
|
||||
return MM_UNSUPPORTED_TYPE;
|
||||
|
||||
if ((ret_code = mm_read_mtx_crd_size(f, M, N, nz)) != 0)
|
||||
return ret_code;
|
||||
|
||||
|
||||
*I = (int *) malloc(*nz * sizeof(int));
|
||||
*J = (int *) malloc(*nz * sizeof(int));
|
||||
*val = NULL;
|
||||
|
||||
if (mm_is_complex(*matcode))
|
||||
{
|
||||
*val = (double *) malloc(*nz * 2 * sizeof(double));
|
||||
ret_code = mm_read_mtx_crd_data(f, *M, *N, *nz, *I, *J, *val,
|
||||
*matcode);
|
||||
if (ret_code != 0) return ret_code;
|
||||
}
|
||||
else if (mm_is_real(*matcode) || mm_is_integer(*matcode))
|
||||
{
|
||||
*val = (double *) malloc(*nz * sizeof(double));
|
||||
ret_code = mm_read_mtx_crd_data(f, *M, *N, *nz, *I, *J, *val,
|
||||
*matcode);
|
||||
if (ret_code != 0) return ret_code;
|
||||
}
|
||||
|
||||
else if (mm_is_pattern(*matcode))
|
||||
{
|
||||
ret_code = mm_read_mtx_crd_data(f, *M, *N, *nz, *I, *J, *val,
|
||||
*matcode);
|
||||
if (ret_code != 0) return ret_code;
|
||||
}
|
||||
|
||||
if (f != stdin) fclose(f);
|
||||
return 0;
|
||||
}
|
||||
|
||||
int mm_write_banner(FILE *f, MM_typecode matcode)
|
||||
{
|
||||
char *str = mm_typecode_to_str(matcode);
|
||||
int ret_code;
|
||||
|
||||
ret_code = fprintf(f, "%s %s\n", MatrixMarketBanner, str);
|
||||
free(str);
|
||||
if (ret_code !=2 )
|
||||
return MM_COULD_NOT_WRITE_FILE;
|
||||
else
|
||||
return 0;
|
||||
}
|
||||
|
||||
int mm_write_mtx_crd(char fname[], int M, int N, int nz, int I[], int J[],
|
||||
double val[], MM_typecode matcode)
|
||||
{
|
||||
FILE *f;
|
||||
int i;
|
||||
|
||||
if (strcmp(fname, "stdout") == 0)
|
||||
f = stdout;
|
||||
else
|
||||
if ((f = fopen(fname, "w")) == NULL)
|
||||
return MM_COULD_NOT_WRITE_FILE;
|
||||
|
||||
/* print banner followed by typecode */
|
||||
fprintf(f, "%s ", MatrixMarketBanner);
|
||||
fprintf(f, "%s\n", mm_typecode_to_str(matcode));
|
||||
|
||||
/* print matrix sizes and nonzeros */
|
||||
fprintf(f, "%d %d %d\n", M, N, nz);
|
||||
|
||||
/* print values */
|
||||
if (mm_is_pattern(matcode))
|
||||
for (i=0; i<nz; i++)
|
||||
fprintf(f, "%d %d\n", I[i], J[i]);
|
||||
else
|
||||
if (mm_is_integer(matcode))
|
||||
for (i=0; i<nz; i++)
|
||||
fprintf(f, "%d %d %d\n", I[i], J[i], (int)val[i]);
|
||||
else
|
||||
if (mm_is_real(matcode))
|
||||
for (i=0; i<nz; i++)
|
||||
fprintf(f, "%d %d %20.16g\n", I[i], J[i], val[i]);
|
||||
else
|
||||
if (mm_is_complex(matcode))
|
||||
for (i=0; i<nz; i++)
|
||||
fprintf(f, "%d %d %20.16g %20.16g\n", I[i], J[i], val[2*i],
|
||||
val[2*i+1]);
|
||||
else
|
||||
{
|
||||
if (f != stdout) fclose(f);
|
||||
return MM_UNSUPPORTED_TYPE;
|
||||
}
|
||||
|
||||
if (f !=stdout) fclose(f);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Create a new copy of a string s. mm_strdup() is a common routine, but
|
||||
* not part of ANSI C, so it is included here. Used by mm_typecode_to_str().
|
||||
*
|
||||
*/
|
||||
static char *mm_strdup(const char *s)
|
||||
{
|
||||
size_t len = strlen(s);
|
||||
char *s2 = (char *) malloc((len+1)*sizeof(char));
|
||||
return strcpy(s2, s);
|
||||
}
|
||||
|
||||
char *mm_typecode_to_str(MM_typecode matcode)
|
||||
{
|
||||
char buffer[MM_MAX_LINE_LENGTH];
|
||||
char *types[4];
|
||||
//char *mm_strdup(const char *);
|
||||
//int error =0;
|
||||
|
||||
/* check for MTX type */
|
||||
if (mm_is_matrix(matcode))
|
||||
types[0] = MM_MTX_STR;
|
||||
else
|
||||
return NULL; // error=1;
|
||||
|
||||
/* check for CRD or ARR matrix */
|
||||
if (mm_is_sparse(matcode))
|
||||
types[1] = MM_SPARSE_STR;
|
||||
else
|
||||
if (mm_is_dense(matcode))
|
||||
types[1] = MM_DENSE_STR;
|
||||
else
|
||||
return NULL;
|
||||
|
||||
/* check for element data type */
|
||||
if (mm_is_real(matcode))
|
||||
types[2] = MM_REAL_STR;
|
||||
else
|
||||
if (mm_is_complex(matcode))
|
||||
types[2] = MM_COMPLEX_STR;
|
||||
else
|
||||
if (mm_is_pattern(matcode))
|
||||
types[2] = MM_PATTERN_STR;
|
||||
else
|
||||
if (mm_is_integer(matcode))
|
||||
types[2] = MM_INT_STR;
|
||||
else
|
||||
return NULL;
|
||||
|
||||
|
||||
/* check for symmetry type */
|
||||
if (mm_is_general(matcode))
|
||||
types[3] = MM_GENERAL_STR;
|
||||
else
|
||||
if (mm_is_symmetric(matcode))
|
||||
types[3] = MM_SYMM_STR;
|
||||
else
|
||||
if (mm_is_hermitian(matcode))
|
||||
types[3] = MM_HERM_STR;
|
||||
else
|
||||
if (mm_is_skew(matcode))
|
||||
types[3] = MM_SKEW_STR;
|
||||
else
|
||||
return NULL;
|
||||
|
||||
sprintf(buffer,"%s %s %s %s", types[0], types[1], types[2], types[3]);
|
||||
return mm_strdup(buffer);
|
||||
|
||||
}
|
||||
141
Samples/cuSolverDn_LinearSolver/mmio.h
Normal file
141
Samples/cuSolverDn_LinearSolver/mmio.h
Normal file
@@ -0,0 +1,141 @@
|
||||
/*
|
||||
* Matrix Market I/O library for ANSI C
|
||||
*
|
||||
* See http://math.nist.gov/MatrixMarket for details.
|
||||
*
|
||||
*
|
||||
*/
|
||||
|
||||
#ifndef MM_IO_H
|
||||
#define MM_IO_H
|
||||
|
||||
#if defined(__cplusplus)
|
||||
extern "C" {
|
||||
#endif /* __cplusplus */
|
||||
|
||||
#define MM_MAX_LINE_LENGTH 1025
|
||||
#define MatrixMarketBanner "%%MatrixMarket"
|
||||
#define MM_MAX_TOKEN_LENGTH 64
|
||||
|
||||
typedef char MM_typecode[4];
|
||||
|
||||
char *mm_typecode_to_str(MM_typecode matcode);
|
||||
|
||||
int mm_read_banner(FILE *f, MM_typecode *matcode);
|
||||
int mm_read_mtx_crd_size(FILE *f, int *M, int *N, int *nz);
|
||||
int mm_read_mtx_array_size(FILE *f, int *M, int *N);
|
||||
|
||||
int mm_write_banner(FILE *f, MM_typecode matcode);
|
||||
int mm_write_mtx_crd_size(FILE *f, int M, int N, int nz);
|
||||
int mm_write_mtx_array_size(FILE *f, int M, int N);
|
||||
|
||||
|
||||
/********************* MM_typecode query fucntions ***************************/
|
||||
|
||||
#define mm_is_matrix(typecode) ((typecode)[0]=='M')
|
||||
|
||||
#define mm_is_sparse(typecode) ((typecode)[1]=='C')
|
||||
#define mm_is_coordinate(typecode)((typecode)[1]=='C')
|
||||
#define mm_is_dense(typecode) ((typecode)[1]=='A')
|
||||
#define mm_is_array(typecode) ((typecode)[1]=='A')
|
||||
|
||||
#define mm_is_complex(typecode) ((typecode)[2]=='C')
|
||||
#define mm_is_real(typecode) ((typecode)[2]=='R')
|
||||
#define mm_is_pattern(typecode) ((typecode)[2]=='P')
|
||||
#define mm_is_integer(typecode) ((typecode)[2]=='I')
|
||||
|
||||
#define mm_is_symmetric(typecode)((typecode)[3]=='S')
|
||||
#define mm_is_general(typecode) ((typecode)[3]=='G')
|
||||
#define mm_is_skew(typecode) ((typecode)[3]=='K')
|
||||
#define mm_is_hermitian(typecode)((typecode)[3]=='H')
|
||||
|
||||
int mm_is_valid(MM_typecode matcode); /* too complex for a macro */
|
||||
|
||||
|
||||
/********************* MM_typecode modify fucntions ***************************/
|
||||
|
||||
#define mm_set_matrix(typecode) ((*typecode)[0]='M')
|
||||
#define mm_set_coordinate(typecode) ((*typecode)[1]='C')
|
||||
#define mm_set_array(typecode) ((*typecode)[1]='A')
|
||||
#define mm_set_dense(typecode) mm_set_array(typecode)
|
||||
#define mm_set_sparse(typecode) mm_set_coordinate(typecode)
|
||||
|
||||
#define mm_set_complex(typecode)((*typecode)[2]='C')
|
||||
#define mm_set_real(typecode) ((*typecode)[2]='R')
|
||||
#define mm_set_pattern(typecode)((*typecode)[2]='P')
|
||||
#define mm_set_integer(typecode)((*typecode)[2]='I')
|
||||
|
||||
|
||||
#define mm_set_symmetric(typecode)((*typecode)[3]='S')
|
||||
#define mm_set_general(typecode)((*typecode)[3]='G')
|
||||
#define mm_set_skew(typecode) ((*typecode)[3]='K')
|
||||
#define mm_set_hermitian(typecode)((*typecode)[3]='H')
|
||||
|
||||
#define mm_clear_typecode(typecode) ((*typecode)[0]=(*typecode)[1]= \
|
||||
(*typecode)[2]=' ',(*typecode)[3]='G')
|
||||
|
||||
#define mm_initialize_typecode(typecode) mm_clear_typecode(typecode)
|
||||
|
||||
|
||||
/********************* Matrix Market error codes ***************************/
|
||||
|
||||
|
||||
#define MM_COULD_NOT_READ_FILE 11
|
||||
#define MM_PREMATURE_EOF 12
|
||||
#define MM_NOT_MTX 13
|
||||
#define MM_NO_HEADER 14
|
||||
#define MM_UNSUPPORTED_TYPE 15
|
||||
#define MM_LINE_TOO_LONG 16
|
||||
#define MM_COULD_NOT_WRITE_FILE 17
|
||||
|
||||
|
||||
/******************** Matrix Market internal definitions ********************
|
||||
|
||||
MM_matrix_typecode: 4-character sequence
|
||||
|
||||
ojbect sparse/ data storage
|
||||
dense type scheme
|
||||
|
||||
string position: [0] [1] [2] [3]
|
||||
|
||||
Matrix typecode: M(atrix) C(oord) R(eal) G(eneral)
|
||||
A(array) C(omplex) H(ermitian)
|
||||
P(attern) S(ymmetric)
|
||||
I(nteger) K(kew)
|
||||
|
||||
***********************************************************************/
|
||||
|
||||
#define MM_MTX_STR "matrix"
|
||||
#define MM_ARRAY_STR "array"
|
||||
#define MM_DENSE_STR "array"
|
||||
#define MM_COORDINATE_STR "coordinate"
|
||||
#define MM_SPARSE_STR "coordinate"
|
||||
#define MM_COMPLEX_STR "complex"
|
||||
#define MM_REAL_STR "real"
|
||||
#define MM_INT_STR "integer"
|
||||
#define MM_GENERAL_STR "general"
|
||||
#define MM_SYMM_STR "symmetric"
|
||||
#define MM_HERM_STR "hermitian"
|
||||
#define MM_SKEW_STR "skew-symmetric"
|
||||
#define MM_PATTERN_STR "pattern"
|
||||
|
||||
|
||||
/* high level routines */
|
||||
int mm_read_mtx_crd(char *fname, int *M, int *N, int *nz, int **I, int **J,
|
||||
double **val, MM_typecode *matcode);
|
||||
|
||||
int mm_write_mtx_crd(char fname[], int M, int N, int nz, int I[], int J[],
|
||||
double val[], MM_typecode matcode);
|
||||
int mm_read_mtx_crd_data(FILE *f, int M, int N, int nz, int I[], int J[],
|
||||
double val[], MM_typecode matcode);
|
||||
int mm_read_mtx_crd_entry(FILE *f, int *I, int *J, double *real, double *img,
|
||||
MM_typecode matcode);
|
||||
|
||||
int mm_read_unsymmetric_sparse(const char *fname, int *M_, int *N_, int *nz_,
|
||||
double **val_, int **I_, int **J_);
|
||||
|
||||
#if defined(__cplusplus)
|
||||
}
|
||||
#endif /* __cplusplus */
|
||||
|
||||
#endif
|
||||
529
Samples/cuSolverDn_LinearSolver/mmio_wrapper.cpp
Normal file
529
Samples/cuSolverDn_LinearSolver/mmio_wrapper.cpp
Normal file
@@ -0,0 +1,529 @@
|
||||
|
||||
#include <stdio.h>
|
||||
#include <stdlib.h>
|
||||
#include <math.h>
|
||||
|
||||
#include "mmio.h"
|
||||
|
||||
#include <cusolverDn.h>
|
||||
|
||||
/* avoid Windows warnings (for example: strcpy, fscanf, etc.) */
|
||||
#if defined(_WIN32)
|
||||
#define _CRT_SECURE_NO_WARNINGS
|
||||
#endif
|
||||
|
||||
/* various __inline__ __device__ function to initialize a T_ELEM */
|
||||
template <typename T_ELEM> __inline__ T_ELEM cuGet (int );
|
||||
template <> __inline__ float cuGet<float >(int x)
|
||||
{
|
||||
return float(x);
|
||||
}
|
||||
|
||||
template <> __inline__ double cuGet<double>(int x)
|
||||
{
|
||||
return double(x);
|
||||
}
|
||||
|
||||
template <> __inline__ cuComplex cuGet<cuComplex>(int x)
|
||||
{
|
||||
return (make_cuComplex( float(x), 0.0f ));
|
||||
}
|
||||
|
||||
template <> __inline__ cuDoubleComplex cuGet<cuDoubleComplex>(int x)
|
||||
{
|
||||
return (make_cuDoubleComplex( double(x), 0.0 ));
|
||||
}
|
||||
|
||||
|
||||
template <typename T_ELEM> __inline__ T_ELEM cuGet (int , int );
|
||||
template <> __inline__ float cuGet<float >(int x, int y)
|
||||
{
|
||||
return float(x);
|
||||
}
|
||||
|
||||
template <> __inline__ double cuGet<double>(int x, int y)
|
||||
{
|
||||
return double(x);
|
||||
}
|
||||
|
||||
template <> __inline__ cuComplex cuGet<cuComplex>(int x, int y)
|
||||
{
|
||||
return make_cuComplex( float(x), float(y) );
|
||||
}
|
||||
|
||||
template <> __inline__ cuDoubleComplex cuGet<cuDoubleComplex>(int x, int y)
|
||||
{
|
||||
return (make_cuDoubleComplex( double(x), double(y) ));
|
||||
}
|
||||
|
||||
|
||||
template <typename T_ELEM> __inline__ T_ELEM cuGet (float );
|
||||
template <> __inline__ float cuGet<float >(float x)
|
||||
{
|
||||
return float(x);
|
||||
}
|
||||
|
||||
template <> __inline__ double cuGet<double>(float x)
|
||||
{
|
||||
return double(x);
|
||||
}
|
||||
|
||||
template <> __inline__ cuComplex cuGet<cuComplex>(float x)
|
||||
{
|
||||
return (make_cuComplex( float(x), 0.0f ));
|
||||
}
|
||||
|
||||
template <> __inline__ cuDoubleComplex cuGet<cuDoubleComplex>(float x)
|
||||
{
|
||||
return (make_cuDoubleComplex( double(x), 0.0 ));
|
||||
}
|
||||
|
||||
|
||||
template <typename T_ELEM> __inline__ T_ELEM cuGet (float, float );
|
||||
template <> __inline__ float cuGet<float >(float x, float y)
|
||||
{
|
||||
return float(x);
|
||||
}
|
||||
|
||||
template <> __inline__ double cuGet<double>(float x, float y)
|
||||
{
|
||||
return double(x);
|
||||
}
|
||||
|
||||
template <> __inline__ cuComplex cuGet<cuComplex>(float x, float y)
|
||||
{
|
||||
return (make_cuComplex( float(x), float(y) ));
|
||||
}
|
||||
|
||||
template <> __inline__ cuDoubleComplex cuGet<cuDoubleComplex>(float x, float y)
|
||||
{
|
||||
return (make_cuDoubleComplex( double(x), double(y) ));
|
||||
}
|
||||
|
||||
|
||||
template <typename T_ELEM> __inline__ T_ELEM cuGet (double );
|
||||
template <> __inline__ float cuGet<float >(double x)
|
||||
{
|
||||
return float(x);
|
||||
}
|
||||
|
||||
template <> __inline__ double cuGet<double>(double x)
|
||||
{
|
||||
return double(x);
|
||||
}
|
||||
|
||||
template <> __inline__ cuComplex cuGet<cuComplex>(double x)
|
||||
{
|
||||
return (make_cuComplex( float(x), 0.0f ));
|
||||
}
|
||||
|
||||
template <> __inline__ cuDoubleComplex cuGet<cuDoubleComplex>(double x)
|
||||
{
|
||||
return (make_cuDoubleComplex( double(x), 0.0 ));
|
||||
}
|
||||
|
||||
|
||||
template <typename T_ELEM> __inline__ T_ELEM cuGet (double, double );
|
||||
template <> __inline__ float cuGet<float >(double x, double y)
|
||||
{
|
||||
return float(x);
|
||||
}
|
||||
|
||||
template <> __inline__ double cuGet<double>(double x, double y)
|
||||
{
|
||||
return double(x);
|
||||
}
|
||||
|
||||
template <> __inline__ cuComplex cuGet<cuComplex>(double x, double y)
|
||||
{
|
||||
return (make_cuComplex( float(x), float(y) ));
|
||||
}
|
||||
|
||||
template <> __inline__ cuDoubleComplex cuGet<cuDoubleComplex>(double x, double y)
|
||||
{
|
||||
return (make_cuDoubleComplex( double(x), double(y) ));
|
||||
}
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
static void compress_index(
|
||||
const int *Ind,
|
||||
int nnz,
|
||||
int m,
|
||||
int *Ptr,
|
||||
int base)
|
||||
{
|
||||
int i;
|
||||
|
||||
/* initialize everything to zero */
|
||||
for(i=0; i<m+1; i++){
|
||||
Ptr[i]=0;
|
||||
}
|
||||
/* count elements in every row */
|
||||
Ptr[0]=base;
|
||||
for(i=0; i<nnz; i++){
|
||||
Ptr[Ind[i]+(1-base)]++;
|
||||
}
|
||||
/* add all the values */
|
||||
for(i=0; i<m; i++){
|
||||
Ptr[i+1]+=Ptr[i];
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
struct cooFormat {
|
||||
int i ;
|
||||
int j ;
|
||||
int p ; // permutation
|
||||
};
|
||||
|
||||
|
||||
int cmp_cooFormat_csr( struct cooFormat *s, struct cooFormat *t)
|
||||
{
|
||||
if ( s->i < t->i ){
|
||||
return -1 ;
|
||||
}
|
||||
else if ( s->i > t->i ){
|
||||
return 1 ;
|
||||
}
|
||||
else{
|
||||
return s->j - t->j ;
|
||||
}
|
||||
}
|
||||
|
||||
int cmp_cooFormat_csc( struct cooFormat *s, struct cooFormat *t)
|
||||
{
|
||||
if ( s->j < t->j ){
|
||||
return -1 ;
|
||||
}
|
||||
else if ( s->j > t->j ){
|
||||
return 1 ;
|
||||
}
|
||||
else{
|
||||
return s->i - t->i ;
|
||||
}
|
||||
}
|
||||
|
||||
typedef int (*FUNPTR) (const void*, const void*) ;
|
||||
typedef int (*FUNPTR2) ( struct cooFormat *s, struct cooFormat *t) ;
|
||||
|
||||
static FUNPTR2 fptr_array[2] = {
|
||||
cmp_cooFormat_csr,
|
||||
cmp_cooFormat_csc,
|
||||
};
|
||||
|
||||
|
||||
static int verify_pattern(
|
||||
int m,
|
||||
int nnz,
|
||||
int *csrRowPtr,
|
||||
int *csrColInd)
|
||||
{
|
||||
int i, col, start, end, base_index;
|
||||
int error_found = 0;
|
||||
|
||||
if (nnz != (csrRowPtr[m] - csrRowPtr[0])){
|
||||
fprintf(stderr, "Error (nnz check failed): (csrRowPtr[%d]=%d - csrRowPtr[%d]=%d) != (nnz=%d)\n", 0, csrRowPtr[0], m, csrRowPtr[m], nnz);
|
||||
error_found = 1;
|
||||
}
|
||||
|
||||
base_index = csrRowPtr[0];
|
||||
if ((0 != base_index) && (1 != base_index)){
|
||||
fprintf(stderr, "Error (base index check failed): base index = %d\n", base_index);
|
||||
error_found = 1;
|
||||
}
|
||||
|
||||
for (i=0; (!error_found) && (i<m); i++){
|
||||
start = csrRowPtr[i ] - base_index;
|
||||
end = csrRowPtr[i+1] - base_index;
|
||||
if (start > end){
|
||||
fprintf(stderr, "Error (corrupted row): csrRowPtr[%d] (=%d) > csrRowPtr[%d] (=%d)\n", i, start+base_index, i+1, end+base_index);
|
||||
error_found = 1;
|
||||
}
|
||||
for (col=start; col<end; col++){
|
||||
if (csrColInd[col] < base_index){
|
||||
fprintf(stderr, "Error (column vs. base index check failed): csrColInd[%d] < %d\n", col, base_index);
|
||||
error_found = 1;
|
||||
}
|
||||
if ((col < (end-1)) && (csrColInd[col] >= csrColInd[col+1])){
|
||||
fprintf(stderr, "Error (sorting of the column indecis check failed): (csrColInd[%d]=%d) >= (csrColInd[%d]=%d)\n", col, csrColInd[col], col+1, csrColInd[col+1]);
|
||||
error_found = 1;
|
||||
}
|
||||
}
|
||||
}
|
||||
return error_found ;
|
||||
}
|
||||
|
||||
|
||||
template <typename T_ELEM>
|
||||
int loadMMSparseMatrix(
|
||||
char *filename,
|
||||
char elem_type,
|
||||
bool csrFormat,
|
||||
int *m,
|
||||
int *n,
|
||||
int *nnz,
|
||||
T_ELEM **aVal,
|
||||
int **aRowInd,
|
||||
int **aColInd,
|
||||
int extendSymMatrix)
|
||||
{
|
||||
MM_typecode matcode;
|
||||
double *tempVal;
|
||||
int *tempRowInd,*tempColInd;
|
||||
double *tval;
|
||||
int *trow,*tcol;
|
||||
int *csrRowPtr, *cscColPtr;
|
||||
int i,j,error,base,count;
|
||||
struct cooFormat *work;
|
||||
|
||||
/* read the matrix */
|
||||
error = mm_read_mtx_crd(filename, m, n, nnz, &trow, &tcol, &tval, &matcode);
|
||||
if (error) {
|
||||
fprintf(stderr, "!!!! can not open file: '%s'\n", filename);
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* start error checking */
|
||||
if (mm_is_complex(matcode) && ((elem_type != 'z') && (elem_type != 'c'))) {
|
||||
fprintf(stderr, "!!!! complex matrix requires type 'z' or 'c'\n");
|
||||
return 1;
|
||||
}
|
||||
|
||||
if (mm_is_dense(matcode) || mm_is_array(matcode) || mm_is_pattern(matcode) /*|| mm_is_integer(matcode)*/){
|
||||
fprintf(stderr, "!!!! dense, array, pattern and integer matrices are not supported\n");
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* if necessary symmetrize the pattern (transform from triangular to full) */
|
||||
if ((extendSymMatrix) && (mm_is_symmetric(matcode) || mm_is_hermitian(matcode) || mm_is_skew(matcode))){
|
||||
//count number of non-diagonal elements
|
||||
count=0;
|
||||
for(i=0; i<(*nnz); i++){
|
||||
if (trow[i] != tcol[i]){
|
||||
count++;
|
||||
}
|
||||
}
|
||||
//allocate space for the symmetrized matrix
|
||||
tempRowInd = (int *)malloc((*nnz + count) * sizeof(int));
|
||||
tempColInd = (int *)malloc((*nnz + count) * sizeof(int));
|
||||
if (mm_is_real(matcode) || mm_is_integer(matcode)){
|
||||
tempVal = (double *)malloc((*nnz + count) * sizeof(double));
|
||||
}
|
||||
else{
|
||||
tempVal = (double *)malloc(2 * (*nnz + count) * sizeof(double));
|
||||
}
|
||||
//copy the elements regular and transposed locations
|
||||
for(j=0, i=0; i<(*nnz); i++){
|
||||
tempRowInd[j]=trow[i];
|
||||
tempColInd[j]=tcol[i];
|
||||
if (mm_is_real(matcode) || mm_is_integer(matcode)){
|
||||
tempVal[j]=tval[i];
|
||||
}
|
||||
else{
|
||||
tempVal[2*j] =tval[2*i];
|
||||
tempVal[2*j+1]=tval[2*i+1];
|
||||
}
|
||||
j++;
|
||||
if (trow[i] != tcol[i]){
|
||||
tempRowInd[j]=tcol[i];
|
||||
tempColInd[j]=trow[i];
|
||||
if (mm_is_real(matcode) || mm_is_integer(matcode)){
|
||||
if (mm_is_skew(matcode)){
|
||||
tempVal[j]=-tval[i];
|
||||
}
|
||||
else{
|
||||
tempVal[j]= tval[i];
|
||||
}
|
||||
}
|
||||
else{
|
||||
if(mm_is_hermitian(matcode)){
|
||||
tempVal[2*j] = tval[2*i];
|
||||
tempVal[2*j+1]=-tval[2*i+1];
|
||||
}
|
||||
else{
|
||||
tempVal[2*j] = tval[2*i];
|
||||
tempVal[2*j+1]= tval[2*i+1];
|
||||
}
|
||||
}
|
||||
j++;
|
||||
}
|
||||
}
|
||||
(*nnz)+=count;
|
||||
//free temporary storage
|
||||
free(trow);
|
||||
free(tcol);
|
||||
free(tval);
|
||||
}
|
||||
else{
|
||||
tempRowInd=trow;
|
||||
tempColInd=tcol;
|
||||
tempVal =tval;
|
||||
}
|
||||
// life time of (trow, tcol, tval) is over.
|
||||
// please use COO format (tempRowInd, tempColInd, tempVal)
|
||||
|
||||
// use qsort to sort COO format
|
||||
work = (struct cooFormat *)malloc(sizeof(struct cooFormat)*(*nnz));
|
||||
if (NULL == work){
|
||||
fprintf(stderr, "!!!! allocation error, malloc failed\n");
|
||||
return 1;
|
||||
}
|
||||
for(i=0; i<(*nnz); i++){
|
||||
work[i].i = tempRowInd[i];
|
||||
work[i].j = tempColInd[i];
|
||||
work[i].p = i; // permutation is identity
|
||||
}
|
||||
|
||||
if (csrFormat){
|
||||
/* create row-major ordering of indices (sorted by row and within each row by column) */
|
||||
qsort(work, *nnz, sizeof(struct cooFormat), (FUNPTR)fptr_array[0] );
|
||||
}else{
|
||||
/* create column-major ordering of indices (sorted by column and within each column by row) */
|
||||
qsort(work, *nnz, sizeof(struct cooFormat), (FUNPTR)fptr_array[1] );
|
||||
|
||||
}
|
||||
|
||||
// (tempRowInd, tempColInd) is sorted either by row-major or by col-major
|
||||
for(i=0; i<(*nnz); i++){
|
||||
tempRowInd[i] = work[i].i;
|
||||
tempColInd[i] = work[i].j;
|
||||
}
|
||||
|
||||
// setup base
|
||||
// check if there is any row/col 0, if so base-0
|
||||
// check if there is any row/col equal to matrix dimension m/n, if so base-1
|
||||
int base0 = 0;
|
||||
int base1 = 0;
|
||||
for(i=0; i<(*nnz); i++){
|
||||
const int row = tempRowInd[i];
|
||||
const int col = tempColInd[i];
|
||||
if ( (0 == row) || (0 == col) ){
|
||||
base0 = 1;
|
||||
}
|
||||
if ( (*m == row) || (*n == col) ){
|
||||
base1 = 1;
|
||||
}
|
||||
}
|
||||
if ( base0 && base1 ){
|
||||
printf("Error: input matrix is base-0 and base-1 \n");
|
||||
return 1;
|
||||
}
|
||||
|
||||
base = 0;
|
||||
if (base1){
|
||||
base = 1;
|
||||
}
|
||||
|
||||
/* compress the appropriate indices */
|
||||
if (csrFormat){
|
||||
/* CSR format (assuming row-major format) */
|
||||
csrRowPtr = (int *)malloc(((*m)+1) * sizeof(csrRowPtr[0]));
|
||||
if (!csrRowPtr) return 1;
|
||||
compress_index(tempRowInd, *nnz, *m, csrRowPtr, base);
|
||||
|
||||
*aRowInd = csrRowPtr;
|
||||
*aColInd = (int *)malloc((*nnz) * sizeof(int));
|
||||
}
|
||||
else {
|
||||
/* CSC format (assuming column-major format) */
|
||||
cscColPtr = (int *)malloc(((*n)+1) * sizeof(cscColPtr[0]));
|
||||
if (!cscColPtr) return 1;
|
||||
compress_index(tempColInd, *nnz, *n, cscColPtr, base);
|
||||
|
||||
*aColInd = cscColPtr;
|
||||
*aRowInd = (int *)malloc((*nnz) * sizeof(int));
|
||||
}
|
||||
|
||||
/* transfrom the matrix values of type double into one of the cusparse library types */
|
||||
*aVal = (T_ELEM *)malloc((*nnz) * sizeof(T_ELEM));
|
||||
|
||||
for (i=0; i<(*nnz); i++) {
|
||||
if (csrFormat){
|
||||
(*aColInd)[i] = tempColInd[i];
|
||||
}
|
||||
else{
|
||||
(*aRowInd)[i] = tempRowInd[i];
|
||||
}
|
||||
if (mm_is_real(matcode) || mm_is_integer(matcode)){
|
||||
(*aVal)[i] = cuGet<T_ELEM>( tempVal[ work[i].p ] );
|
||||
}
|
||||
else{
|
||||
(*aVal)[i] = cuGet<T_ELEM>(tempVal[2*work[i].p], tempVal[2*work[i].p+1]);
|
||||
}
|
||||
}
|
||||
|
||||
/* check for corruption */
|
||||
int error_found;
|
||||
if (csrFormat){
|
||||
error_found = verify_pattern(*m, *nnz, *aRowInd, *aColInd);
|
||||
}else{
|
||||
error_found = verify_pattern(*n, *nnz, *aColInd, *aRowInd);
|
||||
}
|
||||
if (error_found){
|
||||
fprintf(stderr, "!!!! verify_pattern failed\n");
|
||||
return 1;
|
||||
}
|
||||
|
||||
/* cleanup and exit */
|
||||
free(work);
|
||||
free(tempVal);
|
||||
free(tempColInd);
|
||||
free(tempRowInd);
|
||||
|
||||
return 0;
|
||||
}
|
||||
|
||||
|
||||
/* specific instantiation */
|
||||
template int loadMMSparseMatrix<float>(
|
||||
char *filename,
|
||||
char elem_type,
|
||||
bool csrFormat,
|
||||
int *m,
|
||||
int *n,
|
||||
int *nnz,
|
||||
float **aVal,
|
||||
int **aRowInd,
|
||||
int **aColInd,
|
||||
int extendSymMatrix);
|
||||
|
||||
template int loadMMSparseMatrix<double>(
|
||||
char *filename,
|
||||
char elem_type,
|
||||
bool csrFormat,
|
||||
int *m,
|
||||
int *n,
|
||||
int *nnz,
|
||||
double **aVal,
|
||||
int **aRowInd,
|
||||
int **aColInd,
|
||||
int extendSymMatrix);
|
||||
|
||||
template int loadMMSparseMatrix<cuComplex>(
|
||||
char *filename,
|
||||
char elem_type,
|
||||
bool csrFormat,
|
||||
int *m,
|
||||
int *n,
|
||||
int *nnz,
|
||||
cuComplex **aVal,
|
||||
int **aRowInd,
|
||||
int **aColInd,
|
||||
int extendSymMatrix);
|
||||
|
||||
template int loadMMSparseMatrix<cuDoubleComplex>(
|
||||
char *filename,
|
||||
char elem_type,
|
||||
bool csrFormat,
|
||||
int *m,
|
||||
int *n,
|
||||
int *nnz,
|
||||
cuDoubleComplex **aVal,
|
||||
int **aRowInd,
|
||||
int **aColInd,
|
||||
int extendSymMatrix);
|
||||
|
||||
|
||||
Reference in New Issue
Block a user