diff --git a/NVIDIA AR SDK Programming Guide.pdf b/NVIDIA AR SDK Programming Guide.pdf deleted file mode 100644 index f155ff3..0000000 Binary files a/NVIDIA AR SDK Programming Guide.pdf and /dev/null differ diff --git a/README.MD b/README.MD deleted file mode 100644 index bccfa5d..0000000 --- a/README.MD +++ /dev/null @@ -1,16 +0,0 @@ -**Nvidia AR SDK: API Source Code and Sample Applications** - -NVIDIA AR SDK enables real-time modeling and tracking of human faces from video. The SDK is powered by NVIDIA graphics processing units (GPUs) with Tensor Cores, and as a result, the algorithm throughput is greatly accelerated, and latency is reduced. - -NVIDIA AR SDK has the following features: - -- **Face detection and tracking**, which detects, localizes, and tracks human faces in images or videos by using bounding boxes. -- **Facial landmark detection and tracking**, which which predicts and tracks the pixel locations of human facial landmark points and head poses in images or videos.The detected facial landmarks follow the _Multi-PIE 68 point mark-ups_ information in [Facial point annotations](https://ibug.doc.ic.ac.uk/resources/facial-point-annotations/). -- **Face 3D mesh and tracking**, which reconstructs and tracks a 3D human face and its head pose from the provided facial landmarks **.** NVIDIA AR SDK provides a sample application that demonstrates the features listed above n real time by using a webcam or offline videos. - -NVIDIA AR SDK is distributed in the following parts: - -- This open source repository that includes the SDK API and proxy linking source code (_add link to the "nvar" folder in this repo_), sample applications and their dependency libraries (_add link to the "samples" folder in this repo)_. -- An Installer hosted on Nvidia Dev Zone (_link to be added_) that installs the SDK DLLs, the models, and the SDK dependency libraries. - -Please refer to SDK programming guide (_add link to programming guide in this repo_) for configuring the SDK, integrating the SDK, compiling and running the sample application. diff --git a/docs/NVIDIA AR SDK Programming Guide.pdf b/docs/NVIDIA AR SDK Programming Guide.pdf new file mode 100644 index 0000000..2a07b52 Binary files /dev/null and b/docs/NVIDIA AR SDK Programming Guide.pdf differ diff --git a/nvar/include/nvAR.h b/nvar/include/nvAR.h index 34a3f3f..f7413cf 100644 --- a/nvar/include/nvAR.h +++ b/nvar/include/nvAR.h @@ -38,6 +38,13 @@ typedef struct CUstream_st *CUstream; typedef struct nvAR_Feature nvAR_Feature; typedef struct nvAR_Feature *NvAR_FeatureHandle; +//! Get the SDK version +//! \param[in,out] version Pointer to an unsigned int set to +//! (major << 24) | (minor << 16) | (build << 8) | 0 +//! \return NVCV_SUCCESS if the version was set +//! \return NVCV_ERR_PARAMETER if version was NULL +NvCV_Status NvAR_API NvAR_GetVersion(unsigned int *version); + //! Create a new feature instantiation. //! \param[in] InFeatureID The selector code for the desired feature. //! \param[out] handle Handle to the feature instance. @@ -90,7 +97,7 @@ NvCV_Status NvAR_API NvAR_GetObject(NvAR_FeatureHandle handle, const char *name, unsigned long typeSize); NvCV_Status NvAR_API NvAR_GetString(NvAR_FeatureHandle handle, const char *name, const char **str); NvCV_Status NvAR_API NvAR_GetCudaStream(NvAR_FeatureHandle handle, const char *name, const CUstream *stream); -NvCV_Status NvAR_API NvAR_GetF32Array(NvAR_FeatureHandle handle, const char *name, const float **vals, int */*count*/); +NvCV_Status NvAR_API NvAR_GetF32Array(NvAR_FeatureHandle handle, const char *name, const float **vals, int* /*count*/); #ifdef __cplusplus } diff --git a/nvar/include/nvAR_defs.h b/nvar/include/nvAR_defs.h index 85951a0..04ea3d5 100644 --- a/nvar/include/nvAR_defs.h +++ b/nvar/include/nvAR_defs.h @@ -136,12 +136,18 @@ NvAR_Parameter_Output(Pose) - OPTIONAL NvAR_Parameter_Output(LandmarksConfidence) - OPTIONAL *******NvAR_Feature_Face3DReconstruction******* -Config +Config: NvAR_Parameter_Config(FeatureDescription) NvAR_Parameter_Config(ModelDir) NvAR_Parameter_Config(Landmarks_Size) NvAR_Parameter_Config(CUDAStream) -OPTIONAL NvAR_Parameter_Config(Temporal) - OPTIONAL +NvAR_Parameter_Config(ModelName) - OPTIONAL +NvAR_Parameter_Config(GPU) - OPTIONAL +NvAR_Parameter_Config(VertexCount) - QUERY +NvAR_Parameter_Config(TriangleCount) - QUERY +NvAR_Parameter_Config(ExpressionCount) - QUERY +NvAR_Parameter_Config(ShapeEigenValueCount) - QUERY Input: NvAR_Parameter_Input(Width) @@ -157,6 +163,8 @@ NvAR_Parameter_Output(BoundingBoxesConfidence) - OPTIONAL NvAR_Parameter_Output(Landmarks) - OPTIONAL NvAR_Parameter_Output(Pose) - OPTIONAL NvAR_Parameter_Output(LandmarksConfidence) - OPTIONAL +NvAR_Parameter_Output(ExpressionCoefficients) - OPTIONAL +NvAR_Parameter_Output(ShapeEigenValues) - OPTIONAL */ #endif // NvAR_DEFS_H diff --git a/nvar/include/nvCVImage.h b/nvar/include/nvCVImage.h index f63b71c..7f07360 100644 --- a/nvar/include/nvCVImage.h +++ b/nvar/include/nvCVImage.h @@ -63,7 +63,7 @@ typedef enum NvCVImage_ComponentType { } NvCVImage_ComponentType; -//! Value for the planar field or isPlanar argument. Two values are currently accommodated for RGB: +//! Value for the planar field or layout argument. Two values are currently accommodated for RGB: //! Interleaved or chunky storage locates all components of a pixel adjacent in memory, //! e.g. RGBRGBRGB... (denoted [RGB]). //! Planar storage locates the same component of all pixels adjacent in memory, @@ -104,12 +104,11 @@ typedef enum NvCVImage_ComponentType { #define NVCV_CHROMA_MPEG2 NVCV_CHROMA_COSITED #define NVCV_CHROMA_MPEG1 NVCV_CHROMA_INTSTITIAL -//! This is the value for the gpuMem field or the onGPU argument. Two values are currently accommodated: -//! CPU indicates standard CPU memory. -//! GPU indicates CUDA buffers. +//! This is the value for the gpuMem field or the memSpace argument. #define NVCV_CPU 0 //!< The buffer is stored in CPU memory. #define NVCV_GPU 1 //!< The buffer is stored in CUDA memory. #define NVCV_CUDA 1 //!< The buffer is stored in CUDA memory. +#define NVCV_CPU_PINNED 2 //!< The buffer is stored in pinned CPU memory. //! Image descriptor. typedef struct @@ -125,8 +124,8 @@ NvCVImage { unsigned char pixelBytes; //!< The number of bytes in a chunky pixel. unsigned char componentBytes; //!< The number of bytes in each pixel component. unsigned char numComponents; //!< The number of components in each pixel. - unsigned char planar; //!< 0=chunky, 1=planar, 2=semi-planar (NV12, NV21). - unsigned char gpuMem; //!< 0=cpu mem, 1=cuda mem, + unsigned char planar; //!< NVCV_CHUNKY, NVCV_PLANAR, NVCV_UYVY, .... + unsigned char gpuMem; //!< NVCV_CPU, NVCV_CPU_PINNED, NVCV_CUDA, NVCV_GPU unsigned char colorspace; //!< an OR of colorspace, range and chroma phase. unsigned char reserved[2]; //!< For structure padding and future expansion. Set to 0. void *pixels; //!< Pointer to pixel(0,0) in the image. @@ -145,14 +144,14 @@ NvCVImage { //! \param[in] height the number of pixels vertically. //! \param[in] format the format of the pixels. //! \param[in] type the type of each pixel component. - //! \param[in] isPlanar One of { NVCV_CHUNKY, NVCV_PLANAR }. - //! \param[in] onGPU One of { NVCV_CPU, NVCV_GPU } + //! \param[in] layout One of { NVCV_CHUNKY, NVCV_PLANAR } or one of the YUV layouts. + //! \param[in] memSpace One of { NVCV_CPU, NVCV_CPU_PINNED, NVCV_GPU, NVCV_CUDA } //! \param[in] alignment row byte alignment. Choose 0 or a power of 2. //! 1: yields no gap whatsoever between scanlines; //! 0: default alignment: 4 on CPU, and cudaMallocPitch's choice on GPU. //! Other common values are 16 or 32 for cache line size. inline NvCVImage(unsigned width, unsigned height, NvCVImage_PixelFormat format, NvCVImage_ComponentType type, - unsigned isPlanar = 0, unsigned onGPU = 0, unsigned alignment = 0); + unsigned layout = NVCV_CHUNKY, unsigned memSpace = NVCV_CPU, unsigned alignment = 0); //! Subimage constructor. //! \param[in] fullImg the full image, from which this subImage view is to be created. @@ -206,12 +205,12 @@ NvCVImage { //! \param[in] pixels a pointer to the pixel buffer. //! \param[in] format the format of the pixels. //! \param[in] type the type of the components of the pixels. -//! \param[in] isPlanar One of { NVCV_CHUNKY, NVCV_PLANAR }. -//! \param[in] onGPU Location of the buffer: one of { NVCV_CPU, NVCV_GPU } +//! \param[in] layout One of { NVCV_CHUNKY, NVCV_PLANAR } or one of the YUV layouts. +//! \param[in] memSpace Location of the buffer: one of { NVCV_CPU, NVCV_CPU_PINNED, NVCV_GPU, NVCV_CUDA } //! \return NVCV_SUCCESS if successful //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not yet accommodated. NvCV_Status NvCV_API NvCVImage_Init(NvCVImage *im, unsigned width, unsigned height, int pitch, void *pixels, - NvCVImage_PixelFormat format, NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU); + NvCVImage_PixelFormat format, NvCVImage_ComponentType type, unsigned layout, unsigned memSpace); //! Initialize a view into a subset of an existing image. @@ -234,8 +233,8 @@ void NvCV_API NvCVImage_InitView(NvCVImage *subImg, NvCVImage *fullImg, int x, i //! \param[in] height the desired height of the image, in pixels. //! \param[in] format the format of the pixels. //! \param[in] type the type of the components of the pixels. -//! \param[in] isPlanar One of { NVCV_CHUNKY, NVCV_PLANAR }. -//! \param[in] onGPU Location of the buffer: one of { NVCV_CPU, NVCV_GPU } +//! \param[in] layout One of { NVCV_CHUNKY, NVCV_PLANAR } or one of the YUV layouts. +//! \param[in] memSpace Location of the buffer: one of { NVCV_CPU, NVCV_CPU_PINNED, NVCV_GPU, NVCV_CUDA } //! \param[in] alignment row byte alignment. Choose 0 or a power of 2. //! 1: yields no gap whatsoever between scanlines; //! 0: default alignment: 4 on CPU, and cudaMallocPitch's choice on GPU. @@ -244,7 +243,7 @@ void NvCV_API NvCVImage_InitView(NvCVImage *subImg, NvCVImage *fullImg, int x, i //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not accommodated. //! \return NVCV_ERR_MEMORY if there is not enough memory to allocate the buffer. NvCV_Status NvCV_API NvCVImage_Alloc(NvCVImage *im, unsigned width, unsigned height, NvCVImage_PixelFormat format, - NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU, unsigned alignment); + NvCVImage_ComponentType type, unsigned layout, unsigned memSpace, unsigned alignment); //! Reallocate memory for, and initialize an image. This assumes that the image is valid. @@ -255,8 +254,8 @@ NvCV_Status NvCV_API NvCVImage_Alloc(NvCVImage *im, unsigned width, unsigned hei //! \param[in] height the desired height of the image, in pixels. //! \param[in] format the format of the pixels. //! \param[in] type the type of the components of the pixels. -//! \param[in] isPlanar One of { NVCV_CHUNKY, NVCV_PLANAR }. -//! \param[in] onGPU Location of the buffer: one of { NVCV_CPU, NVCV_GPU } +//! \param[in] layout One of { NVCV_CHUNKY, NVCV_PLANAR } or one of the YUV layouts. +//! \param[in] memSpace Location of the buffer: one of { NVCV_CPU, NVCV_CPU_PINNED, NVCV_GPU, NVCV_CUDA } //! \param[in] alignment row byte alignment. Choose 0 or a power of 2. //! 1: yields no gap whatsoever between scanlines; //! 0: default alignment: 4 on CPU, and cudaMallocPitch's choice on GPU. @@ -265,7 +264,7 @@ NvCV_Status NvCV_API NvCVImage_Alloc(NvCVImage *im, unsigned width, unsigned hei //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not accommodated. //! \return NVCV_ERR_MEMORY if there is not enough memory to allocate the buffer. NvCV_Status NvCV_API NvCVImage_Realloc(NvCVImage *im, unsigned width, unsigned height, NvCVImage_PixelFormat format, - NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU, unsigned alignment); + NvCVImage_ComponentType type, unsigned layout, unsigned memSpace, unsigned alignment); //! Deallocate the image buffer from the image. The image is not deallocated. @@ -278,8 +277,8 @@ void NvCV_API NvCVImage_Dealloc(NvCVImage *im); //! \param[in] height the desired height of the image, in pixels. //! \param[in] format the format of the pixels. //! \param[in] type the type of the components of the pixels. -//! \param[in] isPlanar One of { NVCV_CHUNKY, NVCV_PLANAR }. -//! \param[in] onGPU Location of the buffer: one of { NVCV_CPU, NVCV_GPU } +//! \param[in] layout One of { NVCV_CHUNKY, NVCV_PLANAR } or one of the YUV layouts. +//! \param[in] memSpace Location of the buffer: one of { NVCV_CPU, NVCV_CPU_PINNED, NVCV_GPU, NVCV_CUDA } //! \param[in] alignment row byte alignment. Choose 0 or a power of 2. //! 1: yields no gap whatsoever between scanlines; //! 0: default alignment: 4 on CPU, and cudaMallocPitch's choice on GPU. @@ -289,7 +288,7 @@ void NvCV_API NvCVImage_Dealloc(NvCVImage *im); //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not accommodated. //! \return NVCV_ERR_MEMORY if there is not enough memory to allocate the buffer. NvCV_Status NvCV_API NvCVImage_Create(unsigned width, unsigned height, NvCVImage_PixelFormat format, - NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU, unsigned alignment, NvCVImage **out); + NvCVImage_ComponentType type, unsigned layout, unsigned memSpace, unsigned alignment, NvCVImage **out); //! Deallocate the image allocated with NvCVImage_Create() (C-style destructor). @@ -308,22 +307,38 @@ void NvCV_API NvCVImage_ComponentOffsets(NvCVImage_PixelFormat format, int *rOff //! Transfer one image to another, with a limited set of conversions. +//! //! If any of the images resides on the GPU, it may run asynchronously, //! so cudaStreamSynchronize() should be called if it is necessary to run synchronously. -//! Conversions are between -//! - RGBu8 --> RGBu8, where the RGB components are in any order; -//! - RGBu8 <--> RGBf32, where the RGB components are in any order (e.g. BGR, RGB); -//! - RGBu8 --> GRAYf32, where the RGB components are in any order; -//! - RGBf32 --> RGBAu8, setting A=255, where the RGB and RGBA components are in any order; -//! - RGBAu8 --> RGBu8, by removing the alpha component; -//! - GRAYf32 --> GRAYu8; -//! - GRAYu8 --> GRAYu8, where the RGB components are in any order (also works with ALPHAu8); -//! - ALPHAu8 --> RGBAu8, (insertion) without touching the RGB components. -//! - GRAYu8 --> RGBu8, by replicating gray into all RGB components. -//! - YUVu8 --> RGBu8, though colorspace field needs to be set manually prior to calling. -//! - chunky <--> planar; -//! - CPU <--> GPU; -//! Additionally, when the src and dst formats are the same, all formats are accommodated on CPU and GPU, +//! The following table indicates the currently-implemented conversions: +//! +------------------+-------------+-------------+-------------+-------------+ +//! | | u8 --> u8 | u8 --> f32 | f32 --> u8 | f32 --> f32 | +//! +------------------+-------------+-------------+-------------+-------------+ +//! | Y -- > Y | X | | X | X | +//! | Y -- > A | X | | X | X | +//! | Y -- > RGB | X | X | X | X | +//! | Y -- > RGBA | X | X | X | X | +//! | A -- > Y | X | | X | X | +//! | A -- > A | X | | X | X | +//! | A -- > RGB | X | X | X | X | +//! | A -- > RGBA | X | | | | +//! | RGB -- > Y | X | X | | | +//! | RGB -- > A | X | X | | | +//! | RGB -- > RGB | X | X | X | X | +//! | RGB -- > RGBA | X | X | X | X | +//! | RGBA -- > Y | X | X | | | +//! | RGBA -- > A | | X | | | +//! | RGBA -- > RGB | X | X | X | X | +//! | RGBA -- > RGBA | X | | | | +//! | YUV420 -- > RGB | X | | | | +//! | YUV422 -- > RGB | X | | | | +//! +------------------+-------------+-------------+-------------+-------------+ +//! where +//! * Either source or destination can be CHUNKY or PLANAR. +//! * Either source or destination can reside on the CPU or the GPU. +//! * The RGB components are in any order (i.e. RGB or BGR; RGBA or BGRA). +//! * YUV requires that the colorspace field be set manually prior to Transfer. +//! * Additionally, when the src and dst formats are the same, all formats are accommodated on CPU and GPU, //! and this can be used as a replacement for cudaMemcpy2DAsync() (which it utilizes). //! //! When there is some kind of conversion AND the src and dst reside on different processors (CPU, GPU), @@ -354,14 +369,15 @@ NvCV_Status NvCV_API NvCVImage_Transfer( //! Composite one BGRu8 source image over another using the given matte. -//! \param[in] src the source BGRu8 (or RGBu8) image. +//! \param[in] fg the foreground source BGRu8 (or RGBu8) image. +//! \param[in] bg the background source BGRu8 (or RGBu8) image. //! \param[in] mat the matte Yu8 (or Au8) image, indicating where the src should come through. -//! \param[out] dst the destination BGRu8 (or RGBu8) image. +//! \param[out] dst the destination BGRu8 (or RGBu8) image. This can be the same as fg or bg. //! \return NVCV_SUCCESS if the operation was successful. //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not accommodated. -//! \bug This is only implemented for 3-component u8 src and dst, and 1-component mat, +//! \bug This is only implemented for 3-component u8 fg, bg and dst, and 1-component u8 mat, //! where all images are resident on the CPU. -NvCV_Status NvCV_API NvCVImage_Composite(const NvCVImage *src, const NvCVImage *mat, NvCVImage *dst); +NvCV_Status NvCV_API NvCVImage_Composite(const NvCVImage *fg, const NvCVImage *bg, const NvCVImage *mat, NvCVImage *dst); //! Composite a BGRu8 source image over a constant color field using the given matte. @@ -423,9 +439,9 @@ NvCVImage::NvCVImage() { ********************************************************************************/ NvCVImage::NvCVImage(unsigned width, unsigned height, NvCVImage_PixelFormat format, NvCVImage_ComponentType type, - unsigned isPlanar, unsigned onGPU, unsigned alignment) { + unsigned layout, unsigned memSpace, unsigned alignment) { pixels = nullptr; - (void)NvCVImage_Alloc(this, width, height, format, type, isPlanar, onGPU, alignment); + (void)NvCVImage_Alloc(this, width, height, format, type, layout, memSpace, alignment); } /******************************************************************************** diff --git a/nvar/include/nvCVStatus.h b/nvar/include/nvCVStatus.h index 5c32654..17997dc 100644 --- a/nvar/include/nvCVStatus.h +++ b/nvar/include/nvCVStatus.h @@ -61,6 +61,9 @@ typedef enum NvCV_Status { NVCV_ERR_FEATURENOTFOUND = -14, //!< The requested feature was not found NVCV_ERR_MISSINGINPUT = -15, //!< A required parameter was not set NVCV_ERR_RESOLUTION = -16, //!< The specified image resolution is not supported. + NVCV_ERR_UNSUPPORTEDGPU = -17, //!< The GPU is not supported + NVCV_ERR_WRONGGPU = -18, //!< The current GPU is not the one selected. + NVCV_ERR_UNSUPPORTEDDRIVER = -19, //!< The currently installed graphics driver is not supported NVCV_ERR_CUDA_MEMORY = -20, //!< There is not enough CUDA memory for the requested operation. NVCV_ERR_CUDA_VALUE = -21, //!< A CUDA parameter is not within the acceptable range. diff --git a/nvar/src/nvARProxy.cpp b/nvar/src/nvARProxy.cpp index 8368ca9..44f1c02 100644 --- a/nvar/src/nvARProxy.cpp +++ b/nvar/src/nvARProxy.cpp @@ -62,7 +62,7 @@ inline int nvFreeLibrary(HINSTANCE handle) { HINSTANCE getNvARLib() { - TCHAR path[MAX_PATH], fullPath[2*MAX_PATH]; + TCHAR path[MAX_PATH], fullPath[MAX_PATH]; // There can be multiple apps on the system, // some might include the SDK in the app package and @@ -80,6 +80,13 @@ HINSTANCE getNvARLib() { return NvArLib; } +NvCV_Status NvAR_API NvAR_GetVersion(unsigned int* version) { + static const auto funcPtr = (decltype(NvAR_GetVersion)*)nvGetProcAddress(getNvARLib(), "NvAR_GetVersion"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(version); +} + NvCV_Status NvAR_API NvCVImage_Init(NvCVImage* im, unsigned width, unsigned height, int pitch, void* pixels, NvCVImage_PixelFormat format, NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU) { @@ -150,11 +157,11 @@ NvCV_Status NvAR_API NvCVImage_Transfer(const NvCVImage* src, NvCVImage* dst, fl return funcPtr(src, dst, scale, stream, tmp); } -NvCV_Status NvAR_API NvCVImage_Composite(const NvCVImage* src, const NvCVImage* mat, NvCVImage* dst) { +NvCV_Status NvAR_API NvCVImage_Composite(const NvCVImage* fg, const NvCVImage* bg, const NvCVImage* mat, NvCVImage* dst) { static const auto funcPtr = (decltype(NvCVImage_Composite)*)nvGetProcAddress(getNvARLib(), "NvCVImage_Composite"); if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; - return funcPtr(src, mat, dst); + return funcPtr(fg, bg, mat, dst); } NvCV_Status NvAR_API NvCVImage_CompositeOverConstant(const NvCVImage* src, const NvCVImage* mat, @@ -307,7 +314,7 @@ NvCV_Status NvAR_API NvAR_GetCudaStream(NvAR_FeatureHandle handle, const char* n } NvCV_Status NvAR_API NvAR_GetF32Array(NvAR_FeatureHandle handle, const char* name, const float** vals, int* count) { - static const auto funcPtr = (decltype(NvAR_GetF32Array)*)nvGetProcAddress(getNvARLib(), "NvAR_GetCNvAR_GetF32ArrayudaStream"); + static const auto funcPtr = (decltype(NvAR_GetF32Array)*)nvGetProcAddress(getNvARLib(), "NvAR_GetF32Array"); if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; return funcPtr(handle, name, vals, count); diff --git a/samples/FaceTrack/CMakeLists.txt b/samples/FaceTrack/CMakeLists.txt index 49f8858..8b3eca2 100644 --- a/samples/FaceTrack/CMakeLists.txt +++ b/samples/FaceTrack/CMakeLists.txt @@ -1,4 +1,10 @@ -set(SOURCE_FILES FaceEngine.cpp FaceTrack.cpp ../utils/RenderingUtils.cpp ../../nvar/src/nvARProxy.cpp) +set(SOURCE_FILES FaceEngine.cpp +FaceTrack.cpp +../utils/RenderingUtils.cpp +../../nvar/src/nvARProxy.cpp +../utils/FeatureVertexName.cpp +../utils/FeatureVertexName.h +) set(HEADER_FILES FaceEngine.h) # Set Visual Studio source filters @@ -16,10 +22,14 @@ target_link_libraries(FaceTrack PUBLIC GLM ) +set(ARSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin) -set(PATH_STR "PATH=%PATH%" ${OPENCV_PATH_STR}) +set(PATH_STR "PATH=%PATH%" ${ARSDK_PATH_STR} ${OPENCV_PATH_STR}) +set(CMD_ARG_STR "--model_path=\"${CMAKE_CURRENT_SOURCE_DIR}/../../bin/models\"") set_target_properties(FaceTrack PROPERTIES FOLDER SampleApps VS_DEBUGGER_ENVIRONMENT "${PATH_STR}" + VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}" ) + diff --git a/samples/FaceTrack/FaceEngine.cpp b/samples/FaceTrack/FaceEngine.cpp index ddab3d9..83dc3e4 100644 --- a/samples/FaceTrack/FaceEngine.cpp +++ b/samples/FaceTrack/FaceEngine.cpp @@ -23,31 +23,10 @@ #include "FaceEngine.h" #include "RenderingUtils.h" -const char* NvCV_StatusStringFromCode(NvCV_Status code) { - struct TabEntry { - NvCV_Status code; - const char* str; - }; - static const TabEntry lut[] = { - {NVCV_SUCCESS, "no error"}, - {NVCV_ERR_GENERAL, "unspecified failure"}, - {NVCV_ERR_FEATURENOTFOUND, "Feature not found"}, - {NVCV_ERR_PARAMETER, "invalid parameter"}, - {NVCV_ERR_MEMORY, "provided buffer too small"}, - {NVCV_ERR_INITIALIZATION, "not initialized"}, - {NVCV_ERR_MISSINGINPUT, "missing input"}, - {NVCV_ERR_INITIALIZATION, "unable to initialize feature"}, - {NVCV_ERR_CUDA_MEMORY, "out of GPU memory"}, - {NVCV_ERR_SELECTOR, "unsupported parameter"}, - }; - for (const TabEntry* p = lut; p != &lut[sizeof(lut) / sizeof(lut[0])]; ++p) - if (p->code == code) return p->str; - return "UNKNOWN ERROR"; -} bool CheckResult(NvCV_Status nvErr, unsigned line) { if (NVCV_SUCCESS == nvErr) return true; - std::cout << "ERROR: " << NvCV_StatusStringFromCode(nvErr) << ", line " << line << std::endl; + std::cout << "ERROR: " << NvCV_GetErrorStringFromCode(nvErr) << ", line " << line << std::endl; return false; } @@ -67,7 +46,7 @@ FaceEngine::Err FaceEngine::fitFaceModel(cv::Mat& frame) { nvErr = NvAR_Run(faceFitHandle); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errRun); - if (getAverageLandmarksConfidence() < LANDMARK_CONF_THRESH) return FaceEngine::Err::errRun; + if (getAverageLandmarksConfidence() < confidenceThreshold) return FaceEngine::Err::errRun; bail: return err; @@ -81,30 +60,44 @@ FaceEngine::Err FaceEngine::createFeatures(const char* modelPath, unsigned int _ FaceEngine::Err err = FaceEngine::Err::errNone; NvCV_Status cuErr = NvAR_CudaStreamCreate(&stream); - if (appMode == faceDetection) + if (NVCV_SUCCESS != cuErr) { + printf("Cannot create a cuda stream: %s\n", NvCV_GetErrorStringFromCode(cuErr)); + return errInitialization; + } + if (appMode == faceDetection) { err = createFaceDetectionFeature(modelPath, stream); - else if (appMode == landmarkDetection) + if (err != Err::errNone) { + printf("ERROR: An error has occured while initializing Face Detection\n"); + } + } else if (appMode == landmarkDetection) { err = createLandmarkDetectionFeature(modelPath, _batchSize, stream); - else if (appMode == faceMeshGeneration) + if (err != Err::errNone) { + printf("ERROR: An error has occured while initializing Landmark Detection\n"); + } + } else if (appMode == faceMeshGeneration) { err = createFaceFittingFeature(modelPath, stream); + if (err != Err::errNone) { + printf("ERROR: An error has occured while initializing Face Fitting\n"); + } + } return err; } -FaceEngine::Err FaceEngine::createFaceDetectionFeature(const char* modelPath, CUstream stream) { +FaceEngine::Err FaceEngine::createFaceDetectionFeature(const char* modelPath, CUstream str) { FaceEngine::Err err = FaceEngine::Err::errNone; NvCV_Status nvErr; nvErr = NvAR_Create(NvAR_Feature_FaceDetection, &faceDetectHandle); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errEffect); nvErr = NvAR_SetString(faceDetectHandle, NvAR_Parameter_Config(ModelDir), modelPath); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); - nvErr = NvAR_SetCudaStream(faceDetectHandle, NvAR_Parameter_Config(CUDAStream), stream); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetCudaStream(faceDetectHandle, NvAR_Parameter_Config(CUDAStream), str); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetU32(faceDetectHandle, NvAR_Parameter_Config(Temporal), bStabilizeFace); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_Load(faceDetectHandle); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); @@ -114,25 +107,31 @@ bail: } FaceEngine::Err FaceEngine::createLandmarkDetectionFeature(const char* modelPath, unsigned int _batchSize, - CUstream stream) { + CUstream str) { FaceEngine::Err err = FaceEngine::Err::errNone; NvCV_Status nvErr; batchSize = _batchSize; nvErr = NvAR_Create(NvAR_Feature_LandmarkDetection, &landmarkDetectHandle); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errEffect); nvErr = NvAR_SetString(landmarkDetectHandle, NvAR_Parameter_Config(ModelDir), modelPath); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); - nvErr = NvAR_SetCudaStream(landmarkDetectHandle, NvAR_Parameter_Config(CUDAStream), stream); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetCudaStream(landmarkDetectHandle, NvAR_Parameter_Config(CUDAStream), str); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetU32(landmarkDetectHandle, NvAR_Parameter_Config(BatchSize), batchSize); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetU32(landmarkDetectHandle, NvAR_Parameter_Config(Temporal), bStabilizeFace); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + + nvErr = NvAR_SetU32(landmarkDetectHandle, NvAR_Parameter_Config(Landmarks_Size), numLandmarks); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + + nvErr = NvAR_SetU32(landmarkDetectHandle, NvAR_Parameter_Config(LandmarksConfidence_Size), numLandmarks); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_Load(landmarkDetectHandle); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); @@ -141,18 +140,26 @@ bail: return err; } -FaceEngine::Err FaceEngine::createFaceFittingFeature(const char* modelPath, CUstream stream) { +FaceEngine::Err FaceEngine::createFaceFittingFeature(const char* modelPath, CUstream str) { FaceEngine::Err err = FaceEngine::Err::errNone; NvCV_Status nvErr; nvErr = NvAR_Create(NvAR_Feature_Face3DReconstruction, &faceFitHandle); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errEffect); nvErr = NvAR_SetString(faceFitHandle, NvAR_Parameter_Config(ModelDir), modelPath); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); - nvErr = NvAR_SetCudaStream(faceFitHandle, NvAR_Parameter_Config(CUDAStream), stream); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetCudaStream(faceFitHandle, NvAR_Parameter_Config(CUDAStream), str); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + + nvErr = NvAR_SetU32(faceFitHandle, NvAR_Parameter_Config(Landmarks_Size), numLandmarks); // TODO: Check if nonzero?? + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + + if (!face_model.empty()) { + nvErr = NvAR_SetString(faceFitHandle, NvAR_Parameter_Config(ModelName), face_model.c_str()); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + } nvErr = NvAR_Load(faceFitHandle); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); @@ -169,55 +176,64 @@ FaceEngine::Err FaceEngine::initFeatureIOParams() { BAIL_IF_CVERR(cvErr, err, FaceEngine::Err::errInitialization); - if (appMode == faceDetection) + if (appMode == faceDetection) { err = initFaceDetectionIOParams(&inputImageBuffer); - else if (appMode == landmarkDetection) + if (err != Err::errNone) { + printf("ERROR: An error has occured while setting input, output parmeters for Face Detection\n"); + } + } else if (appMode == landmarkDetection) { err = initLandmarkDetectionIOParams(&inputImageBuffer); - else if (appMode == faceMeshGeneration) + if (err != Err::errNone) { + printf("ERROR: An error has occured while setting input, output parmeters for Landmark Detection\n"); + } + } else if (appMode == faceMeshGeneration) { err = initFaceFittingIOParams(&inputImageBuffer); - + if (err != Err::errNone) { + printf("ERROR: An error has occured while setting input, output parmeters for Face Fitting\n"); + } + } return err; bail: return err; } -FaceEngine::Err FaceEngine::initFaceDetectionIOParams(NvCVImage* _inputImageBuffer) { +FaceEngine::Err FaceEngine::initFaceDetectionIOParams(NvCVImage* inBuf) { NvCV_Status nvErr = NVCV_SUCCESS; FaceEngine::Err err = FaceEngine::Err::errNone; - nvErr = NvAR_SetObject(faceDetectHandle, NvAR_Parameter_Input(Image), &inputImageBuffer, sizeof(NvCVImage)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetObject(faceDetectHandle, NvAR_Parameter_Input(Image), inBuf, sizeof(NvCVImage)); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); output_bbox_data.assign(25, {0.f, 0.f, 0.f, 0.f}); output_bbox_conf_data.assign(25, 0.f); output_bboxes.boxes = output_bbox_data.data(); - output_bboxes.max_boxes = output_bbox_data.size(); + output_bboxes.max_boxes = (uint8_t)output_bbox_data.size(); output_bboxes.num_boxes = 0; nvErr = NvAR_SetObject(faceDetectHandle, NvAR_Parameter_Output(BoundingBoxes), &output_bboxes, sizeof(NvAR_BBoxes)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetF32Array(faceDetectHandle, NvAR_Parameter_Output(BoundingBoxesConfidence), output_bbox_conf_data.data(), output_bboxes.max_boxes); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); bail: return err; } -FaceEngine::Err FaceEngine::initLandmarkDetectionIOParams(NvCVImage* _inputImageBuffer) { +FaceEngine::Err FaceEngine::initLandmarkDetectionIOParams(NvCVImage* inBuf) { NvCV_Status nvErr = NVCV_SUCCESS; FaceEngine::Err err = FaceEngine::Err::errNone; - nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Input(Image), &inputImageBuffer, sizeof(NvCVImage)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Input(Image), inBuf, sizeof(NvCVImage)); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); unsigned int OUTPUT_SIZE_KPTS, OUTPUT_SIZE_KPTS_CONF; nvErr = NvAR_GetU32(landmarkDetectHandle, NvAR_Parameter_Config(Landmarks_Size), &OUTPUT_SIZE_KPTS); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_GetU32(landmarkDetectHandle, NvAR_Parameter_Config(LandmarksConfidence_Size), &OUTPUT_SIZE_KPTS_CONF); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); facial_landmarks.assign(batchSize * OUTPUT_SIZE_KPTS, {0.f, 0.f}); facial_pose.assign(batchSize, {0.f, 0.f, 0.f, 0.f}); @@ -225,74 +241,78 @@ FaceEngine::Err FaceEngine::initLandmarkDetectionIOParams(NvCVImage* _inputImage nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Output(Landmarks), facial_landmarks.data(), sizeof(NvAR_Point2f)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Output(Pose), facial_pose.data(), sizeof(NvAR_Quaternion)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetF32Array(landmarkDetectHandle, NvAR_Parameter_Output(LandmarksConfidence), facial_landmarks_confidence.data(), batchSize * OUTPUT_SIZE_KPTS); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); uint output_bbox_size = batchSize; if (!bStabilizeFace) output_bbox_size = 25; output_bbox_data.assign(output_bbox_size, {0.f, 0.f, 0.f, 0.f}); output_bboxes.boxes = output_bbox_data.data(); - output_bboxes.max_boxes = output_bbox_size; - output_bboxes.num_boxes = output_bbox_size; + output_bboxes.max_boxes = (uint8_t)output_bbox_size; + output_bboxes.num_boxes = (uint8_t)output_bbox_size; nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Output(BoundingBoxes), &output_bboxes, sizeof(NvAR_BBoxes)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); bail: return err; } -FaceEngine::Err FaceEngine::initFaceFittingIOParams(NvCVImage* inputImageBuffer) { +FaceEngine::Err FaceEngine::initFaceFittingIOParams(NvCVImage* inBuf) { NvCV_Status nvErr = NVCV_SUCCESS; FaceEngine::Err err = FaceEngine::Err::errNone; face_mesh = new NvAR_FaceMesh(); - face_mesh->vertices = new NvAR_Vector3f[FACE_MODEL_NUM_VERTICES]; - face_mesh->tvi = new NvAR_Vector3u16[FACE_MODEL_NUM_INDICES]; + face_mesh->vertices = nullptr; //new NvAR_Vector3f[FACE_MODEL_NUM_VERTICES]; + face_mesh->tvi = nullptr; // new NvAR_Vector3u16[FACE_MODEL_NUM_INDICES]; rendering_params = new NvAR_RenderingParams(); - nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Input(Image), inputImageBuffer, sizeof(NvCVImage)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Input(Image), inBuf, sizeof(NvCVImage)); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetS32(faceFitHandle, NvAR_Parameter_Input(Width), input_image_width); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetS32(faceFitHandle, NvAR_Parameter_Input(Height), input_image_height); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); unsigned int OUTPUT_SIZE_KPTS; nvErr = NvAR_GetU32(faceFitHandle, NvAR_Parameter_Config(Landmarks_Size), &OUTPUT_SIZE_KPTS); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); facial_landmarks.assign(batchSize * OUTPUT_SIZE_KPTS, {0.f, 0.f}); nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Output(Landmarks), facial_landmarks.data(), sizeof(NvAR_Point2f)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); facial_landmarks_confidence.assign(batchSize * OUTPUT_SIZE_KPTS, 0.f); nvErr = NvAR_SetF32Array(faceFitHandle, NvAR_Parameter_Output(LandmarksConfidence), facial_landmarks_confidence.data(), batchSize * OUTPUT_SIZE_KPTS); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + + facial_pose.assign(batchSize, {0.f, 0.f, 0.f, 0.f}); + nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Output(Pose), facial_pose.data(), sizeof(NvAR_Quaternion)); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); output_bbox_data.assign(batchSize, {0.f, 0.f, 0.f, 0.f}); output_bboxes.boxes = output_bbox_data.data(); - output_bboxes.max_boxes = batchSize; - output_bboxes.num_boxes = batchSize; + output_bboxes.max_boxes = (uint8_t)batchSize; + output_bboxes.num_boxes = (uint8_t)batchSize; nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Output(BoundingBoxes), &output_bboxes, sizeof(NvAR_BBoxes)); nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Output(FaceMesh), face_mesh, sizeof(NvAR_FaceMesh)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Output(RenderingParams), rendering_params, sizeof(NvAR_RenderingParams)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); bail: return err; @@ -352,6 +372,7 @@ void FaceEngine::releaseFaceFittingIOParams() { if (!output_bbox_data.empty()) output_bbox_data.clear(); if (!facial_landmarks.empty()) facial_landmarks.clear(); if (!facial_landmarks_confidence.empty()) facial_landmarks_confidence.clear(); + if (!facial_pose.empty()) facial_pose.clear(); NvCVImage_Dealloc(&inputImageBuffer); if (rendering_params) { delete rendering_params; @@ -461,6 +482,7 @@ void FaceEngine::jiggleBox(std::mt19937& ran, float minMag, float maxMag, const #endif // None of these makes a significant difference } +#ifdef UNUSED /** Intersect a rectangle with an image. * @param[in] srcRect the source rectangle. * @param[in] src the source image. @@ -493,6 +515,7 @@ static bool IntersectRectWithImage(const cv::Rect& srcRect, const cv::Mat& src, clipRect.height = rect[1].y - rect[0].y; return result; } +#endif // UNUSED void FaceEngine::DrawPose(const cv::Mat& src, const NvAR_Quaternion* pose) { float R[3][3]; @@ -505,17 +528,19 @@ void FaceEngine::DrawPose(const cv::Mat& src, const NvAR_Quaternion* pose) { float x3 = radius * R[1][2] * -1.f; float y3 = radius * R[2][2] * -1.f; - // 30th point is the tip of the nose - const int nose_tip = 30; + int nose_tip = 0; + const char* kNoseTipName = "nose-tip"; + nose_tip = FindLandmarkIndexFromName(numLandmarks, kNoseTipName); + int width = src.cols; int height = src.rows; NvAR_Point2f cxy = *(facial_landmarks.data() + nose_tip); - float cx1 = std::min(std::max(0, int(cxy.x + x1)), width - 1); - float cy1 = std::min(std::max(0, int(cxy.y + y1)), height - 1); - float cx2 = std::min(std::max(0, int(cxy.x + x2)), width - 1); - float cy2 = std::min(std::max(0, int(cxy.y + y2)), height - 1); - float cx3 = std::min(std::max(0, int(cxy.x + x3)), width - 1); - float cy3 = std::min(std::max(0, int(cxy.y + y3)), height - 1); + float cx1 = (float)std::min(std::max(0, int(cxy.x + x1)), width - 1); + float cy1 = (float)std::min(std::max(0, int(cxy.y + y1)), height - 1); + float cx2 = (float)std::min(std::max(0, int(cxy.x + x2)), width - 1); + float cy2 = (float)std::min(std::max(0, int(cxy.y + y2)), height - 1); + float cx3 = (float)std::min(std::max(0, int(cxy.x + x3)), width - 1); + float cy3 = (float)std::min(std::max(0, int(cxy.y + y3)), height - 1); cv::line(src, cv::Point((int)cxy.x, (int)cxy.y), cv::Point((int)cx1, (int)cy1), cv::Scalar(0, 0, 255), 2); cv::line(src, cv::Point((int)cxy.x, (int)cxy.y), cv::Point((int)cx2, (int)cy2), cv::Scalar(0, 255, 0), 2); @@ -530,16 +555,16 @@ NvCV_Status FaceEngine::findLandmarks() { return nvErr; } - if (getAverageLandmarksConfidence() < LANDMARK_CONF_THRESH) { + if (getAverageLandmarksConfidence() < confidenceThreshold) { return NVCV_ERR_GENERAL; } else { average_poses(getPose(), batchSize); NvAR_Point2f *pt, *endPt; int i = 0; - for (endPt = (pt = getLandmarks()) + NUM_LANDMARKS; pt != endPt; ++pt, i += 2) { + for (endPt = (pt = getLandmarks()) + numLandmarks; pt != endPt; ++pt, i += 2) { for (int j = 1; j < batchSize; j++) { - pt->x += pt[j * NUM_LANDMARKS].x; - pt->y += pt[j * NUM_LANDMARKS].y; + pt->x += pt[j * numLandmarks].x; + pt->y += pt[j * numLandmarks].y; } // average batch of inferences to generate final result landmark points pt->x /= batchSize; @@ -557,11 +582,11 @@ float FaceEngine::getAverageLandmarksConfidence() { float average_confidence = 0.0f; float* keypoints_landmarks_confidence = getLandmarksConfidence(); for (int i = 0; i < batchSize; i++) { - for (int j = 0; j < NUM_LANDMARKS; j++) { - average_confidence += keypoints_landmarks_confidence[i * NUM_LANDMARKS + j]; + for (int j = 0; j < numLandmarks; j++) { + average_confidence += keypoints_landmarks_confidence[i * numLandmarks + j]; } } - average_confidence /= batchSize * NUM_LANDMARKS; + average_confidence /= batchSize * numLandmarks; return average_confidence; } @@ -598,7 +623,7 @@ unsigned FaceEngine::acquireFaceBox(cv::Mat& src, NvAR_Rect& faceBox, int varian return n; } -unsigned FaceEngine::acquireFaceBoxAndLandmarks(cv::Mat& src, NvAR_Point2f* refMarks, NvAR_Rect& faceBox, int variant) { +unsigned FaceEngine::acquireFaceBoxAndLandmarks(cv::Mat& src, NvAR_Point2f* refMarks, NvAR_Rect& faceBox, int /*variant*/) { unsigned n = 0; NvCVImage fxSrcChunkyCPU; (void)NVWrapperForCVMat(&src, &fxSrcChunkyCPU); @@ -611,11 +636,24 @@ unsigned FaceEngine::acquireFaceBoxAndLandmarks(cv::Mat& src, NvAR_Point2f* refM if (findLandmarks() != NVCV_SUCCESS) return 0; faceBox = output_bboxes.boxes[0]; n = 1; - memcpy(refMarks, getLandmarks(), sizeof(NvAR_Point2f) * FaceEngine::NUM_LANDMARKS); + memcpy(refMarks, getLandmarks(), sizeof(NvAR_Point2f) * numLandmarks); return n; } void FaceEngine::setFaceStabilization(bool _bStabilizeFace) { bStabilizeFace = _bStabilizeFace; } +FaceEngine::Err FaceEngine::setNumLandmarks(int n) { + FaceEngine::Err err = errNone; + for (auto const& info : LANDMARKS_INFO) { + if (n == info.numPoints) { + numLandmarks = info.numPoints; + confidenceThreshold = info.confidence_threshold; + return err; + } + } + err = errGeneral; + return err; +} + void FaceEngine::setAppMode(FaceEngine::mode _mode) { appMode = _mode; } diff --git a/samples/FaceTrack/FaceEngine.h b/samples/FaceTrack/FaceEngine.h index 743e016..2cbb055 100644 --- a/samples/FaceTrack/FaceEngine.h +++ b/samples/FaceTrack/FaceEngine.h @@ -27,6 +27,7 @@ #include "nvAR.h" #include "nvCVOpenCV.h" #include "opencv2/opencv.hpp" +#include "FeatureVertexName.h" #define FITFACE_PRIVATE class KalmanFilter1D { @@ -78,10 +79,15 @@ class KalmanFilter1D { } }; -const char* NvCV_StatusStringFromCode(NvCV_Status code); - bool CheckResult(NvCV_Status nvErr, unsigned line); +#define BAIL_IF_ERR(err) \ +do { \ + if (0!=err) { \ + goto bail; \ + } \ + } while (0) + #define BAIL_IF_CVERR(nvErr, err, code) \ do { \ if (!CheckResult(nvErr, __LINE__)) { \ @@ -90,23 +96,30 @@ bool CheckResult(NvCV_Status nvErr, unsigned line); } \ } while (0) +typedef struct LandmarksProperties { + int numPoints; + float confidence_threshold; +}LandmarksProperties; + /******************************************************************************** * FaceEngine ********************************************************************************/ class FaceEngine { public: - enum Err { errNone, errGeneral, errRun, errInitialization, errRead }; + enum Err { errNone, errGeneral, errRun, errInitialization, errRead, errEffect, errParameter }; int input_image_width, input_image_height, input_image_pitch; - static const int NUM_LANDMARKS = 68; // TODO: get this instead from the SDK. - static const int FACE_MODEL_NUM_VERTICES = 3448, FACE_MODEL_NUM_INDICES = 6736; - static const long LANDMARK_CONF_THRESH = 10.f; + const LandmarksProperties LANDMARKS_INFO[2] = { + { 68, 10.0f }, // number of landmark points, confidence threshold value + { 126, 5.0f} + }; void setInputImageWidth(int width) { input_image_width = width; } void setInputImageHeight(int height) { input_image_height = height; } int getInputImageWidth() { return input_image_width; } int getInputImageHeight() { return input_image_height; } int getInputImagePitch() { return input_image_pitch = input_image_width * 3 * sizeof(unsigned char); } + void setFaceModel(const char *faceModel) { face_model = faceModel; } Err createFeatures(const char* modelPath, unsigned int _batchSize = 1); Err createFaceDetectionFeature(const char* modelPath, CUstream stream); @@ -142,6 +155,8 @@ class FaceEngine { NvAR_FaceMesh* getFaceMesh(); NvAR_RenderingParams* getRenderingParams(); void setFaceStabilization(bool); + Err setNumLandmarks(int); + int getNumLandmarks() { return numLandmarks; } void DrawPose(const cv::Mat& src, const NvAR_Quaternion* pose); NvCVImage inputImageBuffer{}, tmpImage{}; @@ -157,13 +172,17 @@ class FaceEngine { NvAR_BBoxes output_bboxes{}; int batchSize; std::mt19937 ran; + int numLandmarks; + float confidenceThreshold; + std::string face_model; bool bStabilizeFace; - NvAR_Point2f prevLandmark[NUM_LANDMARKS] = {0}; FaceEngine() { batchSize = 1; bStabilizeFace = true; + numLandmarks = LANDMARKS_INFO[0].numPoints; + confidenceThreshold = LANDMARKS_INFO[0].confidence_threshold; appMode = faceMeshGeneration; input_image_width = 640; input_image_height = 480; diff --git a/samples/FaceTrack/FaceTrack.cpp b/samples/FaceTrack/FaceTrack.cpp index 2ba4a46..cbe6cf2 100644 --- a/samples/FaceTrack/FaceTrack.cpp +++ b/samples/FaceTrack/FaceTrack.cpp @@ -63,10 +63,10 @@ ********************************************************************************/ bool FLAG_debug = false, FLAG_verbose = false, FLAG_temporal = true, FLAG_captureOutputs = false, - FLAG_offlineMode = false; + FLAG_offlineMode = false, FLAG_isNumLandmarks126 = false; std::string FLAG_outDir, FLAG_inFile, FLAG_outFile, FLAG_modelPath, FLAG_landmarks, FLAG_proxyWireframe, - FLAG_captureCodec = "avc1", FLAG_camRes; -unsigned int FLAG_batch = 1; + FLAG_captureCodec = "avc1", FLAG_camRes, FLAG_faceModel; +unsigned int FLAG_batch = 1, FLAG_appMode = 2; /******************************************************************************** * Usage @@ -88,8 +88,12 @@ static void Usage() { " --out_file= specify the output file\n" " --out= specify the output file\n" " --model_path= specify the directory containing the TRT models\n" + " --landmarks_126[=(true|false)] set the number of facial landmark points to 126, otherwise default to 68\n" + " --face_model= specify the name of the face model\n" " --wireframe_mesh= specify the path to a proxy wireframe mesh\n" - " --batch= 1 - 8, used for batch inferencing in landmark detector " + " --batch= 1 - 8, used for batch inferencing in landmark detector\n" + " --app_mode[=(0|1|2)] App mode. 0: Face detection, 1: Landmark detection, 2: Face fitting " + "(Default)." " --benchmarks[=] run benchmarks\n"); } @@ -181,11 +185,13 @@ static int ParseMyArgs(int argc, char **argv) { GetFlagArgVal("in", arg, &FLAG_inFile) || GetFlagArgVal("in_file", arg, &FLAG_inFile) || GetFlagArgVal("out", arg, &FLAG_outFile) || GetFlagArgVal("out_file", arg, &FLAG_outFile) || GetFlagArgVal("offline_mode", arg, &FLAG_offlineMode) || + GetFlagArgVal("landmarks_126", arg, &FLAG_isNumLandmarks126) || GetFlagArgVal("capture_outputs", arg, &FLAG_captureOutputs) || GetFlagArgVal("cam_res", arg, &FLAG_camRes) || GetFlagArgVal("codec", arg, &FLAG_captureCodec) || GetFlagArgVal("landmarks", arg, &FLAG_landmarks) || GetFlagArgVal("model_path", arg, &FLAG_modelPath) || GetFlagArgVal("wireframe_mesh", arg, &FLAG_proxyWireframe) || - GetFlagArgVal("temporal", arg, &FLAG_temporal))) { + GetFlagArgVal("face_model", arg, &FLAG_faceModel) || + GetFlagArgVal("app_mode", arg, &FLAG_appMode) || GetFlagArgVal("temporal", arg, &FLAG_temporal))) { continue; } else if (GetFlagArgVal("help", arg, &help)) { Usage(); @@ -251,7 +257,13 @@ std::string getCalendarTime() { class DoApp { public: enum Err { - errNone, + errNone = FaceEngine::Err::errNone, + errGeneral = FaceEngine::Err::errGeneral, + errRun = FaceEngine::Err::errRun, + errInitialization = FaceEngine::Err::errInitialization, + errRead = FaceEngine::Err::errRead, + errEffect = FaceEngine::Err::errEffect, + errParameter = FaceEngine::Err::errParameter, errUnimplemented, errMissing, errVideo, @@ -268,15 +280,15 @@ class DoApp { errSDK, errCuda, errCancel, - errInitFaceEngine + errCamera }; - + Err doAppErr(FaceEngine::Err status) { return (Err)status; } FaceEngine face_ar_engine; DoApp(); ~DoApp(); void stop(); - Err initFaceEngine(const char *modelPath = nullptr); + Err initFaceEngine(const char *modelPath = nullptr, bool isLandmarks126 = false); Err initCamera(const char *camRes = nullptr); Err initOfflineMode(const char *inputFilename = nullptr, const char *outputFilename = nullptr); Err acquireFrame(); @@ -287,7 +299,7 @@ class DoApp { void showFaceFitErrorMessage(); void drawFPS(cv::Mat &img); void DrawBBoxes(const cv::Mat &src, NvAR_Rect *output_bbox); - void DrawLandmarkPoints(const cv::Mat &src, NvAR_Point2f *facial_landmarks); + void DrawLandmarkPoints(const cv::Mat &src, NvAR_Point2f *facial_landmarks, int numLandmarks); void DrawFaceMesh(const cv::Mat &src, NvAR_FaceMesh *face_mesh); void drawKalmanStatus(cv::Mat &img); void drawVideoCaptureStatus(cv::Mat &img); @@ -320,7 +332,7 @@ class DoApp { }; DoApp *gApp = nullptr; -const char DoApp::windowTitle[] = "WINDOW"; +const char DoApp::windowTitle[] = "FaceTrack App"; void DoApp::processKey(int key) { switch (key) { @@ -368,21 +380,23 @@ void DoApp::processKey(int key) { } } -DoApp::Err DoApp::initFaceEngine(const char *modelPath) { +DoApp::Err DoApp::initFaceEngine(const char *modelPath, bool isNumLandmarks126) { Err err = errNone; if (!cap.isOpened()) return errVideo; + int numLandmarkPoints = isNumLandmarks126 ? 126 : 68; + face_ar_engine.setNumLandmarks(numLandmarkPoints); + nvErr = face_ar_engine.createFeatures(modelPath); if (nvErr != FaceEngine::Err::errNone) { if (nvErr == FaceEngine::Err::errInitialization && face_ar_engine.appMode == FaceEngine::mode::faceMeshGeneration) { showFaceFitErrorMessage(); + printf("WARNING: face fitting has failed, trying to initialize Landmark Detection\n"); face_ar_engine.destroyFeatures(); face_ar_engine.setAppMode(FaceEngine::mode::landmarkDetection); nvErr = face_ar_engine.createFeatures(modelPath); } - if (nvErr != FaceEngine::Err::errNone) - err = errInitFaceEngine; } #ifdef DEBUG @@ -396,7 +410,7 @@ DoApp::Err DoApp::initFaceEngine(const char *modelPath) { frameIndex = 0; - return err; + return doAppErr(nvErr); } void DoApp::stop() { @@ -428,29 +442,29 @@ void DoApp::showFaceFitErrorMessage() { } void DoApp::DrawBBoxes(const cv::Mat &src, NvAR_Rect *output_bbox) { - cv::Mat frame; + cv::Mat frm; if (FLAG_offlineMode) - frame = src.clone(); + frm = src.clone(); else - frame = src; + frm = src; if (output_bbox) - cv::rectangle(frame, cv::Point((int)output_bbox->x, (int)output_bbox->y), - cv::Point((int)output_bbox->x + output_bbox->width, (int)output_bbox->y + output_bbox->height), + cv::rectangle(frm, cv::Point(lround(output_bbox->x), lround(output_bbox->y)), + cv::Point(lround(output_bbox->x + output_bbox->width), lround(output_bbox->y + output_bbox->height)), cv::Scalar(255, 0, 0), 2); - if (FLAG_offlineMode) faceDetectOutputVideo.write(frame); + if (FLAG_offlineMode) faceDetectOutputVideo.write(frm); } -void DoApp::writeVideoAndEstResults(const cv::Mat &frame, NvAR_BBoxes output_bboxes, NvAR_Point2f *landmarks) { +void DoApp::writeVideoAndEstResults(const cv::Mat &frm, NvAR_BBoxes output_bboxes, NvAR_Point2f *landmarks) { if (captureVideo) { if (!capturedVideo.isOpened()) { const std::string currentCalendarTime = getCalendarTime(); const std::string capturedOutputFileName = currentCalendarTime + ".mp4"; getFPS(); if (frameTime) { - float fps = 1. / frameTime; + float fps = (float)(1.0 / frameTime); capturedVideo.open(capturedOutputFileName, StringToFourcc(FLAG_captureCodec), fps, - cv::Size(frame.cols, frame.rows)); + cv::Size(frm.cols, frm.rows)); if (!capturedVideo.isOpened()) { std::cout << "Error: Could not open video: \"" << capturedOutputFileName << "\"\n"; return; @@ -473,7 +487,7 @@ void DoApp::writeVideoAndEstResults(const cv::Mat &frame, NvAR_BBoxes output_bbo << "// kNumFaces, (bbox_x, bbox_y, bbox_w, bbox_h){ kNumFaces}, kNumLMs, [lm_x, lm_y]{kNumLMs}\n"; } // Write each frame to the Video - capturedVideo << frame; + capturedVideo << frm; writeEstResults(faceEngineVideoOutputFile, output_bboxes, landmarks); } else { if (capturedVideo.isOpened()) { @@ -517,11 +531,12 @@ void DoApp::writeEstResults(std::ofstream &outputFile, NvAR_BBoxes output_bboxes outputFile << "0,"; } if (landmarkDetectOn && output_bboxes.num_boxes) { + int numLandmarks = face_ar_engine.getNumLandmarks(); // Append number of landmarks - outputFile << FaceEngine::NUM_LANDMARKS << ","; - // Append NUM_LANDMARKS * 2 points + outputFile << numLandmarks << ","; + // Append 2 * number of landmarks values NvAR_Point2f *pt, *endPt; - for (endPt = (pt = (NvAR_Point2f *)landmarks) + FaceEngine::NUM_LANDMARKS; pt < endPt; ++pt) + for (endPt = (pt = (NvAR_Point2f *)landmarks) + numLandmarks; pt < endPt; ++pt) outputFile << pt->x << "," << pt->y << ","; } else { outputFile << "0,"; @@ -530,11 +545,11 @@ void DoApp::writeEstResults(std::ofstream &outputFile, NvAR_BBoxes output_bboxes outputFile << "\n"; } -void DoApp::writeFrameAndEstResults(const cv::Mat &frame, NvAR_BBoxes output_bboxes, NvAR_Point2f *landmarks) { +void DoApp::writeFrameAndEstResults(const cv::Mat &frm, NvAR_BBoxes output_bboxes, NvAR_Point2f *landmarks) { if (captureFrame) { const std::string currentCalendarTime = getCalendarTime(); const std::string capturedFrame = currentCalendarTime + ".png"; - cv::imwrite(capturedFrame, frame); + cv::imwrite(capturedFrame, frm); if (FLAG_verbose) { std::cout << "Captured the frame" << std::endl; } @@ -555,19 +570,19 @@ void DoApp::writeFrameAndEstResults(const cv::Mat &frame, NvAR_BBoxes output_bbo } } -void DoApp::DrawLandmarkPoints(const cv::Mat &src, NvAR_Point2f *facial_landmarks) { - cv::Mat frame; +void DoApp::DrawLandmarkPoints(const cv::Mat &src, NvAR_Point2f *facial_landmarks, int numLandmarks) { + cv::Mat frm; if (FLAG_offlineMode) - frame = src.clone(); + frm = src.clone(); else - frame = src; + frm = src; NvAR_Point2f *pt, *endPt; - for (endPt = (pt = (NvAR_Point2f *)facial_landmarks) + FaceEngine::NUM_LANDMARKS; pt < endPt; ++pt) - cv::circle(frame, cv::Point(lround(pt->x), lround(pt->y)), 1, cv::Scalar(0, 0, 255), -1); + for (endPt = (pt = (NvAR_Point2f *)facial_landmarks) + numLandmarks; pt < endPt; ++pt) + cv::circle(frm, cv::Point(lround(pt->x), lround(pt->y)), 1, cv::Scalar(0, 0, 255), -1); NvAR_Quaternion *pose = face_ar_engine.getPose(); if (pose) - face_ar_engine.DrawPose(frame, pose); - if (FLAG_offlineMode) landMarkOutputVideo.write(frame); + face_ar_engine.DrawPose(frm, pose); + if (FLAG_offlineMode) landMarkOutputVideo.write(frm); } void DoApp::DrawFaceMesh(const cv::Mat &src, NvAR_FaceMesh *face_mesh) { @@ -608,12 +623,14 @@ DoApp::Err DoApp::acquireFrame() { // frames we try to read are empty. So we try to re-initialize the camera with the same resolution settings. If the // resolution has changed, you will need to destroy and create the features again with the new camera resolution (not // done here) as well as reallocate memory accordingly with FaceEngine::initFeatureIOParams() - cap >> frame; // get a new frame from camera + cap >> frame; // get a new frame from camera into the class variable frame. if (frame.empty()) { // if in Offline mode, this means end of video,so we return if (FLAG_offlineMode) return errVideo; // try Init one more time if reading frames from camera - initCamera(FLAG_camRes.c_str()); + err = initCamera(FLAG_camRes.c_str()); + if (err != errNone) + return err; cap >> frame; if (frame.empty()) return errVideo; } @@ -653,30 +670,30 @@ DoApp::Err DoApp::acquireFaceBox() { DoApp::Err DoApp::acquireFaceBoxAndLandmarks() { Err err = errNone; - + int numLandmarks = face_ar_engine.getNumLandmarks(); NvAR_Rect output_bbox; - NvAR_Point2f facial_landmarks[FaceEngine::NUM_LANDMARKS]; + std::vector facial_landmarks(numLandmarks); // get landmarks in original image resolution coordinate space - unsigned n = face_ar_engine.acquireFaceBoxAndLandmarks(frame, facial_landmarks, output_bbox, 0); + unsigned n = face_ar_engine.acquireFaceBoxAndLandmarks(frame, facial_landmarks.data(), output_bbox, 0); if (n && FLAG_verbose && face_ar_engine.appMode != FaceEngine::mode::faceDetection) { printf("Landmarks: [\n"); - NvAR_Point2f *pt, *endPt; - for (endPt = (pt = (NvAR_Point2f *)facial_landmarks) + FaceEngine::NUM_LANDMARKS; pt < endPt; ++pt) - printf("%7.1f%7.1f\n", pt->x, pt->y); + for (const auto &pt : facial_landmarks) { + printf("%7.1f%7.1f\n", pt.x, pt.y); + } printf("]\n"); } if (FLAG_captureOutputs) { - writeFrameAndEstResults(frame, face_ar_engine.output_bboxes, facial_landmarks); - writeVideoAndEstResults(frame, face_ar_engine.output_bboxes, facial_landmarks); + writeFrameAndEstResults(frame, face_ar_engine.output_bboxes, facial_landmarks.data()); + writeVideoAndEstResults(frame, face_ar_engine.output_bboxes, facial_landmarks.data()); } if (0 == n) return errNoFace; #ifdef VISUALIZE if (drawVisualization) { - DrawLandmarkPoints(frame, facial_landmarks); + DrawLandmarkPoints(frame, facial_landmarks.data(), numLandmarks); if (FLAG_offlineMode) { DrawBBoxes(frame, &output_bbox); } @@ -707,24 +724,25 @@ DoApp::Err DoApp::initCamera(const char *camRes) { if (inputWidth) cap.set(CV_CAP_PROP_FRAME_WIDTH, inputWidth); if (inputHeight) cap.set(CV_CAP_PROP_FRAME_HEIGHT, inputHeight); - inputWidth = cap.get(CV_CAP_PROP_FRAME_WIDTH); - inputHeight = cap.get(CV_CAP_PROP_FRAME_HEIGHT); + inputWidth = (int)cap.get(CV_CAP_PROP_FRAME_WIDTH); + inputHeight = (int)cap.get(CV_CAP_PROP_FRAME_HEIGHT); face_ar_engine.setInputImageWidth(inputWidth); face_ar_engine.setInputImageHeight(inputHeight); } } else - return errVideo; + return errCamera; return errNone; } DoApp::Err DoApp::initOfflineMode(const char *inputFilename, const char *outputFilename) { if (cap.open(inputFilename)) { - inputWidth = cap.get(CV_CAP_PROP_FRAME_WIDTH); - inputHeight = cap.get(CV_CAP_PROP_FRAME_HEIGHT); + inputWidth = (int)cap.get(CV_CAP_PROP_FRAME_WIDTH); + inputHeight = (int)cap.get(CV_CAP_PROP_FRAME_HEIGHT); face_ar_engine.setInputImageWidth(inputWidth); face_ar_engine.setInputImageHeight(inputHeight); } else { - return Err::errNotFound; + printf("ERROR: Unable to open the input video file \"%s\" \n", inputFilename); + return Err::errVideo; } std::string fdOutputVideoName, fldOutputVideoName, ffOutputVideoName; @@ -740,14 +758,20 @@ DoApp::Err DoApp::initOfflineMode(const char *inputFilename, const char *outputF ffOutputVideoName = outputFilePrefix + "_faceModel.mp4"; if (!faceDetectOutputVideo.open(fdOutputVideoName, StringToFourcc(FLAG_captureCodec), cap.get(CV_CAP_PROP_FPS), - cv::Size(inputWidth, inputHeight))) - return Err::errSDK; + cv::Size(inputWidth, inputHeight))) { + printf("ERROR: Unable to open the output video file \"%s\" \n", fdOutputVideoName.c_str()); + return Err::errGeneral; + } if (!landMarkOutputVideo.open(fldOutputVideoName, StringToFourcc(FLAG_captureCodec), cap.get(CV_CAP_PROP_FPS), - cv::Size(inputWidth, inputHeight))) - return Err::errSDK; + cv::Size(inputWidth, inputHeight))) { + printf("ERROR: Unable to open the output video file \"%s\" \n", fldOutputVideoName.c_str()); + return Err::errGeneral; + } if (!faceFittingOutputVideo.open(ffOutputVideoName, StringToFourcc(FLAG_captureCodec), cap.get(CV_CAP_PROP_FPS), - cv::Size(inputWidth, inputHeight))) - return Err::errSDK; + cv::Size(inputWidth, inputHeight))) { + printf("ERROR: Unable to open the output video file \"%s\" \n", ffOutputVideoName.c_str()); + return Err::errGeneral; + } return Err::errNone; } @@ -766,7 +790,7 @@ DoApp::Err DoApp::fitFaceModel() { if (drawVisualization) { DrawFaceMesh(frame, face_ar_engine.getFaceMesh()); if (FLAG_offlineMode) { - DrawLandmarkPoints(frame, face_ar_engine.getLandmarks()); + DrawLandmarkPoints(frame, face_ar_engine.getLandmarks(), face_ar_engine.getNumLandmarks()); DrawBBoxes(frame, face_ar_engine.getLargestBox()); } } @@ -835,11 +859,20 @@ void DoApp::drawVideoCaptureStatus(cv::Mat &img) { DoApp::Err DoApp::run() { DoApp::Err doErr = errNone; - face_ar_engine.initFeatureIOParams(); + FaceEngine::Err err = face_ar_engine.initFeatureIOParams(); + if (err != FaceEngine::Err::errNone ) { + return doAppErr(err); + } while (1) { doErr = acquireFrame(); - if (doErr != DoApp::errNone) return doErr; + if (frame.empty() && FLAG_offlineMode) { + // We have reached the end of the video + // so return without any error. + return DoApp::errNone; + } else if (doErr != DoApp::errNone) { + return doErr; + } if (face_ar_engine.appMode == FaceEngine::mode::faceDetection) { doErr = acquireFaceBox(); } else if (face_ar_engine.appMode == FaceEngine::mode::landmarkDetection) { @@ -898,6 +931,12 @@ const char *DoApp::errorStringFromCode(DoApp::Err code) { }; static const LUTEntry lut[] = { {errNone, "no error"}, + {errGeneral, "an error has occured"}, + {errRun, "an error has occured while the feature is running"}, + {errInitialization, "Initializing Face Engine failed"}, + {errRead, "an error has occured while reading a file"}, + {errEffect, "an error has occured while creating a feature"}, + {errParameter, "an error has occured while setting a parameter for a feature"}, {errUnimplemented, "the feature is unimplemented"}, {errMissing, "missing input parameter"}, {errVideo, "no video source has been found"}, @@ -914,7 +953,7 @@ const char *DoApp::errorStringFromCode(DoApp::Err code) { {errSDK, "an SDK error has occurred"}, {errCuda, "a CUDA error has occurred"}, {errCancel, "the user cancelled"}, - {errInitFaceEngine, "an error occurred while initializing the Face Engine"}, + {errCamera, "unable to connect to the camera"}, }; for (const LUTEntry *p = lut; p < &lut[sizeof(lut) / sizeof(lut[0])]; ++p) if (p->code == code) return p->str; @@ -930,19 +969,23 @@ const char *DoApp::errorStringFromCode(DoApp::Err code) { int main(int argc, char **argv) { DoApp app; DoApp::Err doErr; - NvCV_Status nvErr; // Parse the arguments if (0 != ParseMyArgs(argc, argv)) return -100; + app.face_ar_engine.setAppMode(FaceEngine::mode(FLAG_appMode)); + if (FLAG_verbose) printf("Enable temporal optimizations in detecting face and landmarks = %d\n", FLAG_temporal); app.face_ar_engine.setFaceStabilization(FLAG_temporal); doErr = DoApp::errFaceModelInit; if (FLAG_modelPath.empty()) { - printf("WARNING: Model path not specified. Please set --model_path=/path/to/trt/and/face/models\n" - "SDK will attempt to load the models from NVAR_MODEL_DIR environment variable"); + printf("WARNING: Model path not specified. Please set --model_path=/path/to/trt/and/face/models, " + "SDK will attempt to load the models from NVAR_MODEL_DIR environment variable, " + "please restart your application after the SDK Installation. \n"); } + if (!FLAG_faceModel.empty()) + app.face_ar_engine.setFaceModel(FLAG_faceModel.c_str()); if (FLAG_offlineMode) { if (FLAG_inFile.empty()) { @@ -950,21 +993,23 @@ int main(int argc, char **argv) { printf("ERROR: %s, please specify input file using --in_file or --in \n", app.errorStringFromCode(doErr)); goto bail; } - app.initOfflineMode(FLAG_inFile.c_str(), FLAG_outFile.c_str()); + doErr = app.initOfflineMode(FLAG_inFile.c_str(), FLAG_outFile.c_str()); } else { - app.initCamera(FLAG_camRes.c_str()); - } - doErr = app.initFaceEngine(FLAG_modelPath.c_str()); - if (DoApp::errNone != doErr) { - printf("ERROR: %s\n", app.errorStringFromCode(doErr)); - goto bail; + doErr = app.initCamera(FLAG_camRes.c_str()); } + BAIL_IF_ERR(doErr); + + doErr = app.initFaceEngine(FLAG_modelPath.c_str(), FLAG_isNumLandmarks126); + BAIL_IF_ERR(doErr); if (!FLAG_proxyWireframe.empty()) app.setProxyWireframe(FLAG_proxyWireframe.c_str()); doErr = app.run(); + BAIL_IF_ERR(doErr); bail: + if(doErr) + printf("ERROR: %s\n", app.errorStringFromCode(doErr)); app.stop(); return (int)doErr; } diff --git a/samples/FaceTrack/FaceTrack.exe b/samples/FaceTrack/FaceTrack.exe index e149aed..1714066 100644 Binary files a/samples/FaceTrack/FaceTrack.exe and b/samples/FaceTrack/FaceTrack.exe differ diff --git a/samples/FaceTrack/Readme.txt b/samples/FaceTrack/Readme.txt index 4dc0475..e1cc66d 100644 --- a/samples/FaceTrack/Readme.txt +++ b/samples/FaceTrack/Readme.txt @@ -47,4 +47,5 @@ Either the forward slash (/) or back slash (\) can be used as a separator betwee The ConvertSurreyFaceModel.exe file is distributed in the https://github.com/nvidia/BROADCAST-AR-SDK repo. 3) The sample application provided with NVIDIA AR SDK requires that the model file be named face_model0.nvf. -Place the face_model0.nvf file in the /bin/models folder. +Place the face_model0.nvf file in the model folder. By default the models folder is your_sdk_install_path/models +where all models (including *.trtpkg files) are installed, for example: C:\Program Files\NVIDIA Corporation\NVIDIA AR SDK\models. diff --git a/samples/utils/FeatureVertexName.cpp b/samples/utils/FeatureVertexName.cpp new file mode 100644 index 0000000..c811647 --- /dev/null +++ b/samples/utils/FeatureVertexName.cpp @@ -0,0 +1,187 @@ +/*############################################################################### +# +# Copyright(c) 2019 NVIDIA CORPORATION.All Rights Reserved. +# +# NVIDIA CORPORATION and its licensors retain all intellectual property +# and proprietary rights in and to this software, related documentation +# and any modifications thereto.Any use, reproduction, disclosure or +# distribution of this software and related documentation without an express +# license agreement from NVIDIA CORPORATION is strictly prohibited. +# +###############################################################################*/ + +#include "FeatureVertexName.h" +#include +#include + + +// TODO: We should read this from a file +const LandmarkEOSMap LandmarkMapEOS[] = { + { 33, "chin bottom" }, + { 225, "right eyebrow outer-corner" }, + { 229, "right eyebrow between middle and outer corner" }, + { 233, "right eyebrow middle, vertical middle" }, + { 2086, "right eyebrow between middle and inner corner" }, + { 157, "right eyebrow inner-corner" }, + { 590, "left eyebrow inner-corner" }, + { 2091, "left eyebrow between inner corner and middle" }, + { 666, "left eyebrow middle" }, + { 662, "left eyebrow between middle and outer corner" }, + { 658, "left eyebrow outer-corner" }, + { 2842, "bridge of the nose (parallel to upper eye lids)" }, + { 379, "middle of the nose, a bit below the lower eye lids" }, + { 272, "above nose-tip (1cm or so)" }, + { 114, "nose-tip" }, + { 100, "right nostril, below nose, nose-lip junction" }, + { 2794, "nose-lip junction" }, + { 270, "nose-lip junction" }, + { 2797, "nose-lip junction" }, + { 537, "left nostril, below nose, nose-lip junction" }, + { 177, "right eye outer-corner" }, + { 172, "right eye pupil top right (from subject's perspective)" }, + { 191, "right eye pupil top left" }, + { 181, "right eye inner-corner" }, + { 173, "right eye pupil bottom left" }, + { 174, "right eye pupil bottom right" }, + { 614, "left eye inner-corner" }, + { 624, "left eye pupil top right" }, + { 605, "left eye pupil top left" }, + { 610, "left eye outer-corner" }, + { 607, "left eye pupil bottom left" }, + { 606, "left eye pupil bottom right" }, + { 398, "right mouth corner" }, + { 315, "upper lip right top outer" }, + { 413, "upper lip middle top right" }, + { 329, "upper lip middle top" }, + { 825, "upper lip middle top left" }, + { 736, "upper lip left top outer" }, + { 812, "left mouth corner" }, + { 841, "lower lip left bottom outer" }, + { 693, "lower lip middle bottom left" }, + { 411, "lower lip middle bottom" }, + { 264, "lower lip middle bottom right" }, + { 431, "lower lip right bottom outer" }, + { 416, "upper lip right bottom outer" }, + { 423, "upper lip middle bottom" }, + { 828, "upper lip left bottom outer" }, + { 817, "lower lip left top outer" }, + { 442, "lower lip middle top" }, + { 404, "lower lip right top outer" }, + { 0xFFFF, nullptr } +}; + +static const LandmarksMap LandmarkMap[] = { + { 0, 0, "right contour point 1" }, + { 1, 2, "right contour point 2" }, + { 2, 4, "right contour point 3" }, + { 3, 6, "right contour point 4" }, + { 4, 8, "right contour point 5" }, + { 5, 10, "right contour point 6" }, + { 6, 12, "right contour point 7" }, + { 7, 14, "right contour point 8" }, + { 8, 16, "chin bottom" }, + { 9, 18, "left contour point 1" }, + { 10, 20, "left contour point 2" }, + { 11, 22, "left contour point 3" }, + { 12, 24, "left contour point 4" }, + { 13, 26, "left contour point 5" }, + { 14, 28, "left contour point 6" }, + { 15, 30, "left contour point 7" }, + { 16, 32, "left contour point 8" }, + { 17, 33, "right eyebrow outer-corner" }, + { 18, 34, "right eyebrow between middle and outer corner" }, + { 19, 35, "right eyebrow middle, vertical middle" }, + { 20, 36, "right eyebrow between middle and inner corner" }, + { 21, 37, "right eyebrow inner-corner" }, + { 22, 42, "left eyebrow inner-corner" }, + { 23, 43, "left eyebrow between inner corner and middle" }, + { 24, 44, "left eyebrow middle" }, + { 25, 45, "left eyebrow between middle and outer corner" }, + { 26, 46, "left eyebrow outer-corner" }, + { 27, 51, "bridge of the nose (parallel to upper eye lids)" }, + { 28, 52, "middle of the nose, a bit below the lower eye lids" }, + { 29, 53, "above nose-tip (1cm or so)" }, + { 30, 54, "nose-tip" }, + { 31, 57, "right nostril, below nose, nose-lip junction" }, + { 32, 58, "nose-lip junction" }, + { 33, 59, "nose-lip junction" }, + { 34, 60, "nose-lip junction" }, + { 35, 61, "left nostril, below nose, nose-lip junction" }, + { 36, 64, "right eye outer-corner" }, + { 37, 65, "right eye pupil top right (from subject's perspective)" }, + { 38, 67, "right eye pupil top left" }, + { 39, 68, "right eye inner-corner" }, + { 40, 69, "right eye pupil bottom left" }, + { 41, 71, "right eye pupil bottom right" }, + { 42, 81, "left eye inner-corner" }, + { 43, 82, "left eye pupil top right" }, + { 44, 84, "left eye pupil top left" }, + { 45, 85, "left eye outer-corner" }, + { 46, 86, "left eye pupil bottom left" }, + { 47, 88, "left eye pupil bottom right" }, + { 48, 98, "right mouth corner" }, + { 49, 99, "upper lip right top outer" }, + { 50, 100, "upper lip middle top right" }, + { 51, 101, "upper lip middle top" }, + { 52, 102, "upper lip middle top left" }, + { 53, 103, "upper lip left top outer" }, + { 54, 104, "left mouth corner" }, + { 55, 105, "lower lip left bottom outer" }, + { 56, 106, "lower lip middle bottom left" }, + { 57, 107, "lower lip middle bottom" }, + { 58, 108, "lower lip middle bottom right" }, + { 59, 109, "lower lip right bottom outer" }, + { 60, 110, "right inner mouth corner "}, + { 61, 111, "upper lip right bottom outer" }, + { 62, 112, "upper lip middle bottom" }, + { 63, 113, "upper lip left bottom outer" }, + { 64, 114, "left inner mouth corner"}, + { 65, 115, "lower lip left top outer" }, + { 66, 116, "lower lip middle top" }, + { 67, 117, "lower lip right top outer" }, + { 0xFFFF, 0xFFFF, nullptr } +}; + + +unsigned short FindEOSLandmarkIndexFromName(const char* name) +{ + if (!name) + return 0xFFFF; + switch (name[0]) { + case '#': // 1-based index ... + return (unsigned short)(strtol(name + 1, nullptr, 10) - 1); // ... gets converted into a 0-based index + case '@': // 0-based index + return (unsigned short)strtol(name + 1, nullptr, 10); + default: + break; + } + const LandmarkEOSMap* lmList = LandmarkMapEOS; + for (; lmList->name != nullptr; ++lmList) + if (!strcmp(name, lmList->name)) + break; + return lmList->index; +} + +unsigned short FindLandmarkIndexFromName(const unsigned int numLandmarks, const char* name) { + if (!name) + return 0xFFFF; + switch (name[0]) { + case '#': // 1-based index ... + return (unsigned short)(strtol(name + 1, nullptr, 10) - 1); // ... gets converted into a 0-based index + case '@': // 0-based index + return (unsigned short)strtol(name + 1, nullptr, 10); + default: + break; + } + const LandmarksMap* lmList = LandmarkMap; + for (; lmList->name != nullptr; ++lmList) + if (!strcmp(name, lmList->name)) + break; + if (numLandmarks == 68) { + return lmList->index_68; + } else if (numLandmarks == 126) { + return lmList->index_126; + } else { + return 0xFFFF; + } +} diff --git a/samples/utils/FeatureVertexName.h b/samples/utils/FeatureVertexName.h new file mode 100644 index 0000000..d61acc3 --- /dev/null +++ b/samples/utils/FeatureVertexName.h @@ -0,0 +1,23 @@ +/*############################################################################### +# +# Copyright(c) 2019 NVIDIA CORPORATION.All Rights Reserved. +# +# NVIDIA CORPORATION and its licensors retain all intellectual property +# and proprietary rights in and to this software, related documentation +# and any modifications thereto.Any use, reproduction, disclosure or +# distribution of this software and related documentation without an express +# license agreement from NVIDIA CORPORATION is strictly prohibited. +# +###############################################################################*/ + +#ifndef __FEATURE_VERTEX_NAME__ +#define __FEATURE_VERTEX_NAME__ + + +struct LandmarkEOSMap { unsigned short index; const char *name; }; +struct LandmarksMap { unsigned short index_68; unsigned short index_126; const char* name; }; + +unsigned short FindEOSLandmarkIndexFromName(const char *name); +unsigned short FindLandmarkIndexFromName(const unsigned int numLandmarks, const char* name); + +#endif /* __FEATURE_VERTEX_NAME__ */ diff --git a/samples/utils/nvCVOpenCV.h b/samples/utils/nvCVOpenCV.h index 188012f..cdfbe1a 100644 --- a/samples/utils/nvCVOpenCV.h +++ b/samples/utils/nvCVOpenCV.h @@ -21,6 +21,9 @@ # ###############################################################################*/ +#ifndef __NVCVOPENCV_H__ +#define __NVCVOPENCV_H__ + #include "nvCVImage.h" #include "opencv2/opencv.hpp" @@ -73,3 +76,5 @@ inline void NVWrapperForCVMat(const cv::Mat *cvIm, NvCVImage *nvcvIm) { nvcvIm->reserved[0] = 0; nvcvIm->reserved[1] = 0; } + +#endif // __NVCVOPENCV_H__ \ No newline at end of file diff --git a/tools/ConvertSurreyFaceModel.exe b/tools/ConvertSurreyFaceModel.exe index 0ae6758..205b48a 100644 Binary files a/tools/ConvertSurreyFaceModel.exe and b/tools/ConvertSurreyFaceModel.exe differ diff --git a/version.h b/version.h index 2a24909..75f9976 100644 --- a/version.h +++ b/version.h @@ -22,12 +22,12 @@ ###############################################################################*/ #define NVIDIA_AR_SDK_VERSION_MAJOR 0 -#define NVIDIA_AR_SDK_VERSION_MINOR 5 -#define NVIDIA_AR_SDK_VERSION_RELEASE 0 +#define NVIDIA_AR_SDK_VERSION_MINOR 6 +#define NVIDIA_AR_SDK_VERSION_RELEASE 1 -#define NVIDIA_AR_SDK_VERSION 0,5,0,0 -#define NVIDIA_AR_SDK_VERSION_MAJOR_MINOR 0,5 -#define NVIDIA_AR_SDK_VERSION_STRING "0.5.0.0" -#define NVIDIA_AR_SDK_VERSION_STRING_SHORT "0.5.0" -#define NVIDIA_AR_SDK_VERSION_STRING_MAJOR_MINOR "0.5" +#define NVIDIA_AR_SDK_VERSION 0,6,1,0 +#define NVIDIA_AR_SDK_VERSION_MAJOR_MINOR 0,6 +#define NVIDIA_AR_SDK_VERSION_STRING "0.6.1.0" +#define NVIDIA_AR_SDK_VERSION_STRING_SHORT "0.6.1" +#define NVIDIA_AR_SDK_VERSION_STRING_MAJOR_MINOR "0.6"