diff --git a/NVIDIA AR SDK Programming Guide.pdf b/NVIDIA AR SDK Programming Guide.pdf deleted file mode 100644 index f155ff3..0000000 Binary files a/NVIDIA AR SDK Programming Guide.pdf and /dev/null differ diff --git a/README.MD b/README.MD index 16291fe..98b6752 100644 --- a/README.MD +++ b/README.MD @@ -1,18 +1,64 @@ -**Nvidia AR SDK: API Source Code and Sample Applications** - -NVIDIA AR SDK enables real-time modeling and tracking of human faces from video. The SDK is powered by NVIDIA graphics processing units (GPUs) with Tensor Cores, and as a result, the algorithm throughput is greatly accelerated, and latency is reduced. - -NVIDIA AR SDK has the following features: - -- **Face detection and tracking**, which detects, localizes, and tracks human faces in images or videos by using bounding boxes. -- **Facial landmark detection and tracking**, which predicts and tracks the pixel locations of human facial landmark points and head poses in images or videos.The detected facial landmarks follow the _Multi-PIE 68 point mark-ups_ information in [facial point annotations](https://ibug.doc.ic.ac.uk/resources/facial-point-annotations/). -- **Face 3D mesh and tracking**, which reconstructs and tracks a 3D human face and its head pose from the provided facial landmarks **.** - -NVIDIA AR SDK provides a sample application that demonstrates the features listed above in real time by using a webcam or offline videos. - -NVIDIA AR SDK is distributed in the following parts: - -- This open source repository that includes the [SDK API and proxy linking source code](https://github.com/NVIDIA/BROADCAST-AR-SDK/tree/master/nvar), and [sample applications and their dependency libraries](https://github.com/NVIDIA/BROADCAST-AR-SDK/tree/master/samples). -- An installer hosted on [RTX broadcast engine developer page](https://developer.nvidia.com/rtx-broadcast-engine) that installs the SDK DLLs, the models, and the SDK dependency libraries. - -Please refer to [SDK programming guide](https://github.com/NVIDIA/BROADCAST-AR-SDK/blob/master/NVIDIA%20AR%20SDK%20Programming%20Guide.pdf) for configuring and integrating the SDK, compiling and running the sample applications. +# README +## NVIDIA AR SDK: API Source Code and Sample Applications + +NVIDIA AR SDK enables real-time modeling and tracking of human faces from video. The SDK is powered by NVIDIA graphics processing units (GPUs) with Tensor Cores, and as a result, the algorithm throughput is greatly accelerated, and latency is reduced. + +NVIDIA AR SDK has the following features: + +- **Face detection and tracking**, which detects, localizes, and tracks human faces in images or videos by using bounding boxes. +- **Facial landmark detection and tracking**, which predicts and tracks the pixel locations of human facial landmark points and head poses in images or videos. It can predict 68 and 126 landmark points. The 68 detected facial landmarks follow the _Multi-PIE 68 point mark-ups_ information in [facial point annotations](https://ibug.doc.ic.ac.uk/resources/facial-point-annotations/). The 126 landmark points detector can predict more points on the cheeks, the eyes, and on laugh lines. +- **Face 3D mesh and tracking**, which reconstructs and tracks a 3D human face and its head pose from the provided facial landmarks. + +

+Face detection and tracking +Facial landmark detection and tracking - 68 pts +

+Facial landmark detection and tracking - 126 pts +Face 3D mesh and tracking +

+ +NVIDIA AR SDK provides a sample application that demonstrates the features listed above in real time by using a webcam or offline videos. + +NVIDIA AR SDK is distributed in the following parts: + +- This open source repository that includes the [SDK API and proxy linking source code](https://github.com/NVIDIA/BROADCAST-AR-SDK/tree/master/nvar), and [sample applications and their dependency libraries](https://github.com/NVIDIA/BROADCAST-AR-SDK/tree/master/samples). +- An installer hosted on [RTX broadcast engine developer page](https://developer.nvidia.com/rtx-broadcast-engine) that installs the SDK DLLs, the models, and the SDK dependency libraries. + +Please refer to [SDK programming guide](https://github.com/NVIDIA/BROADCAST-AR-SDK/blob/master/NVIDIA%20AR%20SDK%20Programming%20Guide.pdf) for configuring and integrating the SDK, compiling and running the sample applications. + +## System requirements +The SDK is supported on NVIDIA GPUs that are based on the NVIDIA® Turing™ architecture. Although the SDK can run on Turing™ GPUs without Tensor Cores, it is optimized for much higher performance on GPUs with Tensor Cores. + +* Windows OS supported: 64-bit Windows 10 +* Microsoft Visual Studio: 2015 (MSVC14.0) or later +* CMake: v3.12 or later +* NVIDIA Graphics Driver for Windows: 455.57 or later +* NVIDIA CUDA Toolkit: 11.1 or later +* NVIDIA TensorRT: 7.2.0 or later + +## NVIDIA Branding Guidelines +If you integrate an NVIDIA Broadcast Engine SDK within your product, please follow the required branding guidelines that are available [here]( +https://nvidia.frontify.com/d/uAobRitG8H8B) + +## Compiling the sample app + +### Steps + +The open source repository includes the source code to build the sample application, and a proxy file nvARProxy.cpp to enable compilation without explicitly linking against the SDK DLL. + +**Note: To download the models and runtime dependencies required by the features, you need to run the [SDK Installer](https://developer.nvidia.com/rtx-broadcast-engine).** + +1. In the root folder of the downloaded source code, start the CMake GUI and specify the source folder and a build folder for the binary files. +* For the source folder, ensure that the path ends in OSS. +* For the build folder, ensure that the path ends in OSS/build. +2. Use CMake to configure and generate the Visual Studio solution file. +* Click Configure. +* When prompted to confirm that CMake can create the build folder, click OK. +* Select Visual Studio for the generator and x64 for the platform. +* To complete configuring the Visual Studio solution file, click Finish. +* To generate the Visual Studio Solution file, click Generate. +* Verify that the build folder contains the NvAR_SDK.sln file. +3. Use Visual Studio to generate the FaceTrack.exe file from the NvAR_SDK.sln file. +* In CMake, to open Visual Studio, click Open Project. +* In Visual Studio, select Build > Build Solution. + diff --git a/docs/NVIDIA AR SDK Programming Guide.pdf b/docs/NVIDIA AR SDK Programming Guide.pdf new file mode 100644 index 0000000..2a07b52 Binary files /dev/null and b/docs/NVIDIA AR SDK Programming Guide.pdf differ diff --git a/nvar/include/nvAR.h b/nvar/include/nvAR.h index 34a3f3f..f7413cf 100644 --- a/nvar/include/nvAR.h +++ b/nvar/include/nvAR.h @@ -38,6 +38,13 @@ typedef struct CUstream_st *CUstream; typedef struct nvAR_Feature nvAR_Feature; typedef struct nvAR_Feature *NvAR_FeatureHandle; +//! Get the SDK version +//! \param[in,out] version Pointer to an unsigned int set to +//! (major << 24) | (minor << 16) | (build << 8) | 0 +//! \return NVCV_SUCCESS if the version was set +//! \return NVCV_ERR_PARAMETER if version was NULL +NvCV_Status NvAR_API NvAR_GetVersion(unsigned int *version); + //! Create a new feature instantiation. //! \param[in] InFeatureID The selector code for the desired feature. //! \param[out] handle Handle to the feature instance. @@ -90,7 +97,7 @@ NvCV_Status NvAR_API NvAR_GetObject(NvAR_FeatureHandle handle, const char *name, unsigned long typeSize); NvCV_Status NvAR_API NvAR_GetString(NvAR_FeatureHandle handle, const char *name, const char **str); NvCV_Status NvAR_API NvAR_GetCudaStream(NvAR_FeatureHandle handle, const char *name, const CUstream *stream); -NvCV_Status NvAR_API NvAR_GetF32Array(NvAR_FeatureHandle handle, const char *name, const float **vals, int */*count*/); +NvCV_Status NvAR_API NvAR_GetF32Array(NvAR_FeatureHandle handle, const char *name, const float **vals, int* /*count*/); #ifdef __cplusplus } diff --git a/nvar/include/nvAR_defs.h b/nvar/include/nvAR_defs.h index 85951a0..04ea3d5 100644 --- a/nvar/include/nvAR_defs.h +++ b/nvar/include/nvAR_defs.h @@ -136,12 +136,18 @@ NvAR_Parameter_Output(Pose) - OPTIONAL NvAR_Parameter_Output(LandmarksConfidence) - OPTIONAL *******NvAR_Feature_Face3DReconstruction******* -Config +Config: NvAR_Parameter_Config(FeatureDescription) NvAR_Parameter_Config(ModelDir) NvAR_Parameter_Config(Landmarks_Size) NvAR_Parameter_Config(CUDAStream) -OPTIONAL NvAR_Parameter_Config(Temporal) - OPTIONAL +NvAR_Parameter_Config(ModelName) - OPTIONAL +NvAR_Parameter_Config(GPU) - OPTIONAL +NvAR_Parameter_Config(VertexCount) - QUERY +NvAR_Parameter_Config(TriangleCount) - QUERY +NvAR_Parameter_Config(ExpressionCount) - QUERY +NvAR_Parameter_Config(ShapeEigenValueCount) - QUERY Input: NvAR_Parameter_Input(Width) @@ -157,6 +163,8 @@ NvAR_Parameter_Output(BoundingBoxesConfidence) - OPTIONAL NvAR_Parameter_Output(Landmarks) - OPTIONAL NvAR_Parameter_Output(Pose) - OPTIONAL NvAR_Parameter_Output(LandmarksConfidence) - OPTIONAL +NvAR_Parameter_Output(ExpressionCoefficients) - OPTIONAL +NvAR_Parameter_Output(ShapeEigenValues) - OPTIONAL */ #endif // NvAR_DEFS_H diff --git a/nvar/include/nvCVImage.h b/nvar/include/nvCVImage.h index f63b71c..7f07360 100644 --- a/nvar/include/nvCVImage.h +++ b/nvar/include/nvCVImage.h @@ -63,7 +63,7 @@ typedef enum NvCVImage_ComponentType { } NvCVImage_ComponentType; -//! Value for the planar field or isPlanar argument. Two values are currently accommodated for RGB: +//! Value for the planar field or layout argument. Two values are currently accommodated for RGB: //! Interleaved or chunky storage locates all components of a pixel adjacent in memory, //! e.g. RGBRGBRGB... (denoted [RGB]). //! Planar storage locates the same component of all pixels adjacent in memory, @@ -104,12 +104,11 @@ typedef enum NvCVImage_ComponentType { #define NVCV_CHROMA_MPEG2 NVCV_CHROMA_COSITED #define NVCV_CHROMA_MPEG1 NVCV_CHROMA_INTSTITIAL -//! This is the value for the gpuMem field or the onGPU argument. Two values are currently accommodated: -//! CPU indicates standard CPU memory. -//! GPU indicates CUDA buffers. +//! This is the value for the gpuMem field or the memSpace argument. #define NVCV_CPU 0 //!< The buffer is stored in CPU memory. #define NVCV_GPU 1 //!< The buffer is stored in CUDA memory. #define NVCV_CUDA 1 //!< The buffer is stored in CUDA memory. +#define NVCV_CPU_PINNED 2 //!< The buffer is stored in pinned CPU memory. //! Image descriptor. typedef struct @@ -125,8 +124,8 @@ NvCVImage { unsigned char pixelBytes; //!< The number of bytes in a chunky pixel. unsigned char componentBytes; //!< The number of bytes in each pixel component. unsigned char numComponents; //!< The number of components in each pixel. - unsigned char planar; //!< 0=chunky, 1=planar, 2=semi-planar (NV12, NV21). - unsigned char gpuMem; //!< 0=cpu mem, 1=cuda mem, + unsigned char planar; //!< NVCV_CHUNKY, NVCV_PLANAR, NVCV_UYVY, .... + unsigned char gpuMem; //!< NVCV_CPU, NVCV_CPU_PINNED, NVCV_CUDA, NVCV_GPU unsigned char colorspace; //!< an OR of colorspace, range and chroma phase. unsigned char reserved[2]; //!< For structure padding and future expansion. Set to 0. void *pixels; //!< Pointer to pixel(0,0) in the image. @@ -145,14 +144,14 @@ NvCVImage { //! \param[in] height the number of pixels vertically. //! \param[in] format the format of the pixels. //! \param[in] type the type of each pixel component. - //! \param[in] isPlanar One of { NVCV_CHUNKY, NVCV_PLANAR }. - //! \param[in] onGPU One of { NVCV_CPU, NVCV_GPU } + //! \param[in] layout One of { NVCV_CHUNKY, NVCV_PLANAR } or one of the YUV layouts. + //! \param[in] memSpace One of { NVCV_CPU, NVCV_CPU_PINNED, NVCV_GPU, NVCV_CUDA } //! \param[in] alignment row byte alignment. Choose 0 or a power of 2. //! 1: yields no gap whatsoever between scanlines; //! 0: default alignment: 4 on CPU, and cudaMallocPitch's choice on GPU. //! Other common values are 16 or 32 for cache line size. inline NvCVImage(unsigned width, unsigned height, NvCVImage_PixelFormat format, NvCVImage_ComponentType type, - unsigned isPlanar = 0, unsigned onGPU = 0, unsigned alignment = 0); + unsigned layout = NVCV_CHUNKY, unsigned memSpace = NVCV_CPU, unsigned alignment = 0); //! Subimage constructor. //! \param[in] fullImg the full image, from which this subImage view is to be created. @@ -206,12 +205,12 @@ NvCVImage { //! \param[in] pixels a pointer to the pixel buffer. //! \param[in] format the format of the pixels. //! \param[in] type the type of the components of the pixels. -//! \param[in] isPlanar One of { NVCV_CHUNKY, NVCV_PLANAR }. -//! \param[in] onGPU Location of the buffer: one of { NVCV_CPU, NVCV_GPU } +//! \param[in] layout One of { NVCV_CHUNKY, NVCV_PLANAR } or one of the YUV layouts. +//! \param[in] memSpace Location of the buffer: one of { NVCV_CPU, NVCV_CPU_PINNED, NVCV_GPU, NVCV_CUDA } //! \return NVCV_SUCCESS if successful //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not yet accommodated. NvCV_Status NvCV_API NvCVImage_Init(NvCVImage *im, unsigned width, unsigned height, int pitch, void *pixels, - NvCVImage_PixelFormat format, NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU); + NvCVImage_PixelFormat format, NvCVImage_ComponentType type, unsigned layout, unsigned memSpace); //! Initialize a view into a subset of an existing image. @@ -234,8 +233,8 @@ void NvCV_API NvCVImage_InitView(NvCVImage *subImg, NvCVImage *fullImg, int x, i //! \param[in] height the desired height of the image, in pixels. //! \param[in] format the format of the pixels. //! \param[in] type the type of the components of the pixels. -//! \param[in] isPlanar One of { NVCV_CHUNKY, NVCV_PLANAR }. -//! \param[in] onGPU Location of the buffer: one of { NVCV_CPU, NVCV_GPU } +//! \param[in] layout One of { NVCV_CHUNKY, NVCV_PLANAR } or one of the YUV layouts. +//! \param[in] memSpace Location of the buffer: one of { NVCV_CPU, NVCV_CPU_PINNED, NVCV_GPU, NVCV_CUDA } //! \param[in] alignment row byte alignment. Choose 0 or a power of 2. //! 1: yields no gap whatsoever between scanlines; //! 0: default alignment: 4 on CPU, and cudaMallocPitch's choice on GPU. @@ -244,7 +243,7 @@ void NvCV_API NvCVImage_InitView(NvCVImage *subImg, NvCVImage *fullImg, int x, i //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not accommodated. //! \return NVCV_ERR_MEMORY if there is not enough memory to allocate the buffer. NvCV_Status NvCV_API NvCVImage_Alloc(NvCVImage *im, unsigned width, unsigned height, NvCVImage_PixelFormat format, - NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU, unsigned alignment); + NvCVImage_ComponentType type, unsigned layout, unsigned memSpace, unsigned alignment); //! Reallocate memory for, and initialize an image. This assumes that the image is valid. @@ -255,8 +254,8 @@ NvCV_Status NvCV_API NvCVImage_Alloc(NvCVImage *im, unsigned width, unsigned hei //! \param[in] height the desired height of the image, in pixels. //! \param[in] format the format of the pixels. //! \param[in] type the type of the components of the pixels. -//! \param[in] isPlanar One of { NVCV_CHUNKY, NVCV_PLANAR }. -//! \param[in] onGPU Location of the buffer: one of { NVCV_CPU, NVCV_GPU } +//! \param[in] layout One of { NVCV_CHUNKY, NVCV_PLANAR } or one of the YUV layouts. +//! \param[in] memSpace Location of the buffer: one of { NVCV_CPU, NVCV_CPU_PINNED, NVCV_GPU, NVCV_CUDA } //! \param[in] alignment row byte alignment. Choose 0 or a power of 2. //! 1: yields no gap whatsoever between scanlines; //! 0: default alignment: 4 on CPU, and cudaMallocPitch's choice on GPU. @@ -265,7 +264,7 @@ NvCV_Status NvCV_API NvCVImage_Alloc(NvCVImage *im, unsigned width, unsigned hei //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not accommodated. //! \return NVCV_ERR_MEMORY if there is not enough memory to allocate the buffer. NvCV_Status NvCV_API NvCVImage_Realloc(NvCVImage *im, unsigned width, unsigned height, NvCVImage_PixelFormat format, - NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU, unsigned alignment); + NvCVImage_ComponentType type, unsigned layout, unsigned memSpace, unsigned alignment); //! Deallocate the image buffer from the image. The image is not deallocated. @@ -278,8 +277,8 @@ void NvCV_API NvCVImage_Dealloc(NvCVImage *im); //! \param[in] height the desired height of the image, in pixels. //! \param[in] format the format of the pixels. //! \param[in] type the type of the components of the pixels. -//! \param[in] isPlanar One of { NVCV_CHUNKY, NVCV_PLANAR }. -//! \param[in] onGPU Location of the buffer: one of { NVCV_CPU, NVCV_GPU } +//! \param[in] layout One of { NVCV_CHUNKY, NVCV_PLANAR } or one of the YUV layouts. +//! \param[in] memSpace Location of the buffer: one of { NVCV_CPU, NVCV_CPU_PINNED, NVCV_GPU, NVCV_CUDA } //! \param[in] alignment row byte alignment. Choose 0 or a power of 2. //! 1: yields no gap whatsoever between scanlines; //! 0: default alignment: 4 on CPU, and cudaMallocPitch's choice on GPU. @@ -289,7 +288,7 @@ void NvCV_API NvCVImage_Dealloc(NvCVImage *im); //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not accommodated. //! \return NVCV_ERR_MEMORY if there is not enough memory to allocate the buffer. NvCV_Status NvCV_API NvCVImage_Create(unsigned width, unsigned height, NvCVImage_PixelFormat format, - NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU, unsigned alignment, NvCVImage **out); + NvCVImage_ComponentType type, unsigned layout, unsigned memSpace, unsigned alignment, NvCVImage **out); //! Deallocate the image allocated with NvCVImage_Create() (C-style destructor). @@ -308,22 +307,38 @@ void NvCV_API NvCVImage_ComponentOffsets(NvCVImage_PixelFormat format, int *rOff //! Transfer one image to another, with a limited set of conversions. +//! //! If any of the images resides on the GPU, it may run asynchronously, //! so cudaStreamSynchronize() should be called if it is necessary to run synchronously. -//! Conversions are between -//! - RGBu8 --> RGBu8, where the RGB components are in any order; -//! - RGBu8 <--> RGBf32, where the RGB components are in any order (e.g. BGR, RGB); -//! - RGBu8 --> GRAYf32, where the RGB components are in any order; -//! - RGBf32 --> RGBAu8, setting A=255, where the RGB and RGBA components are in any order; -//! - RGBAu8 --> RGBu8, by removing the alpha component; -//! - GRAYf32 --> GRAYu8; -//! - GRAYu8 --> GRAYu8, where the RGB components are in any order (also works with ALPHAu8); -//! - ALPHAu8 --> RGBAu8, (insertion) without touching the RGB components. -//! - GRAYu8 --> RGBu8, by replicating gray into all RGB components. -//! - YUVu8 --> RGBu8, though colorspace field needs to be set manually prior to calling. -//! - chunky <--> planar; -//! - CPU <--> GPU; -//! Additionally, when the src and dst formats are the same, all formats are accommodated on CPU and GPU, +//! The following table indicates the currently-implemented conversions: +//! +------------------+-------------+-------------+-------------+-------------+ +//! | | u8 --> u8 | u8 --> f32 | f32 --> u8 | f32 --> f32 | +//! +------------------+-------------+-------------+-------------+-------------+ +//! | Y -- > Y | X | | X | X | +//! | Y -- > A | X | | X | X | +//! | Y -- > RGB | X | X | X | X | +//! | Y -- > RGBA | X | X | X | X | +//! | A -- > Y | X | | X | X | +//! | A -- > A | X | | X | X | +//! | A -- > RGB | X | X | X | X | +//! | A -- > RGBA | X | | | | +//! | RGB -- > Y | X | X | | | +//! | RGB -- > A | X | X | | | +//! | RGB -- > RGB | X | X | X | X | +//! | RGB -- > RGBA | X | X | X | X | +//! | RGBA -- > Y | X | X | | | +//! | RGBA -- > A | | X | | | +//! | RGBA -- > RGB | X | X | X | X | +//! | RGBA -- > RGBA | X | | | | +//! | YUV420 -- > RGB | X | | | | +//! | YUV422 -- > RGB | X | | | | +//! +------------------+-------------+-------------+-------------+-------------+ +//! where +//! * Either source or destination can be CHUNKY or PLANAR. +//! * Either source or destination can reside on the CPU or the GPU. +//! * The RGB components are in any order (i.e. RGB or BGR; RGBA or BGRA). +//! * YUV requires that the colorspace field be set manually prior to Transfer. +//! * Additionally, when the src and dst formats are the same, all formats are accommodated on CPU and GPU, //! and this can be used as a replacement for cudaMemcpy2DAsync() (which it utilizes). //! //! When there is some kind of conversion AND the src and dst reside on different processors (CPU, GPU), @@ -354,14 +369,15 @@ NvCV_Status NvCV_API NvCVImage_Transfer( //! Composite one BGRu8 source image over another using the given matte. -//! \param[in] src the source BGRu8 (or RGBu8) image. +//! \param[in] fg the foreground source BGRu8 (or RGBu8) image. +//! \param[in] bg the background source BGRu8 (or RGBu8) image. //! \param[in] mat the matte Yu8 (or Au8) image, indicating where the src should come through. -//! \param[out] dst the destination BGRu8 (or RGBu8) image. +//! \param[out] dst the destination BGRu8 (or RGBu8) image. This can be the same as fg or bg. //! \return NVCV_SUCCESS if the operation was successful. //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not accommodated. -//! \bug This is only implemented for 3-component u8 src and dst, and 1-component mat, +//! \bug This is only implemented for 3-component u8 fg, bg and dst, and 1-component u8 mat, //! where all images are resident on the CPU. -NvCV_Status NvCV_API NvCVImage_Composite(const NvCVImage *src, const NvCVImage *mat, NvCVImage *dst); +NvCV_Status NvCV_API NvCVImage_Composite(const NvCVImage *fg, const NvCVImage *bg, const NvCVImage *mat, NvCVImage *dst); //! Composite a BGRu8 source image over a constant color field using the given matte. @@ -423,9 +439,9 @@ NvCVImage::NvCVImage() { ********************************************************************************/ NvCVImage::NvCVImage(unsigned width, unsigned height, NvCVImage_PixelFormat format, NvCVImage_ComponentType type, - unsigned isPlanar, unsigned onGPU, unsigned alignment) { + unsigned layout, unsigned memSpace, unsigned alignment) { pixels = nullptr; - (void)NvCVImage_Alloc(this, width, height, format, type, isPlanar, onGPU, alignment); + (void)NvCVImage_Alloc(this, width, height, format, type, layout, memSpace, alignment); } /******************************************************************************** diff --git a/nvar/include/nvCVStatus.h b/nvar/include/nvCVStatus.h index 5c32654..17997dc 100644 --- a/nvar/include/nvCVStatus.h +++ b/nvar/include/nvCVStatus.h @@ -61,6 +61,9 @@ typedef enum NvCV_Status { NVCV_ERR_FEATURENOTFOUND = -14, //!< The requested feature was not found NVCV_ERR_MISSINGINPUT = -15, //!< A required parameter was not set NVCV_ERR_RESOLUTION = -16, //!< The specified image resolution is not supported. + NVCV_ERR_UNSUPPORTEDGPU = -17, //!< The GPU is not supported + NVCV_ERR_WRONGGPU = -18, //!< The current GPU is not the one selected. + NVCV_ERR_UNSUPPORTEDDRIVER = -19, //!< The currently installed graphics driver is not supported NVCV_ERR_CUDA_MEMORY = -20, //!< There is not enough CUDA memory for the requested operation. NVCV_ERR_CUDA_VALUE = -21, //!< A CUDA parameter is not within the acceptable range. diff --git a/nvar/src/nvARProxy.cpp b/nvar/src/nvARProxy.cpp index 8368ca9..44f1c02 100644 --- a/nvar/src/nvARProxy.cpp +++ b/nvar/src/nvARProxy.cpp @@ -62,7 +62,7 @@ inline int nvFreeLibrary(HINSTANCE handle) { HINSTANCE getNvARLib() { - TCHAR path[MAX_PATH], fullPath[2*MAX_PATH]; + TCHAR path[MAX_PATH], fullPath[MAX_PATH]; // There can be multiple apps on the system, // some might include the SDK in the app package and @@ -80,6 +80,13 @@ HINSTANCE getNvARLib() { return NvArLib; } +NvCV_Status NvAR_API NvAR_GetVersion(unsigned int* version) { + static const auto funcPtr = (decltype(NvAR_GetVersion)*)nvGetProcAddress(getNvARLib(), "NvAR_GetVersion"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(version); +} + NvCV_Status NvAR_API NvCVImage_Init(NvCVImage* im, unsigned width, unsigned height, int pitch, void* pixels, NvCVImage_PixelFormat format, NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU) { @@ -150,11 +157,11 @@ NvCV_Status NvAR_API NvCVImage_Transfer(const NvCVImage* src, NvCVImage* dst, fl return funcPtr(src, dst, scale, stream, tmp); } -NvCV_Status NvAR_API NvCVImage_Composite(const NvCVImage* src, const NvCVImage* mat, NvCVImage* dst) { +NvCV_Status NvAR_API NvCVImage_Composite(const NvCVImage* fg, const NvCVImage* bg, const NvCVImage* mat, NvCVImage* dst) { static const auto funcPtr = (decltype(NvCVImage_Composite)*)nvGetProcAddress(getNvARLib(), "NvCVImage_Composite"); if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; - return funcPtr(src, mat, dst); + return funcPtr(fg, bg, mat, dst); } NvCV_Status NvAR_API NvCVImage_CompositeOverConstant(const NvCVImage* src, const NvCVImage* mat, @@ -307,7 +314,7 @@ NvCV_Status NvAR_API NvAR_GetCudaStream(NvAR_FeatureHandle handle, const char* n } NvCV_Status NvAR_API NvAR_GetF32Array(NvAR_FeatureHandle handle, const char* name, const float** vals, int* count) { - static const auto funcPtr = (decltype(NvAR_GetF32Array)*)nvGetProcAddress(getNvARLib(), "NvAR_GetCNvAR_GetF32ArrayudaStream"); + static const auto funcPtr = (decltype(NvAR_GetF32Array)*)nvGetProcAddress(getNvARLib(), "NvAR_GetF32Array"); if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; return funcPtr(handle, name, vals, count); diff --git a/resources/ar_001.png b/resources/ar_001.png new file mode 100644 index 0000000..271c5d6 Binary files /dev/null and b/resources/ar_001.png differ diff --git a/resources/ar_002.png b/resources/ar_002.png new file mode 100644 index 0000000..c037e92 Binary files /dev/null and b/resources/ar_002.png differ diff --git a/resources/ar_003.png b/resources/ar_003.png new file mode 100644 index 0000000..fd54ec1 Binary files /dev/null and b/resources/ar_003.png differ diff --git a/resources/ar_004.png b/resources/ar_004.png new file mode 100644 index 0000000..26708be Binary files /dev/null and b/resources/ar_004.png differ diff --git a/samples/FaceTrack/CMakeLists.txt b/samples/FaceTrack/CMakeLists.txt index 49f8858..8b3eca2 100644 --- a/samples/FaceTrack/CMakeLists.txt +++ b/samples/FaceTrack/CMakeLists.txt @@ -1,4 +1,10 @@ -set(SOURCE_FILES FaceEngine.cpp FaceTrack.cpp ../utils/RenderingUtils.cpp ../../nvar/src/nvARProxy.cpp) +set(SOURCE_FILES FaceEngine.cpp +FaceTrack.cpp +../utils/RenderingUtils.cpp +../../nvar/src/nvARProxy.cpp +../utils/FeatureVertexName.cpp +../utils/FeatureVertexName.h +) set(HEADER_FILES FaceEngine.h) # Set Visual Studio source filters @@ -16,10 +22,14 @@ target_link_libraries(FaceTrack PUBLIC GLM ) +set(ARSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin) -set(PATH_STR "PATH=%PATH%" ${OPENCV_PATH_STR}) +set(PATH_STR "PATH=%PATH%" ${ARSDK_PATH_STR} ${OPENCV_PATH_STR}) +set(CMD_ARG_STR "--model_path=\"${CMAKE_CURRENT_SOURCE_DIR}/../../bin/models\"") set_target_properties(FaceTrack PROPERTIES FOLDER SampleApps VS_DEBUGGER_ENVIRONMENT "${PATH_STR}" + VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}" ) + diff --git a/samples/FaceTrack/FaceEngine.cpp b/samples/FaceTrack/FaceEngine.cpp index ddab3d9..83dc3e4 100644 --- a/samples/FaceTrack/FaceEngine.cpp +++ b/samples/FaceTrack/FaceEngine.cpp @@ -23,31 +23,10 @@ #include "FaceEngine.h" #include "RenderingUtils.h" -const char* NvCV_StatusStringFromCode(NvCV_Status code) { - struct TabEntry { - NvCV_Status code; - const char* str; - }; - static const TabEntry lut[] = { - {NVCV_SUCCESS, "no error"}, - {NVCV_ERR_GENERAL, "unspecified failure"}, - {NVCV_ERR_FEATURENOTFOUND, "Feature not found"}, - {NVCV_ERR_PARAMETER, "invalid parameter"}, - {NVCV_ERR_MEMORY, "provided buffer too small"}, - {NVCV_ERR_INITIALIZATION, "not initialized"}, - {NVCV_ERR_MISSINGINPUT, "missing input"}, - {NVCV_ERR_INITIALIZATION, "unable to initialize feature"}, - {NVCV_ERR_CUDA_MEMORY, "out of GPU memory"}, - {NVCV_ERR_SELECTOR, "unsupported parameter"}, - }; - for (const TabEntry* p = lut; p != &lut[sizeof(lut) / sizeof(lut[0])]; ++p) - if (p->code == code) return p->str; - return "UNKNOWN ERROR"; -} bool CheckResult(NvCV_Status nvErr, unsigned line) { if (NVCV_SUCCESS == nvErr) return true; - std::cout << "ERROR: " << NvCV_StatusStringFromCode(nvErr) << ", line " << line << std::endl; + std::cout << "ERROR: " << NvCV_GetErrorStringFromCode(nvErr) << ", line " << line << std::endl; return false; } @@ -67,7 +46,7 @@ FaceEngine::Err FaceEngine::fitFaceModel(cv::Mat& frame) { nvErr = NvAR_Run(faceFitHandle); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errRun); - if (getAverageLandmarksConfidence() < LANDMARK_CONF_THRESH) return FaceEngine::Err::errRun; + if (getAverageLandmarksConfidence() < confidenceThreshold) return FaceEngine::Err::errRun; bail: return err; @@ -81,30 +60,44 @@ FaceEngine::Err FaceEngine::createFeatures(const char* modelPath, unsigned int _ FaceEngine::Err err = FaceEngine::Err::errNone; NvCV_Status cuErr = NvAR_CudaStreamCreate(&stream); - if (appMode == faceDetection) + if (NVCV_SUCCESS != cuErr) { + printf("Cannot create a cuda stream: %s\n", NvCV_GetErrorStringFromCode(cuErr)); + return errInitialization; + } + if (appMode == faceDetection) { err = createFaceDetectionFeature(modelPath, stream); - else if (appMode == landmarkDetection) + if (err != Err::errNone) { + printf("ERROR: An error has occured while initializing Face Detection\n"); + } + } else if (appMode == landmarkDetection) { err = createLandmarkDetectionFeature(modelPath, _batchSize, stream); - else if (appMode == faceMeshGeneration) + if (err != Err::errNone) { + printf("ERROR: An error has occured while initializing Landmark Detection\n"); + } + } else if (appMode == faceMeshGeneration) { err = createFaceFittingFeature(modelPath, stream); + if (err != Err::errNone) { + printf("ERROR: An error has occured while initializing Face Fitting\n"); + } + } return err; } -FaceEngine::Err FaceEngine::createFaceDetectionFeature(const char* modelPath, CUstream stream) { +FaceEngine::Err FaceEngine::createFaceDetectionFeature(const char* modelPath, CUstream str) { FaceEngine::Err err = FaceEngine::Err::errNone; NvCV_Status nvErr; nvErr = NvAR_Create(NvAR_Feature_FaceDetection, &faceDetectHandle); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errEffect); nvErr = NvAR_SetString(faceDetectHandle, NvAR_Parameter_Config(ModelDir), modelPath); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); - nvErr = NvAR_SetCudaStream(faceDetectHandle, NvAR_Parameter_Config(CUDAStream), stream); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetCudaStream(faceDetectHandle, NvAR_Parameter_Config(CUDAStream), str); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetU32(faceDetectHandle, NvAR_Parameter_Config(Temporal), bStabilizeFace); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_Load(faceDetectHandle); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); @@ -114,25 +107,31 @@ bail: } FaceEngine::Err FaceEngine::createLandmarkDetectionFeature(const char* modelPath, unsigned int _batchSize, - CUstream stream) { + CUstream str) { FaceEngine::Err err = FaceEngine::Err::errNone; NvCV_Status nvErr; batchSize = _batchSize; nvErr = NvAR_Create(NvAR_Feature_LandmarkDetection, &landmarkDetectHandle); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errEffect); nvErr = NvAR_SetString(landmarkDetectHandle, NvAR_Parameter_Config(ModelDir), modelPath); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); - nvErr = NvAR_SetCudaStream(landmarkDetectHandle, NvAR_Parameter_Config(CUDAStream), stream); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetCudaStream(landmarkDetectHandle, NvAR_Parameter_Config(CUDAStream), str); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetU32(landmarkDetectHandle, NvAR_Parameter_Config(BatchSize), batchSize); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetU32(landmarkDetectHandle, NvAR_Parameter_Config(Temporal), bStabilizeFace); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + + nvErr = NvAR_SetU32(landmarkDetectHandle, NvAR_Parameter_Config(Landmarks_Size), numLandmarks); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + + nvErr = NvAR_SetU32(landmarkDetectHandle, NvAR_Parameter_Config(LandmarksConfidence_Size), numLandmarks); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_Load(landmarkDetectHandle); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); @@ -141,18 +140,26 @@ bail: return err; } -FaceEngine::Err FaceEngine::createFaceFittingFeature(const char* modelPath, CUstream stream) { +FaceEngine::Err FaceEngine::createFaceFittingFeature(const char* modelPath, CUstream str) { FaceEngine::Err err = FaceEngine::Err::errNone; NvCV_Status nvErr; nvErr = NvAR_Create(NvAR_Feature_Face3DReconstruction, &faceFitHandle); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errEffect); nvErr = NvAR_SetString(faceFitHandle, NvAR_Parameter_Config(ModelDir), modelPath); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); - nvErr = NvAR_SetCudaStream(faceFitHandle, NvAR_Parameter_Config(CUDAStream), stream); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetCudaStream(faceFitHandle, NvAR_Parameter_Config(CUDAStream), str); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + + nvErr = NvAR_SetU32(faceFitHandle, NvAR_Parameter_Config(Landmarks_Size), numLandmarks); // TODO: Check if nonzero?? + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + + if (!face_model.empty()) { + nvErr = NvAR_SetString(faceFitHandle, NvAR_Parameter_Config(ModelName), face_model.c_str()); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + } nvErr = NvAR_Load(faceFitHandle); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); @@ -169,55 +176,64 @@ FaceEngine::Err FaceEngine::initFeatureIOParams() { BAIL_IF_CVERR(cvErr, err, FaceEngine::Err::errInitialization); - if (appMode == faceDetection) + if (appMode == faceDetection) { err = initFaceDetectionIOParams(&inputImageBuffer); - else if (appMode == landmarkDetection) + if (err != Err::errNone) { + printf("ERROR: An error has occured while setting input, output parmeters for Face Detection\n"); + } + } else if (appMode == landmarkDetection) { err = initLandmarkDetectionIOParams(&inputImageBuffer); - else if (appMode == faceMeshGeneration) + if (err != Err::errNone) { + printf("ERROR: An error has occured while setting input, output parmeters for Landmark Detection\n"); + } + } else if (appMode == faceMeshGeneration) { err = initFaceFittingIOParams(&inputImageBuffer); - + if (err != Err::errNone) { + printf("ERROR: An error has occured while setting input, output parmeters for Face Fitting\n"); + } + } return err; bail: return err; } -FaceEngine::Err FaceEngine::initFaceDetectionIOParams(NvCVImage* _inputImageBuffer) { +FaceEngine::Err FaceEngine::initFaceDetectionIOParams(NvCVImage* inBuf) { NvCV_Status nvErr = NVCV_SUCCESS; FaceEngine::Err err = FaceEngine::Err::errNone; - nvErr = NvAR_SetObject(faceDetectHandle, NvAR_Parameter_Input(Image), &inputImageBuffer, sizeof(NvCVImage)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetObject(faceDetectHandle, NvAR_Parameter_Input(Image), inBuf, sizeof(NvCVImage)); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); output_bbox_data.assign(25, {0.f, 0.f, 0.f, 0.f}); output_bbox_conf_data.assign(25, 0.f); output_bboxes.boxes = output_bbox_data.data(); - output_bboxes.max_boxes = output_bbox_data.size(); + output_bboxes.max_boxes = (uint8_t)output_bbox_data.size(); output_bboxes.num_boxes = 0; nvErr = NvAR_SetObject(faceDetectHandle, NvAR_Parameter_Output(BoundingBoxes), &output_bboxes, sizeof(NvAR_BBoxes)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetF32Array(faceDetectHandle, NvAR_Parameter_Output(BoundingBoxesConfidence), output_bbox_conf_data.data(), output_bboxes.max_boxes); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); bail: return err; } -FaceEngine::Err FaceEngine::initLandmarkDetectionIOParams(NvCVImage* _inputImageBuffer) { +FaceEngine::Err FaceEngine::initLandmarkDetectionIOParams(NvCVImage* inBuf) { NvCV_Status nvErr = NVCV_SUCCESS; FaceEngine::Err err = FaceEngine::Err::errNone; - nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Input(Image), &inputImageBuffer, sizeof(NvCVImage)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Input(Image), inBuf, sizeof(NvCVImage)); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); unsigned int OUTPUT_SIZE_KPTS, OUTPUT_SIZE_KPTS_CONF; nvErr = NvAR_GetU32(landmarkDetectHandle, NvAR_Parameter_Config(Landmarks_Size), &OUTPUT_SIZE_KPTS); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_GetU32(landmarkDetectHandle, NvAR_Parameter_Config(LandmarksConfidence_Size), &OUTPUT_SIZE_KPTS_CONF); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); facial_landmarks.assign(batchSize * OUTPUT_SIZE_KPTS, {0.f, 0.f}); facial_pose.assign(batchSize, {0.f, 0.f, 0.f, 0.f}); @@ -225,74 +241,78 @@ FaceEngine::Err FaceEngine::initLandmarkDetectionIOParams(NvCVImage* _inputImage nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Output(Landmarks), facial_landmarks.data(), sizeof(NvAR_Point2f)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Output(Pose), facial_pose.data(), sizeof(NvAR_Quaternion)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetF32Array(landmarkDetectHandle, NvAR_Parameter_Output(LandmarksConfidence), facial_landmarks_confidence.data(), batchSize * OUTPUT_SIZE_KPTS); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); uint output_bbox_size = batchSize; if (!bStabilizeFace) output_bbox_size = 25; output_bbox_data.assign(output_bbox_size, {0.f, 0.f, 0.f, 0.f}); output_bboxes.boxes = output_bbox_data.data(); - output_bboxes.max_boxes = output_bbox_size; - output_bboxes.num_boxes = output_bbox_size; + output_bboxes.max_boxes = (uint8_t)output_bbox_size; + output_bboxes.num_boxes = (uint8_t)output_bbox_size; nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Output(BoundingBoxes), &output_bboxes, sizeof(NvAR_BBoxes)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); bail: return err; } -FaceEngine::Err FaceEngine::initFaceFittingIOParams(NvCVImage* inputImageBuffer) { +FaceEngine::Err FaceEngine::initFaceFittingIOParams(NvCVImage* inBuf) { NvCV_Status nvErr = NVCV_SUCCESS; FaceEngine::Err err = FaceEngine::Err::errNone; face_mesh = new NvAR_FaceMesh(); - face_mesh->vertices = new NvAR_Vector3f[FACE_MODEL_NUM_VERTICES]; - face_mesh->tvi = new NvAR_Vector3u16[FACE_MODEL_NUM_INDICES]; + face_mesh->vertices = nullptr; //new NvAR_Vector3f[FACE_MODEL_NUM_VERTICES]; + face_mesh->tvi = nullptr; // new NvAR_Vector3u16[FACE_MODEL_NUM_INDICES]; rendering_params = new NvAR_RenderingParams(); - nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Input(Image), inputImageBuffer, sizeof(NvCVImage)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Input(Image), inBuf, sizeof(NvCVImage)); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetS32(faceFitHandle, NvAR_Parameter_Input(Width), input_image_width); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetS32(faceFitHandle, NvAR_Parameter_Input(Height), input_image_height); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); unsigned int OUTPUT_SIZE_KPTS; nvErr = NvAR_GetU32(faceFitHandle, NvAR_Parameter_Config(Landmarks_Size), &OUTPUT_SIZE_KPTS); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); facial_landmarks.assign(batchSize * OUTPUT_SIZE_KPTS, {0.f, 0.f}); nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Output(Landmarks), facial_landmarks.data(), sizeof(NvAR_Point2f)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); facial_landmarks_confidence.assign(batchSize * OUTPUT_SIZE_KPTS, 0.f); nvErr = NvAR_SetF32Array(faceFitHandle, NvAR_Parameter_Output(LandmarksConfidence), facial_landmarks_confidence.data(), batchSize * OUTPUT_SIZE_KPTS); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); + + facial_pose.assign(batchSize, {0.f, 0.f, 0.f, 0.f}); + nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Output(Pose), facial_pose.data(), sizeof(NvAR_Quaternion)); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); output_bbox_data.assign(batchSize, {0.f, 0.f, 0.f, 0.f}); output_bboxes.boxes = output_bbox_data.data(); - output_bboxes.max_boxes = batchSize; - output_bboxes.num_boxes = batchSize; + output_bboxes.max_boxes = (uint8_t)batchSize; + output_bboxes.num_boxes = (uint8_t)batchSize; nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Output(BoundingBoxes), &output_bboxes, sizeof(NvAR_BBoxes)); nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Output(FaceMesh), face_mesh, sizeof(NvAR_FaceMesh)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); nvErr = NvAR_SetObject(faceFitHandle, NvAR_Parameter_Output(RenderingParams), rendering_params, sizeof(NvAR_RenderingParams)); - BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errInitialization); + BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); bail: return err; @@ -352,6 +372,7 @@ void FaceEngine::releaseFaceFittingIOParams() { if (!output_bbox_data.empty()) output_bbox_data.clear(); if (!facial_landmarks.empty()) facial_landmarks.clear(); if (!facial_landmarks_confidence.empty()) facial_landmarks_confidence.clear(); + if (!facial_pose.empty()) facial_pose.clear(); NvCVImage_Dealloc(&inputImageBuffer); if (rendering_params) { delete rendering_params; @@ -461,6 +482,7 @@ void FaceEngine::jiggleBox(std::mt19937& ran, float minMag, float maxMag, const #endif // None of these makes a significant difference } +#ifdef UNUSED /** Intersect a rectangle with an image. * @param[in] srcRect the source rectangle. * @param[in] src the source image. @@ -493,6 +515,7 @@ static bool IntersectRectWithImage(const cv::Rect& srcRect, const cv::Mat& src, clipRect.height = rect[1].y - rect[0].y; return result; } +#endif // UNUSED void FaceEngine::DrawPose(const cv::Mat& src, const NvAR_Quaternion* pose) { float R[3][3]; @@ -505,17 +528,19 @@ void FaceEngine::DrawPose(const cv::Mat& src, const NvAR_Quaternion* pose) { float x3 = radius * R[1][2] * -1.f; float y3 = radius * R[2][2] * -1.f; - // 30th point is the tip of the nose - const int nose_tip = 30; + int nose_tip = 0; + const char* kNoseTipName = "nose-tip"; + nose_tip = FindLandmarkIndexFromName(numLandmarks, kNoseTipName); + int width = src.cols; int height = src.rows; NvAR_Point2f cxy = *(facial_landmarks.data() + nose_tip); - float cx1 = std::min(std::max(0, int(cxy.x + x1)), width - 1); - float cy1 = std::min(std::max(0, int(cxy.y + y1)), height - 1); - float cx2 = std::min(std::max(0, int(cxy.x + x2)), width - 1); - float cy2 = std::min(std::max(0, int(cxy.y + y2)), height - 1); - float cx3 = std::min(std::max(0, int(cxy.x + x3)), width - 1); - float cy3 = std::min(std::max(0, int(cxy.y + y3)), height - 1); + float cx1 = (float)std::min(std::max(0, int(cxy.x + x1)), width - 1); + float cy1 = (float)std::min(std::max(0, int(cxy.y + y1)), height - 1); + float cx2 = (float)std::min(std::max(0, int(cxy.x + x2)), width - 1); + float cy2 = (float)std::min(std::max(0, int(cxy.y + y2)), height - 1); + float cx3 = (float)std::min(std::max(0, int(cxy.x + x3)), width - 1); + float cy3 = (float)std::min(std::max(0, int(cxy.y + y3)), height - 1); cv::line(src, cv::Point((int)cxy.x, (int)cxy.y), cv::Point((int)cx1, (int)cy1), cv::Scalar(0, 0, 255), 2); cv::line(src, cv::Point((int)cxy.x, (int)cxy.y), cv::Point((int)cx2, (int)cy2), cv::Scalar(0, 255, 0), 2); @@ -530,16 +555,16 @@ NvCV_Status FaceEngine::findLandmarks() { return nvErr; } - if (getAverageLandmarksConfidence() < LANDMARK_CONF_THRESH) { + if (getAverageLandmarksConfidence() < confidenceThreshold) { return NVCV_ERR_GENERAL; } else { average_poses(getPose(), batchSize); NvAR_Point2f *pt, *endPt; int i = 0; - for (endPt = (pt = getLandmarks()) + NUM_LANDMARKS; pt != endPt; ++pt, i += 2) { + for (endPt = (pt = getLandmarks()) + numLandmarks; pt != endPt; ++pt, i += 2) { for (int j = 1; j < batchSize; j++) { - pt->x += pt[j * NUM_LANDMARKS].x; - pt->y += pt[j * NUM_LANDMARKS].y; + pt->x += pt[j * numLandmarks].x; + pt->y += pt[j * numLandmarks].y; } // average batch of inferences to generate final result landmark points pt->x /= batchSize; @@ -557,11 +582,11 @@ float FaceEngine::getAverageLandmarksConfidence() { float average_confidence = 0.0f; float* keypoints_landmarks_confidence = getLandmarksConfidence(); for (int i = 0; i < batchSize; i++) { - for (int j = 0; j < NUM_LANDMARKS; j++) { - average_confidence += keypoints_landmarks_confidence[i * NUM_LANDMARKS + j]; + for (int j = 0; j < numLandmarks; j++) { + average_confidence += keypoints_landmarks_confidence[i * numLandmarks + j]; } } - average_confidence /= batchSize * NUM_LANDMARKS; + average_confidence /= batchSize * numLandmarks; return average_confidence; } @@ -598,7 +623,7 @@ unsigned FaceEngine::acquireFaceBox(cv::Mat& src, NvAR_Rect& faceBox, int varian return n; } -unsigned FaceEngine::acquireFaceBoxAndLandmarks(cv::Mat& src, NvAR_Point2f* refMarks, NvAR_Rect& faceBox, int variant) { +unsigned FaceEngine::acquireFaceBoxAndLandmarks(cv::Mat& src, NvAR_Point2f* refMarks, NvAR_Rect& faceBox, int /*variant*/) { unsigned n = 0; NvCVImage fxSrcChunkyCPU; (void)NVWrapperForCVMat(&src, &fxSrcChunkyCPU); @@ -611,11 +636,24 @@ unsigned FaceEngine::acquireFaceBoxAndLandmarks(cv::Mat& src, NvAR_Point2f* refM if (findLandmarks() != NVCV_SUCCESS) return 0; faceBox = output_bboxes.boxes[0]; n = 1; - memcpy(refMarks, getLandmarks(), sizeof(NvAR_Point2f) * FaceEngine::NUM_LANDMARKS); + memcpy(refMarks, getLandmarks(), sizeof(NvAR_Point2f) * numLandmarks); return n; } void FaceEngine::setFaceStabilization(bool _bStabilizeFace) { bStabilizeFace = _bStabilizeFace; } +FaceEngine::Err FaceEngine::setNumLandmarks(int n) { + FaceEngine::Err err = errNone; + for (auto const& info : LANDMARKS_INFO) { + if (n == info.numPoints) { + numLandmarks = info.numPoints; + confidenceThreshold = info.confidence_threshold; + return err; + } + } + err = errGeneral; + return err; +} + void FaceEngine::setAppMode(FaceEngine::mode _mode) { appMode = _mode; } diff --git a/samples/FaceTrack/FaceEngine.h b/samples/FaceTrack/FaceEngine.h index 743e016..2cbb055 100644 --- a/samples/FaceTrack/FaceEngine.h +++ b/samples/FaceTrack/FaceEngine.h @@ -27,6 +27,7 @@ #include "nvAR.h" #include "nvCVOpenCV.h" #include "opencv2/opencv.hpp" +#include "FeatureVertexName.h" #define FITFACE_PRIVATE class KalmanFilter1D { @@ -78,10 +79,15 @@ class KalmanFilter1D { } }; -const char* NvCV_StatusStringFromCode(NvCV_Status code); - bool CheckResult(NvCV_Status nvErr, unsigned line); +#define BAIL_IF_ERR(err) \ +do { \ + if (0!=err) { \ + goto bail; \ + } \ + } while (0) + #define BAIL_IF_CVERR(nvErr, err, code) \ do { \ if (!CheckResult(nvErr, __LINE__)) { \ @@ -90,23 +96,30 @@ bool CheckResult(NvCV_Status nvErr, unsigned line); } \ } while (0) +typedef struct LandmarksProperties { + int numPoints; + float confidence_threshold; +}LandmarksProperties; + /******************************************************************************** * FaceEngine ********************************************************************************/ class FaceEngine { public: - enum Err { errNone, errGeneral, errRun, errInitialization, errRead }; + enum Err { errNone, errGeneral, errRun, errInitialization, errRead, errEffect, errParameter }; int input_image_width, input_image_height, input_image_pitch; - static const int NUM_LANDMARKS = 68; // TODO: get this instead from the SDK. - static const int FACE_MODEL_NUM_VERTICES = 3448, FACE_MODEL_NUM_INDICES = 6736; - static const long LANDMARK_CONF_THRESH = 10.f; + const LandmarksProperties LANDMARKS_INFO[2] = { + { 68, 10.0f }, // number of landmark points, confidence threshold value + { 126, 5.0f} + }; void setInputImageWidth(int width) { input_image_width = width; } void setInputImageHeight(int height) { input_image_height = height; } int getInputImageWidth() { return input_image_width; } int getInputImageHeight() { return input_image_height; } int getInputImagePitch() { return input_image_pitch = input_image_width * 3 * sizeof(unsigned char); } + void setFaceModel(const char *faceModel) { face_model = faceModel; } Err createFeatures(const char* modelPath, unsigned int _batchSize = 1); Err createFaceDetectionFeature(const char* modelPath, CUstream stream); @@ -142,6 +155,8 @@ class FaceEngine { NvAR_FaceMesh* getFaceMesh(); NvAR_RenderingParams* getRenderingParams(); void setFaceStabilization(bool); + Err setNumLandmarks(int); + int getNumLandmarks() { return numLandmarks; } void DrawPose(const cv::Mat& src, const NvAR_Quaternion* pose); NvCVImage inputImageBuffer{}, tmpImage{}; @@ -157,13 +172,17 @@ class FaceEngine { NvAR_BBoxes output_bboxes{}; int batchSize; std::mt19937 ran; + int numLandmarks; + float confidenceThreshold; + std::string face_model; bool bStabilizeFace; - NvAR_Point2f prevLandmark[NUM_LANDMARKS] = {0}; FaceEngine() { batchSize = 1; bStabilizeFace = true; + numLandmarks = LANDMARKS_INFO[0].numPoints; + confidenceThreshold = LANDMARKS_INFO[0].confidence_threshold; appMode = faceMeshGeneration; input_image_width = 640; input_image_height = 480; diff --git a/samples/FaceTrack/FaceTrack.cpp b/samples/FaceTrack/FaceTrack.cpp index 2ba4a46..cbe6cf2 100644 --- a/samples/FaceTrack/FaceTrack.cpp +++ b/samples/FaceTrack/FaceTrack.cpp @@ -63,10 +63,10 @@ ********************************************************************************/ bool FLAG_debug = false, FLAG_verbose = false, FLAG_temporal = true, FLAG_captureOutputs = false, - FLAG_offlineMode = false; + FLAG_offlineMode = false, FLAG_isNumLandmarks126 = false; std::string FLAG_outDir, FLAG_inFile, FLAG_outFile, FLAG_modelPath, FLAG_landmarks, FLAG_proxyWireframe, - FLAG_captureCodec = "avc1", FLAG_camRes; -unsigned int FLAG_batch = 1; + FLAG_captureCodec = "avc1", FLAG_camRes, FLAG_faceModel; +unsigned int FLAG_batch = 1, FLAG_appMode = 2; /******************************************************************************** * Usage @@ -88,8 +88,12 @@ static void Usage() { " --out_file= specify the output file\n" " --out= specify the output file\n" " --model_path= specify the directory containing the TRT models\n" + " --landmarks_126[=(true|false)] set the number of facial landmark points to 126, otherwise default to 68\n" + " --face_model= specify the name of the face model\n" " --wireframe_mesh= specify the path to a proxy wireframe mesh\n" - " --batch= 1 - 8, used for batch inferencing in landmark detector " + " --batch= 1 - 8, used for batch inferencing in landmark detector\n" + " --app_mode[=(0|1|2)] App mode. 0: Face detection, 1: Landmark detection, 2: Face fitting " + "(Default)." " --benchmarks[=] run benchmarks\n"); } @@ -181,11 +185,13 @@ static int ParseMyArgs(int argc, char **argv) { GetFlagArgVal("in", arg, &FLAG_inFile) || GetFlagArgVal("in_file", arg, &FLAG_inFile) || GetFlagArgVal("out", arg, &FLAG_outFile) || GetFlagArgVal("out_file", arg, &FLAG_outFile) || GetFlagArgVal("offline_mode", arg, &FLAG_offlineMode) || + GetFlagArgVal("landmarks_126", arg, &FLAG_isNumLandmarks126) || GetFlagArgVal("capture_outputs", arg, &FLAG_captureOutputs) || GetFlagArgVal("cam_res", arg, &FLAG_camRes) || GetFlagArgVal("codec", arg, &FLAG_captureCodec) || GetFlagArgVal("landmarks", arg, &FLAG_landmarks) || GetFlagArgVal("model_path", arg, &FLAG_modelPath) || GetFlagArgVal("wireframe_mesh", arg, &FLAG_proxyWireframe) || - GetFlagArgVal("temporal", arg, &FLAG_temporal))) { + GetFlagArgVal("face_model", arg, &FLAG_faceModel) || + GetFlagArgVal("app_mode", arg, &FLAG_appMode) || GetFlagArgVal("temporal", arg, &FLAG_temporal))) { continue; } else if (GetFlagArgVal("help", arg, &help)) { Usage(); @@ -251,7 +257,13 @@ std::string getCalendarTime() { class DoApp { public: enum Err { - errNone, + errNone = FaceEngine::Err::errNone, + errGeneral = FaceEngine::Err::errGeneral, + errRun = FaceEngine::Err::errRun, + errInitialization = FaceEngine::Err::errInitialization, + errRead = FaceEngine::Err::errRead, + errEffect = FaceEngine::Err::errEffect, + errParameter = FaceEngine::Err::errParameter, errUnimplemented, errMissing, errVideo, @@ -268,15 +280,15 @@ class DoApp { errSDK, errCuda, errCancel, - errInitFaceEngine + errCamera }; - + Err doAppErr(FaceEngine::Err status) { return (Err)status; } FaceEngine face_ar_engine; DoApp(); ~DoApp(); void stop(); - Err initFaceEngine(const char *modelPath = nullptr); + Err initFaceEngine(const char *modelPath = nullptr, bool isLandmarks126 = false); Err initCamera(const char *camRes = nullptr); Err initOfflineMode(const char *inputFilename = nullptr, const char *outputFilename = nullptr); Err acquireFrame(); @@ -287,7 +299,7 @@ class DoApp { void showFaceFitErrorMessage(); void drawFPS(cv::Mat &img); void DrawBBoxes(const cv::Mat &src, NvAR_Rect *output_bbox); - void DrawLandmarkPoints(const cv::Mat &src, NvAR_Point2f *facial_landmarks); + void DrawLandmarkPoints(const cv::Mat &src, NvAR_Point2f *facial_landmarks, int numLandmarks); void DrawFaceMesh(const cv::Mat &src, NvAR_FaceMesh *face_mesh); void drawKalmanStatus(cv::Mat &img); void drawVideoCaptureStatus(cv::Mat &img); @@ -320,7 +332,7 @@ class DoApp { }; DoApp *gApp = nullptr; -const char DoApp::windowTitle[] = "WINDOW"; +const char DoApp::windowTitle[] = "FaceTrack App"; void DoApp::processKey(int key) { switch (key) { @@ -368,21 +380,23 @@ void DoApp::processKey(int key) { } } -DoApp::Err DoApp::initFaceEngine(const char *modelPath) { +DoApp::Err DoApp::initFaceEngine(const char *modelPath, bool isNumLandmarks126) { Err err = errNone; if (!cap.isOpened()) return errVideo; + int numLandmarkPoints = isNumLandmarks126 ? 126 : 68; + face_ar_engine.setNumLandmarks(numLandmarkPoints); + nvErr = face_ar_engine.createFeatures(modelPath); if (nvErr != FaceEngine::Err::errNone) { if (nvErr == FaceEngine::Err::errInitialization && face_ar_engine.appMode == FaceEngine::mode::faceMeshGeneration) { showFaceFitErrorMessage(); + printf("WARNING: face fitting has failed, trying to initialize Landmark Detection\n"); face_ar_engine.destroyFeatures(); face_ar_engine.setAppMode(FaceEngine::mode::landmarkDetection); nvErr = face_ar_engine.createFeatures(modelPath); } - if (nvErr != FaceEngine::Err::errNone) - err = errInitFaceEngine; } #ifdef DEBUG @@ -396,7 +410,7 @@ DoApp::Err DoApp::initFaceEngine(const char *modelPath) { frameIndex = 0; - return err; + return doAppErr(nvErr); } void DoApp::stop() { @@ -428,29 +442,29 @@ void DoApp::showFaceFitErrorMessage() { } void DoApp::DrawBBoxes(const cv::Mat &src, NvAR_Rect *output_bbox) { - cv::Mat frame; + cv::Mat frm; if (FLAG_offlineMode) - frame = src.clone(); + frm = src.clone(); else - frame = src; + frm = src; if (output_bbox) - cv::rectangle(frame, cv::Point((int)output_bbox->x, (int)output_bbox->y), - cv::Point((int)output_bbox->x + output_bbox->width, (int)output_bbox->y + output_bbox->height), + cv::rectangle(frm, cv::Point(lround(output_bbox->x), lround(output_bbox->y)), + cv::Point(lround(output_bbox->x + output_bbox->width), lround(output_bbox->y + output_bbox->height)), cv::Scalar(255, 0, 0), 2); - if (FLAG_offlineMode) faceDetectOutputVideo.write(frame); + if (FLAG_offlineMode) faceDetectOutputVideo.write(frm); } -void DoApp::writeVideoAndEstResults(const cv::Mat &frame, NvAR_BBoxes output_bboxes, NvAR_Point2f *landmarks) { +void DoApp::writeVideoAndEstResults(const cv::Mat &frm, NvAR_BBoxes output_bboxes, NvAR_Point2f *landmarks) { if (captureVideo) { if (!capturedVideo.isOpened()) { const std::string currentCalendarTime = getCalendarTime(); const std::string capturedOutputFileName = currentCalendarTime + ".mp4"; getFPS(); if (frameTime) { - float fps = 1. / frameTime; + float fps = (float)(1.0 / frameTime); capturedVideo.open(capturedOutputFileName, StringToFourcc(FLAG_captureCodec), fps, - cv::Size(frame.cols, frame.rows)); + cv::Size(frm.cols, frm.rows)); if (!capturedVideo.isOpened()) { std::cout << "Error: Could not open video: \"" << capturedOutputFileName << "\"\n"; return; @@ -473,7 +487,7 @@ void DoApp::writeVideoAndEstResults(const cv::Mat &frame, NvAR_BBoxes output_bbo << "// kNumFaces, (bbox_x, bbox_y, bbox_w, bbox_h){ kNumFaces}, kNumLMs, [lm_x, lm_y]{kNumLMs}\n"; } // Write each frame to the Video - capturedVideo << frame; + capturedVideo << frm; writeEstResults(faceEngineVideoOutputFile, output_bboxes, landmarks); } else { if (capturedVideo.isOpened()) { @@ -517,11 +531,12 @@ void DoApp::writeEstResults(std::ofstream &outputFile, NvAR_BBoxes output_bboxes outputFile << "0,"; } if (landmarkDetectOn && output_bboxes.num_boxes) { + int numLandmarks = face_ar_engine.getNumLandmarks(); // Append number of landmarks - outputFile << FaceEngine::NUM_LANDMARKS << ","; - // Append NUM_LANDMARKS * 2 points + outputFile << numLandmarks << ","; + // Append 2 * number of landmarks values NvAR_Point2f *pt, *endPt; - for (endPt = (pt = (NvAR_Point2f *)landmarks) + FaceEngine::NUM_LANDMARKS; pt < endPt; ++pt) + for (endPt = (pt = (NvAR_Point2f *)landmarks) + numLandmarks; pt < endPt; ++pt) outputFile << pt->x << "," << pt->y << ","; } else { outputFile << "0,"; @@ -530,11 +545,11 @@ void DoApp::writeEstResults(std::ofstream &outputFile, NvAR_BBoxes output_bboxes outputFile << "\n"; } -void DoApp::writeFrameAndEstResults(const cv::Mat &frame, NvAR_BBoxes output_bboxes, NvAR_Point2f *landmarks) { +void DoApp::writeFrameAndEstResults(const cv::Mat &frm, NvAR_BBoxes output_bboxes, NvAR_Point2f *landmarks) { if (captureFrame) { const std::string currentCalendarTime = getCalendarTime(); const std::string capturedFrame = currentCalendarTime + ".png"; - cv::imwrite(capturedFrame, frame); + cv::imwrite(capturedFrame, frm); if (FLAG_verbose) { std::cout << "Captured the frame" << std::endl; } @@ -555,19 +570,19 @@ void DoApp::writeFrameAndEstResults(const cv::Mat &frame, NvAR_BBoxes output_bbo } } -void DoApp::DrawLandmarkPoints(const cv::Mat &src, NvAR_Point2f *facial_landmarks) { - cv::Mat frame; +void DoApp::DrawLandmarkPoints(const cv::Mat &src, NvAR_Point2f *facial_landmarks, int numLandmarks) { + cv::Mat frm; if (FLAG_offlineMode) - frame = src.clone(); + frm = src.clone(); else - frame = src; + frm = src; NvAR_Point2f *pt, *endPt; - for (endPt = (pt = (NvAR_Point2f *)facial_landmarks) + FaceEngine::NUM_LANDMARKS; pt < endPt; ++pt) - cv::circle(frame, cv::Point(lround(pt->x), lround(pt->y)), 1, cv::Scalar(0, 0, 255), -1); + for (endPt = (pt = (NvAR_Point2f *)facial_landmarks) + numLandmarks; pt < endPt; ++pt) + cv::circle(frm, cv::Point(lround(pt->x), lround(pt->y)), 1, cv::Scalar(0, 0, 255), -1); NvAR_Quaternion *pose = face_ar_engine.getPose(); if (pose) - face_ar_engine.DrawPose(frame, pose); - if (FLAG_offlineMode) landMarkOutputVideo.write(frame); + face_ar_engine.DrawPose(frm, pose); + if (FLAG_offlineMode) landMarkOutputVideo.write(frm); } void DoApp::DrawFaceMesh(const cv::Mat &src, NvAR_FaceMesh *face_mesh) { @@ -608,12 +623,14 @@ DoApp::Err DoApp::acquireFrame() { // frames we try to read are empty. So we try to re-initialize the camera with the same resolution settings. If the // resolution has changed, you will need to destroy and create the features again with the new camera resolution (not // done here) as well as reallocate memory accordingly with FaceEngine::initFeatureIOParams() - cap >> frame; // get a new frame from camera + cap >> frame; // get a new frame from camera into the class variable frame. if (frame.empty()) { // if in Offline mode, this means end of video,so we return if (FLAG_offlineMode) return errVideo; // try Init one more time if reading frames from camera - initCamera(FLAG_camRes.c_str()); + err = initCamera(FLAG_camRes.c_str()); + if (err != errNone) + return err; cap >> frame; if (frame.empty()) return errVideo; } @@ -653,30 +670,30 @@ DoApp::Err DoApp::acquireFaceBox() { DoApp::Err DoApp::acquireFaceBoxAndLandmarks() { Err err = errNone; - + int numLandmarks = face_ar_engine.getNumLandmarks(); NvAR_Rect output_bbox; - NvAR_Point2f facial_landmarks[FaceEngine::NUM_LANDMARKS]; + std::vector facial_landmarks(numLandmarks); // get landmarks in original image resolution coordinate space - unsigned n = face_ar_engine.acquireFaceBoxAndLandmarks(frame, facial_landmarks, output_bbox, 0); + unsigned n = face_ar_engine.acquireFaceBoxAndLandmarks(frame, facial_landmarks.data(), output_bbox, 0); if (n && FLAG_verbose && face_ar_engine.appMode != FaceEngine::mode::faceDetection) { printf("Landmarks: [\n"); - NvAR_Point2f *pt, *endPt; - for (endPt = (pt = (NvAR_Point2f *)facial_landmarks) + FaceEngine::NUM_LANDMARKS; pt < endPt; ++pt) - printf("%7.1f%7.1f\n", pt->x, pt->y); + for (const auto &pt : facial_landmarks) { + printf("%7.1f%7.1f\n", pt.x, pt.y); + } printf("]\n"); } if (FLAG_captureOutputs) { - writeFrameAndEstResults(frame, face_ar_engine.output_bboxes, facial_landmarks); - writeVideoAndEstResults(frame, face_ar_engine.output_bboxes, facial_landmarks); + writeFrameAndEstResults(frame, face_ar_engine.output_bboxes, facial_landmarks.data()); + writeVideoAndEstResults(frame, face_ar_engine.output_bboxes, facial_landmarks.data()); } if (0 == n) return errNoFace; #ifdef VISUALIZE if (drawVisualization) { - DrawLandmarkPoints(frame, facial_landmarks); + DrawLandmarkPoints(frame, facial_landmarks.data(), numLandmarks); if (FLAG_offlineMode) { DrawBBoxes(frame, &output_bbox); } @@ -707,24 +724,25 @@ DoApp::Err DoApp::initCamera(const char *camRes) { if (inputWidth) cap.set(CV_CAP_PROP_FRAME_WIDTH, inputWidth); if (inputHeight) cap.set(CV_CAP_PROP_FRAME_HEIGHT, inputHeight); - inputWidth = cap.get(CV_CAP_PROP_FRAME_WIDTH); - inputHeight = cap.get(CV_CAP_PROP_FRAME_HEIGHT); + inputWidth = (int)cap.get(CV_CAP_PROP_FRAME_WIDTH); + inputHeight = (int)cap.get(CV_CAP_PROP_FRAME_HEIGHT); face_ar_engine.setInputImageWidth(inputWidth); face_ar_engine.setInputImageHeight(inputHeight); } } else - return errVideo; + return errCamera; return errNone; } DoApp::Err DoApp::initOfflineMode(const char *inputFilename, const char *outputFilename) { if (cap.open(inputFilename)) { - inputWidth = cap.get(CV_CAP_PROP_FRAME_WIDTH); - inputHeight = cap.get(CV_CAP_PROP_FRAME_HEIGHT); + inputWidth = (int)cap.get(CV_CAP_PROP_FRAME_WIDTH); + inputHeight = (int)cap.get(CV_CAP_PROP_FRAME_HEIGHT); face_ar_engine.setInputImageWidth(inputWidth); face_ar_engine.setInputImageHeight(inputHeight); } else { - return Err::errNotFound; + printf("ERROR: Unable to open the input video file \"%s\" \n", inputFilename); + return Err::errVideo; } std::string fdOutputVideoName, fldOutputVideoName, ffOutputVideoName; @@ -740,14 +758,20 @@ DoApp::Err DoApp::initOfflineMode(const char *inputFilename, const char *outputF ffOutputVideoName = outputFilePrefix + "_faceModel.mp4"; if (!faceDetectOutputVideo.open(fdOutputVideoName, StringToFourcc(FLAG_captureCodec), cap.get(CV_CAP_PROP_FPS), - cv::Size(inputWidth, inputHeight))) - return Err::errSDK; + cv::Size(inputWidth, inputHeight))) { + printf("ERROR: Unable to open the output video file \"%s\" \n", fdOutputVideoName.c_str()); + return Err::errGeneral; + } if (!landMarkOutputVideo.open(fldOutputVideoName, StringToFourcc(FLAG_captureCodec), cap.get(CV_CAP_PROP_FPS), - cv::Size(inputWidth, inputHeight))) - return Err::errSDK; + cv::Size(inputWidth, inputHeight))) { + printf("ERROR: Unable to open the output video file \"%s\" \n", fldOutputVideoName.c_str()); + return Err::errGeneral; + } if (!faceFittingOutputVideo.open(ffOutputVideoName, StringToFourcc(FLAG_captureCodec), cap.get(CV_CAP_PROP_FPS), - cv::Size(inputWidth, inputHeight))) - return Err::errSDK; + cv::Size(inputWidth, inputHeight))) { + printf("ERROR: Unable to open the output video file \"%s\" \n", ffOutputVideoName.c_str()); + return Err::errGeneral; + } return Err::errNone; } @@ -766,7 +790,7 @@ DoApp::Err DoApp::fitFaceModel() { if (drawVisualization) { DrawFaceMesh(frame, face_ar_engine.getFaceMesh()); if (FLAG_offlineMode) { - DrawLandmarkPoints(frame, face_ar_engine.getLandmarks()); + DrawLandmarkPoints(frame, face_ar_engine.getLandmarks(), face_ar_engine.getNumLandmarks()); DrawBBoxes(frame, face_ar_engine.getLargestBox()); } } @@ -835,11 +859,20 @@ void DoApp::drawVideoCaptureStatus(cv::Mat &img) { DoApp::Err DoApp::run() { DoApp::Err doErr = errNone; - face_ar_engine.initFeatureIOParams(); + FaceEngine::Err err = face_ar_engine.initFeatureIOParams(); + if (err != FaceEngine::Err::errNone ) { + return doAppErr(err); + } while (1) { doErr = acquireFrame(); - if (doErr != DoApp::errNone) return doErr; + if (frame.empty() && FLAG_offlineMode) { + // We have reached the end of the video + // so return without any error. + return DoApp::errNone; + } else if (doErr != DoApp::errNone) { + return doErr; + } if (face_ar_engine.appMode == FaceEngine::mode::faceDetection) { doErr = acquireFaceBox(); } else if (face_ar_engine.appMode == FaceEngine::mode::landmarkDetection) { @@ -898,6 +931,12 @@ const char *DoApp::errorStringFromCode(DoApp::Err code) { }; static const LUTEntry lut[] = { {errNone, "no error"}, + {errGeneral, "an error has occured"}, + {errRun, "an error has occured while the feature is running"}, + {errInitialization, "Initializing Face Engine failed"}, + {errRead, "an error has occured while reading a file"}, + {errEffect, "an error has occured while creating a feature"}, + {errParameter, "an error has occured while setting a parameter for a feature"}, {errUnimplemented, "the feature is unimplemented"}, {errMissing, "missing input parameter"}, {errVideo, "no video source has been found"}, @@ -914,7 +953,7 @@ const char *DoApp::errorStringFromCode(DoApp::Err code) { {errSDK, "an SDK error has occurred"}, {errCuda, "a CUDA error has occurred"}, {errCancel, "the user cancelled"}, - {errInitFaceEngine, "an error occurred while initializing the Face Engine"}, + {errCamera, "unable to connect to the camera"}, }; for (const LUTEntry *p = lut; p < &lut[sizeof(lut) / sizeof(lut[0])]; ++p) if (p->code == code) return p->str; @@ -930,19 +969,23 @@ const char *DoApp::errorStringFromCode(DoApp::Err code) { int main(int argc, char **argv) { DoApp app; DoApp::Err doErr; - NvCV_Status nvErr; // Parse the arguments if (0 != ParseMyArgs(argc, argv)) return -100; + app.face_ar_engine.setAppMode(FaceEngine::mode(FLAG_appMode)); + if (FLAG_verbose) printf("Enable temporal optimizations in detecting face and landmarks = %d\n", FLAG_temporal); app.face_ar_engine.setFaceStabilization(FLAG_temporal); doErr = DoApp::errFaceModelInit; if (FLAG_modelPath.empty()) { - printf("WARNING: Model path not specified. Please set --model_path=/path/to/trt/and/face/models\n" - "SDK will attempt to load the models from NVAR_MODEL_DIR environment variable"); + printf("WARNING: Model path not specified. Please set --model_path=/path/to/trt/and/face/models, " + "SDK will attempt to load the models from NVAR_MODEL_DIR environment variable, " + "please restart your application after the SDK Installation. \n"); } + if (!FLAG_faceModel.empty()) + app.face_ar_engine.setFaceModel(FLAG_faceModel.c_str()); if (FLAG_offlineMode) { if (FLAG_inFile.empty()) { @@ -950,21 +993,23 @@ int main(int argc, char **argv) { printf("ERROR: %s, please specify input file using --in_file or --in \n", app.errorStringFromCode(doErr)); goto bail; } - app.initOfflineMode(FLAG_inFile.c_str(), FLAG_outFile.c_str()); + doErr = app.initOfflineMode(FLAG_inFile.c_str(), FLAG_outFile.c_str()); } else { - app.initCamera(FLAG_camRes.c_str()); - } - doErr = app.initFaceEngine(FLAG_modelPath.c_str()); - if (DoApp::errNone != doErr) { - printf("ERROR: %s\n", app.errorStringFromCode(doErr)); - goto bail; + doErr = app.initCamera(FLAG_camRes.c_str()); } + BAIL_IF_ERR(doErr); + + doErr = app.initFaceEngine(FLAG_modelPath.c_str(), FLAG_isNumLandmarks126); + BAIL_IF_ERR(doErr); if (!FLAG_proxyWireframe.empty()) app.setProxyWireframe(FLAG_proxyWireframe.c_str()); doErr = app.run(); + BAIL_IF_ERR(doErr); bail: + if(doErr) + printf("ERROR: %s\n", app.errorStringFromCode(doErr)); app.stop(); return (int)doErr; } diff --git a/samples/FaceTrack/FaceTrack.exe b/samples/FaceTrack/FaceTrack.exe index e149aed..4d57e8f 100644 Binary files a/samples/FaceTrack/FaceTrack.exe and b/samples/FaceTrack/FaceTrack.exe differ diff --git a/samples/FaceTrack/Readme.txt b/samples/FaceTrack/Readme.txt index 4dc0475..e1cc66d 100644 --- a/samples/FaceTrack/Readme.txt +++ b/samples/FaceTrack/Readme.txt @@ -47,4 +47,5 @@ Either the forward slash (/) or back slash (\) can be used as a separator betwee The ConvertSurreyFaceModel.exe file is distributed in the https://github.com/nvidia/BROADCAST-AR-SDK repo. 3) The sample application provided with NVIDIA AR SDK requires that the model file be named face_model0.nvf. -Place the face_model0.nvf file in the /bin/models folder. +Place the face_model0.nvf file in the model folder. By default the models folder is your_sdk_install_path/models +where all models (including *.trtpkg files) are installed, for example: C:\Program Files\NVIDIA Corporation\NVIDIA AR SDK\models. diff --git a/samples/utils/FeatureVertexName.cpp b/samples/utils/FeatureVertexName.cpp new file mode 100644 index 0000000..c811647 --- /dev/null +++ b/samples/utils/FeatureVertexName.cpp @@ -0,0 +1,187 @@ +/*############################################################################### +# +# Copyright(c) 2019 NVIDIA CORPORATION.All Rights Reserved. +# +# NVIDIA CORPORATION and its licensors retain all intellectual property +# and proprietary rights in and to this software, related documentation +# and any modifications thereto.Any use, reproduction, disclosure or +# distribution of this software and related documentation without an express +# license agreement from NVIDIA CORPORATION is strictly prohibited. +# +###############################################################################*/ + +#include "FeatureVertexName.h" +#include +#include + + +// TODO: We should read this from a file +const LandmarkEOSMap LandmarkMapEOS[] = { + { 33, "chin bottom" }, + { 225, "right eyebrow outer-corner" }, + { 229, "right eyebrow between middle and outer corner" }, + { 233, "right eyebrow middle, vertical middle" }, + { 2086, "right eyebrow between middle and inner corner" }, + { 157, "right eyebrow inner-corner" }, + { 590, "left eyebrow inner-corner" }, + { 2091, "left eyebrow between inner corner and middle" }, + { 666, "left eyebrow middle" }, + { 662, "left eyebrow between middle and outer corner" }, + { 658, "left eyebrow outer-corner" }, + { 2842, "bridge of the nose (parallel to upper eye lids)" }, + { 379, "middle of the nose, a bit below the lower eye lids" }, + { 272, "above nose-tip (1cm or so)" }, + { 114, "nose-tip" }, + { 100, "right nostril, below nose, nose-lip junction" }, + { 2794, "nose-lip junction" }, + { 270, "nose-lip junction" }, + { 2797, "nose-lip junction" }, + { 537, "left nostril, below nose, nose-lip junction" }, + { 177, "right eye outer-corner" }, + { 172, "right eye pupil top right (from subject's perspective)" }, + { 191, "right eye pupil top left" }, + { 181, "right eye inner-corner" }, + { 173, "right eye pupil bottom left" }, + { 174, "right eye pupil bottom right" }, + { 614, "left eye inner-corner" }, + { 624, "left eye pupil top right" }, + { 605, "left eye pupil top left" }, + { 610, "left eye outer-corner" }, + { 607, "left eye pupil bottom left" }, + { 606, "left eye pupil bottom right" }, + { 398, "right mouth corner" }, + { 315, "upper lip right top outer" }, + { 413, "upper lip middle top right" }, + { 329, "upper lip middle top" }, + { 825, "upper lip middle top left" }, + { 736, "upper lip left top outer" }, + { 812, "left mouth corner" }, + { 841, "lower lip left bottom outer" }, + { 693, "lower lip middle bottom left" }, + { 411, "lower lip middle bottom" }, + { 264, "lower lip middle bottom right" }, + { 431, "lower lip right bottom outer" }, + { 416, "upper lip right bottom outer" }, + { 423, "upper lip middle bottom" }, + { 828, "upper lip left bottom outer" }, + { 817, "lower lip left top outer" }, + { 442, "lower lip middle top" }, + { 404, "lower lip right top outer" }, + { 0xFFFF, nullptr } +}; + +static const LandmarksMap LandmarkMap[] = { + { 0, 0, "right contour point 1" }, + { 1, 2, "right contour point 2" }, + { 2, 4, "right contour point 3" }, + { 3, 6, "right contour point 4" }, + { 4, 8, "right contour point 5" }, + { 5, 10, "right contour point 6" }, + { 6, 12, "right contour point 7" }, + { 7, 14, "right contour point 8" }, + { 8, 16, "chin bottom" }, + { 9, 18, "left contour point 1" }, + { 10, 20, "left contour point 2" }, + { 11, 22, "left contour point 3" }, + { 12, 24, "left contour point 4" }, + { 13, 26, "left contour point 5" }, + { 14, 28, "left contour point 6" }, + { 15, 30, "left contour point 7" }, + { 16, 32, "left contour point 8" }, + { 17, 33, "right eyebrow outer-corner" }, + { 18, 34, "right eyebrow between middle and outer corner" }, + { 19, 35, "right eyebrow middle, vertical middle" }, + { 20, 36, "right eyebrow between middle and inner corner" }, + { 21, 37, "right eyebrow inner-corner" }, + { 22, 42, "left eyebrow inner-corner" }, + { 23, 43, "left eyebrow between inner corner and middle" }, + { 24, 44, "left eyebrow middle" }, + { 25, 45, "left eyebrow between middle and outer corner" }, + { 26, 46, "left eyebrow outer-corner" }, + { 27, 51, "bridge of the nose (parallel to upper eye lids)" }, + { 28, 52, "middle of the nose, a bit below the lower eye lids" }, + { 29, 53, "above nose-tip (1cm or so)" }, + { 30, 54, "nose-tip" }, + { 31, 57, "right nostril, below nose, nose-lip junction" }, + { 32, 58, "nose-lip junction" }, + { 33, 59, "nose-lip junction" }, + { 34, 60, "nose-lip junction" }, + { 35, 61, "left nostril, below nose, nose-lip junction" }, + { 36, 64, "right eye outer-corner" }, + { 37, 65, "right eye pupil top right (from subject's perspective)" }, + { 38, 67, "right eye pupil top left" }, + { 39, 68, "right eye inner-corner" }, + { 40, 69, "right eye pupil bottom left" }, + { 41, 71, "right eye pupil bottom right" }, + { 42, 81, "left eye inner-corner" }, + { 43, 82, "left eye pupil top right" }, + { 44, 84, "left eye pupil top left" }, + { 45, 85, "left eye outer-corner" }, + { 46, 86, "left eye pupil bottom left" }, + { 47, 88, "left eye pupil bottom right" }, + { 48, 98, "right mouth corner" }, + { 49, 99, "upper lip right top outer" }, + { 50, 100, "upper lip middle top right" }, + { 51, 101, "upper lip middle top" }, + { 52, 102, "upper lip middle top left" }, + { 53, 103, "upper lip left top outer" }, + { 54, 104, "left mouth corner" }, + { 55, 105, "lower lip left bottom outer" }, + { 56, 106, "lower lip middle bottom left" }, + { 57, 107, "lower lip middle bottom" }, + { 58, 108, "lower lip middle bottom right" }, + { 59, 109, "lower lip right bottom outer" }, + { 60, 110, "right inner mouth corner "}, + { 61, 111, "upper lip right bottom outer" }, + { 62, 112, "upper lip middle bottom" }, + { 63, 113, "upper lip left bottom outer" }, + { 64, 114, "left inner mouth corner"}, + { 65, 115, "lower lip left top outer" }, + { 66, 116, "lower lip middle top" }, + { 67, 117, "lower lip right top outer" }, + { 0xFFFF, 0xFFFF, nullptr } +}; + + +unsigned short FindEOSLandmarkIndexFromName(const char* name) +{ + if (!name) + return 0xFFFF; + switch (name[0]) { + case '#': // 1-based index ... + return (unsigned short)(strtol(name + 1, nullptr, 10) - 1); // ... gets converted into a 0-based index + case '@': // 0-based index + return (unsigned short)strtol(name + 1, nullptr, 10); + default: + break; + } + const LandmarkEOSMap* lmList = LandmarkMapEOS; + for (; lmList->name != nullptr; ++lmList) + if (!strcmp(name, lmList->name)) + break; + return lmList->index; +} + +unsigned short FindLandmarkIndexFromName(const unsigned int numLandmarks, const char* name) { + if (!name) + return 0xFFFF; + switch (name[0]) { + case '#': // 1-based index ... + return (unsigned short)(strtol(name + 1, nullptr, 10) - 1); // ... gets converted into a 0-based index + case '@': // 0-based index + return (unsigned short)strtol(name + 1, nullptr, 10); + default: + break; + } + const LandmarksMap* lmList = LandmarkMap; + for (; lmList->name != nullptr; ++lmList) + if (!strcmp(name, lmList->name)) + break; + if (numLandmarks == 68) { + return lmList->index_68; + } else if (numLandmarks == 126) { + return lmList->index_126; + } else { + return 0xFFFF; + } +} diff --git a/samples/utils/FeatureVertexName.h b/samples/utils/FeatureVertexName.h new file mode 100644 index 0000000..d61acc3 --- /dev/null +++ b/samples/utils/FeatureVertexName.h @@ -0,0 +1,23 @@ +/*############################################################################### +# +# Copyright(c) 2019 NVIDIA CORPORATION.All Rights Reserved. +# +# NVIDIA CORPORATION and its licensors retain all intellectual property +# and proprietary rights in and to this software, related documentation +# and any modifications thereto.Any use, reproduction, disclosure or +# distribution of this software and related documentation without an express +# license agreement from NVIDIA CORPORATION is strictly prohibited. +# +###############################################################################*/ + +#ifndef __FEATURE_VERTEX_NAME__ +#define __FEATURE_VERTEX_NAME__ + + +struct LandmarkEOSMap { unsigned short index; const char *name; }; +struct LandmarksMap { unsigned short index_68; unsigned short index_126; const char* name; }; + +unsigned short FindEOSLandmarkIndexFromName(const char *name); +unsigned short FindLandmarkIndexFromName(const unsigned int numLandmarks, const char* name); + +#endif /* __FEATURE_VERTEX_NAME__ */ diff --git a/samples/utils/nvCVOpenCV.h b/samples/utils/nvCVOpenCV.h index 188012f..cdfbe1a 100644 --- a/samples/utils/nvCVOpenCV.h +++ b/samples/utils/nvCVOpenCV.h @@ -21,6 +21,9 @@ # ###############################################################################*/ +#ifndef __NVCVOPENCV_H__ +#define __NVCVOPENCV_H__ + #include "nvCVImage.h" #include "opencv2/opencv.hpp" @@ -73,3 +76,5 @@ inline void NVWrapperForCVMat(const cv::Mat *cvIm, NvCVImage *nvcvIm) { nvcvIm->reserved[0] = 0; nvcvIm->reserved[1] = 0; } + +#endif // __NVCVOPENCV_H__ \ No newline at end of file diff --git a/tools/ConvertSurreyFaceModel.exe b/tools/ConvertSurreyFaceModel.exe index 0ae6758..ab36c7b 100644 Binary files a/tools/ConvertSurreyFaceModel.exe and b/tools/ConvertSurreyFaceModel.exe differ diff --git a/version.h b/version.h index 2a24909..75f9976 100644 --- a/version.h +++ b/version.h @@ -22,12 +22,12 @@ ###############################################################################*/ #define NVIDIA_AR_SDK_VERSION_MAJOR 0 -#define NVIDIA_AR_SDK_VERSION_MINOR 5 -#define NVIDIA_AR_SDK_VERSION_RELEASE 0 +#define NVIDIA_AR_SDK_VERSION_MINOR 6 +#define NVIDIA_AR_SDK_VERSION_RELEASE 1 -#define NVIDIA_AR_SDK_VERSION 0,5,0,0 -#define NVIDIA_AR_SDK_VERSION_MAJOR_MINOR 0,5 -#define NVIDIA_AR_SDK_VERSION_STRING "0.5.0.0" -#define NVIDIA_AR_SDK_VERSION_STRING_SHORT "0.5.0" -#define NVIDIA_AR_SDK_VERSION_STRING_MAJOR_MINOR "0.5" +#define NVIDIA_AR_SDK_VERSION 0,6,1,0 +#define NVIDIA_AR_SDK_VERSION_MAJOR_MINOR 0,6 +#define NVIDIA_AR_SDK_VERSION_STRING "0.6.1.0" +#define NVIDIA_AR_SDK_VERSION_STRING_SHORT "0.6.1" +#define NVIDIA_AR_SDK_VERSION_STRING_MAJOR_MINOR "0.6"