diff --git a/CMakeLists.txt b/CMakeLists.txt index de57aab..b232a98 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -24,6 +24,13 @@ set(SDK_INCLUDES_PATH ${CMAKE_CURRENT_SOURCE_DIR}/nvar/include) add_library(nvARPose INTERFACE) target_include_directories(nvARPose INTERFACE ${SDK_INCLUDES_PATH}) +# Add target for NVCVImage +add_library(NVCVImage INTERFACE) +target_include_directories(NVCVImage INTERFACE ${SDK_INCLUDES_PATH}) +if(UNIX) + target_link_libraries(NVCVImage INTERFACE ${CMAKE_CURRENT_SOURCE_DIR}/bin/libNVCVImage.so) +endif(UNIX) + set(ENABLE_SAMPLES TRUE) add_subdirectory(samples) diff --git a/LICENSE b/LICENSE index 93bd1cc..58871f4 100644 --- a/LICENSE +++ b/LICENSE @@ -1,6 +1,6 @@ The MIT License (MIT) -Copyright (c) 2020 NVIDIA Corporation +Copyright (c) 2021 NVIDIA Corporation Permission is hereby granted, free of charge, to any person obtaining a copy of this software and associated documentation files (the "Software"), to deal in diff --git a/README.MD b/README.MD index 98b6752..bfd8fbd 100644 --- a/README.MD +++ b/README.MD @@ -1,44 +1,50 @@ # README -## NVIDIA AR SDK: API Source Code and Sample Applications +## NVIDIA MAXINE AR SDK: API Source Code and Sample Applications -NVIDIA AR SDK enables real-time modeling and tracking of human faces from video. The SDK is powered by NVIDIA graphics processing units (GPUs) with Tensor Cores, and as a result, the algorithm throughput is greatly accelerated, and latency is reduced. +NVIDIA MAXINE AR SDK enables real-time modeling and tracking of human faces from video. The SDK is powered by NVIDIA graphics processing units (GPUs) with Tensor Cores, and as a result, the algorithm throughput is greatly accelerated, and latency is reduced. -NVIDIA AR SDK has the following features: +The SDK has the following features: - **Face detection and tracking**, which detects, localizes, and tracks human faces in images or videos by using bounding boxes. - **Facial landmark detection and tracking**, which predicts and tracks the pixel locations of human facial landmark points and head poses in images or videos. It can predict 68 and 126 landmark points. The 68 detected facial landmarks follow the _Multi-PIE 68 point mark-ups_ information in [facial point annotations](https://ibug.doc.ic.ac.uk/resources/facial-point-annotations/). The 126 landmark points detector can predict more points on the cheeks, the eyes, and on laugh lines. - **Face 3D mesh and tracking**, which reconstructs and tracks a 3D human face and its head pose from the provided facial landmarks. +- **3D Body Pose and tracking**, which predicts and tracks the 3D human pose from images or videos. It predicts 34 keypoints of body pose in 2D and 3D.

-Face detection and tracking -Facial landmark detection and tracking - 68 pts +Face detection and tracking +Facial landmark detection and tracking - 68 pts

-Facial landmark detection and tracking - 126 pts -Face 3D mesh and tracking +Facial landmark detection and tracking - 126 pts +Face 3D mesh and tracking +

+

+Body 3D Pose and tracking

-NVIDIA AR SDK provides a sample application that demonstrates the features listed above in real time by using a webcam or offline videos. +The SDK provides two sample applications that demonstrate the features listed above in real time by using a webcam or offline videos. +- **FaceTrack App** which demonstrates the face tracking, landmark tracking and 3D mesh tracking features. +- **BodyTrack App** which demonstrates the 3D Body Pose tracking feature. -NVIDIA AR SDK is distributed in the following parts: +NVIDIA MAXINE AR SDK is distributed in the following parts: -- This open source repository that includes the [SDK API and proxy linking source code](https://github.com/NVIDIA/BROADCAST-AR-SDK/tree/master/nvar), and [sample applications and their dependency libraries](https://github.com/NVIDIA/BROADCAST-AR-SDK/tree/master/samples). -- An installer hosted on [RTX broadcast engine developer page](https://developer.nvidia.com/rtx-broadcast-engine) that installs the SDK DLLs, the models, and the SDK dependency libraries. +- This open source repository that includes the [SDK API and proxy linking source code](https://github.com/NVIDIA/MAXINE-AR-SDK/tree/master/nvar), and [sample applications and their dependency libraries](https://github.com/NVIDIA/MAXINE-AR-SDK/tree/master/samples). +- An installer hosted on [NVIDIA Maxine developer page](https://www.nvidia.com/broadcast-sdk-resources) that installs the SDK DLLs, the models, and the SDK dependency libraries. -Please refer to [SDK programming guide](https://github.com/NVIDIA/BROADCAST-AR-SDK/blob/master/NVIDIA%20AR%20SDK%20Programming%20Guide.pdf) for configuring and integrating the SDK, compiling and running the sample applications. +Please refer to [SDK programming guide](https://github.com/NVIDIA/MAXINE-AR-SDK/blob/master/NVIDIA%20AR%20SDK%20Programming%20Guide.pdf) for configuring and integrating the SDK, compiling and running the sample applications. Please visit the [NVIDIA MAXINE AR SDK](https://developer.nvidia.com/maxine-getting-started) webpage for more information about the SDK. ## System requirements -The SDK is supported on NVIDIA GPUs that are based on the NVIDIA® Turing™ architecture. Although the SDK can run on Turing™ GPUs without Tensor Cores, it is optimized for much higher performance on GPUs with Tensor Cores. +The SDK is supported on NVIDIA GPUs that are based on the NVIDIA® Turing™ or Ampere™ architecture and have Tensor Cores. * Windows OS supported: 64-bit Windows 10 * Microsoft Visual Studio: 2015 (MSVC14.0) or later * CMake: v3.12 or later * NVIDIA Graphics Driver for Windows: 455.57 or later * NVIDIA CUDA Toolkit: 11.1 or later -* NVIDIA TensorRT: 7.2.0 or later +* NVIDIA TensorRT: 7.2.2.5 -## NVIDIA Branding Guidelines -If you integrate an NVIDIA Broadcast Engine SDK within your product, please follow the required branding guidelines that are available [here]( -https://nvidia.frontify.com/d/uAobRitG8H8B) +## NVIDIA MAXINE Branding Guidelines +If you integrate an NVIDIA MAXINE SDK within your product, please follow the required branding guidelines that are available [here]( +https://www.nvidia.com/maxine-sdk-guidelines) ## Compiling the sample app @@ -46,7 +52,7 @@ https://nvidia.frontify.com/d/uAobRitG8H8B) The open source repository includes the source code to build the sample application, and a proxy file nvARProxy.cpp to enable compilation without explicitly linking against the SDK DLL. -**Note: To download the models and runtime dependencies required by the features, you need to run the [SDK Installer](https://developer.nvidia.com/rtx-broadcast-engine).** +**Note: To download the models and runtime dependencies required by the features, you need to run the [SDK Installer](https://www.nvidia.com/broadcast-sdk-resources).** 1. In the root folder of the downloaded source code, start the CMake GUI and specify the source folder and a build folder for the binary files. * For the source folder, ensure that the path ends in OSS. @@ -58,7 +64,7 @@ The open source repository includes the source code to build the sample applicat * To complete configuring the Visual Studio solution file, click Finish. * To generate the Visual Studio Solution file, click Generate. * Verify that the build folder contains the NvAR_SDK.sln file. -3. Use Visual Studio to generate the FaceTrack.exe file from the NvAR_SDK.sln file. +3. Use Visual Studio to generate the FaceTrack.exe or BodyTrack.exe file from the NvAR_SDK.sln file. * In CMake, to open Visual Studio, click Open Project. * In Visual Studio, select Build > Build Solution. diff --git a/docs/NVIDIA AR SDK Programming Guide.pdf b/docs/NVIDIA AR SDK Programming Guide.pdf index 2a07b52..88073a4 100644 Binary files a/docs/NVIDIA AR SDK Programming Guide.pdf and b/docs/NVIDIA AR SDK Programming Guide.pdf differ diff --git a/nvar/include/nvAR_defs.h b/nvar/include/nvAR_defs.h index 04ea3d5..e7eb59f 100644 --- a/nvar/include/nvAR_defs.h +++ b/nvar/include/nvAR_defs.h @@ -23,20 +23,22 @@ #ifndef NvAR_DEFS_H #define NvAR_DEFS_H +#include #include #include #include #ifdef _WIN32 -#ifdef NVAR_API_EXPORT -#define NvAR_API __declspec(dllexport) __cdecl + #ifdef NVAR_API_EXPORT + #define NvAR_API __declspec(dllexport) __cdecl + #else + #define NvAR_API + #endif #else -#define NvAR_API -#endif -#elif linux -// TODO: Linux code goes here -#endif + #define NvAR_API +#endif // OS dependencies +// TODO: Change the representation to x,y,z instead of array typedef struct NvAR_Vector3f { float vec[3]; @@ -78,6 +80,10 @@ typedef struct NvAR_Point2f { float x, y; } NvAR_Point2f; +typedef struct NvAR_Point3f { + float x, y, z; +} NvAR_Point3f; + typedef struct NvAR_Vector2f { float x, y; } NvAR_Vector2f; @@ -93,10 +99,13 @@ typedef const char* NvAR_FeatureID; #define NvAR_Feature_FaceDetection "FaceDetection" #define NvAR_Feature_LandmarkDetection "LandmarkDetection" #define NvAR_Feature_Face3DReconstruction "Face3DReconstruction" +#define NvAR_Feature_BodyDetection "BodyDetection" +#define NvAR_Feature_BodyPoseEstimation "BodyPoseEstimation" #define NvAR_Parameter_Input(Name) "NvAR_Parameter_Input_" #Name #define NvAR_Parameter_Output(Name) "NvAR_Parameter_Output_" #Name #define NvAR_Parameter_Config(Name) "NvAR_Parameter_Config_" #Name +#define NvAR_Parameter_InOut(Name) "NvAR_Parameter_InOut_" #Name /* Parameters supported by each NvAR_FeatureID @@ -167,4 +176,5 @@ NvAR_Parameter_Output(ExpressionCoefficients) - OPTIONAL NvAR_Parameter_Output(ShapeEigenValues) - OPTIONAL */ + #endif // NvAR_DEFS_H diff --git a/nvar/include/nvCVImage.h b/nvar/include/nvCVImage.h index 7f07360..c5ed9ed 100644 --- a/nvar/include/nvCVImage.h +++ b/nvar/include/nvCVImage.h @@ -1,6 +1,6 @@ /*############################################################################### # -# Copyright 2020 NVIDIA Corporation +# Copyright 2020-2021 NVIDIA Corporation # # Permission is hereby granted, free of charge, to any person obtaining a copy of # this software and associated documentation files (the "Software"), to deal in @@ -30,6 +30,12 @@ extern "C" { #endif // ___cplusplus + +#ifndef RTX_CAMERA_IMAGE // Compile with -DRTX_CAMERA_IMAGE=0 to get more functionality and bug fixes. + #define RTX_CAMERA_IMAGE 0 // Set to 1 for RTXCamera, which needs an old version, that avoids new functionality +#endif // RTX_CAMERA_IMAGE + + struct CUstream_st; // typedef struct CUstream_st *CUstream; //! The format of pixels in an image. @@ -42,8 +48,16 @@ typedef enum NvCVImage_PixelFormat { NVCV_BGR = 5, //!< { Red, Green, Blue } NVCV_RGBA = 6, //!< { Red, Green, Blue, Alpha } NVCV_BGRA = 7, //!< { Red, Green, Blue, Alpha } +#if RTX_CAMERA_IMAGE NVCV_YUV420 = 8, //!< Luminance and subsampled Chrominance { Y, Cb, Cr } NVCV_YUV422 = 9, //!< Luminance and subsampled Chrominance { Y, Cb, Cr } +#else // !RTX_CAMERA_IMAGE + NVCV_ARGB = 8, //!< { Red, Green, Blue, Alpha } + NVCV_ABGR = 9, //!< { Red, Green, Blue, Alpha } + NVCV_YUV420 = 10, //!< Luminance and subsampled Chrominance { Y, Cb, Cr } + NVCV_YUV422 = 11, //!< Luminance and subsampled Chrominance { Y, Cb, Cr } +#endif // !RTX_CAMERA_IMAGE + NVCV_YUV444 = 12, //!< Luminance and full bandwidth Chrominance { Y, Cb, Cr } } NvCVImage_PixelFormat; @@ -80,35 +94,51 @@ typedef enum NvCVImage_ComponentType { #define NVCV_VYUY 4 //!< [VYUY] Chunky 4:2:2 #define NVCV_YUYV 6 //!< [YUYV] Chunky 4:2:2 #define NVCV_YVYU 8 //!< [YVYU] Chunky 4:2:2 -#define NVCV_YUV 3 //!< [Y][U][V] Planar 4:2:2 or 4:2:0 -#define NVCV_YVU 5 //!< [Y][V][U] Planar 4:2:2 or 4:2:0 +#define NVCV_CYUV 10 //!< [YUV] Chunky 4:4:4 +#define NVCV_CYVU 12 //!< [YVU] Chunky 4:4:4 +#define NVCV_YUV 3 //!< [Y][U][V] Planar 4:2:2 or 4:2:0 or 4:4:4 +#define NVCV_YVU 5 //!< [Y][V][U] Planar 4:2:2 or 4:2:0 or 4:4:4 #define NVCV_YCUV 7 //!< [Y][UV] Semi-planar 4:2:2 or 4:2:0 (default for 4:2:0) #define NVCV_YCVU 9 //!< [Y][VU] Semi-planar 4:2:2 or 4:2:0 + +//! The following are FOURCC aliases for specific layouts. Note that it is still required to specify the format as well +//! as the layout, e.g. NVCV_YUV420 and NVCV_NV12, even though the NV12 layout is only associated with YUV420 sampling. +#define NVCV_I420 NVCV_YUV //!< [Y][U][V] Planar 4:2:0 +#define NVCV_IYUV NVCV_YUV //!< [Y][U][V] Planar 4:2:0 +#define NVCV_YV12 NVCV_YVU //!< [Y][V][U] Planar 4:2:0 +#define NVCV_NV12 NVCV_YCUV //!< [Y][UV] Semi-planar 4:2:0 (default for 4:2:0) +#define NVCV_NV21 NVCV_YCVU //!< [Y][VU] Semi-planar 4:2:0 #define NVCV_YUY2 NVCV_YUYV //!< [YUYV] Chunky 4:2:2 -#define NVCV_I420 NVCV_YUV //!< [Y][U][V] Planar 4:2:2 or 4:2:0 -#define NVCV_IYUV NVCV_YUV //!< [Y][U][V] Planar 4:2:2 or 4:2:0 -#define NVCV_YV12 NVCV_YVU //!< [Y][V][U] Planar 4:2:2 or 4:2:0 -#define NVCV_NV12 NVCV_YCUV //!< [Y][UV] Semi-planar 4:2:2 or 4:2:0 (default for 4:2:0) -#define NVCV_NV21 NVCV_YCVU //!< [Y][VU] Semi-planar 4:2:2 or 4:2:0 +#define NVCV_I444 NVCV_YUV //!< [Y][U][V] Planar 4:4:4 +#define NVCV_YM24 NVCV_YUV //!< [Y][U][V] Planar 4:4:4 +#define NVCV_YM42 NVCV_YVU //!< [Y][V][U] Planar 4:4:4 +#define NVCV_NV24 NVCV_YCUV //!< [Y][UV] Semi-planar 4:4:4 +#define NVCV_NV42 NVCV_YCVU //!< [Y][VU] Semi-planar 4:4:4 //! The following are ORed together for the colorspace field for YUV. //! NVCV_601 and NVCV_709 describe the color axes of YUV. //! NVCV_VIDEO_RANGE and NVCV_VIDEO_RANGE describe the range, [16, 235] or [0, 255], respectively. //! NVCV_CHROMA_COSITED and NVCV_CHROMA_INTSTITIAL describe the location of the chroma samples. -#define NVCV_601 0 //!< The Rec.601 YUV colorspace, typically used for SD. -#define NVCV_709 1 //!< The Rec.709 YUV colorspace, typically used for HD. -#define NVCV_VIDEO_RANGE 0 //!< The video range is [16, 235]. -#define NVCV_FULL_RANGE 4 //!< The video range is [ 0, 255]. -#define NVCV_CHROMA_COSITED 0 //!< The chroma is sampled at the same location as the luma samples horizontally. -#define NVCV_CHROMA_INTSTITIAL 8 //!< The chroma is sampled between luma samples horizontally. -#define NVCV_CHROMA_MPEG2 NVCV_CHROMA_COSITED +#define NVCV_601 0x00 //!< The Rec.601 YUV colorspace, typically used for SD. +#define NVCV_709 0x01 //!< The Rec.709 YUV colorspace, typically used for HD. +#define NVCV_2020 0x02 //!< The Rec.2020 YUV colorspace. +#define NVCV_VIDEO_RANGE 0x00 //!< The video range is [16, 235]. +#define NVCV_FULL_RANGE 0x04 //!< The video range is [ 0, 255]. +#define NVCV_CHROMA_COSITED 0x00 //!< The chroma is sampled at the same location as the luma samples horizontally. +#define NVCV_CHROMA_INTSTITIAL 0x08 //!< The chroma is sampled between luma samples horizontally. +#define NVCV_CHROMA_TOPLEFT 0x10 //!< The chroma is sampled at the same location as the luma samples horizontally and vertically. +#define NVCV_CHROMA_MPEG2 NVCV_CHROMA_COSITED //!< As is most video. #define NVCV_CHROMA_MPEG1 NVCV_CHROMA_INTSTITIAL +#define NVCV_CHROMA_JPEG NVCV_CHROMA_INTSTITIAL +#define NVCV_CHROMA_H261 NVCV_CHROMA_INTSTITIAL +#define NVCV_CHROMA_INTERSTITIAL NVCV_CHROMA_INTSTITIAL //!< Correct spelling //! This is the value for the gpuMem field or the memSpace argument. -#define NVCV_CPU 0 //!< The buffer is stored in CPU memory. -#define NVCV_GPU 1 //!< The buffer is stored in CUDA memory. -#define NVCV_CUDA 1 //!< The buffer is stored in CUDA memory. +#define NVCV_CPU 0 //!< The buffer is stored in CPU memory. +#define NVCV_GPU 1 //!< The buffer is stored in CUDA memory. +#define NVCV_CUDA 1 //!< The buffer is stored in CUDA memory. #define NVCV_CPU_PINNED 2 //!< The buffer is stored in pinned CPU memory. +#define NVCV_CUDA_ARRAY 3 //!< A CUDA array is used for storage. //! Image descriptor. typedef struct @@ -126,7 +156,7 @@ NvCVImage { unsigned char numComponents; //!< The number of components in each pixel. unsigned char planar; //!< NVCV_CHUNKY, NVCV_PLANAR, NVCV_UYVY, .... unsigned char gpuMem; //!< NVCV_CPU, NVCV_CPU_PINNED, NVCV_CUDA, NVCV_GPU - unsigned char colorspace; //!< an OR of colorspace, range and chroma phase. + unsigned char colorspace; //!< An OR of colorspace, range and chroma phase. unsigned char reserved[2]; //!< For structure padding and future expansion. Set to 0. void *pixels; //!< Pointer to pixel(0,0) in the image. void *deletePtr; //!< Buffer memory to be deleted (can be NULL). @@ -179,8 +209,6 @@ NvCVImage { //! \return NVCV_ERR_MISMATCH if the formats are different //! \return NVCV_ERR_CUDA if a CUDA error occurred //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not yet accommodated. - //! \bug This does not work for planar or semi-planar formats, neither RGB nor YUV. - //! \note This does work for all chunky formats, including UYVY, VYUY, YUYV, YVYU. inline NvCV_Status copyFrom(const NvCVImage *src, int srcX, int srcY, int dstX, int dstY, unsigned width, unsigned height); //! Copy from one image to another. This works for CPU->CPU, CPU->GPU, GPU->GPU, and GPU->CPU. @@ -196,6 +224,22 @@ NvCVImage { } NvCVImage; +//! Integer rectangle. +typedef struct NvCVRect2i { + int x; //!< The left edge of the rectangle. + int y; //!< The top edge of the rectangle. + int width; //!< The width of the rectangle. + int height; //!< The height of the rectangle. +} NvCVRect2i; + + +//! Integer point. +typedef struct NvCVPoint2i { + int x; //!< The horizontal coordinate. + int y; //!< The vertical coordinate +} NvCVPoint2i; + + //! Initialize an image. The C++ constructors can initialize this appropriately. //! This is called by the C++ constructor, but C code should call this explicitly. //! \param[in,out] im the image to initialize. @@ -221,8 +265,12 @@ NvCV_Status NvCV_API NvCVImage_Init(NvCVImage *im, unsigned width, unsigned heig //! \param[in] y the top edge of the sub-image, as coordinate of the full image. //! \param[in] width the desired width of the subImage, in pixels. //! \param[in] height the desired height of the subImage, in pixels. -//! \bug This does not work for planar or semi-planar formats, neither RGB nor YUV. +//! \bug This does not work in general for planar or semi-planar formats, neither RGB nor YUV. +//! However, it does work for all formats with the full image, to make a shallow copy, e.g. +//! NvCVImage_InitView(&subImg, &fullImg, 0, 0, fullImage.width, fullImage.height). +//! Cropping a planar or semi-planar image can be accomplished with NvCVImage_TransferRect(). //! \note This does work for all chunky formats, including UYVY, VYUY, YUYV, YVYU. +//! \sa { NvCVImage_TransferRect } void NvCV_API NvCVImage_InitView(NvCVImage *subImg, NvCVImage *fullImg, int x, int y, unsigned width, unsigned height); @@ -310,36 +358,52 @@ void NvCV_API NvCVImage_ComponentOffsets(NvCVImage_PixelFormat format, int *rOff //! //! If any of the images resides on the GPU, it may run asynchronously, //! so cudaStreamSynchronize() should be called if it is necessary to run synchronously. -//! The following table indicates the currently-implemented conversions: -//! +------------------+-------------+-------------+-------------+-------------+ -//! | | u8 --> u8 | u8 --> f32 | f32 --> u8 | f32 --> f32 | -//! +------------------+-------------+-------------+-------------+-------------+ -//! | Y -- > Y | X | | X | X | -//! | Y -- > A | X | | X | X | -//! | Y -- > RGB | X | X | X | X | -//! | Y -- > RGBA | X | X | X | X | -//! | A -- > Y | X | | X | X | -//! | A -- > A | X | | X | X | -//! | A -- > RGB | X | X | X | X | -//! | A -- > RGBA | X | | | | -//! | RGB -- > Y | X | X | | | -//! | RGB -- > A | X | X | | | -//! | RGB -- > RGB | X | X | X | X | -//! | RGB -- > RGBA | X | X | X | X | -//! | RGBA -- > Y | X | X | | | -//! | RGBA -- > A | | X | | | -//! | RGBA -- > RGB | X | X | X | X | -//! | RGBA -- > RGBA | X | | | | -//! | YUV420 -- > RGB | X | | | | -//! | YUV422 -- > RGB | X | | | | -//! +------------------+-------------+-------------+-------------+-------------+ +//! The following table indicates (with X) the currently-implemented conversions: +//! +-------------------+-------------+-------------+-------------+-------------+ +//! | | u8 --> u8 | u8 --> f32 | f32 --> u8 | f32 --> f32 | +//! +-------------------+-------------+-------------+-------------+-------------+ +//! | Y --> Y | X | | X | X | +//! | Y --> A | X | | X | X | +//! | Y --> RGB | X | X | X | X | +//! | Y --> RGBA | X | X | X | X | +//! | A --> Y | X | | X | X | +//! | A --> A | X | | X | X | +//! | A --> RGB | X | X | X | X | +//! | A --> RGBA | X | | | | +//! | RGB --> Y | X | X | | | +//! | RGB --> A | X | X | | | +//! | RGB --> RGB | X | X | X | X | +//! | RGB --> RGBA | X | X | X | X | +//! | RGBA --> Y | X | X | | | +//! | RGBA --> A | | X | | | +//! | RGBA --> RGB | X | X | X | X | +//! | RGBA --> RGBA | X | X | X | X | +//! | RGB --> YUV420 | X | | X | | +//! | RGBA --> YUV420 | X | | X | | +//! | RGB --> YUV422 | X | | X | | +//! | RGBA --> YUV422 | X | | X | | +//! | RGB --> YUV444 | X | | X | | +//! | RGBA --> YUV444 | X | | X | | +//! | YUV420 --> RGB | X | X | | | +//! | YUV420 --> RGBA | X | X | | | +//! | YUV422 --> RGB | X | X | | | +//! | YUV422 --> RGBA | X | X | | | +//! | YUV444 --> RGB | X | X | | | +//! | YUV444 --> RGBA | X | X | | | +//! +-------------------+-------------+-------------+-------------+-------------+ //! where //! * Either source or destination can be CHUNKY or PLANAR. //! * Either source or destination can reside on the CPU or the GPU. //! * The RGB components are in any order (i.e. RGB or BGR; RGBA or BGRA). -//! * YUV requires that the colorspace field be set manually prior to Transfer. +//! * For RGBA (or BGRA) destinations, most implementations do not change the alpha channel, so it is recommended to +//! set it at initialization time with [cuda]memset(im.pixels, -1, im.pitch * im.height) or +//! [cuda]memset(im.pixels, -1, im.pitch * im.height * im.numComponents) for chunky and planar images respectively. +//! * YUV requires that the colorspace field be set manually prior to Transfer, e.g. typical for layout=NVCV_NV12: +//! image.colorspace = NVCV_709 | NVCV_VIDEO_RANGE | NVCV_CHROMA_INTSTITIAL; +//! * There are also RGBf16-->RGBf32 and RGBf32-->RGBf16 transfers. //! * Additionally, when the src and dst formats are the same, all formats are accommodated on CPU and GPU, -//! and this can be used as a replacement for cudaMemcpy2DAsync() (which it utilizes). +//! and this can be used as a replacement for cudaMemcpy2DAsync() (which it utilizes). This is also true for YUV, +//! whose src and dst must share the same format, layout and colorspace. //! //! When there is some kind of conversion AND the src and dst reside on different processors (CPU, GPU), //! it is necessary to have a temporary GPU buffer, which is reshaped as needed to match the characteristics @@ -368,17 +432,139 @@ NvCV_Status NvCV_API NvCVImage_Transfer( const NvCVImage *src, NvCVImage *dst, float scale, struct CUstream_st *stream, NvCVImage *tmp); -//! Composite one BGRu8 source image over another using the given matte. -//! \param[in] fg the foreground source BGRu8 (or RGBu8) image. -//! \param[in] bg the background source BGRu8 (or RGBu8) image. +//! Transfer a rectangular portion of an image. +//! See NvCVImage_Transfer() for the pixel format combinations that are implemented. +//! \param[in] src the source image. +//! \param[in] srcRect the subRect of the src to be transferred (NULL implies the whole image). +//! \param[out] dst the destination image. +//! \param[in] dstPt location to which the srcRect is to be copied (NULL implies (0,0)). +//! \param[in] scale scale factor applied to the magnitude during transfer, typically 1, 255 or 1/255. +//! \param[in] stream the CUDA stream. +//! \param[in] tmp a staging image. +//! \return NVCV_SUCCESS if the operation was completed successfully. +//! \note The actual transfer region may be smaller, because the rects are clipped against the images. +NvCV_Status NvCV_API NvCVImage_TransferRect( + const NvCVImage *src, const NvCVRect2i *srcRect, NvCVImage *dst, const NvCVPoint2i *dstPt, + float scale, struct CUstream_st *stream, NvCVImage *tmp); + + +//! Transfer from a YUV image. +//! YUVu8 --> RGBu8 and YUVu8 --> RGBf32 are currently available. +//! \param[in] y pointer to pixel(0,0) of the luminance channel. +//! \param[in] yPixBytes the byte stride between y pixels horizontally. +//! \param[in] yPitch the byte stride between y pixels vertically. +//! \param[in] u pointer to pixel(0,0) of the u (Cb) chrominance channel. +//! \param[in] v pointer to pixel(0,0) of the v (Cr) chrominance channel. +//! \param[in] uvPixBytes the byte stride between u or v pixels horizontally. +//! \param[in] uvPitch the byte stride between u or v pixels vertically. +//! \param[in] yuvColorSpace the yuv colorspace, specifying range, chromaticities, and chrominance phase. +//! \param[in] yuvMemSpace the memory space where the pixel buffers reside. +//! \param[out] dst the destination image. +//! \param[in] dstRect the destination rectangle (NULL implies the whole image). +//! \param[in] scale scale factor applied to the magnitude during transfer, typically 1, 255 or 1/255. +//! \param[in] stream the CUDA stream. +//! \param[in] tmp a staging image. +//! \return NVCV_SUCCESS if the operation was completed successfully. +//! \note The actual transfer region may be smaller, because the rects are clipped against the images. +NvCV_Status NvCV_API NvCVImage_TransferFromYUV( + const void *y, int yPixBytes, int yPitch, + const void *u, const void *v, int uvPixBytes, int uvPitch, + NvCVImage_PixelFormat yuvFormat, NvCVImage_ComponentType yuvType, + unsigned yuvColorSpace, unsigned yuvMemSpace, + NvCVImage *dst, const NvCVRect2i *dstRect, float scale, struct CUstream_st *stream, NvCVImage *tmp); + + +//! Transfer to a YUV image. +//! RGBu8 --> YUVu8 and RGBf32 --> YUVu8 are currently available. +//! \param[in] src the source image. +//! \param[in] srcRect the destination rectangle (NULL implies the whole image). +//! \param[out] y pointer to pixel(0,0) of the luminance channel. +//! \param[in] yPixBytes the byte stride between y pixels horizontally. +//! \param[in] yPitch the byte stride between y pixels vertically. +//! \param[out] u pointer to pixel(0,0) of the u (Cb) chrominance channel. +//! \param[out] v pointer to pixel(0,0) of the v (Cr) chrominance channel. +//! \param[in] uvPixBytes the byte stride between u or v pixels horizontally. +//! \param[in] uvPitch the byte stride between u or v pixels vertically. +//! \param[in] yuvColorSpace the yuv colorspace, specifying range, chromaticities, and chrominance phase. +//! \param[in] yuvMemSpace the memory space where the pixel buffers reside. +//! \param[in] scale scale factor applied to the magnitude during transfer, typically 1, 255 or 1/255. +//! \param[in] stream the CUDA stream. +//! \param[in] tmp a staging image. +//! \return NVCV_SUCCESS if the operation was completed successfully. +//! \note The actual transfer region may be smaller, because the rects are clipped against the images. +NvCV_Status NvCV_API NvCVImage_TransferToYUV( + const NvCVImage *src, const NvCVRect2i *srcRect, + const void *y, int yPixBytes, int yPitch, + const void *u, const void *v, int uvPixBytes, int uvPitch, + NvCVImage_PixelFormat yuvFormat, NvCVImage_ComponentType yuvType, + unsigned yuvColorSpace, unsigned yuvMemSpace, + float scale, struct CUstream_st *stream, NvCVImage *tmp); + + +//! Between rendering by a graphics system and Transfer by CUDA, it is necessary to map the texture resource. +//! There is a fair amount of overhead, so its use should be minimized. +//! Every call to NvCVImage_MapResource() should be matched by a subsequent call to NvCVImage_UnmapResource(). +//! \param[in,out] im the image to be mapped. +//! \param[in] stream the stream on which the mapping is to be performed. +//! \return NVCV_SUCCESS is the operation was completed successfully. +NvCV_Status NvCV_API NvCVImage_MapResource(NvCVImage *im, struct CUstream_st *stream); + + +//! After transfer by CUDA, the texture resource must be unmapped in order to be used by the graphics system again. +//! There is a fair amount of overhead, so its use should be minimized. +//! Every call to NvCVImage_UnmapResource() should correspond to a preceding call to NvCVImage_MapResource(). +//! \param[in,out] im the image to be mapped. +//! \param[in] stream the CUDA stream on which the mapping is to be performed. +//! \return NVCV_SUCCESS is the operation was completed successfully. +NvCV_Status NvCV_API NvCVImage_UnmapResource(NvCVImage *im, struct CUstream_st *stream); + + +//! Composite one source image over another using the given matte. +//! This accommodates all RGB and RGBA formats, with u8 and f32 components. +//! \param[in] fg the foreground source image. +//! \param[in] bg the background source image. //! \param[in] mat the matte Yu8 (or Au8) image, indicating where the src should come through. -//! \param[out] dst the destination BGRu8 (or RGBu8) image. This can be the same as fg or bg. +//! \param[out] dst the destination image. This can be the same as fg or bg. +//! \param[in] stream the CUDA stream on which the composition is to be performed. //! \return NVCV_SUCCESS if the operation was successful. //! \return NVCV_ERR_PIXELFORMAT if the pixel format is not accommodated. -//! \bug This is only implemented for 3-component u8 fg, bg and dst, and 1-component u8 mat, -//! where all images are resident on the CPU. +//! \return NVCV_ERR_MISMATCH if either the fg & bg & dst formats do not match, or if fg & bg & dst & mat are not +//! in the same address space (CPU or GPU). +#if RTX_CAMERA_IMAGE == 0 +NvCV_Status NvCV_API NvCVImage_Composite(const NvCVImage *fg, const NvCVImage *bg, const NvCVImage *mat, NvCVImage *dst, + struct CUstream_st *stream); +#else // RTX_CAMERA_IMAGE == 1 // No GPU acceleration NvCV_Status NvCV_API NvCVImage_Composite(const NvCVImage *fg, const NvCVImage *bg, const NvCVImage *mat, NvCVImage *dst); +#endif // RTX_CAMERA_IMAGE == 1 +//! Composite one source image over another using the given matte. +//! Not all pixel format combinations are accommodated. +//! \param[in] fg the foreground source image. +//! \param[in] fgOrg the upper-left corner of the fg image to be composited (NULL implies (0,0)). +//! \param[in] bg the background source image. +//! \param[in] bgOrg the upper-left corner of the bg image to be composited (NULL implies (0,0)). +//! \param[in] mat the matte image, indicating where the src should come through. +//! This determines the size of the rectangle to be composited. +//! If this is multi-channel, the alpha channel is used as the matte. +//! \param[in] mode the composition mode. Only 0 (straight alpha over) is implemented at this time. +//! \param[out] dst the destination image. This can be the same as fg or bg. +//! \param[in] dstOrg the upper-left corner of the dst image to be updated (NULL implies (0,0)). +//! \param[in] stream the CUDA stream on which the composition is to be performed. +//! \note If a smaller region of a matte is desired, a window can be created using +//! NvCVImage_InitView() for chunky or NvCVImage_Init() for planar pixels. +//! \return NVCV_SUCCESS if the operation was successful. +//! \return NVCV_ERR_PIXELFORMAT if the pixel format is not accommodated. +//! \return NVCV_ERR_MISMATCH if either the fg & bg & dst formats do not match, or if fg & bg & dst & mat are not +//! in the same address space (CPU or GPU). +//! \bug Though RGBA destinations are accommodated, the A channel is not updated at all. +//! \todo Accommodate premultiplied alpha, either as a flag in NvCVImage or as a different mode. +//! \todo If the destination has an A channel, update it as per Adobe and Pixar. +NvCV_Status NvCV_API NvCVImage_CompositeRect( + const NvCVImage *fg, const NvCVPoint2i *fgOrg, + const NvCVImage *bg, const NvCVPoint2i *bgOrg, + const NvCVImage *mat, unsigned mode, + NvCVImage *dst, const NvCVPoint2i *dstOrg, + struct CUstream_st *stream); //! Composite a BGRu8 source image over a constant color field using the given matte. //! \param[in] src the source BGRu8 (or RGBu8) image. @@ -464,10 +650,16 @@ NvCVImage::~NvCVImage() { NvCVImage_Dealloc(this); } NvCV_Status NvCVImage::copyFrom(const NvCVImage *src, int srcX, int srcY, int dstX, int dstY, unsigned wd, unsigned ht) { +#if RTX_CAMERA_IMAGE // This only works for chunky images NvCVImage srcView, dstView; NvCVImage_InitView(&srcView, const_cast(src), srcX, srcY, wd, ht); NvCVImage_InitView(&dstView, this, dstX, dstY, wd, ht); return NvCVImage_Transfer(&srcView, &dstView, 1.f, 0, nullptr); +#else // !RTX_CAMERA_IMAGE bug fix for non-chunky images + NvCVRect2i srcRect = { (int)srcX, (int)srcY, (int)wd, (int)ht }; + NvCVPoint2i dstPt = { (int)dstX, (int)dstY }; + return NvCVImage_TransferRect(src, &srcRect, this, &dstPt, 1.f, 0, nullptr); +#endif // RTX_CAMERA_IMAGE } /******************************************************************************** diff --git a/nvar/include/nvCVStatus.h b/nvar/include/nvCVStatus.h index 17997dc..dd47ba3 100644 --- a/nvar/include/nvCVStatus.h +++ b/nvar/include/nvCVStatus.h @@ -64,17 +64,33 @@ typedef enum NvCV_Status { NVCV_ERR_UNSUPPORTEDGPU = -17, //!< The GPU is not supported NVCV_ERR_WRONGGPU = -18, //!< The current GPU is not the one selected. NVCV_ERR_UNSUPPORTEDDRIVER = -19, //!< The currently installed graphics driver is not supported + NVCV_ERR_MODELDEPENDENCIES = -20, //!< There is no model with dependencies that match this system + NVCV_ERR_PARSE = -21, //!< There has been a parsing or syntax error while reading a file + NVCV_ERR_MODELSUBSTITUTION = -22, //!< The specified model does not exist and has been substituted. + NVCV_ERR_READ = -23, //!< An error occurred while reading a file. + NVCV_ERR_WRITE = -24, //!< An error occurred while writing a file. + NVCV_ERR_PARAMREADONLY = -25, //!< The selected parameter is read-only. + NVCV_ERR_TRT_ENQUEUE = -26, //!< TensorRT enqueue failed. + NVCV_ERR_TRT_BINDINGS = -27, //!< Unexpected TensorRT bindings. + NVCV_ERR_TRT_CONTEXT = -28, //!< An error occurred while creating a TensorRT context. + NVCV_ERR_TRT_INFER = -29, ///< The was a problem creating the inference engine. + NVCV_ERR_TRT_ENGINE = -30, ///< There was a problem deserializing the inference runtime engine. + NVCV_ERR_NPP = -31, //!< An error has occurred in the NPP library. + NVCV_ERR_CONFIG = -32, //!< No suitable model exists for the specified parameter configuration. - NVCV_ERR_CUDA_MEMORY = -20, //!< There is not enough CUDA memory for the requested operation. - NVCV_ERR_CUDA_VALUE = -21, //!< A CUDA parameter is not within the acceptable range. - NVCV_ERR_CUDA_PITCH = -22, //!< A CUDA pitch is not within the acceptable range. - NVCV_ERR_CUDA_INIT = -23, //!< The CUDA driver and runtime could not be initialized. - NVCV_ERR_CUDA_LAUNCH = -24, //!< The CUDA kernel launch has failed. - NVCV_ERR_CUDA_KERNEL = -25, //!< No suitable kernel image is available for the device. - NVCV_ERR_CUDA_DRIVER = -26, //!< The installed NVIDIA CUDA driver is older than the CUDA runtime library. - NVCV_ERR_CUDA_UNSUPPORTED = -27, //!< The CUDA operation is not supported on the current system or device. - NVCV_ERR_CUDA_ILLEGAL_ADDRESS = -28, //!< CUDA tried to load or store on an invalid memory address. - NVCV_ERR_CUDA = -30, //!< An otherwise unspecified CUDA error has been reported. + NVCV_ERR_DIRECT3D = -99, //!< A Direct3D error has occurred. + + NVCV_ERR_CUDA_BASE = -100, //!< CUDA errors are offset from this value. + NVCV_ERR_CUDA_VALUE = -101, //!< A CUDA parameter is not within the acceptable range. + NVCV_ERR_CUDA_MEMORY = -102, //!< There is not enough CUDA memory for the requested operation. + NVCV_ERR_CUDA_PITCH = -112, //!< A CUDA pitch is not within the acceptable range. + NVCV_ERR_CUDA_INIT = -127, //!< The CUDA driver and runtime could not be initialized. + NVCV_ERR_CUDA_LAUNCH = -819, //!< The CUDA kernel launch has failed. + NVCV_ERR_CUDA_KERNEL = -309, //!< No suitable kernel image is available for the device. + NVCV_ERR_CUDA_DRIVER = -135, //!< The installed NVIDIA CUDA driver is older than the CUDA runtime library. + NVCV_ERR_CUDA_UNSUPPORTED = -901, //!< The CUDA operation is not supported on the current system or device. + NVCV_ERR_CUDA_ILLEGAL_ADDRESS = -800, //!< CUDA tried to load or store on an invalid memory address. + NVCV_ERR_CUDA = -1099, //!< An otherwise unspecified CUDA error has been reported. } NvCV_Status; diff --git a/nvar/include/nvTransferD3D.h b/nvar/include/nvTransferD3D.h new file mode 100644 index 0000000..e914eb5 --- /dev/null +++ b/nvar/include/nvTransferD3D.h @@ -0,0 +1,72 @@ +/*############################################################################### +# +# Copyright(c) 2021 NVIDIA CORPORATION.All Rights Reserved. +# +# NVIDIA CORPORATION and its licensors retain all intellectual property +# and proprietary rights in and to this software, related documentation +# and any modifications thereto.Any use, reproduction, disclosure or +# distribution of this software and related documentation without an express +# license agreement from NVIDIA CORPORATION is strictly prohibited. +# +###############################################################################*/ + +#ifndef __NVTRANSFER_D3D_H__ +#define __NVTRANSFER_D3D_H__ + +#ifndef _WINDOWS_ + #define WIN32_LEAN_AND_MEAN + #include +#endif // _WINDOWS_ +#include +#include "nvCVImage.h" + +#ifdef __cplusplus +extern "C" { +#endif // ___cplusplus + + + +//! Utility to determine the D3D format from the NvCVImage format, type and layout. +//! \param[in] format the pixel format. +//! \param[in] type the component type. +//! \param[in] layout the layout. +//! \param[out] d3dFormat a place to store the corresponding D3D format. +//! \return NVCV_SUCCESS if successful. +NvCV_Status NvCV_API NvCVImage_ToD3DFormat(NvCVImage_PixelFormat format, NvCVImage_ComponentType type, unsigned layout, DXGI_FORMAT *d3dFormat); + + +//! Utility to determine the NvCVImage format, component type and layout from a D3D format. +//! \param[in] d3dFormat the D3D format to translate. +//! \param[out] format a place to store the NvCVImage pixel format. +//! \param[out] type a place to store the NvCVImage component type. +//! \param[out] layout a place to store the NvCVImage layout. +//! \return NVCV_SUCCESS if successful. +NvCV_Status NvCV_API NvCVImage_FromD3DFormat(DXGI_FORMAT d3dFormat, NvCVImage_PixelFormat *format, NvCVImage_ComponentType *type, unsigned char *layout); + + +#ifdef __dxgicommon_h__ + +//! Utility to determine the D3D color space from the NvCVImage color space. +//! \param[in] nvcvColorSpace the NvCVImage colro space. +//! \param[out] pD3dColorSpace a place to store the resultant D3D color space. +//! \return NVCV_SUCCESS if successful. +//! \return NVCV_ERR_PIXELFORMAT if there is no equivalent color space. +NvCV_Status NvCV_API NvCVImage_ToD3DColorSpace(unsigned char nvcvColorSpace, DXGI_COLOR_SPACE_TYPE *pD3dColorSpace); + + +//! Utility to determine the NvCVImage color space from the D3D color space. +//! \param[in] d3dColorSpace the D3D color space. +//! \param[out] pNvcvColorSpace a place to store the resultant NvCVImage color space. +//! \return NVCV_SUCCESS if successful. +//! \return NVCV_ERR_PIXELFORMAT if there is no equivalent color space. +NvCV_Status NvCV_API NvCVImage_FromD3DColorSpace(DXGI_COLOR_SPACE_TYPE d3dColorSpace, unsigned char *pNvcvColorSpace); + +#endif // __dxgicommon_h__ + + +#ifdef __cplusplus +} // extern "C" +#endif // __cplusplus + +#endif // __NVTRANSFER_D3D_H__ + diff --git a/nvar/include/nvTransferD3D11.h b/nvar/include/nvTransferD3D11.h new file mode 100644 index 0000000..fabf067 --- /dev/null +++ b/nvar/include/nvTransferD3D11.h @@ -0,0 +1,44 @@ +/*############################################################################### +# +# Copyright(c) 2021 NVIDIA CORPORATION.All Rights Reserved. +# +# NVIDIA CORPORATION and its licensors retain all intellectual property +# and proprietary rights in and to this software, related documentation +# and any modifications thereto.Any use, reproduction, disclosure or +# distribution of this software and related documentation without an express +# license agreement from NVIDIA CORPORATION is strictly prohibited. +# +###############################################################################*/ + +#ifndef __NVTRANSFER_D3D11_H__ +#define __NVTRANSFER_D3D11_H__ + +#include +#include "nvCVImage.h" +#include "nvTransferD3D.h" // for NvCVImage_ToD3DFormat() and NvCVImage_FromD3DFormat() + +#ifdef __cplusplus +extern "C" { +#endif // ___cplusplus + + + +//! Initialize an NvCVImage from a D3D11 texture. +//! The pixelFormat and component types with be transferred over, and a cudaGraphicsResource will be registered; +//! the NvCVImage destructor will unregister the resource. +//! This is designed to work with NvCVImage_TransferFromArray() (and eventually NvCVImage_Transfer()); +//! however it is necessary to call NvCVImage_MapResource beforehand, and NvCVImage_UnmapResource +//! before allowing D3D to render into it. +//! \param[in,out] im the image to be initialized. +//! \param[in] tx the texture to be used for initialization. +//! \return NVCV_SUCCESS if successful. +NvCV_Status NvCV_API NvCVImage_InitFromD3D11Texture(NvCVImage *im, struct ID3D11Texture2D *tx); + + + +#ifdef __cplusplus +} // extern "C" +#endif // __cplusplus + +#endif // __NVTRANSFER_D3D11_H__ + diff --git a/nvar/src/nvARProxy.cpp b/nvar/src/nvARProxy.cpp index 44f1c02..ad73ea1 100644 --- a/nvar/src/nvARProxy.cpp +++ b/nvar/src/nvARProxy.cpp @@ -87,99 +87,6 @@ NvCV_Status NvAR_API NvAR_GetVersion(unsigned int* version) { return funcPtr(version); } -NvCV_Status NvAR_API NvCVImage_Init(NvCVImage* im, unsigned width, unsigned height, int pitch, void* pixels, - NvCVImage_PixelFormat format, NvCVImage_ComponentType type, unsigned isPlanar, - unsigned onGPU) { - static const auto funcPtr = (decltype(NvCVImage_Init)*)nvGetProcAddress(getNvARLib(), "NvCVImage_Init"); - - if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; - return funcPtr(im, width, height, pitch, pixels, format, type, isPlanar, onGPU); -} - -void NvAR_API NvCVImage_InitView(NvCVImage* subImg, NvCVImage* fullImg, int x, int y, unsigned width, - unsigned height) { - static const auto funcPtr = (decltype(NvCVImage_InitView)*)nvGetProcAddress(getNvARLib(), "NvCVImage_InitView"); - - if (nullptr != funcPtr) funcPtr(subImg, fullImg, x, y, width, height); -} - -NvCV_Status NvAR_API NvCVImage_Alloc(NvCVImage* im, unsigned width, unsigned height, NvCVImage_PixelFormat format, - NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU, unsigned alignment) { - static const auto funcPtr = (decltype(NvCVImage_Alloc)*)nvGetProcAddress(getNvARLib(), "NvCVImage_Alloc"); - - if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; - return funcPtr(im, width, height, format, type, isPlanar, onGPU, alignment); -} - -NvCV_Status NvAR_API NvCVImage_Realloc(NvCVImage* im, unsigned width, unsigned height, - NvCVImage_PixelFormat format, NvCVImage_ComponentType type, - unsigned isPlanar, unsigned onGPU, unsigned alignment) { - static const auto funcPtr = (decltype(NvCVImage_Realloc)*)nvGetProcAddress(getNvARLib(), "NvCVImage_Realloc"); - - if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; - return funcPtr(im, width, height, format, type, isPlanar, onGPU, alignment); -} - -void NvAR_API NvCVImage_Dealloc(NvCVImage* im) { - static const auto funcPtr = (decltype(NvCVImage_Dealloc)*)nvGetProcAddress(getNvARLib(), "NvCVImage_Dealloc"); - - if (nullptr != funcPtr) funcPtr(im); -} - -NvCV_Status NvAR_API NvCVImage_Create(unsigned width, unsigned height, NvCVImage_PixelFormat format, - NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU, - unsigned alignment, NvCVImage** out) { - static const auto funcPtr = (decltype(NvCVImage_Create)*)nvGetProcAddress(getNvARLib(), "NvCVImage_Create"); - - if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; - return funcPtr(width, height, format, type, isPlanar, onGPU, alignment, out); -} - -void NvAR_API NvCVImage_Destroy(NvCVImage* im) { - static const auto funcPtr = (decltype(NvCVImage_Destroy)*)nvGetProcAddress(getNvARLib(), "NvCVImage_Destroy"); - - if (nullptr != funcPtr) funcPtr(im); -} - -void NvAR_API NvCVImage_ComponentOffsets(NvCVImage_PixelFormat format, int* rOff, int* gOff, int* bOff, int* aOff, - int* yOff) { - static const auto funcPtr = - (decltype(NvCVImage_ComponentOffsets)*)nvGetProcAddress(getNvARLib(), "NvCVImage_ComponentOffsets"); - - if (nullptr != funcPtr) funcPtr(format, rOff, gOff, bOff, aOff, yOff); -} - -NvCV_Status NvAR_API NvCVImage_Transfer(const NvCVImage* src, NvCVImage* dst, float scale, CUstream_st* stream, - NvCVImage* tmp) { - static const auto funcPtr = (decltype(NvCVImage_Transfer)*)nvGetProcAddress(getNvARLib(), "NvCVImage_Transfer"); - - if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; - return funcPtr(src, dst, scale, stream, tmp); -} - -NvCV_Status NvAR_API NvCVImage_Composite(const NvCVImage* fg, const NvCVImage* bg, const NvCVImage* mat, NvCVImage* dst) { - static const auto funcPtr = (decltype(NvCVImage_Composite)*)nvGetProcAddress(getNvARLib(), "NvCVImage_Composite"); - - if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; - return funcPtr(fg, bg, mat, dst); -} - -NvCV_Status NvAR_API NvCVImage_CompositeOverConstant(const NvCVImage* src, const NvCVImage* mat, - const unsigned char bgColor[3], NvCVImage* dst) { - static const auto funcPtr = - (decltype(NvCVImage_CompositeOverConstant)*)nvGetProcAddress(getNvARLib(), "NvCVImage_CompositeOverConstant"); - - if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; - return funcPtr(src, mat, bgColor, dst); -} - -NvCV_Status NvAR_API NvCVImage_FlipY(const NvCVImage* src, NvCVImage* dst) { - static const auto funcPtr = (decltype(NvCVImage_FlipY)*)nvGetProcAddress(getNvARLib(), "NvCVImage_FlipY"); - - if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; - return funcPtr(src, dst); -} - NvCV_Status NvAR_API NvAR_Create(NvAR_FeatureID featureID, NvAR_FeatureHandle* handle) { static const auto funcPtr = (decltype(NvAR_Create)*)nvGetProcAddress(getNvARLib(), "NvAR_Create"); @@ -348,17 +255,4 @@ NvCV_Status NvAR_API NvAR_CudaStreamDestroy(CUstream stream) { if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; return funcPtr(stream); -} - -#ifdef _WIN32 -__declspec(dllexport) const char* __cdecl -#else -const char* -#endif // _WIN32 or linux - NvCV_GetErrorStringFromCode(NvCV_Status code) { - static const auto funcPtr = - (decltype(NvCV_GetErrorStringFromCode)*)nvGetProcAddress(getNvARLib(), "NvCV_GetErrorStringFromCode"); - - if (nullptr == funcPtr) return "Cannot find nvARPose DLL or its dependencies"; - return funcPtr(code); -} +} \ No newline at end of file diff --git a/nvar/src/nvCVImageProxy.cpp b/nvar/src/nvCVImageProxy.cpp new file mode 100644 index 0000000..f724d7a --- /dev/null +++ b/nvar/src/nvCVImageProxy.cpp @@ -0,0 +1,311 @@ +#if defined(linux) || defined(unix) || defined(__linux) +#warning nvCVImageProxy.cpp not ported +#else +/*############################################################################### +# +# Copyright 2020 NVIDIA Corporation +# +# Permission is hereby granted, free of charge, to any person obtaining a copy of +# this software and associated documentation files (the "Software"), to deal in +# the Software without restriction, including without limitation the rights to +# use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +# the Software, and to permit persons to whom the Software is furnished to do so, +# subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +# FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +# COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +# IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +# CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +# +###############################################################################*/ +#include +#include "nvCVImage.h" + +#ifdef _WIN32 + #define _WINSOCKAPI_ + #include + #include + #include "nvTransferD3D.h" + #include "nvTransferD3D11.h" +#else // !_WIN32 + #include + typedef void* HMODULE; + typedef void* HANDLE; + typedef void* HINSTANCE; +#endif // _WIN32 + +// Parameter string does not include the file extension +#ifdef _WIN32 +#define nvLoadLibrary(library) LoadLibrary(TEXT(library ".dll")) +#else // !_WIN32 +#define nvLoadLibrary(library) dlopen("lib" library ".so", RTLD_LAZY) +#endif // _WIN32 + + +inline void* nvGetProcAddress(HINSTANCE handle, const char* proc) { + if (nullptr == handle) return nullptr; +#ifdef _WIN32 + return GetProcAddress(handle, proc); +#else // !_WIN32 + return dlsym(handle, proc); +#endif // _WIN32 +} + +inline int nvFreeLibrary(HINSTANCE handle) { +#ifdef _WIN32 + return FreeLibrary(handle); +#else + return dlclose(handle); +#endif +} + +HINSTANCE getNvCVImageLib() { + TCHAR path[MAX_PATH], tmpPath[MAX_PATH], fullPath[MAX_PATH]; + static HINSTANCE nvCVImageLib = NULL; + static bool bSDKPathSet = false; + if (!bSDKPathSet) { + // There can be multiple apps on the system, + // some might include the SDK in the app package and + // others might expect the SDK to be installed in Program Files + GetEnvironmentVariable(TEXT("NV_VIDEO_EFFECTS_PATH"), path, MAX_PATH); + GetEnvironmentVariable(TEXT("NV_AR_SDK_PATH"), tmpPath, MAX_PATH); + if (_tcscmp(path, TEXT("USE_APP_PATH")) && _tcscmp(tmpPath, TEXT("USE_APP_PATH"))) { + // App has not set environment variable to "USE_APP_PATH" + // So pick up the SDK dll and dependencies from Program Files + GetEnvironmentVariable(TEXT("ProgramFiles"), path, MAX_PATH); + size_t max_len = sizeof(fullPath) / sizeof(TCHAR); + _stprintf_s(fullPath, max_len, TEXT("%s\\NVIDIA Corporation\\NVIDIA Video Effects\\"), path); + SetDllDirectory(fullPath); + nvCVImageLib = nvLoadLibrary("NVCVImage"); + if (!nvCVImageLib) { + _stprintf_s(fullPath, max_len, TEXT("%s\\NVIDIA Corporation\\NVIDIA AR SDK\\"), path); + SetDllDirectory(fullPath); + nvCVImageLib = nvLoadLibrary("NVCVImage"); + } + } + bSDKPathSet = true; + } + return nvCVImageLib; +} + +NvCV_Status NvCV_API NvCVImage_Init(NvCVImage* im, unsigned width, unsigned height, int pitch, void* pixels, + NvCVImage_PixelFormat format, NvCVImage_ComponentType type, unsigned isPlanar, + unsigned onGPU) { + static const auto funcPtr = (decltype(NvCVImage_Init)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_Init"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(im, width, height, pitch, pixels, format, type, isPlanar, onGPU); +} + +void NvCV_API NvCVImage_InitView(NvCVImage* subImg, NvCVImage* fullImg, int x, int y, unsigned width, + unsigned height) { + static const auto funcPtr = (decltype(NvCVImage_InitView)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_InitView"); + + if (nullptr != funcPtr) funcPtr(subImg, fullImg, x, y, width, height); +} + +NvCV_Status NvCV_API NvCVImage_Alloc(NvCVImage* im, unsigned width, unsigned height, NvCVImage_PixelFormat format, + NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU, unsigned alignment) { + static const auto funcPtr = (decltype(NvCVImage_Alloc)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_Alloc"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(im, width, height, format, type, isPlanar, onGPU, alignment); +} + +NvCV_Status NvCV_API NvCVImage_Realloc(NvCVImage* im, unsigned width, unsigned height, + NvCVImage_PixelFormat format, NvCVImage_ComponentType type, + unsigned isPlanar, unsigned onGPU, unsigned alignment) { + static const auto funcPtr = (decltype(NvCVImage_Realloc)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_Realloc"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(im, width, height, format, type, isPlanar, onGPU, alignment); +} + +void NvCV_API NvCVImage_Dealloc(NvCVImage* im) { + static const auto funcPtr = (decltype(NvCVImage_Dealloc)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_Dealloc"); + + if (nullptr != funcPtr) funcPtr(im); +} + +NvCV_Status NvCV_API NvCVImage_Create(unsigned width, unsigned height, NvCVImage_PixelFormat format, + NvCVImage_ComponentType type, unsigned isPlanar, unsigned onGPU, + unsigned alignment, NvCVImage** out) { + static const auto funcPtr = (decltype(NvCVImage_Create)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_Create"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(width, height, format, type, isPlanar, onGPU, alignment, out); +} + +void NvCV_API NvCVImage_Destroy(NvCVImage* im) { + static const auto funcPtr = (decltype(NvCVImage_Destroy)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_Destroy"); + + if (nullptr != funcPtr) funcPtr(im); +} + +void NvCV_API NvCVImage_ComponentOffsets(NvCVImage_PixelFormat format, int* rOff, int* gOff, int* bOff, int* aOff, + int* yOff) { + static const auto funcPtr = + (decltype(NvCVImage_ComponentOffsets)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_ComponentOffsets"); + + if (nullptr != funcPtr) funcPtr(format, rOff, gOff, bOff, aOff, yOff); +} + +NvCV_Status NvCV_API NvCVImage_Transfer(const NvCVImage* src, NvCVImage* dst, float scale, CUstream_st* stream, + NvCVImage* tmp) { + static const auto funcPtr = (decltype(NvCVImage_Transfer)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_Transfer"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(src, dst, scale, stream, tmp); +} + +NvCV_Status NvCV_API NvCVImage_TransferRect(const NvCVImage *src, const NvCVRect2i *srcRect, NvCVImage *dst, + const NvCVPoint2i *dstPt, float scale, struct CUstream_st *stream, NvCVImage *tmp) { + static const auto funcPtr = (decltype(NvCVImage_TransferRect)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_TransferRect"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(src, srcRect, dst, dstPt, scale, stream, tmp); +} + +NvCV_Status NvCV_API NvCVImage_TransferFromYUV(const void *y, int yPixBytes, int yPitch, const void *u, const void *v, + int uvPixBytes, int uvPitch, NvCVImage_PixelFormat yuvFormat, NvCVImage_ComponentType yuvType, unsigned yuvColorSpace, + unsigned yuvMemSpace, NvCVImage *dst, const NvCVRect2i *dstRect, float scale, struct CUstream_st *stream, NvCVImage *tmp) { + static const auto funcPtr = (decltype(NvCVImage_TransferFromYUV)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_TransferFromYUV"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(y, yPixBytes, yPitch, u, v, uvPixBytes, uvPitch, yuvFormat, yuvType, yuvColorSpace, yuvMemSpace, dst, + dstRect, scale, stream, tmp); +} + +NvCV_Status NvCV_API NvCVImage_TransferToYUV(const NvCVImage *src, const NvCVRect2i *srcRect, + const void *y, int yPixBytes, int yPitch, const void *u, const void *v, int uvPixBytes, int uvPitch, + NvCVImage_PixelFormat yuvFormat, NvCVImage_ComponentType yuvType, unsigned yuvColorSpace, unsigned yuvMemSpace, + float scale, struct CUstream_st *stream, NvCVImage *tmp) { + static const auto funcPtr = (decltype(NvCVImage_TransferToYUV)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_TransferToYUV"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(src, srcRect, y, yPixBytes, yPitch, u, v, uvPixBytes, uvPitch, yuvFormat, yuvType, yuvColorSpace, yuvMemSpace, scale, stream, tmp); +} + +NvCV_Status NvCV_API NvCVImage_MapResource(NvCVImage *im, struct CUstream_st *stream) { + static const auto funcPtr = (decltype(NvCVImage_MapResource)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_MapResource"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(im, stream); +} + +NvCV_Status NvCV_API NvCVImage_UnmapResource(NvCVImage *im, struct CUstream_st *stream) { + static const auto funcPtr = (decltype(NvCVImage_UnmapResource)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_UnmapResource"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(im, stream); +} + +#if RTX_CAMERA_IMAGE == 0 +NvCV_Status NvCV_API NvCVImage_Composite(const NvCVImage* fg, const NvCVImage* bg, const NvCVImage* mat, NvCVImage* dst, + struct CUstream_st *stream) { + static const auto funcPtr = (decltype(NvCVImage_Composite)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_Composite"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(fg, bg, mat, dst, stream); +} +#else // RTX_CAMERA_IMAGE == 1 +NvCV_Status NvCV_API NvCVImage_Composite(const NvCVImage* fg, const NvCVImage* bg, const NvCVImage* mat, NvCVImage* dst) { + static const auto funcPtr = (decltype(NvCVImage_Composite)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_Composite"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(fg, bg, mat, dst); +} +#endif // RTX_CAMERA_IMAGE + +NvCV_Status NvCV_API NvCVImage_CompositeRect( + const NvCVImage *fg, const NvCVPoint2i *fgOrg, + const NvCVImage *bg, const NvCVPoint2i *bgOrg, + const NvCVImage *mat, unsigned mode, + NvCVImage *dst, const NvCVPoint2i *dstOrg, + struct CUstream_st *stream) { + static const auto funcPtr = (decltype(NvCVImage_CompositeRect)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_CompositeRect"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(fg, fgOrg, bg, bgOrg, mat, mode, dst, dstOrg, stream); +} + +NvCV_Status NvCV_API NvCVImage_CompositeOverConstant(const NvCVImage* src, const NvCVImage* mat, + const unsigned char bgColor[3], NvCVImage* dst) { + static const auto funcPtr = + (decltype(NvCVImage_CompositeOverConstant)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_CompositeOverConstant"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(src, mat, bgColor, dst); +} + +NvCV_Status NvCV_API NvCVImage_FlipY(const NvCVImage* src, NvCVImage* dst) { + static const auto funcPtr = (decltype(NvCVImage_FlipY)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_FlipY"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(src, dst); +} + +#ifdef _WIN32 +__declspec(dllexport) const char* __cdecl +#else +const char* +#endif // _WIN32 or linux + NvCV_GetErrorStringFromCode(NvCV_Status code) { + static const auto funcPtr = + (decltype(NvCV_GetErrorStringFromCode)*)nvGetProcAddress(getNvCVImageLib(), "NvCV_GetErrorStringFromCode"); + + if (nullptr == funcPtr) return "Cannot find nvCVImage DLL or its dependencies"; + return funcPtr(code); +} + + + +#ifdef _WIN32 // Direct 3D + +NvCV_Status NvCV_API NvCVImage_InitFromD3D11Texture(NvCVImage *im, struct ID3D11Texture2D *tx) { + static const auto funcPtr = (decltype(NvCVImage_InitFromD3D11Texture)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_InitFromD3D11Texture"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(im, tx); +} + +NvCV_Status NvCV_API NvCVImage_ToD3DFormat(NvCVImage_PixelFormat format, NvCVImage_ComponentType type, unsigned layout, DXGI_FORMAT *d3dFormat) { + static const auto funcPtr = (decltype(NvCVImage_ToD3DFormat)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_ToD3DFormat"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(format, type, layout, d3dFormat); +} + +NvCV_Status NvCV_API NvCVImage_FromD3DFormat(DXGI_FORMAT d3dFormat, NvCVImage_PixelFormat *format, NvCVImage_ComponentType *type, unsigned char *layout) { + static const auto funcPtr = (decltype(NvCVImage_FromD3DFormat)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_FromD3DFormat"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(d3dFormat, format, type, layout); +} + +#ifdef __dxgicommon_h__ + +NvCV_Status NvCV_API NvCVImage_ToD3DColorSpace(unsigned char nvcvColorSpace, DXGI_COLOR_SPACE_TYPE *pD3dColorSpace) { + static const auto funcPtr = (decltype(NvCVImage_ToD3DColorSpace)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_ToD3DColorSpace"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(nvcvColorSpace, pD3dColorSpace); +} + +NvCV_Status NvCV_API NvCVImage_FromD3DColorSpace(DXGI_COLOR_SPACE_TYPE d3dColorSpace, unsigned char *pNvcvColorSpace) { + static const auto funcPtr = (decltype(NvCVImage_FromD3DColorSpace)*)nvGetProcAddress(getNvCVImageLib(), "NvCVImage_FromD3DColorSpace"); + + if (nullptr == funcPtr) return NVCV_ERR_LIBRARY; + return funcPtr(d3dColorSpace, pNvcvColorSpace); +} + +#endif // __dxgicommon_h__ + +#endif // _WIN32 Direct 3D + +#endif // enabling for this file diff --git a/resources/ar_005.png b/resources/ar_005.png new file mode 100644 index 0000000..06c8007 Binary files /dev/null and b/resources/ar_005.png differ diff --git a/samples/BodyTrack/BodyEngine.cpp b/samples/BodyTrack/BodyEngine.cpp new file mode 100644 index 0000000..7061cbf --- /dev/null +++ b/samples/BodyTrack/BodyEngine.cpp @@ -0,0 +1,458 @@ +/*############################################################################### +# +# Copyright 2020 NVIDIA Corporation +# +# Permission is hereby granted, free of charge, to any person obtaining a copy of +# this software and associated documentation files (the "Software"), to deal in +# the Software without restriction, including without limitation the rights to +# use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +# the Software, and to permit persons to whom the Software is furnished to do so, +# subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +# FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +# COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +# IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +# CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +# +###############################################################################*/ +#include "BodyEngine.h" + + +bool CheckResult(NvCV_Status nvErr, unsigned line) { + if (NVCV_SUCCESS == nvErr) return true; + std::cout << "ERROR: " << NvCV_GetErrorStringFromCode(nvErr) << ", line " << line << std::endl; + return false; +} + +BodyEngine::Err BodyEngine::createFeatures(const char* modelPath, unsigned int _batchSize) { + BodyEngine::Err err = BodyEngine::Err::errNone; + + NvCV_Status cuErr = NvAR_CudaStreamCreate(&stream); + if (NVCV_SUCCESS != cuErr) { + printf("Cannot create a cuda stream: %s\n", NvCV_GetErrorStringFromCode(cuErr)); + return errInitialization; + } + if (appMode == bodyDetection) { + err = createBodyDetectionFeature(modelPath, stream); + if (err != Err::errNone) { + printf("ERROR: An error has occured while initializing Body Detection\n"); + } + } + else if (appMode == keyPointDetection) { + err = createKeyPointDetectionFeature(modelPath, _batchSize, stream); + if (err != Err::errNone) { + printf("ERROR: An error has occured while initializing KeyPoint Detection\n"); + } + } + return err; +} + +BodyEngine::Err BodyEngine::createBodyDetectionFeature(const char* modelPath, CUstream str) { + BodyEngine::Err err = BodyEngine::Err::errNone; + NvCV_Status nvErr; + + nvErr = NvAR_Create(NvAR_Feature_BodyDetection, &bodyDetectHandle); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errEffect); + + if (bUseOTAU && (!modelPath || !modelPath[0])) { + nvErr = NvAR_SetString(bodyDetectHandle, NvAR_Parameter_Config(ModelDir), this->bdOTAModelPath); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + } else { + nvErr = NvAR_SetString(bodyDetectHandle, NvAR_Parameter_Config(ModelDir), modelPath); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + } + + nvErr = NvAR_SetCudaStream(bodyDetectHandle, NvAR_Parameter_Config(CUDAStream), str); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetU32(bodyDetectHandle, NvAR_Parameter_Config(Temporal), bStabilizeBody); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_Load(bodyDetectHandle); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errInitialization); + +bail: + return err; +} + +BodyEngine::Err BodyEngine::createKeyPointDetectionFeature(const char* modelPath, unsigned int _batchSize, + CUstream str) { + BodyEngine::Err err = BodyEngine::Err::errNone; + NvCV_Status nvErr; + + batchSize = _batchSize; + nvErr = NvAR_Create(NvAR_Feature_BodyPoseEstimation, &keyPointDetectHandle); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errEffect); + + if (bUseOTAU && (!modelPath || !modelPath[0])) { + nvErr = NvAR_SetString(keyPointDetectHandle, NvAR_Parameter_Config(ModelDir), this->ldOTAModelPath); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + } + else { + nvErr = NvAR_SetString(keyPointDetectHandle, NvAR_Parameter_Config(ModelDir), modelPath); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + } + + nvErr = NvAR_SetCudaStream(keyPointDetectHandle, NvAR_Parameter_Config(CUDAStream), str); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetU32(keyPointDetectHandle, NvAR_Parameter_Config(BatchSize), batchSize); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetU32(keyPointDetectHandle, NvAR_Parameter_Config(NVAR_MODE), nvARMode); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetU32(keyPointDetectHandle, NvAR_Parameter_Config(Temporal), bStabilizeBody); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetF32(keyPointDetectHandle, NvAR_Parameter_Config(FocalLength), bFocalLength); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetF32(keyPointDetectHandle, NvAR_Parameter_Config(UseCudaGraph), bUseCudaGraph); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_Load(keyPointDetectHandle); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errInitialization); + +bail: + return err; +} + +BodyEngine::Err BodyEngine::initFeatureIOParams() { + BodyEngine::Err err = BodyEngine::Err::errNone; + + NvCV_Status cvErr = NvCVImage_Alloc(&inputImageBuffer, input_image_width, input_image_height, NVCV_BGR, NVCV_U8, + NVCV_CHUNKY, NVCV_GPU, 1); + + BAIL_IF_CVERR(cvErr, err, BodyEngine::Err::errInitialization); + + if (appMode == bodyDetection) { + err = initBodyDetectionIOParams(&inputImageBuffer); + if (err != Err::errNone) { + printf("ERROR: An error has occured while setting input, output parmeters for Body Detection\n"); + } + } + else if (appMode == keyPointDetection) { + err = initKeyPointDetectionIOParams(&inputImageBuffer); + if (err != Err::errNone) { + printf("ERROR: An error has occured while setting input, output parmeters for KeyPoint Detection\n"); + } + } + return err; + +bail: + return err; +} + +BodyEngine::Err BodyEngine::initBodyDetectionIOParams(NvCVImage* inBuf) { + NvCV_Status nvErr = NVCV_SUCCESS; + BodyEngine::Err err = BodyEngine::Err::errNone; + + nvErr = NvAR_SetObject(bodyDetectHandle, NvAR_Parameter_Input(Image), inBuf, sizeof(NvCVImage)); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + output_bbox_data.assign(25, {0.f, 0.f, 0.f, 0.f}); + output_bbox_conf_data.assign(25, 0.f); + output_bboxes.boxes = output_bbox_data.data(); + output_bboxes.max_boxes = (uint8_t)output_bbox_data.size(); + output_bboxes.num_boxes = 0; + nvErr = NvAR_SetObject(bodyDetectHandle, NvAR_Parameter_Output(BoundingBoxes), &output_bboxes, sizeof(NvAR_BBoxes)); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetF32Array(bodyDetectHandle, NvAR_Parameter_Output(BoundingBoxesConfidence), + output_bbox_conf_data.data(), output_bboxes.max_boxes); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + +bail: + return err; +} + +BodyEngine::Err BodyEngine::initKeyPointDetectionIOParams(NvCVImage* inBuf) { + NvCV_Status nvErr = NVCV_SUCCESS; + BodyEngine::Err err = BodyEngine::Err::errNone; + uint output_bbox_size; + + nvErr = NvAR_SetObject(keyPointDetectHandle, NvAR_Parameter_Input(Image), inBuf, sizeof(NvCVImage)); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_GetU32(keyPointDetectHandle, NvAR_Parameter_Config(NumKeyPoints), &numKeyPoints); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + keypoints.assign(batchSize * numKeyPoints, {0.f, 0.f}); + keypoints3D.assign(batchSize * numKeyPoints, {0.f, 0.f, 0.f}); + jointAngles.assign(batchSize * numKeyPoints, {0.f, 0.f, 0.f, 1.f}); + keypoints_confidence.assign(batchSize * numKeyPoints, 0.f); + referencePose.assign(numKeyPoints, {0.f, 0.f, 0.f}); + + const void* pReferencePose; + nvErr = NvAR_GetObject(keyPointDetectHandle, NvAR_Parameter_Config(ReferencePose), &pReferencePose, + sizeof(NvAR_Point3f)); + memcpy(referencePose.data(), pReferencePose, sizeof(NvAR_Point3f) * numKeyPoints); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetObject(keyPointDetectHandle, NvAR_Parameter_Output(KeyPoints), keypoints.data(), + sizeof(NvAR_Point2f)); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetObject(keyPointDetectHandle, NvAR_Parameter_Output(KeyPoints3D), keypoints3D.data(), + sizeof(NvAR_Point3f)); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetObject(keyPointDetectHandle, NvAR_Parameter_Output(JointAngles), jointAngles.data(), + sizeof(NvAR_Quaternion)); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + nvErr = NvAR_SetF32Array(keyPointDetectHandle, NvAR_Parameter_Output(KeyPointsConfidence), + keypoints_confidence.data(), batchSize * numKeyPoints); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + + output_bbox_size = batchSize; + if (!bStabilizeBody) output_bbox_size = 25; + output_bbox_data.assign(output_bbox_size, {0.f, 0.f, 0.f, 0.f}); + output_bboxes.boxes = output_bbox_data.data(); + output_bboxes.max_boxes = (uint8_t)output_bbox_size; + output_bboxes.num_boxes = (uint8_t)output_bbox_size; + nvErr = + NvAR_SetObject(keyPointDetectHandle, NvAR_Parameter_Output(BoundingBoxes), &output_bboxes, sizeof(NvAR_BBoxes)); + BAIL_IF_CVERR(nvErr, err, BodyEngine::Err::errParameter); + +bail: + return err; +} + +void BodyEngine::destroyFeatures() { + if (stream) { + NvAR_CudaStreamDestroy(stream); + stream = 0; + } + releaseFeatureIOParams(); + destroyBodyDetectionFeature(); + destroyKeyPointDetectionFeature(); +} + +void BodyEngine::destroyBodyDetectionFeature() { + if (bodyDetectHandle) { + (void)NvAR_Destroy(bodyDetectHandle); + bodyDetectHandle = nullptr; + } +} +void BodyEngine::destroyKeyPointDetectionFeature() { + if (keyPointDetectHandle) { + (void)NvAR_Destroy(keyPointDetectHandle); + keyPointDetectHandle = nullptr; + } +} + +void BodyEngine::releaseFeatureIOParams() { + releaseBodyDetectionIOParams(); + releaseKeyPointDetectionIOParams(); +} + +void BodyEngine::releaseBodyDetectionIOParams() { + NvCVImage_Dealloc(&inputImageBuffer); + if (!output_bbox_data.empty()) output_bbox_data.clear(); + if (!output_bbox_conf_data.empty()) output_bbox_conf_data.clear(); +} + +void BodyEngine::releaseKeyPointDetectionIOParams() { + NvCVImage_Dealloc(&inputImageBuffer); + if (!output_bbox_data.empty()) output_bbox_data.clear(); + if (!keypoints.empty()) keypoints.clear(); + if (!keypoints3D.empty()) keypoints3D.clear(); + if (!jointAngles.empty()) jointAngles.clear(); + if (!keypoints_confidence.empty()) keypoints_confidence.clear(); +} + +unsigned BodyEngine::findBodyBoxes() { + NvCV_Status nvErr = NvAR_Run(bodyDetectHandle); + if (NVCV_SUCCESS != nvErr) return 0; + return (unsigned)output_bboxes.num_boxes; +} + +NvAR_Rect* BodyEngine::getLargestBox() { + NvAR_Rect *box, *bigBox, *lastBox; + float maxArea, area; + for (lastBox = (box = &output_bboxes.boxes[0]) + output_bboxes.num_boxes, bigBox = nullptr, maxArea = 0; + box != lastBox; ++box) { + if (maxArea < (area = box->width * box->height)) { + maxArea = area; + bigBox = box; + } + } + return bigBox; +} + +NvAR_BBoxes* BodyEngine::getBoundingBoxes() { return &output_bboxes; } + +void BodyEngine::enlargeAndSquarifyImageBox(float enlarge, NvAR_Rect& box, int FLAG_variant) { + NvAR_Vector2f size = {box.width * .5f, box.height * .5f}; + NvAR_Point2f center = {box.x + size.x, box.y + size.y}; + float t; + + size.x *= (1.f + enlarge); + size.y *= (1.f + enlarge); + + if (!(FLAG_variant & 1)) /* Default: enforce square bounding box */ + { + if (size.x < size.y) /* Make square */ + size.x = size.y; + else + size.y = size.x; + } + + if (center.x < size.x) /* Shift box into image left-right */ + center.x = size.x; + else if (center.x > (t = input_image_width - size.x)) + center.x = t; + + if (center.y < size.y) /* Shift box into image up-down */ + center.y = size.y; + else if (center.y > (t = input_image_height - size.y)) + center.y = t; + + // TODO: Above we assume that the box is smaller than the image. + + box.width = roundf(size.x * 2.f); /* Integral box */ + box.height = roundf(size.y * 2.f); + box.x = roundf(center.x - box.width * .5f); + box.y = roundf(center.y - box.height * .5f); +} + + NvCV_Status BodyEngine::findKeyPoints() { + NvCV_Status nvErr; +#ifdef DEBUG_PERF_RUNTIME + auto start = std::chrono::high_resolution_clock::now(); +#endif + nvErr = NvAR_Run(keyPointDetectHandle); + if (NVCV_SUCCESS != nvErr) { + return nvErr; + } +#ifdef DEBUG_PERF_RUNTIME + auto end = std::chrono::high_resolution_clock::now(); + auto duration = std::chrono::duration_cast(end - start); + std::cout << "[bodypose] > NvAR_Run(keyPointDetectHandle): " << duration.count() << " microseconds" << std::endl; +#endif + + if (getAverageKeyPointsConfidence() < confidenceThreshold) { + return NVCV_ERR_GENERAL; + } else { + NvAR_Point2f *pt, *endPt; + int i = 0; + for (endPt = (pt = getKeyPoints()) + numKeyPoints; pt != endPt; ++pt, i += 2) { + for (int j = 1; j < batchSize; j++) { + pt->x += pt[j * numKeyPoints].x; + pt->y += pt[j * numKeyPoints].y; + } + // average batch of inferences to generate final result keypoints + pt->x /= batchSize; + pt->y /= batchSize; + } + } +#ifdef DEBUG_PERF_RUNTIME + end = std::chrono::high_resolution_clock::now(); + duration = std::chrono::duration_cast(end - start); + std::cout << "[bodypose] inside findKeyPoints(): " << duration.count() << " microseconds" << std::endl; +#endif + return NVCV_SUCCESS; + } + +NvAR_Point2f* BodyEngine::getKeyPoints() { return keypoints.data(); } + +NvAR_Point3f* BodyEngine::getKeyPoints3D() { return keypoints3D.data(); } + +NvAR_Quaternion* BodyEngine::getJointAngles() { return jointAngles.data(); } + + float* BodyEngine::getKeyPointsConfidence() { return keypoints_confidence.data(); } + + float BodyEngine::getAverageKeyPointsConfidence() { + float average_confidence = 0.0f; + float* keypoints_confidence_all = getKeyPointsConfidence(); + for (int i = 0; i < batchSize; i++) { + for (int j = 0; j < numKeyPoints; j++) { + average_confidence += keypoints_confidence_all[i * numKeyPoints + j]; + } + } + average_confidence /= batchSize * numKeyPoints; + return average_confidence; + } + +unsigned BodyEngine::findLargestBodyBox(NvAR_Rect& bodyBox, int variant) { + unsigned n; + NvAR_Rect* pBodyBox; + + n = findBodyBoxes(); + if (n >= 1) { + pBodyBox = getLargestBox(); + if (nullptr == pBodyBox) { + bodyBox.x = bodyBox.y = bodyBox.width = bodyBox.height = 0.0f; + } else { + bodyBox = *pBodyBox; + } + //enlargeAndSquarifyImageBox(.2f, bodyBox, variant); + } + return n; +} + +unsigned BodyEngine::acquireBodyBox(cv::Mat& src, NvAR_Rect& bodyBox, int variant) { + unsigned n = 0; + NvCVImage fxSrcChunkyCPU; + (void)NVWrapperForCVMat(&src, &fxSrcChunkyCPU); + NvCV_Status cvErr = NvCVImage_Transfer(&fxSrcChunkyCPU, &inputImageBuffer, 1.0f, stream, &tmpImage); + + if (NVCV_SUCCESS != cvErr) { + return n; + } + + n = findLargestBodyBox(bodyBox, variant); + return n; +} + +unsigned BodyEngine::acquireBodyBoxAndKeyPoints(cv::Mat& src, NvAR_Point2f* refMarks, NvAR_Point3f* refKeyPoints3D, + NvAR_Quaternion* refJointAngles, NvAR_Rect& bodyBox, int /*variant*/) { + unsigned n = 0; + NvCVImage fxSrcChunkyCPU; + (void)NVWrapperForCVMat(&src, &fxSrcChunkyCPU); + NvCV_Status cvErr = NvCVImage_Transfer(&fxSrcChunkyCPU, &inputImageBuffer, 1.0f, stream, &tmpImage); + + if (NVCV_SUCCESS != cvErr) { + return n; + } +#ifdef DEBUG_PERF_RUNTIME + auto start = std::chrono::high_resolution_clock::now(); +#endif + if (findKeyPoints() != NVCV_SUCCESS) return 0; + bodyBox = output_bboxes.boxes[0]; + n = 1; +#ifdef DEBUG_PERF_RUNTIME + auto start2 = std::chrono::high_resolution_clock::now(); +#endif + memcpy(refMarks, getKeyPoints(), sizeof(NvAR_Point2f) * numKeyPoints); + memcpy(refKeyPoints3D, getKeyPoints3D(), sizeof(NvAR_Point3f) * numKeyPoints); + memcpy(refJointAngles, getJointAngles(), sizeof(NvAR_Quaternion) * numKeyPoints); + +#ifdef DEBUG_PERF_RUNTIME + auto end = std::chrono::high_resolution_clock::now(); + auto duration3 = std::chrono::duration_cast(start2 - start); + std::cout << "[bodypose] run findKeyPoints(): " << duration3.count() << " microseconds" << std::endl; + auto duration2 = std::chrono::duration_cast(end - start2); + std::cout << "[bodypose] keypoint copy time: " << duration2.count() << " microseconds" << std::endl; + auto duration = std::chrono::duration_cast(end - start); + std::cout << "[bodypose] end-to-end time: " << duration.count() << " microseconds" << std::endl; +#endif + return n; +} + +void BodyEngine::setBodyStabilization(bool _bStabilizeBody) { bStabilizeBody = _bStabilizeBody; } + +void BodyEngine::setMode(int _mode) { nvARMode = _mode; } + +void BodyEngine::setFocalLength(float _bFocalLength) { bFocalLength = _bFocalLength; } + +void BodyEngine::useCudaGraph(bool _bUseCudaGraph) { bUseCudaGraph = _bUseCudaGraph; } + +void BodyEngine::setAppMode(BodyEngine::mode _mode) { appMode = _mode; } diff --git a/samples/BodyTrack/BodyEngine.h b/samples/BodyTrack/BodyEngine.h new file mode 100644 index 0000000..a8b7618 --- /dev/null +++ b/samples/BodyTrack/BodyEngine.h @@ -0,0 +1,200 @@ +/*############################################################################### +# +# Copyright 2020 NVIDIA Corporation +# +# Permission is hereby granted, free of charge, to any person obtaining a copy of +# this software and associated documentation files (the "Software"), to deal in +# the Software without restriction, including without limitation the rights to +# use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +# the Software, and to permit persons to whom the Software is furnished to do so, +# subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +# FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +# COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +# IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +# CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +# +###############################################################################*/ +#ifndef __BODY_ENGINE__ +#define __BODY_ENGINE__ + +#include +#include +#include "nvAR.h" +#include "nvCVOpenCV.h" +#include "opencv2/opencv.hpp" +// #include "FeatureVertexName.h" +#define FITBODY_PRIVATE + +class KalmanFilter1D { + private: + float Q_; // Covariance of the process noise + float xhat_; // Current prediction + float xhatminus_; // Previous prediction + float P_; // Estimated accuracy of xhat_ + float Pminus_; // Previous P_ + float K_; // Kalman gain + float R_; // Covariance of the observation noise + bool bFirstUse_; + + public: + KalmanFilter1D() { reset(); } + + KalmanFilter1D(float Q, float R) { reset(Q, R); } + + void reset() { + R_ = 0.005f * 0.005f; + Q_ = 1e-5f; + xhat_ = 0.0f; + xhatminus_ = 0.0f; + P_ = 1; + bFirstUse_ = true; + Pminus_ = 0.0f; + K_ = 0.0f; + } + + void reset(float Q, float R) { + reset(); + Q_ = Q; + R_ = R; + } + + float update(float val) { + if (bFirstUse_) { + xhat_ = val; + bFirstUse_ = false; + } + + xhatminus_ = xhat_; + Pminus_ = P_ + Q_; + K_ = Pminus_ / (Pminus_ + R_); + xhat_ = xhatminus_ + K_ * (val - xhatminus_); + P_ = (1 - K_) * Pminus_; + + return xhat_; + } +}; + +bool CheckResult(NvCV_Status nvErr, unsigned line); + +#define BAIL_IF_ERR(err) \ +do { \ + if (0!=err) { \ + goto bail; \ + } \ + } while (0) + +#define BAIL_IF_CVERR(nvErr, err, code) \ + do { \ + if (!CheckResult(nvErr, __LINE__)) { \ + err = code; \ + goto bail; \ + } \ + } while (0) + +typedef struct KeyPointsProperties { + int numPoints; + float confidence_threshold; +}KeyPointsProperties; + +// This default focal length matches a logitech webcam +static const float FOCAL_LENGTH_DEFAULT = 800.f; + +/******************************************************************************** + * BodyEngine + ********************************************************************************/ + +class BodyEngine { + public: + enum Err { errNone, errGeneral, errRun, errInitialization, errRead, errEffect, errParameter }; + int input_image_width, input_image_height, input_image_pitch; + + void setInputImageWidth(int width) { input_image_width = width; } + void setInputImageHeight(int height) { input_image_height = height; } + int getInputImageWidth() { return input_image_width; } + int getInputImageHeight() { return input_image_height; } + int getInputImagePitch() { return input_image_pitch = input_image_width * 3 * sizeof(unsigned char); } + void setBodyModel(const char *bodyModel) { body_model = bodyModel; } + + Err createFeatures(const char* modelPath, unsigned int _batchSize = 1); + Err createBodyDetectionFeature(const char* modelPath, CUstream stream); + Err createKeyPointDetectionFeature(const char* modelPath, unsigned int batchSize, CUstream stream); + void destroyFeatures(); + void destroyBodyDetectionFeature(); + void destroyKeyPointDetectionFeature(); + Err initFeatureIOParams(); + Err initBodyDetectionIOParams(NvCVImage* _inputImageBuffer); + Err initKeyPointDetectionIOParams(NvCVImage* _inputImageBuffer); + void releaseFeatureIOParams(); + void releaseBodyDetectionIOParams(); + void releaseKeyPointDetectionIOParams(); + + unsigned findBodyBoxes(); + NvAR_Rect* getLargestBox(); + NvCV_Status findKeyPoints(); + NvAR_BBoxes* getBoundingBoxes(); + NvAR_Point2f* getKeyPoints(); + NvAR_Point3f* getKeyPoints3D(); + NvAR_Quaternion* getJointAngles(); + float* getKeyPointsConfidence(); + float getAverageKeyPointsConfidence(); + void enlargeAndSquarifyImageBox(float enlarge, NvAR_Rect& box, int FLAG_variant); + unsigned findLargestBodyBox(NvAR_Rect& bodyBox, int variant = 0); + unsigned acquireBodyBox(cv::Mat& src, NvAR_Rect& bodyBox, int variant = 0); + unsigned acquireBodyBoxAndKeyPoints(cv::Mat& src, NvAR_Point2f* refMarks, NvAR_Point3f* refKeyPoints3D, + NvAR_Quaternion* refJointAngles, NvAR_Rect& bodyBox, int variant = 0); + void setBodyStabilization(bool); + void setMode(int); + void setFocalLength(float); + void useCudaGraph(bool); // Using cuda graph improves model latency + int getNumKeyPoints() { return numKeyPoints; } + std::vector getReferencePose() { return referencePose; } + + NvCVImage inputImageBuffer{}, tmpImage{}; + NvAR_FeatureHandle bodyDetectHandle{}, keyPointDetectHandle{}; + std::vector keypoints; + std::vector keypoints_confidence; + std::vector keypoints3D; + std::vector jointAngles; + CUstream stream{}; + std::vector output_bbox_data; + std::vector output_bbox_conf_data; + NvAR_BBoxes output_bboxes{}; + int batchSize; + int nvARMode; + std::mt19937 ran; + unsigned int numKeyPoints; + std::vector referencePose; + float confidenceThreshold; + std::string body_model; + + bool bStabilizeBody; + bool bUseOTAU; + char *bdOTAModelPath, *ldOTAModelPath; + float bFocalLength; + bool bUseCudaGraph; + + BodyEngine() { + batchSize = 1; + nvARMode = 1; + bStabilizeBody = true; + bUseCudaGraph = true; + bFocalLength = FOCAL_LENGTH_DEFAULT; + confidenceThreshold = 0.f; + appMode = keyPointDetection; + input_image_width = 960; + input_image_height = 544; + input_image_pitch = 3 * input_image_width * sizeof(unsigned char); // RGB + bUseOTAU = false; + bdOTAModelPath = NULL; + ldOTAModelPath = NULL; + } + enum mode { bodyDetection = 0, keyPointDetection } appMode; + void setAppMode(BodyEngine::mode _mAppMode); +}; +#endif diff --git a/samples/BodyTrack/BodyTrack.cpp b/samples/BodyTrack/BodyTrack.cpp new file mode 100644 index 0000000..01eed8d --- /dev/null +++ b/samples/BodyTrack/BodyTrack.cpp @@ -0,0 +1,1022 @@ +/*############################################################################### +# +# Copyright 2020 NVIDIA Corporation +# +# Permission is hereby granted, free of charge, to any person obtaining a copy of +# this software and associated documentation files (the "Software"), to deal in +# the Software without restriction, including without limitation the rights to +# use, copy, modify, merge, publish, distribute, sublicense, and/or sell copies of +# the Software, and to permit persons to whom the Software is furnished to do so, +# subject to the following conditions: +# +# The above copyright notice and this permission notice shall be included in all +# copies or substantial portions of the Software. +# +# THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR +# IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY, FITNESS +# FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR +# COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, WHETHER +# IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM, OUT OF OR IN +# CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. +# +###############################################################################*/ +#include +#include +#include +#include +#include +#include +#include +#include + +#include "BodyEngine.h" +#include "RenderingUtils.h" +#include "nvAR.h" +#include "nvAR_defs.h" +#include "opencv2/opencv.hpp" + +#ifndef M_PI +#define M_PI 3.1415926535897932385 +#endif /* M_PI */ +#ifndef M_2PI +#define M_2PI 6.2831853071795864769 +#endif /* M_2PI */ +#ifndef M_PI_2 +#define M_PI_2 1.5707963267948966192 +#endif /* M_PI_2 */ +#define F_PI ((float)M_PI) +#define F_PI_2 ((float)M_PI_2) +#define F_2PI ((float)M_2PI) + +#ifdef _MSC_VER +#define strcasecmp _stricmp +#endif /* _MSC_VER */ + +#define BAIL(err, code) \ + do { \ + err = code; \ + goto bail; \ + } while (0) + +#define DEBUG_RUNTIME + +/******************************************************************************** + * Command-line arguments + ********************************************************************************/ + +bool FLAG_debug = false, FLAG_verbose = false, FLAG_temporal = true, FLAG_captureOutputs = false, + FLAG_offlineMode = false, FLAG_useCudaGraph = true; +std::string FLAG_outDir, FLAG_inFile, FLAG_outFile, FLAG_modelPath, FLAG_captureCodec = "avc1", + FLAG_camRes, FLAG_bodyModel; +unsigned int FLAG_batch = 1, FLAG_appMode = 1, FLAG_mode = 1; + +/******************************************************************************** + * Usage + ********************************************************************************/ + +static void Usage() { + printf( + "BodyTrack [ ...]\n" + "where is\n" + " --verbose[=(true|false)] report interesting info\n" + " --debug[=(true|false)] report debugging info\n" + " --temporal[=(true|false)] temporally optimize body rect and keypoints\n" + " --use_cuda_graph[=(true|false)] enable faster execution by using cuda graph to capture engine execution\n" + " --capture_outputs[=(true|false)] enables video/image capture and writing body detection/keypoints outputs\n" + " --offline_mode[=(true|false)] disables webcam, reads video from file and writes output video results\n" + " --cam_res=[WWWx]HHH specify resolution as height or width x height\n" + " --in_file= specify the input file\n" + " --codec= FOURCC code for the desired codec (default H264)\n" + " --in= specify the input file\n" + " --out_file= specify the output file\n" + " --out= specify the output file\n" + " --model_path= specify the directory containing the TRT models\n" + " --batch= 1 - 8, used for batch inferencing in keypoints detector \n" + " --mode[=0|1] Model Mode. 0: High Quality, 1: High Performance\n" + " --app_mode[=(0|1)] App mode. 0: Body detection, 1: Keypoint detection " + "(Default).\n" + " --benchmarks[=] run benchmarks\n"); +} + +static bool GetFlagArgVal(const char *flag, const char *arg, const char **val) { + if (*arg != '-') { + return false; + } + while (*++arg == '-') { + continue; + } + const char *s = strchr(arg, '='); + if (s == NULL) { + if (strcmp(flag, arg) != 0) { + return false; + } + *val = NULL; + return true; + } + unsigned n = (unsigned)(s - arg); + if ((strlen(flag) != n) || (strncmp(flag, arg, n) != 0)) { + return false; + } + *val = s + 1; + return true; +} + +static bool GetFlagArgVal(const char *flag, const char *arg, std::string *val) { + const char *valStr; + if (!GetFlagArgVal(flag, arg, &valStr)) return false; + val->assign(valStr ? valStr : ""); + return true; +} + +static bool GetFlagArgVal(const char *flag, const char *arg, bool *val) { + const char *valStr; + bool success = GetFlagArgVal(flag, arg, &valStr); + if (success) { + *val = (valStr == NULL || strcasecmp(valStr, "true") == 0 || strcasecmp(valStr, "on") == 0 || + strcasecmp(valStr, "yes") == 0 || strcasecmp(valStr, "1") == 0); + } + return success; +} + +bool GetFlagArgVal(const char *flag, const char *arg, long *val) { + const char *valStr; + bool success = GetFlagArgVal(flag, arg, &valStr); + if (success) { + *val = strtol(valStr, NULL, 10); + } + return success; +} + +static bool GetFlagArgVal(const char *flag, const char *arg, unsigned *val) { + long longVal; + bool success = GetFlagArgVal(flag, arg, &longVal); + if (success) { + *val = (unsigned)longVal; + } + return success; +} + +/******************************************************************************** + * StringToFourcc + ********************************************************************************/ + +static int StringToFourcc(const std::string &str) { + union chint { + int i; + char c[4]; + }; + chint x = {0}; + for (int n = (str.size() < 4) ? (int)str.size() : 4; n--;) x.c[n] = str[n]; + return x.i; +} + +/******************************************************************************** + * ParseMyArgs + ********************************************************************************/ + +static int ParseMyArgs(int argc, char **argv) { + int errs = 0; + for (--argc, ++argv; argc--; ++argv) { + bool help; + const char *arg = *argv; + if (arg[0] != '-') { + continue; + } else if ((arg[1] == '-') && + (GetFlagArgVal("verbose", arg, &FLAG_verbose) || GetFlagArgVal("debug", arg, &FLAG_debug) || + GetFlagArgVal("in", arg, &FLAG_inFile) || GetFlagArgVal("in_file", arg, &FLAG_inFile) || + GetFlagArgVal("out", arg, &FLAG_outFile) || GetFlagArgVal("out_file", arg, &FLAG_outFile) || + GetFlagArgVal("offline_mode", arg, &FLAG_offlineMode) || + GetFlagArgVal("capture_outputs", arg, &FLAG_captureOutputs) || + GetFlagArgVal("cam_res", arg, &FLAG_camRes) || GetFlagArgVal("codec", arg, &FLAG_captureCodec) || + GetFlagArgVal("model_path", arg, &FLAG_modelPath) || + GetFlagArgVal("app_mode", arg, &FLAG_appMode) || + GetFlagArgVal("mode", arg, &FLAG_mode) || + GetFlagArgVal("use_cuda_graph", arg, &FLAG_useCudaGraph) || + GetFlagArgVal("temporal", arg, &FLAG_temporal))) { + continue; + } else if (GetFlagArgVal("help", arg, &help)) { + Usage(); + } else if (arg[1] != '-') { + for (++arg; *arg; ++arg) { + if (*arg == 'v') { + FLAG_verbose = true; + } else { + // printf("Unknown flag: \"-%c\"\n", *arg); + } + } + continue; + } else { + // printf("Unknown flag: \"%s\"\n", arg); + } + } + return errs; +} + +enum { + myErrNone = 0, + myErrShader = -1, + myErrProgram = -2, + myErrTexture = -3, +}; + +static const cv::Scalar cv_colors[] = { cv::Scalar(0, 0, 255), cv::Scalar(0, 255, 0), cv::Scalar(255, 0, 0) }; +enum { + kColorRed = 0, + kColorGreen = 1, + kColorBlue = 2 +}; + +#if 1 +class MyTimer { + public: + void start() { t0 = std::chrono::high_resolution_clock::now(); } /**< Start the timer. */ + void pause() { dt = std::chrono::high_resolution_clock::now() - t0; } /**< Pause the timer. */ + void resume() { t0 = std::chrono::high_resolution_clock::now() - dt; } /**< Resume the timer. */ + void stop() { pause(); } /**< Stop the timer. */ + double elapsedTimeFloat() const { + return std::chrono::duration(dt).count(); + } /**< Report the elapsed time as a float. */ + private: + std::chrono::high_resolution_clock::time_point t0; + std::chrono::high_resolution_clock::duration dt; +}; +#endif + +std::string getCalendarTime() { + // Get the current time + std::chrono::system_clock::time_point currentTimePoint = std::chrono::system_clock::now(); + // Convert to time_t from time_point + std::time_t currentTime = std::chrono::system_clock::to_time_t(currentTimePoint); + // Convert to tm to get structure holding a calendar date and time broken down into its components. + std::tm brokenTime = *std::localtime(¤tTime); + std::ostringstream calendarTime; + calendarTime << std::put_time( + &brokenTime, + "%Y-%m-%d-%H-%M-%S"); // (YYYY-MM-DD-HH-mm-ss)----- + // Get the time since epoch 0(Thu Jan 1 00:00:00 1970) and the remainder after division is + // our milliseconds + std::chrono::milliseconds currentMilliseconds = + std::chrono::duration_cast(currentTimePoint.time_since_epoch()) % 1000; + // Append the milliseconds to the stream + calendarTime << "-" << std::setfill('0') << std::setw(3) << currentMilliseconds.count(); // milliseconds + return calendarTime.str(); +} + +class DoApp { + public: + enum Err { + errNone = BodyEngine::Err::errNone, + errGeneral = BodyEngine::Err::errGeneral, + errRun = BodyEngine::Err::errRun, + errInitialization = BodyEngine::Err::errInitialization, + errRead = BodyEngine::Err::errRead, + errEffect = BodyEngine::Err::errEffect, + errParameter = BodyEngine::Err::errParameter, + errUnimplemented, + errMissing, + errVideo, + errImageSize, + errNotFound, + errBodyModelInit, + errGLFWInit, + errGLInit, + errRendererInit, + errGLResource, + errGLGeneric, + errBodyFit, + errNoBody, + errSDK, + errCuda, + errCancel, + errCamera + }; + Err doAppErr(BodyEngine::Err status) { return (Err)status; } + BodyEngine body_ar_engine; + DoApp(); + ~DoApp(); + + void stop(); + Err initBodyEngine(const char *modelPath = nullptr); + Err initCamera(const char *camRes = nullptr); + Err initOfflineMode(const char *inputFilename = nullptr, const char *outputFilename = nullptr); + Err acquireFrame(); + Err acquireBodyBox(); + Err acquireBodyBoxAndKeyPoints(); + Err run(); + void drawFPS(cv::Mat &img); + void DrawBBoxes(const cv::Mat &src, NvAR_Rect *output_bbox); + void DrawKeyPointLine(const cv::Mat& src, NvAR_Point2f* keypoints, int point1, int point2, int color); + void DrawKeyPointsAndEdges(const cv::Mat &src, NvAR_Point2f *keypoints, int numKeyPoints, NvAR_Rect* output_bbox); + void drawKalmanStatus(cv::Mat &img); + void drawVideoCaptureStatus(cv::Mat &img); + void processKey(int key); + void writeVideoAndEstResults(const cv::Mat &frame, NvAR_BBoxes output_bboxes, NvAR_Point2f *keypoints = NULL); + void writeFrameAndEstResults(const cv::Mat &frame, NvAR_BBoxes output_bboxes, NvAR_Point2f *keypoints = NULL); + void writeEstResults(std::ofstream &outputFile, NvAR_BBoxes output_bboxes, NvAR_Point2f *keypoints = NULL); + void getFPS(); + static const char *errorStringFromCode(Err code); + + cv::VideoCapture cap{}; + cv::Mat frame; + int inputWidth, inputHeight; + cv::VideoWriter bodyDetectOutputVideo{}, keyPointsOutputVideo{}; + int frameIndex; + static const char windowTitle[]; + double frameTime; + // std::chrono::high_resolution_clock::time_point frameTimer; + MyTimer frameTimer; + cv::VideoWriter capturedVideo; + std::ofstream bodyEngineVideoOutputFile; + + BodyEngine::Err nvErr; + float expr[6]; + bool drawVisualization, showFPS, captureVideo, captureFrame; + float scaleOffsetXY[4]; +}; + +DoApp *gApp = nullptr; +const char DoApp::windowTitle[] = "BodyTrack App"; + +void DoApp::processKey(int key) { + switch (key) { + case '2': + body_ar_engine.destroyFeatures(); + body_ar_engine.setAppMode(BodyEngine::mode::keyPointDetection); + body_ar_engine.createFeatures(FLAG_modelPath.c_str()); + body_ar_engine.initFeatureIOParams(); + break; + case '1': + body_ar_engine.destroyFeatures(); + body_ar_engine.setAppMode(BodyEngine::mode::bodyDetection); + body_ar_engine.createFeatures(FLAG_modelPath.c_str()); + body_ar_engine.initFeatureIOParams(); + break; + case 'C': + case 'c': + captureVideo = !captureVideo; + break; + case 'S': + case 's': + captureFrame = !captureFrame; + break; + case 'W': + case 'w': + drawVisualization = !drawVisualization; + break; + case 'F': + case 'f': + showFPS = !showFPS; + break; + default: + break; + } +} + +DoApp::Err DoApp::initBodyEngine(const char *modelPath) { + if (!cap.isOpened()) return errVideo; + + int numKeyPoints = body_ar_engine.getNumKeyPoints(); + + nvErr = body_ar_engine.createFeatures(modelPath); + +#ifdef DEBUG + detector->setOutputLocation(outputDir); +#endif // DEBUG + +#define VISUALIZE +#ifdef VISUALIZE + if (!FLAG_offlineMode) cv::namedWindow(windowTitle, 1); +#endif // VISUALIZE + + frameIndex = 0; + + return doAppErr(nvErr); +} + +void DoApp::stop() { + body_ar_engine.destroyFeatures(); + + if (FLAG_offlineMode) { + bodyDetectOutputVideo.release(); + keyPointsOutputVideo.release(); + } + cap.release(); +#ifdef VISUALIZE + cv::destroyAllWindows(); +#endif // VISUALIZE +} + +void DoApp::DrawBBoxes(const cv::Mat &src, NvAR_Rect *output_bbox) { + cv::Mat frm; + if (FLAG_offlineMode) + frm = src.clone(); + else + frm = src; + + if (output_bbox) + cv::rectangle(frm, cv::Point(lround(output_bbox->x), lround(output_bbox->y)), + cv::Point(lround(output_bbox->x + output_bbox->width), lround(output_bbox->y + output_bbox->height)), + cv::Scalar(255, 0, 0), 2); + if (FLAG_offlineMode) bodyDetectOutputVideo.write(frm); +} + +void DoApp::writeVideoAndEstResults(const cv::Mat &frm, NvAR_BBoxes output_bboxes, NvAR_Point2f* keypoints) { + if (captureVideo) { + if (!capturedVideo.isOpened()) { + const std::string currentCalendarTime = getCalendarTime(); + const std::string capturedOutputFileName = currentCalendarTime + ".mp4"; + getFPS(); + if (frameTime) { + float fps = (float)(1.0 / frameTime); + capturedVideo.open(capturedOutputFileName, StringToFourcc(FLAG_captureCodec), fps, + cv::Size(frm.cols, frm.rows)); + if (!capturedVideo.isOpened()) { + std::cout << "Error: Could not open video: \"" << capturedOutputFileName << "\"\n"; + return; + } + if (FLAG_verbose) { + std::cout << "Capturing video started" << std::endl; + } + } else { // If frameTime is 0.f, returns without writing the frame to the Video + return; + } + const std::string outputsFileName = currentCalendarTime + ".txt"; + bodyEngineVideoOutputFile.open(outputsFileName, std::ios_base::out); + if (!bodyEngineVideoOutputFile.is_open()) { + std::cout << "Error: Could not open file: \"" << outputsFileName << "\"\n"; + return; + } + std::string keyPointDetectionMode = (keypoints == NULL) ? "Off" : "On"; + bodyEngineVideoOutputFile << "// BodyDetectOn, KeyPointDetect" << keyPointDetectionMode << "\n "; + bodyEngineVideoOutputFile + << "// kNumPeople, (bbox_x, bbox_y, bbox_w, bbox_h){ kNumPeople}, kNumLMs, [lm_x, lm_y]{kNumLMs}\n"; + } + // Write each frame to the Video + capturedVideo << frm; + writeEstResults(bodyEngineVideoOutputFile, output_bboxes, keypoints); + } else { + if (capturedVideo.isOpened()) { + if (FLAG_verbose) { + std::cout << "Capturing video ended" << std::endl; + } + capturedVideo.release(); + if (bodyEngineVideoOutputFile.is_open()) bodyEngineVideoOutputFile.close(); + } + } +} + +void DoApp::writeEstResults(std::ofstream &outputFile, NvAR_BBoxes output_bboxes, NvAR_Point2f* keypoints) { + /** + * Output File Format : + * BodyDetectOn, KeyPointDetectOn + * kNumPeople, (bbox_x, bbox_y, bbox_w, bbox_h){ kNumPeople}, kNumKPs, [j_x, j_y]{kNumKPs} + */ + + int bodyDetectOn = (body_ar_engine.appMode == BodyEngine::mode::bodyDetection || + body_ar_engine.appMode == BodyEngine::mode::keyPointDetection) + ? 1 + : 0; + int keyPointDetectOn = (body_ar_engine.appMode == BodyEngine::mode::keyPointDetection) + ? 1 + : 0; + outputFile << bodyDetectOn << "," << keyPointDetectOn << "\n"; + + if (bodyDetectOn && output_bboxes.num_boxes) { + // Append number of bodies detected in the current frame + outputFile << unsigned(output_bboxes.num_boxes) << ","; + // write outputbboxes to outputFile + for (size_t i = 0; i < output_bboxes.num_boxes; i++) { + int x1 = (int)output_bboxes.boxes[i].x, y1 = (int)output_bboxes.boxes[i].y, + width = (int)output_bboxes.boxes[i].width, height = (int)output_bboxes.boxes[i].height; + outputFile << x1 << "," << y1 << "," << width << "," << height << ","; + } + } else { + outputFile << "0,"; + } + if (keyPointDetectOn && output_bboxes.num_boxes) { + int numKeyPoints = body_ar_engine.getNumKeyPoints(); + // Append number of keypoints + outputFile << numKeyPoints << ","; + // Append 2 * number of keypoint values + NvAR_Point2f *pt, *endPt; + for (endPt = (pt = (NvAR_Point2f *)keypoints) + numKeyPoints; pt < endPt; ++pt) + outputFile << pt->x << "," << pt->y << ","; + } else { + outputFile << "0,"; + } + + outputFile << "\n"; +} + +void DoApp::writeFrameAndEstResults(const cv::Mat &frm, NvAR_BBoxes output_bboxes, NvAR_Point2f* keypoints) { + if (captureFrame) { + const std::string currentCalendarTime = getCalendarTime(); + const std::string capturedFrame = currentCalendarTime + ".png"; + cv::imwrite(capturedFrame, frm); + if (FLAG_verbose) { + std::cout << "Captured the frame" << std::endl; + } + // Write Body Engine Outputs + const std::string outputFilename = currentCalendarTime + ".txt"; + std::ofstream outputFile; + outputFile.open(outputFilename, std::ios_base::out); + if (!outputFile.is_open()) { + std::cout << "Error: Could not open file: \"" << outputFilename << "\"\n"; + return; + } + std::string keyPointDetectionMode = (keypoints == NULL) ? "Off" : "On"; + outputFile << "// BodyDetectOn, KeyPointDetect" << keyPointDetectionMode << "\n"; + outputFile << "// kNumPeople, (bbox_x, bbox_y, bbox_w, bbox_h){ kNumPeople}, kNumLMs, [lm_x, lm_y]{kNumLMs}\n"; + writeEstResults(outputFile, output_bboxes, keypoints); + if (outputFile.is_open()) outputFile.close(); + captureFrame = false; + } +} + +void DoApp::DrawKeyPointLine(const cv::Mat& src, NvAR_Point2f* keypoints, int point1, int point2, int color) { + NvAR_Point2f point1_pos = *(keypoints + point1); + NvAR_Point2f point2_pos = *(keypoints + point2); + cv::line(src, cv::Point((int)point1_pos.x, (int)point1_pos.y), cv::Point((int)point2_pos.x, (int)point2_pos.y), cv_colors[color], 2); + +} + +void DoApp::DrawKeyPointsAndEdges(const cv::Mat& src, NvAR_Point2f* keypoints, int numKeyPoints, NvAR_Rect* output_bbox) { + cv::Mat frm; + if (FLAG_offlineMode) + frm = src.clone(); + else + frm = src; + NvAR_Point2f *pt, *endPt; + for (endPt = (pt = (NvAR_Point2f *)keypoints) + numKeyPoints; pt < endPt; ++pt) + cv::circle(frm, cv::Point(lround(pt->x), lround(pt->y)), 4, cv::Scalar(180, 180, 180), -1); + + if (output_bbox) + cv::rectangle(frm, cv::Point(lround(output_bbox->x), lround(output_bbox->y)), + cv::Point(lround(output_bbox->x + output_bbox->width), lround(output_bbox->y + output_bbox->height)), + cv::Scalar(255, 0, 0), 2); + + int pelvis = 0; + int left_hip = 1; + int right_hip = 2; + int torso = 3; + int left_knee = 4; + int right_knee = 5; + int neck = 6; + int left_ankle = 7; + int right_ankle = 8; + int left_big_toe = 9; + int right_big_toe = 10; + int left_small_toe = 11; + int right_small_toe = 12; + int left_heel = 13; + int right_heel = 14; + int nose = 15; + int left_eye = 16; + int right_eye = 17; + int left_ear = 18; + int right_ear = 19; + int left_shoulder = 20; + int right_shoulder = 21; + int left_elbow = 22; + int right_elbow = 23; + int left_wrist = 24; + int right_wrist = 25; + int left_pinky_knuckle = 26; + int right_pinky_knuckle = 27; + int left_middle_tip = 28; + int right_middle_tip = 29; + int left_index_knuckle = 30; + int right_index_knuckle = 31; + int left_thumb_tip = 32; + int right_thumb_tip = 33; + + // center body + DrawKeyPointLine(frm, keypoints, pelvis, torso, kColorGreen); + DrawKeyPointLine(frm, keypoints, torso, neck, kColorGreen); + DrawKeyPointLine(frm, keypoints, neck, pelvis, kColorGreen); + + // right side + DrawKeyPointLine(frm, keypoints, right_ankle, right_knee, kColorRed); + DrawKeyPointLine(frm, keypoints, right_knee, right_hip, kColorRed); + DrawKeyPointLine(frm, keypoints, right_hip, pelvis, kColorRed); + DrawKeyPointLine(frm, keypoints, right_hip, right_shoulder, kColorRed); + DrawKeyPointLine(frm, keypoints, right_shoulder, right_elbow, kColorRed); + DrawKeyPointLine(frm, keypoints, right_elbow, right_wrist, kColorRed); + DrawKeyPointLine(frm, keypoints, right_shoulder, neck, kColorRed); + + // right side hand and feet + DrawKeyPointLine(frm, keypoints, right_wrist, right_pinky_knuckle, kColorRed); + DrawKeyPointLine(frm, keypoints, right_wrist, right_middle_tip, kColorRed); + DrawKeyPointLine(frm, keypoints, right_wrist, right_index_knuckle, kColorRed); + DrawKeyPointLine(frm, keypoints, right_wrist, right_thumb_tip, kColorRed); + DrawKeyPointLine(frm, keypoints, right_ankle, right_heel, kColorRed); + DrawKeyPointLine(frm, keypoints, right_ankle, right_big_toe, kColorRed); + DrawKeyPointLine(frm, keypoints, right_big_toe, right_small_toe, kColorRed); + + //left side + DrawKeyPointLine(frm, keypoints, left_ankle, left_knee, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_knee, left_hip, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_hip, pelvis, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_hip, left_shoulder, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_shoulder, left_elbow, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_elbow, left_wrist, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_shoulder, neck, kColorBlue); + + // left side hand and feet + DrawKeyPointLine(frm, keypoints, left_wrist, left_pinky_knuckle, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_wrist, left_middle_tip, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_wrist, left_index_knuckle, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_wrist, left_thumb_tip, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_ankle, left_heel, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_ankle, left_big_toe, kColorBlue); + DrawKeyPointLine(frm, keypoints, left_big_toe, left_small_toe, kColorBlue); + + // head + DrawKeyPointLine(frm, keypoints, neck, nose, kColorGreen); + DrawKeyPointLine(frm, keypoints, nose, right_eye, kColorGreen); + DrawKeyPointLine(frm, keypoints, right_eye, right_ear, kColorGreen); + DrawKeyPointLine(frm, keypoints, nose, left_eye, kColorGreen); + DrawKeyPointLine(frm, keypoints, left_eye, left_ear, kColorGreen); + + if (FLAG_offlineMode) keyPointsOutputVideo.write(frm); +} + +DoApp::Err DoApp::acquireFrame() { + Err err = errNone; + + // If the machine goes to sleep with the app running and then wakes up, the camera object is not destroyed but the + // frames we try to read are empty. So we try to re-initialize the camera with the same resolution settings. If the + // resolution has changed, you will need to destroy and create the features again with the new camera resolution (not + // done here) as well as reallocate memory accordingly with BodyEngine::initFeatureIOParams() + cap >> frame; // get a new frame from camera into the class variable frame. + if (frame.empty()) { + // if in Offline mode, this means end of video,so we return + if (FLAG_offlineMode) return errVideo; + // try Init one more time if reading frames from camera + err = initCamera(FLAG_camRes.c_str()); + if (err != errNone) + return err; + cap >> frame; + if (frame.empty()) return errVideo; + } + + return err; +} + +DoApp::Err DoApp::acquireBodyBox() { + Err err = errNone; + NvAR_Rect output_bbox; + + // get keypoints in original image resolution coordinate space + unsigned n = body_ar_engine.acquireBodyBox(frame, output_bbox, 0); + + if (n && FLAG_verbose) { + printf("BodyBox: [\n"); + printf("%7.1f%7.1f%7.1f%7.1f\n", output_bbox.x, output_bbox.y, output_bbox.x + output_bbox.width, + output_bbox.y + output_bbox.height); + printf("]\n"); + } + if (FLAG_captureOutputs) { + writeFrameAndEstResults(frame, body_ar_engine.output_bboxes); + writeVideoAndEstResults(frame, body_ar_engine.output_bboxes); + } + if (0 == n) return errNoBody; + +#ifdef VISUALIZE + + if (drawVisualization) { + DrawBBoxes(frame, &output_bbox); + } +#endif // VISUALIZE + frameIndex++; + + return err; +} + +DoApp::Err DoApp::acquireBodyBoxAndKeyPoints() { + Err err = errNone; + int numKeyPoints = body_ar_engine.getNumKeyPoints(); + NvAR_Rect output_bbox; + std::vector keypoints2D(numKeyPoints); + std::vector keypoints3D(numKeyPoints); + std::vector jointAngles(numKeyPoints); + +#ifdef DEBUG_PERF_RUNTIME + auto start = std::chrono::high_resolution_clock::now(); +#endif + + // get keypoints in original image resolution coordinate space + unsigned n = body_ar_engine.acquireBodyBoxAndKeyPoints(frame, keypoints2D.data(), keypoints3D.data(), + jointAngles.data(), output_bbox, 0); + +#ifdef DEBUG_PERF_RUNTIME + auto end = std::chrono::high_resolution_clock::now(); + auto duration = std::chrono::duration_cast(end - start); + std::cout << "box+keypoints time: " << duration.count() << " microseconds" << std::endl; +#endif + + if (n && FLAG_verbose && body_ar_engine.appMode != BodyEngine::mode::bodyDetection) { + printf("KeyPoints: [\n"); + for (const auto &pt : keypoints2D) { + printf("%7.1f%7.1f\n", pt.x, pt.y); + } + printf("]\n"); + + printf("3d KeyPoints: [\n"); + for (const auto& pt : keypoints3D) { + printf("%7.1f%7.1f%7.1f\n", pt.x, pt.y, pt.z); + } + printf("]\n"); + } + if (FLAG_captureOutputs) { + writeFrameAndEstResults(frame, body_ar_engine.output_bboxes, keypoints2D.data()); + writeVideoAndEstResults(frame, body_ar_engine.output_bboxes, keypoints2D.data()); + } + if (0 == n) return errNoBody; + +#ifdef VISUALIZE + + if (drawVisualization) { + DrawKeyPointsAndEdges(frame, keypoints2D.data(), numKeyPoints, &output_bbox); + if (FLAG_offlineMode) { + DrawBBoxes(frame, &output_bbox); + } + } +#endif // VISUALIZE + frameIndex++; + + return err; +} + +DoApp::Err DoApp::initCamera(const char *camRes) { + if (cap.open(0)) { + if (camRes) { + int n; + n = sscanf(camRes, "%d%*[xX]%d", &inputWidth, &inputHeight); + switch (n) { + case 2: + break; // We have read both width and height + case 1: + inputHeight = inputWidth; + inputWidth = (int)(inputHeight * (4. / 3.) + .5); + break; + default: + inputHeight = 0; + inputWidth = 0; + break; + } + if (inputWidth) cap.set(CV_CAP_PROP_FRAME_WIDTH, inputWidth); + if (inputHeight) cap.set(CV_CAP_PROP_FRAME_HEIGHT, inputHeight); + + inputWidth = (int)cap.get(CV_CAP_PROP_FRAME_WIDTH); + inputHeight = (int)cap.get(CV_CAP_PROP_FRAME_HEIGHT); + body_ar_engine.setInputImageWidth(inputWidth); + body_ar_engine.setInputImageHeight(inputHeight); + } + } else + return errCamera; + return errNone; +} + +DoApp::Err DoApp::initOfflineMode(const char *inputFilename, const char *outputFilename) { + if (cap.open(inputFilename)) { + inputWidth = (int)cap.get(CV_CAP_PROP_FRAME_WIDTH); + inputHeight = (int)cap.get(CV_CAP_PROP_FRAME_HEIGHT); + body_ar_engine.setInputImageWidth(inputWidth); + body_ar_engine.setInputImageHeight(inputHeight); + } else { + printf("ERROR: Unable to open the input video file \"%s\" \n", inputFilename); + return Err::errVideo; + } + + std::string bdOutputVideoName, jdOutputVideoName; + std::string outputFilePrefix; + if (outputFilename && strlen(outputFilename) != 0) { + outputFilePrefix = outputFilename; + } else { + size_t lastindex = std::string(inputFilename).find_last_of("."); + outputFilePrefix = std::string(inputFilename).substr(0, lastindex); + } + bdOutputVideoName = outputFilePrefix + "_bbox.mp4"; + jdOutputVideoName = outputFilePrefix + "_pose.mp4"; + + if (!bodyDetectOutputVideo.open(bdOutputVideoName, StringToFourcc(FLAG_captureCodec), cap.get(CV_CAP_PROP_FPS), + cv::Size(inputWidth, inputHeight))) { + printf("ERROR: Unable to open the output video file \"%s\" \n", bdOutputVideoName.c_str()); + return Err::errGeneral; + } + if (!keyPointsOutputVideo.open(jdOutputVideoName, StringToFourcc(FLAG_captureCodec), cap.get(CV_CAP_PROP_FPS), + cv::Size(inputWidth, inputHeight))) { + printf("ERROR: Unable to open the output video file \"%s\" \n", bdOutputVideoName.c_str()); + return Err::errGeneral; + } + + return Err::errNone; +} + +DoApp::DoApp() { + // Make sure things are initialized properly + gApp = this; + drawVisualization = true; + showFPS = false; + captureVideo = false; + captureFrame = false; + frameTime = 0; + frameIndex = 0; + nvErr = BodyEngine::errNone; + scaleOffsetXY[0] = scaleOffsetXY[2] = 1.f; + scaleOffsetXY[1] = scaleOffsetXY[3] = 0.f; +} + +DoApp::~DoApp() {} + +char *g_nvARSDKPath = NULL; + +int chooseGPU() { + // If the system has multiple supported GPUs then the application + // should use CUDA driver APIs or CUDA runtime APIs to enumerate + // the GPUs and select one based on the application's requirements + + //Cuda device 0 + return 0; + +} + +void DoApp::getFPS() { + const float timeConstant = 16.f; + frameTimer.stop(); + float t = (float)frameTimer.elapsedTimeFloat(); + if (t < 100.f) { + if (frameTime) + frameTime += (t - frameTime) * (1.f / timeConstant); // 1 pole IIR filter + else + frameTime = t; + } else { // Ludicrous time interval; reset + frameTime = 0.f; // WAKE UP + } + frameTimer.start(); +} + +void DoApp::drawFPS(cv::Mat &img) { + getFPS(); + if (frameTime && showFPS) { + char buf[32]; + snprintf(buf, sizeof(buf), "%.1f", 1. / frameTime); + cv::putText(img, buf, cv::Point(img.cols - 80, img.rows - 10), cv::FONT_HERSHEY_SIMPLEX, 1, + cv::Scalar(255, 255, 255), 1); + } +} + +void DoApp::drawKalmanStatus(cv::Mat &img) { + char buf[32]; + snprintf(buf, sizeof(buf), "Kalman %s", (body_ar_engine.bStabilizeBody ? "on" : "off")); + cv::putText(img, buf, cv::Point(10, img.rows - 40), cv::FONT_HERSHEY_SIMPLEX, 1, cv::Scalar(255, 255, 255), 1); +} + +void DoApp::drawVideoCaptureStatus(cv::Mat &img) { + char buf[32]; + snprintf(buf, sizeof(buf), "Video Capturing %s", (captureVideo ? "on" : "off")); + cv::putText(img, buf, cv::Point(10, img.rows - 70), cv::FONT_HERSHEY_SIMPLEX, 1, cv::Scalar(255, 255, 255), 1); +} + +DoApp::Err DoApp::run() { + DoApp::Err doErr = errNone; + + BodyEngine::Err err = body_ar_engine.initFeatureIOParams(); + if (err != BodyEngine::Err::errNone ) { + return doAppErr(err); + } + while (1) { + //printf(">> frame %d \n", framenum++); + doErr = acquireFrame(); + if (frame.empty() && FLAG_offlineMode) { + // We have reached the end of the video + // so return without any error. + return DoApp::errNone; + } + else if (doErr != DoApp::errNone) { + return doErr; + } + if (body_ar_engine.appMode == BodyEngine::mode::bodyDetection) { + doErr = acquireBodyBox(); + } else if (body_ar_engine.appMode == BodyEngine::mode::keyPointDetection) { + doErr = acquireBodyBoxAndKeyPoints(); + } + if ((DoApp::errNoBody == doErr || DoApp::errBodyFit == doErr) && FLAG_offlineMode) { + bodyDetectOutputVideo.write(frame); + keyPointsOutputVideo.write(frame); + } + if (DoApp::errCancel == doErr || DoApp::errVideo == doErr) return doErr; + if (!frame.empty() && !FLAG_offlineMode) { + if (drawVisualization) { + drawFPS(frame); + drawKalmanStatus(frame); + if (FLAG_captureOutputs && captureVideo) drawVideoCaptureStatus(frame); + } + cv::imshow(windowTitle, frame); + } + + if (!FLAG_offlineMode) { + int n = cv::waitKey(1); + if (n >= 0) { + static const int ESC_KEY = 27; + if (n == ESC_KEY) break; + processKey(n); + } + } + } + return doErr; +} + +const char *DoApp::errorStringFromCode(DoApp::Err code) { + struct LUTEntry { + Err code; + const char *str; + }; + static const LUTEntry lut[] = { + {errNone, "no error"}, + {errGeneral, "an error has occured"}, + {errRun, "an error has occured while the feature is running"}, + {errInitialization, "Initializing Body Engine failed"}, + {errRead, "an error has occured while reading a file"}, + {errEffect, "an error has occured while creating a feature"}, + {errParameter, "an error has occured while setting a parameter for a feature"}, + {errUnimplemented, "the feature is unimplemented"}, + {errMissing, "missing input parameter"}, + {errVideo, "no video source has been found"}, + {errImageSize, "the image size cannot be accommodated"}, + {errNotFound, "the item cannot be found"}, + {errBodyModelInit, "body model initialization failed"}, + {errGLFWInit, "GLFW initialization failed"}, + {errGLInit, "OpenGL initialization failed"}, + {errRendererInit, "renderer initialization failed"}, + {errGLResource, "an OpenGL resource could not be found"}, + {errGLGeneric, "an otherwise unspecified OpenGL error has occurred"}, + {errBodyFit, "an error has occurred while body fitting"}, + {errNoBody, "no body has been found"}, + {errSDK, "an SDK error has occurred"}, + {errCuda, "a CUDA error has occurred"}, + {errCancel, "the user cancelled"}, + {errCamera, "unable to connect to the camera"}, + }; + for (const LUTEntry *p = lut; p < &lut[sizeof(lut) / sizeof(lut[0])]; ++p) + if (p->code == code) return p->str; + static char msg[18]; + snprintf(msg, sizeof(msg), "error #%d", code); + return msg; +} + +/******************************************************************************** + * main + ********************************************************************************/ + +int main(int argc, char **argv) { + // Parse the arguments + if (0 != ParseMyArgs(argc, argv)) return -100; + + DoApp app; + DoApp::Err doErr = DoApp::Err::errNone; + + app.body_ar_engine.setAppMode(BodyEngine::mode(FLAG_appMode)); + + app.body_ar_engine.setMode(FLAG_mode); + + if (FLAG_verbose) printf("Enable temporal optimizations in detecting body and keypoints = %d\n", FLAG_temporal); + app.body_ar_engine.setBodyStabilization(FLAG_temporal); + + if (FLAG_useCudaGraph) printf("Enable capturing cuda graph = %d\n", FLAG_useCudaGraph); + app.body_ar_engine.useCudaGraph(FLAG_useCudaGraph); + + doErr = DoApp::errBodyModelInit; + if (FLAG_modelPath.empty()) { + printf("WARNING: Model path not specified. Please set --model_path=/path/to/trt/and/body/models, " + "SDK will attempt to load the models from NVAR_MODEL_DIR environment variable, " + "please restart your application after the SDK Installation. \n"); + } + if (!FLAG_bodyModel.empty()) + app.body_ar_engine.setBodyModel(FLAG_bodyModel.c_str()); + + if (FLAG_offlineMode) { + if (FLAG_inFile.empty()) { + doErr = DoApp::errMissing; + printf("ERROR: %s, please specify input file using --in_file or --in \n", app.errorStringFromCode(doErr)); + goto bail; + } + doErr = app.initOfflineMode(FLAG_inFile.c_str(), FLAG_outFile.c_str()); + } else { + doErr = app.initCamera(FLAG_camRes.c_str()); + } + BAIL_IF_ERR(doErr); + + doErr = app.initBodyEngine(FLAG_modelPath.c_str()); + BAIL_IF_ERR(doErr); + + doErr = app.run(); + BAIL_IF_ERR(doErr); + +bail: + if(doErr) + printf("ERROR: %s\n", app.errorStringFromCode(doErr)); + app.stop(); + return (int)doErr; +} diff --git a/samples/BodyTrack/BodyTrack.exe b/samples/BodyTrack/BodyTrack.exe new file mode 100644 index 0000000..de3f7d0 Binary files /dev/null and b/samples/BodyTrack/BodyTrack.exe differ diff --git a/samples/BodyTrack/CMakeLists.txt b/samples/BodyTrack/CMakeLists.txt new file mode 100644 index 0000000..8bb2db9 --- /dev/null +++ b/samples/BodyTrack/CMakeLists.txt @@ -0,0 +1,52 @@ +set(SOURCE_FILES BodyEngine.cpp + BodyTrack.cpp + ../utils/RenderingUtils.cpp +) + +set(HEADER_FILES BodyEngine.h) +if(MSVC) + set(SOURCE_FILES ${SOURCE_FILES} + ../../nvar/src/nvARProxy.cpp + ../../nvar/src/nvCVImageProxy.cpp) + + set(HEADER_FILES ${HEADER_FILES} + ../utils/RenderingUtils.h) +endif(MSVC) + +# Set Visual Studio source filters +source_group("Source Files" FILES ${SOURCE_FILES}) +source_group("Header Files" FILES ${HEADER_FILES}) + +add_executable(BodyTrack ${SOURCE_FILES} ${HEADER_FILES}) +target_include_directories(BodyTrack PRIVATE ${CMAKE_CURRENT_SOURCE_DIR}) +target_include_directories(BodyTrack PUBLIC + ${SDK_INCLUDES_PATH} +) + +if(MSVC) +target_link_libraries(BodyTrack PUBLIC + opencv346 + utils_sample +) + +set(ARSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) +set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin) +set(PATH_STR "PATH=%PATH%" ${OPENCV_PATH_STR}) +set(CMD_ARG_STR "--model_path=\"${CMAKE_CURRENT_SOURCE_DIR}/../../bin/models\"") + set_target_properties(BodyTrack PROPERTIES + FOLDER SampleApps + VS_DEBUGGER_ENVIRONMENT "${PATH_STR}" + VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}" + ) +elseif(UNIX) + find_package(PNG REQUIRED) + find_package(JPEG REQUIRED) + set(CMAKE_CXX_FLAGS "${CMAKE_CXX_FLAGS} -pthread") + target_link_libraries(BodyTrack PUBLIC + nvARPose + NVCVImage + OpenCV + utils_sample +) +endif(MSVC) + diff --git a/samples/BodyTrack/Readme.txt b/samples/BodyTrack/Readme.txt new file mode 100644 index 0000000..7e190fc --- /dev/null +++ b/samples/BodyTrack/Readme.txt @@ -0,0 +1,9 @@ +BodyTrack is a sample Windows application that demonstrates the person detection and 3D body pose estimation +features of the NVIDIA AR SDK. The application requires a video feed from a camera connected to the computer +running the application, or from a video file, as specified with command-line arguments (enumerated by +executing: BodyTrack.exe --help). +The sample application can run in 2 modes which can be toggled through the '1' and '2' keys on +the keyboard +1 - Person detection +2 - 3D Body Pose Estimation +For more controls and configurations of the sample app, please read the SDK programming guide. \ No newline at end of file diff --git a/samples/BodyTrack/run.bat b/samples/BodyTrack/run.bat new file mode 100644 index 0000000..f911492 --- /dev/null +++ b/samples/BodyTrack/run.bat @@ -0,0 +1,3 @@ +SETLOCAL +SET PATH=%PATH%;..\..\samples\external\opencv\bin;..\..\bin; +BodyTrack.exe \ No newline at end of file diff --git a/samples/CMakeLists.txt b/samples/CMakeLists.txt index 019f900..3198bd1 100644 --- a/samples/CMakeLists.txt +++ b/samples/CMakeLists.txt @@ -5,3 +5,4 @@ target_include_directories(utils_sample INTERFACE ${CMAKE_CURRENT_SOURCE_DIR}/ut target_link_libraries(utils_sample INTERFACE GLM) add_subdirectory(external) add_subdirectory(FaceTrack) +add_subdirectory(BodyTrack) diff --git a/samples/FaceTrack/CMakeLists.txt b/samples/FaceTrack/CMakeLists.txt index 8b3eca2..3742de1 100644 --- a/samples/FaceTrack/CMakeLists.txt +++ b/samples/FaceTrack/CMakeLists.txt @@ -1,10 +1,12 @@ set(SOURCE_FILES FaceEngine.cpp -FaceTrack.cpp -../utils/RenderingUtils.cpp -../../nvar/src/nvARProxy.cpp -../utils/FeatureVertexName.cpp -../utils/FeatureVertexName.h + FaceTrack.cpp + ../utils/RenderingUtils.cpp + ../utils/FeatureVertexName.cpp + ../utils/FeatureVertexName.h ) +if(MSVC) + set(SOURCE_FILES ${SOURCE_FILES} ../../nvar/src/nvARProxy.cpp ../../nvar/src/nvCVImageProxy.cpp) +endif(MSVC) set(HEADER_FILES FaceEngine.h) # Set Visual Studio source filters @@ -19,17 +21,18 @@ target_include_directories(FaceTrack PUBLIC target_link_libraries(FaceTrack PUBLIC opencv346 utils_sample - GLM ) set(ARSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin) set(PATH_STR "PATH=%PATH%" ${ARSDK_PATH_STR} ${OPENCV_PATH_STR}) set(CMD_ARG_STR "--model_path=\"${CMAKE_CURRENT_SOURCE_DIR}/../../bin/models\"") -set_target_properties(FaceTrack PROPERTIES - FOLDER SampleApps - VS_DEBUGGER_ENVIRONMENT "${PATH_STR}" - VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}" -) +if(MSVC) + set_target_properties(FaceTrack PROPERTIES + FOLDER SampleApps + VS_DEBUGGER_ENVIRONMENT "${PATH_STR}" + VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}" + ) +endif(MSVC) diff --git a/samples/FaceTrack/FaceEngine.cpp b/samples/FaceTrack/FaceEngine.cpp index 83dc3e4..86d644f 100644 --- a/samples/FaceTrack/FaceEngine.cpp +++ b/samples/FaceTrack/FaceEngine.cpp @@ -224,11 +224,12 @@ bail: FaceEngine::Err FaceEngine::initLandmarkDetectionIOParams(NvCVImage* inBuf) { NvCV_Status nvErr = NVCV_SUCCESS; FaceEngine::Err err = FaceEngine::Err::errNone; + uint output_bbox_size; + unsigned int OUTPUT_SIZE_KPTS, OUTPUT_SIZE_KPTS_CONF; nvErr = NvAR_SetObject(landmarkDetectHandle, NvAR_Parameter_Input(Image), inBuf, sizeof(NvCVImage)); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); - unsigned int OUTPUT_SIZE_KPTS, OUTPUT_SIZE_KPTS_CONF; nvErr = NvAR_GetU32(landmarkDetectHandle, NvAR_Parameter_Config(Landmarks_Size), &OUTPUT_SIZE_KPTS); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); @@ -251,7 +252,7 @@ FaceEngine::Err FaceEngine::initLandmarkDetectionIOParams(NvCVImage* inBuf) { facial_landmarks_confidence.data(), batchSize * OUTPUT_SIZE_KPTS); BAIL_IF_CVERR(nvErr, err, FaceEngine::Err::errParameter); - uint output_bbox_size = batchSize; + output_bbox_size = batchSize; if (!bStabilizeFace) output_bbox_size = 25; output_bbox_data.assign(output_bbox_size, {0.f, 0.f, 0.f, 0.f}); output_bboxes.boxes = output_bbox_data.data(); diff --git a/samples/FaceTrack/FaceEngine.h b/samples/FaceTrack/FaceEngine.h index 2cbb055..9e26858 100644 --- a/samples/FaceTrack/FaceEngine.h +++ b/samples/FaceTrack/FaceEngine.h @@ -151,7 +151,7 @@ class FaceEngine { unsigned findLargestFaceBox(NvAR_Rect& faceBox, int variant = 0); unsigned acquireFaceBox(cv::Mat& src, NvAR_Rect& faceBox, int variant = 0); unsigned acquireFaceBoxAndLandmarks(cv::Mat& src, NvAR_Point2f* refMarks, NvAR_Rect& faceBox, int variant = 0); - Err fitFaceModel(cv::Mat& frame = cv::Mat()); + Err fitFaceModel(cv::Mat& frame); NvAR_FaceMesh* getFaceMesh(); NvAR_RenderingParams* getRenderingParams(); void setFaceStabilization(bool); diff --git a/samples/FaceTrack/FaceTrack.cpp b/samples/FaceTrack/FaceTrack.cpp index cbe6cf2..cf4bb1c 100644 --- a/samples/FaceTrack/FaceTrack.cpp +++ b/samples/FaceTrack/FaceTrack.cpp @@ -20,6 +20,13 @@ # CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE. # ###############################################################################*/ +#include +#include +#include +#include +#include +#include +#include #include #include #include @@ -174,6 +181,12 @@ static int StringToFourcc(const std::string &str) { ********************************************************************************/ static int ParseMyArgs(int argc, char **argv) { + // query NVAR_MODEL_DIR environment variable first before checking the command line arguments + const char* modelPath = getenv("NVAR_MODEL_DIR"); + if (modelPath) { + FLAG_modelPath = modelPath; + } + int errs = 0; for (--argc, ++argv; argc--; ++argv) { bool help; diff --git a/samples/FaceTrack/FaceTrack.exe b/samples/FaceTrack/FaceTrack.exe index 4d57e8f..7b6dfb1 100644 Binary files a/samples/FaceTrack/FaceTrack.exe and b/samples/FaceTrack/FaceTrack.exe differ diff --git a/samples/FaceTrack/Readme.txt b/samples/FaceTrack/Readme.txt index e1cc66d..8bd30bf 100644 --- a/samples/FaceTrack/Readme.txt +++ b/samples/FaceTrack/Readme.txt @@ -44,7 +44,7 @@ output-path The full or relative path to the folder where you want the output .nvf format file to be written. Either the forward slash (/) or back slash (\) can be used as a separator between path elements. -The ConvertSurreyFaceModel.exe file is distributed in the https://github.com/nvidia/BROADCAST-AR-SDK repo. +The ConvertSurreyFaceModel.exe file is distributed in the https://github.com/nvidia/MAXINE-AR-SDK repo. 3) The sample application provided with NVIDIA AR SDK requires that the model file be named face_model0.nvf. Place the face_model0.nvf file in the model folder. By default the models folder is your_sdk_install_path/models diff --git a/samples/FaceTrack/run.bat b/samples/FaceTrack/run.bat index bc2f255..4235855 100644 --- a/samples/FaceTrack/run.bat +++ b/samples/FaceTrack/run.bat @@ -1,3 +1,3 @@ SETLOCAL -SET PATH=%PATH%;..\..\samples\external\opencv\bin; -FaceTrack.exe +SET PATH=%PATH%;..\..\samples\external\opencv\bin;..\..\bin; +FaceTrack.exe \ No newline at end of file diff --git a/samples/utils/RenderingUtils.h b/samples/utils/RenderingUtils.h index 61afb33..2b44e81 100644 --- a/samples/utils/RenderingUtils.h +++ b/samples/utils/RenderingUtils.h @@ -100,4 +100,4 @@ void average_poses(NvAR_Quaternion *q, unsigned n); void set_rotation_from_quaternion(const NvAR_Quaternion *quat, float M[3*3]); -#endif __RENDERING_UTILS__ \ No newline at end of file +#endif // __RENDERING_UTILS__ \ No newline at end of file diff --git a/samples/utils/nvCVOpenCV.h b/samples/utils/nvCVOpenCV.h index cdfbe1a..ddb9274 100644 --- a/samples/utils/nvCVOpenCV.h +++ b/samples/utils/nvCVOpenCV.h @@ -52,7 +52,7 @@ inline void CVImageSet(cv::Mat *cvIm, int width, int height, int numComps, int c // Wrap an NvCVImage in a cv::Mat inline void CVWrapperForNvCVImage(const NvCVImage *nvcvIm, cv::Mat *cvIm) { static const char cvType[] = { 7, 0, 2, 3, 7, 7, 4, 5, 7, 7, 6 }; - CVImageSet(cvIm, nvcvIm->width, nvcvIm->height, nvcvIm->numComponents, cvType[(int)nvcvIm->pixelFormat], nvcvIm->componentBytes, nvcvIm->pixels, nvcvIm->pitch); + CVImageSet(cvIm, nvcvIm->width, nvcvIm->height, nvcvIm->numComponents, cvType[(int)nvcvIm->componentType], nvcvIm->componentBytes, nvcvIm->pixels, nvcvIm->pitch); } // Wrap a cv::Mat in an NvCVImage. diff --git a/tools/ConvertSurreyFaceModel.exe b/tools/ConvertSurreyFaceModel.exe index ab36c7b..7404c86 100644 Binary files a/tools/ConvertSurreyFaceModel.exe and b/tools/ConvertSurreyFaceModel.exe differ diff --git a/version.h b/version.h index 75f9976..a93b723 100644 --- a/version.h +++ b/version.h @@ -22,12 +22,12 @@ ###############################################################################*/ #define NVIDIA_AR_SDK_VERSION_MAJOR 0 -#define NVIDIA_AR_SDK_VERSION_MINOR 6 -#define NVIDIA_AR_SDK_VERSION_RELEASE 1 +#define NVIDIA_AR_SDK_VERSION_MINOR 7 +#define NVIDIA_AR_SDK_VERSION_RELEASE 5 -#define NVIDIA_AR_SDK_VERSION 0,6,1,0 -#define NVIDIA_AR_SDK_VERSION_MAJOR_MINOR 0,6 -#define NVIDIA_AR_SDK_VERSION_STRING "0.6.1.0" -#define NVIDIA_AR_SDK_VERSION_STRING_SHORT "0.6.1" -#define NVIDIA_AR_SDK_VERSION_STRING_MAJOR_MINOR "0.6" +#define NVIDIA_AR_SDK_VERSION 0,7,5,0 +#define NVIDIA_AR_SDK_VERSION_MAJOR_MINOR 0,7 +#define NVIDIA_AR_SDK_VERSION_STRING "0.7.5.0" +#define NVIDIA_AR_SDK_VERSION_STRING_SHORT "0.7.5" +#define NVIDIA_AR_SDK_VERSION_STRING_MAJOR_MINOR "0.7"