v0.6.5.2 Release

v0.6.5.2
This commit is contained in:
jdsouza90
2021-09-13 08:59:44 -07:00
parent 3df6c37852
commit 7f69da2ee4
41 changed files with 2816 additions and 503 deletions

View File

@@ -63,9 +63,9 @@
bool FLAG_progress = false;
bool FLAG_show = false;
bool FLAG_useOTAU = false;
bool FLAG_verbose = false;
bool FLAG_webcam = false;
bool FLAG_cudaGraph = false;
int FLAG_compMode = 3 /*compWhite*/;
int FLAG_mode = 0;
float FLAG_blurStrength = 0.5;
@@ -75,6 +75,7 @@ std::string FLAG_inFile;
std::string FLAG_modelDir;
std::string FLAG_outDir;
std::string FLAG_outFile;
std::string FLAG_bgFile;
static bool GetFlagArgVal(const char *flag, const char *arg, const char **val) {
if (*arg != '-') return false;
@@ -135,6 +136,7 @@ static void Usage() {
" where args is:\n"
" --in_file=<path> input file to be processed\n"
" --out_file=<path> output file to be written\n"
" --bg_file=<path> background file for composition\n"
" --webcam use a webcam as input\n"
" --cam_res=[WWWx]HHH specify resolution as height or width x height\n"
" --model_dir=<path> the path to the directory that contains the models\n"
@@ -144,8 +146,15 @@ static void Usage() {
" --mode=(0|1) pick one of the green screen modes\n"
" 0 - Best quality\n"
" 1 - Best performance\n"
" --comp_mode choose the composition mode - { compMatte = 0, compLight = 1, compGreen = 2, compWhite = 3, compNone = 4, compBG = 5, compBlur = 6}\n"
" --blur_strength change the blur strength, range is [0, 1]"
" --comp_mode choose the composition mode - {\n"
" 0 (show matte - compMatte),\n"
" 1 (overlay mask on foreground - compLight),\n"
" 2 (composite over green - compGreen),\n"
" 3 (composite over white - compWhite),\n"
" 4 (show input - compNone),\n"
" 5 (composite over a specified background image - compBG),\n"
" 6 (blur the background of the image - compBlur) }\n"
" --cuda_graph Enable cuda graph.\n"
);
}
@@ -160,10 +169,12 @@ static int ParseMyArgs(int argc, char **argv) {
(GetFlagArgVal("verbose", arg, &FLAG_verbose) || GetFlagArgVal("in", arg, &FLAG_inFile) ||
GetFlagArgVal("in_file", arg, &FLAG_inFile) || GetFlagArgVal("out", arg, &FLAG_outFile) ||
GetFlagArgVal("out_file", arg, &FLAG_outFile) || GetFlagArgVal("model_dir", arg, &FLAG_modelDir) ||
GetFlagArgVal("bg_file", arg, &FLAG_bgFile) ||
GetFlagArgVal("codec", arg, &FLAG_codec) || GetFlagArgVal("webcam", arg, &FLAG_webcam) ||
GetFlagArgVal("cam_res", arg, &FLAG_camRes) || GetFlagArgVal("mode", arg, &FLAG_mode) ||
GetFlagArgVal("progress", arg, &FLAG_progress) || GetFlagArgVal("show", arg, &FLAG_show) ||
GetFlagArgVal("comp_mode", arg, &FLAG_compMode) || GetFlagArgVal("blur_strength", arg, &FLAG_blurStrength))) {
GetFlagArgVal("comp_mode", arg, &FLAG_compMode) || GetFlagArgVal("blur_strength", arg, &FLAG_blurStrength) ||
GetFlagArgVal("cuda_graph", arg, &FLAG_cudaGraph) )) {
continue;
} else if (GetFlagArgVal("help", arg, &help)) {
return NVCV_ERR_HELP;
@@ -208,6 +219,10 @@ static bool IsImageFile(const char *str) {
return HasOneOfTheseSuffixes(str, ".bmp", ".jpg", ".jpeg", ".png", nullptr);
}
static bool IsLossyImageFile(const char *str) {
return HasOneOfTheseSuffixes(str, ".jpg", ".jpeg", nullptr);
}
static const char *DurationString(double sc) {
static char buf[16];
int hr, mn;
@@ -246,9 +261,9 @@ static void GetVideoInfo(cv::VideoCapture &reader, const char *fileName, VideoIn
info->height = (int)reader.get(cv::CAP_PROP_FRAME_HEIGHT);
info->frameRate = (double)reader.get(cv::CAP_PROP_FPS);
if(!strcmp(fileName,"webcam"))
info->frameCount = 0;
info->frameCount = 0;
else
info->frameCount = (long long)reader.get(cv::CAP_PROP_FRAME_COUNT);
info->frameCount = (long long)reader.get(cv::CAP_PROP_FRAME_COUNT);
if (FLAG_verbose) PrintVideoInfo(info, fileName);
}
@@ -282,8 +297,8 @@ struct FXApp {
errLibrary = NVCV_ERR_LIBRARY,
errInitialization = NVCV_ERR_INITIALIZATION,
errFileNotFound = NVCV_ERR_FILE,
errFeatureNotFound = NVCV_ERR_FEATURENOTFOUND,
errMissingInput = NVCV_ERR_MISSINGINPUT,
errFeatureNotFound = NVCV_ERR_FEATURENOTFOUND,
errMissingInput = NVCV_ERR_MISSINGINPUT,
errResolution = NVCV_ERR_RESOLUTION,
errUnsupportedGPU = NVCV_ERR_UNSUPPORTEDGPU,
errWrongGPU = NVCV_ERR_WRONGGPU,
@@ -341,6 +356,8 @@ struct FXApp {
NvVFX_Handle _eff, _bgblurEff;
cv::Mat _srcImg;
cv::Mat _dstImg;
cv::Mat _bgImg;
cv::Mat _resizedCroppedBgImg;
NvCVImage _srcVFX;
NvCVImage _dstVFX;
bool _show;
@@ -401,24 +418,26 @@ void FXApp::drawFrameRate(cv::Mat &img) {
void FXApp::nextCompMode() {
switch (_compMode) {
default:
case compBG:
case compMatte:
_compMode = compLight;
break;
case compLight:
_compMode = compGreen;
break;
case compMatte:
_compMode = compNone;
break;
case compGreen:
_compMode = compWhite;
break;
case compWhite:
_compMode = compMatte;
_compMode = compNone;
break;
case compNone:
_compMode = compBG;
break;
case compBG:
_compMode = compBlur;
break;
case compBlur:
_compMode = compLight;
_compMode = compMatte;
break;
}
}
@@ -473,7 +492,7 @@ NvCV_Status FXApp::createAigsEffect() {
if (!FLAG_modelDir.empty()) {
vfxErr = NvVFX_SetString(_eff, NVVFX_MODEL_DIRECTORY, FLAG_modelDir.c_str());
}
}
if (vfxErr != NVCV_SUCCESS) {
std::cerr << "Error setting the model path to \"" << FLAG_modelDir << "\"\n";
return vfxErr;
@@ -493,6 +512,12 @@ NvCV_Status FXApp::createAigsEffect() {
return vfxErr;
}
vfxErr = NvVFX_SetU32(_eff, NVVFX_CUDA_GRAPH, FLAG_cudaGraph?1u:0u);
if (vfxErr != NVCV_SUCCESS) {
std::cerr << "Error enabling cuda graph \n";
return vfxErr;
}
vfxErr = NvVFX_CudaStreamCreate(&_stream);
if (vfxErr != NVCV_SUCCESS) {
std::cerr << "Error creating CUDA stream " << std::endl;
@@ -570,6 +595,8 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
overlay(_srcImg, _dstImg, 0.5, result);
if (!std::string(outFile).empty()) {
if(IsLossyImageFile(outFile))
fprintf(stderr, "WARNING: JPEG output file format will reduce image quality\n");
ok = cv::imwrite(outFile, result);
if (!ok) {
printf("Error writing: \"%s\"\n", outFile);
@@ -647,6 +674,30 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
unsigned int width = (unsigned int)reader.get(cv::CAP_PROP_FRAME_WIDTH);
unsigned int height = (unsigned int)reader.get(cv::CAP_PROP_FRAME_HEIGHT);
if (!FLAG_bgFile.empty())
{
_bgImg = cv::imread(FLAG_bgFile);
if (!_bgImg.data)
{
return errRead;
}
else
{
// Find the scale to resize background such that image can fit into background
float scale = float(height) / float(_bgImg.rows);
if ((scale * _bgImg.cols) < float(width))
{
scale = float(width) / float(_bgImg.cols);
}
cv::Mat resizedBg;
cv::resize(_bgImg, resizedBg, cv::Size(), scale, scale, cv::INTER_AREA);
// Always crop from top left of background.
cv::Rect rect(0, 0, width, height);
_resizedCroppedBgImg = resizedBg(rect);
}
}
// allocate src for GPU
if (!_srcNvVFXImage.pixels)
BAIL_IF_ERR(vfxErr =
@@ -695,6 +746,25 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
case compNone:
_srcImg.copyTo(result);
break;
case compBG: {
if (FLAG_bgFile.empty())
{
_resizedCroppedBgImg = cv::Mat(_srcImg.rows, _srcImg.cols, CV_8UC3, cv::Scalar(118, 185, 0));
size_t startX = _resizedCroppedBgImg.cols/20;
size_t offsetY = _resizedCroppedBgImg.rows/20;
std::string text = "No Background Image!";
for (size_t startY = offsetY; startY < _resizedCroppedBgImg.rows; startY += offsetY)
{
cv::putText(_resizedCroppedBgImg, text, cv::Point(startX, startY),
cv::FONT_HERSHEY_DUPLEX, 1.0, CV_RGB(0, 0, 0), 1);
}
}
NvCVImage bgVFX;
(void)NVWrapperForCVMat(&_resizedCroppedBgImg, &bgVFX);
NvCVImage matVFX;
(void)NVWrapperForCVMat(&result, &matVFX);
NvCVImage_Composite(&_srcVFX, &bgVFX, &_dstVFX, &matVFX, _stream);
} break;
case compLight:
if (inFile) {
overlay(_srcImg, _dstImg, 0.5, result);
@@ -708,13 +778,13 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
const unsigned char bgColor[3] = {0, 255, 0};
NvCVImage matVFX;
(void)NVWrapperForCVMat(&result, &matVFX);
NvCVImage_CompositeOverConstant(&_srcVFX, &_dstVFX, bgColor, &matVFX);
NvCVImage_CompositeOverConstant(&_srcVFX, &_dstVFX, bgColor, &matVFX, _stream);
} break;
case compWhite: {
const unsigned char bgColor[3] = {255, 255, 255};
NvCVImage matVFX;
(void)NVWrapperForCVMat(&result, &matVFX);
NvCVImage_CompositeOverConstant(&_srcVFX, &_dstVFX, bgColor, &matVFX);
NvCVImage_CompositeOverConstant(&_srcVFX, &_dstVFX, bgColor, &matVFX, _stream);
} break;
case compMatte:
cv::cvtColor(_dstImg, result, cv::COLOR_GRAY2BGR);
@@ -726,7 +796,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_bgblurEff, NVVFX_OUTPUT_IMAGE, &_blurNvVFXImage));
BAIL_IF_ERR(vfxErr = NvVFX_Load(_bgblurEff));
BAIL_IF_ERR(vfxErr = NvVFX_Run(_bgblurEff, 0));
NvCVImage matVFX;
(void)NVWrapperForCVMat(&result, &matVFX);
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_blurNvVFXImage, &matVFX, 1.0f, _stream, NULL));
@@ -750,12 +820,11 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
if (errQuit == appErr) break;
}
}
if (_progress) {
if(info.frameCount == 0) // no progress for a webcam
fprintf(stderr, "\b\b\b\b???%%");
else
fprintf(stderr, "\b\b\b\b%3.0f%%", 100.f * frameNum / info.frameCount);
}
if (_progress)
if(info.frameCount == 0) // no progress for a webcam
fprintf(stderr, "\b\b\b\b???%%");
else
fprintf(stderr, "\b\b\b\b%3.0f%%", 100.f * frameNum / info.frameCount);
}
if (_progress) fprintf(stderr, "\n");
@@ -769,17 +838,44 @@ bail:
return appErrFromVfxStatus(vfxErr);
}
// This path is used by nvVideoEffectsProxy.cpp to load the SDK dll
char *g_nvVFXSDKPath = NULL;
int chooseGPU() {
// If the system has multiple supported GPUs then the application
// should use CUDA driver APIs or CUDA runtime APIs to enumerate
// the GPUs and select one based on the application's requirements
// Cuda device 0
return 0;
}
bool isCompModeEnumValid(const FXApp::CompMode& mode)
{
if (mode != FXApp::CompMode::compMatte &&
mode != FXApp::CompMode::compLight &&
mode != FXApp::CompMode::compGreen &&
mode != FXApp::CompMode::compWhite &&
mode != FXApp::CompMode::compNone &&
mode != FXApp::CompMode::compBG &&
mode != FXApp::CompMode::compBlur)
{
return false;
}
return true;
}
int main(int argc, char **argv) {
int nErrs = 0;
FXApp::Err fxErr = FXApp::errNone;
FXApp app;
nErrs = ParseMyArgs(argc, argv);
if (nErrs) {
Usage();
return nErrs;
}
FXApp::Err fxErr = FXApp::errNone;
FXApp app;
if (FLAG_inFile.empty() && !FLAG_webcam) {
std::cerr << "Please specify --in_file=XXX or --webcam\n";
++nErrs;
@@ -793,6 +889,12 @@ int main(int argc, char **argv) {
app.setShow(FLAG_show);
app._compMode = static_cast<FXApp::CompMode>(FLAG_compMode);
if (!isCompModeEnumValid(app._compMode))
{
std::cerr << "Please specify a valid --comp_mode=XXX, valid range is [0,6] check help section\n";
++nErrs;
}
app._blurStrength = FLAG_blurStrength;
if (app._blurStrength < 0) {
app._blurStrength = 0;
@@ -823,4 +925,4 @@ int main(int argc, char **argv) {
if (fxErr) std::cerr << "Error: " << app.errorStringFromCode(fxErr) << std::endl;
return (int)fxErr;
}
}

View File

@@ -22,11 +22,10 @@ target_link_libraries(AigsEffectApp PUBLIC
)
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
set(CMD_ARG_STR "--show --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input_003054.jpg\"")
set(PATH_STR "PATH=%PATH%" ${OPENCV_PATH_STR})
set(CMD_ARG_STR "--show --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input_003054.jpg\" ")
set_target_properties(AigsEffectApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)
)

View File

@@ -21,6 +21,7 @@
#
###############################################################################*/
#include <stdarg.h>
#include <stdio.h>
#include <string.h>
@@ -127,8 +128,7 @@ static void Usage() {
" where flags is:\n"
" --out_file=<path> output image files to be written, default \"BatchOut_%%02u.png\"\n"
" --effect=<effect> the effect to apply\n"
" --strength=<value> strength of an effect, 0 or 1 for super res and artifact reduction,\n"
" and [0.0, 1.0] for upscaling\n"
" --strength=<value> strength of the upscaling effect, [0.0, 1.0]\n"
" --scale=<scale> scale factor to be applied: 1.5, 2, 3, maybe 1.3333333\n"
" --resolution=<height> the desired height (either --scale or --resolution may be used)\n"
" --mode=<mode> mode 0 or 1\n"
@@ -183,6 +183,32 @@ static int ParseMyArgs(int argc, char **argv) {
return errs;
}
static bool HasSuffix(const char *str, const char *suf) {
size_t strSize = strlen(str),
sufSize = strlen(suf);
if (strSize < sufSize)
return false;
return (0 == strcasecmp(suf, str + strSize - sufSize));
}
static bool HasOneOfTheseSuffixes(const char *str, ...) {
bool matches = false;
const char *suf;
va_list ap;
va_start(ap, str);
while (nullptr != (suf = va_arg(ap, const char*))) {
if (HasSuffix(str, suf)) {
matches = true;
break;
}
}
va_end(ap);
return matches;
}
static bool IsLossyImageFile(const char *str) {
return HasOneOfTheseSuffixes(str, ".jpg", ".jpeg", nullptr);
}
class App {
public:
@@ -233,7 +259,7 @@ public:
BAIL_IF_ERR(err = AllocateBatchBuffer(&_src, _batchSize, src->width, src->height, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_CUDA, 1));
BAIL_IF_ERR(err = AllocateBatchBuffer(&_dst, _batchSize, src->width, src->height, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_CUDA, 1));
BAIL_IF_ERR(err = NvVFX_SetString(_eff, NVVFX_MODEL_DIRECTORY, FLAG_modelDir.c_str()));
BAIL_IF_ERR(err = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned)FLAG_strength));
BAIL_IF_ERR(err = NvVFX_SetU32(_eff, NVVFX_MODE, FLAG_mode));
}
#endif // NVVFX_FX_ARTIFACT_REDUCTION
#ifdef NVVFX_FX_SUPER_RES
@@ -241,7 +267,7 @@ public:
BAIL_IF_ERR(err = AllocateBatchBuffer(&_src, _batchSize, src->width, src->height, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_CUDA, 1));
BAIL_IF_ERR(err = AllocateBatchBuffer(&_dst, _batchSize, dw, dh, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_CUDA, 1));
BAIL_IF_ERR(err = NvVFX_SetString(_eff, NVVFX_MODEL_DIRECTORY, FLAG_modelDir.c_str()));
BAIL_IF_ERR(err = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned)FLAG_strength));
BAIL_IF_ERR(err = NvVFX_SetU32(_eff, NVVFX_MODE, FLAG_mode));
}
#endif // NVVFX_FX_SUPER_RES
else {
@@ -339,6 +365,8 @@ NvCV_Status BatchProcessImages(const char* effectName, const std::vector<const c
dstHeight = app._dst.height / batchSize;
BAIL_IF_ERR(err = NvCVImage_Alloc(&nvx, app._dst.width, dstHeight, ((app._dst.numComponents == 1) ? NVCV_Y : NVCV_BGR), NVCV_U8, NVCV_CHUNKY, NVCV_CPU, 0));
CVWrapperForNvCVImage(&nvx, &ocv);
if(IsLossyImageFile(outfilePattern))
fprintf(stderr, "WARNING: JPEG output file format will reduce image quality\n");
for (i = 0; i < batchSize; ++i) {
char fileName[1024];
snprintf(fileName, sizeof(fileName), outfilePattern, i);

View File

@@ -4,6 +4,7 @@ set(SOURCE_FILES
../../nvvfx/src/nvVideoEffectsProxy.cpp
../../nvvfx/src/nvCVImageProxy.cpp)
# Set Visual Studio source filters
source_group("Source Files" FILES ${SOURCE_FILES})
@@ -16,32 +17,20 @@ target_include_directories(BatchEffectApp PUBLIC
${SDK_INCLUDES_PATH}
)
if(MSVC)
target_link_libraries(BatchEffectApp PUBLIC
opencv346
NVVideoEffects
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
)
target_link_libraries(BatchEffectApp PUBLIC
opencv346
NVVideoEffects
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
)
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
set(CMD_ARG_STR "--show --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input_003054.jpg\"")
set_target_properties(BatchEffectApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)
else()
target_link_libraries(BatchEffectApp PUBLIC
NVVideoEffects
NVCVImage
OpenCV
TensorRT
CUDA
)
endif()
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(PATH_STR "PATH=%PATH%" ${OPENCV_PATH_STR})
set(CMD_ARG_STR "--show --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input_003054.jpg\" ")
set_target_properties(BatchEffectApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)
#Batch denoise effect
set(SOURCE_FILES
@@ -62,30 +51,19 @@ target_include_directories(BatchDenoiseEffectApp PUBLIC
${SDK_INCLUDES_PATH}
)
if(MSVC)
target_link_libraries(BatchDenoiseEffectApp PUBLIC
opencv346
NVVideoEffects
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
)
target_include_directories(BatchDenoiseEffectApp PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/include)
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
set(CMD_ARG_STR "video1.mp4 video2.mp4")
set_target_properties(BatchDenoiseEffectApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)
else()
target_link_libraries(BatchDenoiseEffectApp PUBLIC
NVVideoEffects
NVCVImage
OpenCV
TensorRT
CUDA
)
endif()
target_link_libraries(BatchDenoiseEffectApp PUBLIC
opencv346
NVVideoEffects
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
)
target_include_directories(BatchDenoiseEffectApp PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/include)
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(PATH_STR "PATH=%PATH%" ${OPENCV_PATH_STR})
set(CMD_ARG_STR "video1.mp4 video2.mp4 ")
set_target_properties(BatchDenoiseEffectApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)

View File

@@ -7,29 +7,20 @@ add_executable(DenoiseEffectApp ${SOURCE_FILES})
target_include_directories(DenoiseEffectApp PRIVATE ${CMAKE_CURRENT_SOURCE_DIR} ${CMAKE_CURRENT_SOURCE_DIR}/../utils)
target_include_directories(DenoiseEffectApp PUBLIC ${SDK_INCLUDES_PATH})
if(MSVC)
target_link_libraries(DenoiseEffectApp PUBLIC
opencv346
NVVideoEffects
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
)
target_include_directories(DenoiseEffectApp PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/include)
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
set(CMD_ARG_STR "--model_dir=\"${CMAKE_CURRENT_SOURCE_DIR}/../../bin/models\" --show --webcam")
set_target_properties(DenoiseEffectApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)
else()
target_link_libraries(DenoiseEffectApp PUBLIC
NVVideoEffects
NVCVImage
OpenCV
TensorRT
CUDA
)
endif()
target_link_libraries(DenoiseEffectApp PUBLIC
opencv346
NVVideoEffects
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
)
target_include_directories(DenoiseEffectApp PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/include)
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
set(CMD_ARG_STR "--show --webcam")
set_target_properties(DenoiseEffectApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)

View File

@@ -220,6 +220,10 @@ static bool IsImageFile(const char *str) {
return HasOneOfTheseSuffixes(str, ".bmp", ".jpg", ".jpeg", ".png", nullptr);
}
static bool IsLossyImageFile(const char *str) {
return HasOneOfTheseSuffixes(str, ".jpg", ".jpeg", nullptr);
}
static const char* DurationString(double sc) {
static char buf[16];
int hr, mn;
@@ -388,7 +392,7 @@ FXApp::Err FXApp::processKey(int key) {
case 'p': case 'P': case '%':
_progress = !_progress;
case 'e': case 'E':
_enableEffect = !_enableEffect;
_enableEffect = !_enableEffect;
break;
case 'd': case'D':
if (FLAG_webcam)
@@ -474,10 +478,10 @@ NvCV_Status FXApp::allocBuffers(unsigned width, unsigned height) {
_srcImg.create(height, width, CV_8UC3); // src CPU
BAIL_IF_NULL(_srcImg.data, vfxErr, NVCV_ERR_MEMORY);
}
_dstImg.create(_srcImg.rows, _srcImg.cols, _srcImg.type()); //
BAIL_IF_NULL(_dstImg.data, vfxErr, NVCV_ERR_MEMORY); //
BAIL_IF_ERR(vfxErr = NvCVImage_Alloc(&_srcGpuBuf, _srcImg.cols, _srcImg.rows, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_GPU, 1)); // src GPU
BAIL_IF_ERR(vfxErr = NvCVImage_Alloc(&_srcGpuBuf, _srcImg.cols, _srcImg.rows, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_GPU, 1)); // src GPU
BAIL_IF_ERR(vfxErr = NvCVImage_Alloc(&_dstGpuBuf, _srcImg.cols, _srcImg.rows, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_GPU, 1)); //dst GPU
NVWrapperForCVMat(&_srcImg, &_srcVFX); // _srcVFX is an alias for _srcImg
@@ -510,9 +514,9 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
BAIL_IF_ERR(vfxErr = allocBuffers(_srcImg.cols, _srcImg.rows));
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_srcVFX, &_srcGpuBuf, 1.f / 255.f, stream, &_tmpVFX)); // _srcVFX--> _tmpVFX --> _srcGpuBuf
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_INPUT_IMAGE, &_srcGpuBuf));
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
BAIL_IF_ERR(vfxErr = NvVFX_SetF32(_eff, NVVFX_STRENGTH, FLAG_strength));
unsigned int stateSizeInBytes;
BAIL_IF_ERR(vfxErr = NvVFX_GetU32(_eff, NVVFX_STATE_SIZE, &stateSizeInBytes));
cudaMalloc(&state, stateSizeInBytes);
@@ -525,6 +529,8 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_dstGpuBuf, &_dstVFX, 255.f, stream, &_tmpVFX));
if (outFile && outFile[0]) {
if(IsLossyImageFile(outFile))
fprintf(stderr, "WARNING: JPEG output file format will reduce image quality\n");
if (!cv::imwrite(outFile, _dstImg)) {
printf("Error writing: \"%s\"\n", outFile);
return errWrite;
@@ -552,7 +558,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
void* state = nullptr;
void* stateArray[1];
if (inFile && !inFile[0]) inFile = nullptr; // Set file paths to NULL if zero length
if (!FLAG_webcam && inFile) {
@@ -574,7 +580,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
printf("Filters only target H264 videos, not %.4s\n", (char*)&info.codec);
BAIL_IF_ERR(vfxErr = allocBuffers(info.width, info.height));
if (outFile && !outFile[0]) outFile = nullptr;
if (outFile) {
ok = writer.open(outFile, StringToFourcc(FLAG_codec), info.frameRate, cv::Size(_dstVFX.width, _dstVFX.height));
@@ -586,7 +592,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
}
}
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_INPUT_IMAGE, &_srcGpuBuf));
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_INPUT_IMAGE, &_srcGpuBuf));
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
BAIL_IF_ERR(vfxErr = NvVFX_SetF32(_eff, NVVFX_STRENGTH, FLAG_strength));
@@ -598,7 +604,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
BAIL_IF_ERR(vfxErr = NvVFX_SetObject(_eff, NVVFX_STATE, (void*)stateArray));
BAIL_IF_ERR(vfxErr = NvVFX_Load(_eff));
for (frameNum = 0; reader.read(_srcImg); frameNum++) {
if (_enableEffect) {
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_srcVFX, &_srcGpuBuf, 1.f / 255.f, stream, &_tmpVFX));
@@ -613,7 +619,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
writer.write(_dstImg);
if (_show) {
if (_drawVisualization) drawEffectStatus(_dstImg);
if (_drawVisualization) drawEffectStatus(_dstImg);
drawFrameRate(_dstImg);
cv::imshow("Output", _dstImg);
int key= cv::waitKey(1);
@@ -630,7 +636,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
if (_progress) fprintf(stderr, "\n");
reader.release();
if (outFile)
writer.release();
writer.release();
bail:
if (state) cudaFree(state); // release state memory
return appErrFromVfxStatus(vfxErr);

View File

@@ -12,29 +12,19 @@ target_include_directories(UpscalePipelineApp PUBLIC
${SDK_INCLUDES_PATH}
)
if(MSVC)
target_link_libraries(UpscalePipelineApp PUBLIC
opencv346
NVVideoEffects
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
)
target_link_libraries(UpscalePipelineApp PUBLIC
opencv346
NVVideoEffects
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
)
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
set(CMD_ARG_STR "--model_dir=\"${CMAKE_CURRENT_SOURCE_DIR}/../../bin/models\" --show --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input1.jpg\"")
set_target_properties(UpscalePipelineApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)
else()
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
set(CMD_ARG_STR "--show --resolution=1440 --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input1.jpg\"")
set_target_properties(UpscalePipelineApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)
target_link_libraries(UpscalePipelineApp PUBLIC
NVVideoEffects
NVCVImage
OpenCV
TensorRT
CUDA
)
endif()

View File

@@ -66,7 +66,7 @@ bool FLAG_debug = false,
FLAG_show = false,
FLAG_progress = false;
int FLAG_resolution = 0,
FLAG_arStrength = 0;
FLAG_arMode = 0;
float FLAG_upscaleStrength = 0.2f;
std::string FLAG_codec = DEFAULT_CODEC,
FLAG_inFile,
@@ -74,6 +74,11 @@ std::string FLAG_codec = DEFAULT_CODEC,
FLAG_outDir,
FLAG_modelDir;
// Set this when using OTA Updates
// This path is used by nvVideoEffectsProxy.cpp to load the SDK dll
// when using OTA Updates
char *g_nvVFXSDKPath = NULL;
static bool GetFlagArgVal(const char *flag, const char *arg, const char **val) {
if (*arg != '-')
return false;
@@ -146,7 +151,7 @@ static void Usage() {
" --in_file=<path> input file to be processed\n"
" --out_file=<path> output file to be written\n"
" --show display the results in a window\n"
" --ar_strength=(0|1) strength of artifact reduction filter (0: conservative, 1: aggressive, default 0)\n"
" --ar_mode=(0|1) mode of artifact reduction filter (0: conservative, 1: aggressive, default 0)\n"
" --upscale_strength=(0 to 1) strength of upscale filter (float value between 0 to 1)\n"
" --resolution=<height> the desired height of the output\n"
" --out_height=<height> the desired height of the output\n"
@@ -172,7 +177,7 @@ static int ParseMyArgs(int argc, char **argv) {
GetFlagArgVal("out", arg, &FLAG_outFile) ||
GetFlagArgVal("out_file", arg, &FLAG_outFile) ||
GetFlagArgVal("show", arg, &FLAG_show) ||
GetFlagArgVal("ar_strength", arg, &FLAG_arStrength) ||
GetFlagArgVal("ar_mode", arg, &FLAG_arMode) ||
GetFlagArgVal("upscale_strength", arg, &FLAG_upscaleStrength) ||
GetFlagArgVal("resolution", arg, &FLAG_resolution) ||
GetFlagArgVal("model_dir", arg, &FLAG_modelDir) ||
@@ -226,6 +231,10 @@ static bool IsImageFile(const char *str) {
return HasOneOfTheseSuffixes(str, ".bmp", ".jpg", ".jpeg", ".png", nullptr);
}
static bool IsLossyImageFile(const char *str) {
return HasOneOfTheseSuffixes(str, ".jpg", ".jpeg", nullptr);
}
static const char* DurationString(double sc) {
static char buf[16];
int hr, mn;
@@ -488,7 +497,7 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_arEff, NVVFX_INPUT_IMAGE, &_srcGpuBuf));
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_arEff, NVVFX_OUTPUT_IMAGE, &_interGpuBGRf32pl));
BAIL_IF_ERR(vfxErr = NvVFX_SetCudaStream(_arEff, NVVFX_CUDA_STREAM, stream));
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_arEff, NVVFX_STRENGTH, FLAG_arStrength));
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_arEff, NVVFX_MODE, FLAG_arMode));
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_upscaleEff, NVVFX_INPUT_IMAGE, &_interGpuRGBAu8));
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_upscaleEff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
@@ -503,6 +512,8 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_dstGpuBuf, &_dstVFX, 1.f, stream, &_tmpVFX)); // _dstGpuBuf --> _dstTmpVFX --> _dstVFX
if (outFile && outFile[0]) {
if(IsLossyImageFile(outFile))
fprintf(stderr, "WARNING: JPEG output file format will reduce image quality\n");
if (!cv::imwrite(outFile, _dstImg)) {
printf("Error writing: \"%s\"\n", outFile);
return errWrite;
@@ -552,7 +563,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_arEff, NVVFX_INPUT_IMAGE, &_srcGpuBuf));
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_arEff, NVVFX_OUTPUT_IMAGE, &_interGpuBGRf32pl));
BAIL_IF_ERR(vfxErr = NvVFX_SetCudaStream(_arEff, NVVFX_CUDA_STREAM, stream));
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_arEff, NVVFX_STRENGTH, FLAG_arStrength));
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_arEff, NVVFX_MODE, FLAG_arMode));
BAIL_IF_ERR(vfxErr = NvVFX_Load(_arEff));
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_upscaleEff, NVVFX_INPUT_IMAGE, &_interGpuRGBAu8));

View File

@@ -1,6 +1,6 @@
SETLOCAL
SET PATH=%PATH%;..\external\opencv\bin;
REM Use --show to show the output in a window or use --out_file=<filename> to write output to file
UpscalePipelineApp.exe --in_file=..\input\input1.jpg --ar_strength=0 --upscale_strength=0 --resolution=1080 --show --out_file=ar_sr_0.png
UpscalePipelineApp.exe --in_file=..\input\input1.jpg --ar_strength=0 --upscale_strength=1 --resolution=1080 --show --out_file=ar_sr_1.png
UpscalePipelineApp.exe --in_file=..\input\input1.jpg --ar_mode=0 --upscale_strength=0 --resolution=1080 --show --out_file=ar_sr_0.png
UpscalePipelineApp.exe --in_file=..\input\input1.jpg --ar_mode=0 --upscale_strength=1 --resolution=1080 --show --out_file=ar_sr_1.png

View File

@@ -7,29 +7,19 @@ add_executable(VideoEffectsApp ${SOURCE_FILES})
target_include_directories(VideoEffectsApp PRIVATE ${CMAKE_CURRENT_SOURCE_DIR} ${CMAKE_CURRENT_SOURCE_DIR}/../utils)
target_include_directories(VideoEffectsApp PUBLIC ${SDK_INCLUDES_PATH})
if(MSVC)
target_link_libraries(VideoEffectsApp PUBLIC
opencv346
NVVideoEffects
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
)
target_link_libraries(VideoEffectsApp PUBLIC
opencv346
NVVideoEffects
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
)
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
set(CMD_ARG_STR "--model_dir=\"${CMAKE_CURRENT_SOURCE_DIR}/../../bin/models\" --show --effect=SuperRes --resolution=1080 --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input2.png\"")
set_target_properties(VideoEffectsApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)
else()
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
set(CMD_ARG_STR "--show --effect=SuperRes --resolution=1080 --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input1.jpg\"")
set_target_properties(VideoEffectsApp PROPERTIES
FOLDER SampleApps
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
)
target_link_libraries(VideoEffectsApp PUBLIC
NVVideoEffects
NVCVImage
OpenCV
TensorRT
CUDA
)
endif()

View File

@@ -57,6 +57,7 @@ bool FLAG_debug = false,
FLAG_progress = false,
FLAG_webcam = false;
float FLAG_strength = 0.f;
int FLAG_mode = 0;
int FLAG_resolution = 0;
std::string FLAG_codec = DEFAULT_CODEC,
FLAG_camRes = "1280x720",
@@ -66,6 +67,11 @@ std::string FLAG_codec = DEFAULT_CODEC,
FLAG_modelDir,
FLAG_effect;
// Set this when using OTA Updates
// This path is used by nvVideoEffectsProxy.cpp to load the SDK dll
// when using OTA Updates
char *g_nvVFXSDKPath = NULL;
static bool GetFlagArgVal(const char *flag, const char *arg, const char **val) {
if (*arg != '-')
return false;
@@ -140,8 +146,9 @@ static void Usage() {
" --out_file=<path> output file to be written\n"
" --effect=<effect> the effect to apply\n"
" --show display the results in a window (for webcam, it is always true)\n"
" --strength=<value> strength of an effect, 0 or 1 for super res and artifact reduction,\n"
" and [0.0, 1.0] for upscaling\n"
" --strength=<value> strength of the upscaling effect, [0.0, 1.0]\n"
" --mode=<value> mode of the super res or artifact reduction effect, 0 or 1, \n"
" where 0 - conservative and 1 - aggressive\n"
" --cam_res=[WWWx]HHH specify camera resolution as height or width x height\n"
" supports 720 and 1080 resolutions (default \"720\") \n"
" --resolution=<height> the desired height of the output\n"
@@ -176,6 +183,7 @@ static int ParseMyArgs(int argc, char **argv) {
GetFlagArgVal("webcam", arg, &FLAG_webcam) ||
GetFlagArgVal("cam_res", arg, &FLAG_camRes) ||
GetFlagArgVal("strength", arg, &FLAG_strength) ||
GetFlagArgVal("mode", arg, &FLAG_mode) ||
GetFlagArgVal("resolution", arg, &FLAG_resolution) ||
GetFlagArgVal("model_dir", arg, &FLAG_modelDir) ||
GetFlagArgVal("codec", arg, &FLAG_codec) ||
@@ -228,6 +236,11 @@ static bool IsImageFile(const char *str) {
return HasOneOfTheseSuffixes(str, ".bmp", ".jpg", ".jpeg", ".png", nullptr);
}
static bool IsLossyImageFile(const char *str) {
return HasOneOfTheseSuffixes(str, ".jpg", ".jpeg", nullptr);
}
static const char* DurationString(double sc) {
static char buf[16];
int hr, mn;
@@ -561,9 +574,9 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
BAIL_IF_ERR(vfxErr = NvVFX_SetCudaStream(_eff, NVVFX_CUDA_STREAM, stream));
if (!strcmp(_effectName, NVVFX_FX_ARTIFACT_REDUCTION)) {
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned int)FLAG_strength));
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_MODE, (unsigned int)FLAG_mode));
} else if (!strcmp(_effectName, NVVFX_FX_SUPER_RES)) {
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned int)FLAG_strength));
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_MODE, (unsigned int)FLAG_mode));
}
BAIL_IF_ERR(vfxErr = NvVFX_Load(_eff));
@@ -571,6 +584,8 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_dstGpuBuf, &_dstVFX, 255.f, stream, &_tmpVFX)); // _dstGpuBuf --> _tmpVFX --> _dstVFX
if (outFile && outFile[0]) {
if(IsLossyImageFile(outFile))
fprintf(stderr, "WARNING: JPEG output file format will reduce image quality\n");
if (!cv::imwrite(outFile, _dstImg)) {
printf("Error writing: \"%s\"\n", outFile);
return errWrite;
@@ -632,9 +647,9 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
BAIL_IF_ERR(vfxErr = NvVFX_SetCudaStream(_eff, NVVFX_CUDA_STREAM, stream));
if (!strcmp(_effectName, NVVFX_FX_ARTIFACT_REDUCTION)) {
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned int)FLAG_strength));
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_MODE, (unsigned int)FLAG_mode));
} else if (!strcmp(_effectName, NVVFX_FX_SUPER_RES)) {
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned int)FLAG_strength));
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_MODE, (unsigned int)FLAG_mode));
}
BAIL_IF_ERR(vfxErr = NvVFX_Load(_eff));

View File

@@ -1,7 +1,7 @@
SETLOCAL
SET PATH=%PATH%;..\external\opencv\bin;
REM Use --show to show the output in a window or use --out_file=<filename> to write output to file
VideoEffectsApp.exe --in_file=..\input\input1.jpg --out_file=ar_1.png --effect=ArtifactReduction --strength=1 --show
VideoEffectsApp.exe --in_file=..\input\input1.jpg --out_file=ar_0.png --effect=ArtifactReduction --strength=0 --show
VideoEffectsApp.exe --in_file=..\input\input2.jpg --out_file=sr_0.png --effect=SuperRes --resolution=2160 --strength=0 --show
VideoEffectsApp.exe --in_file=..\input\input2.jpg --out_file=sr_1.png --effect=SuperRes --resolution=2160 --strength=1 --show
VideoEffectsApp.exe --in_file=..\input\input1.jpg --out_file=ar_1.png --effect=ArtifactReduction --mode=1 --show
VideoEffectsApp.exe --in_file=..\input\input1.jpg --out_file=ar_0.png --effect=ArtifactReduction --mode=0 --show
VideoEffectsApp.exe --in_file=..\input\input2.jpg --out_file=sr_0.png --effect=SuperRes --resolution=2160 --mode=0 --show
VideoEffectsApp.exe --in_file=..\input\input2.jpg --out_file=sr_1.png --effect=SuperRes --resolution=2160 --mode=1 --show

View File

@@ -40,7 +40,7 @@ else()
message("OpenCV_LIBRARIES ${OpenCV_LIBRARIES}")
message("OpenCV_LIBS ${OpenCV_LIBS}")
find_package(CUDA 11.1 REQUIRED)
find_package(CUDA 11.3 REQUIRED)
add_library(CUDA INTERFACE)
target_include_directories(CUDA INTERFACE ${CUDA_INCLUDE_DIRS})
target_link_libraries(CUDA INTERFACE "${CUDA_LIBRARIES};cuda")
@@ -48,7 +48,7 @@ else()
message("CUDA_INCLUDE_DIRS ${CUDA_INCLUDE_DIRS}")
message("CUDA_LIBRARIES ${CUDA_LIBRARIES}")
find_package(TensorRT 7 REQUIRED)
find_package(TensorRT 8 REQUIRED)
add_library(TensorRT INTERFACE)
target_include_directories(TensorRT INTERFACE ${TensorRT_INCLUDE_DIRS})
target_link_libraries(TensorRT INTERFACE ${TensorRT_LIBRARIES})

View File

@@ -86,7 +86,7 @@
* \endcode
*
* where ::cudaChannelFormatKind is one of ::cudaChannelFormatKindSigned,
* ::cudaChannelFormatKindUnsigned, or ::cudaChannelFormatKindFloat.
* ::cudaChannelFormatKindUnsigned, cudaChannelFormatKindFloat or ::cudaChannelFormatKindNV12.
*
* \return
* Channel descriptor with format \p f
@@ -401,6 +401,12 @@ template<> __inline__ __host__ cudaChannelFormatDesc cudaCreateChannelDesc<float
return cudaCreateChannelDesc(e, e, e, e, cudaChannelFormatKindFloat);
}
static __inline__ __host__ cudaChannelFormatDesc cudaCreateChannelDescNV12(void)
{
int e = (int)sizeof(char) * 8;
return cudaCreateChannelDesc(e, e, e, 0, cudaChannelFormatKindNV12);
}
#endif /* __cplusplus */
/** @} */

View File

@@ -114,8 +114,8 @@
#endif /* __ICC */
#if defined(__PGIC__)
#if ((__PGIC__ != 18) && (__PGIC__ != 19) && (__PGIC__ != 20) && !(__PGIC__ == 99 && __PGIC_MINOR__ == 99))
#error -- unsupported pgc++ configuration! Only pgc++ 18, 19 and 20 are supported! The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
#if ((__PGIC__ != 18) && (__PGIC__ != 19) && (__PGIC__ != 20) && (__PGIC__ != 21) && !(__PGIC__ == 99 && __PGIC_MINOR__ == 99))
#error -- unsupported pgc++ configuration! Only pgc++ 18, 19, 20 and 21 are supported! The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
#endif
#endif /* __PGIC__ */
@@ -143,10 +143,10 @@
#if defined(__clang__) && !defined(__ibmxl_vrm__) && !defined(__ICC) && !defined(__HORIZON__) && !defined(__APPLE__)
#if (__clang_major__ >= 11) || (__clang_major__ < 3) || ((__clang_major__ == 3) && (__clang_minor__ < 3))
#error -- unsupported clang version! clang version must be less than 11 and greater than 3.2 . The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
#if (__clang_major__ >= 12) || (__clang_major__ < 3) || ((__clang_major__ == 3) && (__clang_minor__ < 3))
#error -- unsupported clang version! clang version must be less than 12 and greater than 3.2 . The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
#endif /* (__clang_major__ >= 11) || (__clang_major__ < 3) || ((__clang_major__ == 3) && (__clang_minor__ < 3)) */
#endif /* (__clang_major__ >= 12) || (__clang_major__ < 3) || ((__clang_major__ == 3) && (__clang_minor__ < 3)) */
#endif /* defined(__clang__) && !defined(__ibmxl_vrm__) && !defined(__ICC) && !defined(__HORIZON__) && !defined(__APPLE__) */
@@ -155,15 +155,15 @@
#if defined(_WIN32)
#if _MSC_VER < 1700 || _MSC_VER >= 1930
#if _MSC_VER < 1910 || _MSC_VER >= 1930
#error -- unsupported Microsoft Visual Studio version! Only the versions between 2015 and 2019 (inclusive) are supported! The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
#error -- unsupported Microsoft Visual Studio version! Only the versions between 2017 and 2019 (inclusive) are supported! The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
#elif _MSC_VER >= 1700 && _MSC_VER < 1900
#elif _MSC_VER >= 1910 && _MSC_VER < 1910
#pragma message("support for this version of Microsoft Visual Studio has been deprecated! Only the versions between 2015 and 2019 (inclusive) are supported!")
#pragma message("support for this version of Microsoft Visual Studio has been deprecated! Only the versions between 2017 and 2019 (inclusive) are supported!")
#endif /* (_MSC_VER < 1700 || _MSC_VER >= 1930) || (_MSC_VER >= 1700 && _MSC_VER < 1900) */
#endif /* (_MSC_VER < 1910 || _MSC_VER >= 1930) || (_MSC_VER >= 1910 && _MSC_VER < 1910) */
#endif /* _WIN32 */
#endif /* !__NV_NO_HOST_COMPILER_CHECK */

View File

@@ -97,6 +97,7 @@
#define __location__(a) \
__annotate__(a)
#define CUDARTAPI
#define CUDARTAPI_CDECL
#elif defined(_MSC_VER)
@@ -133,6 +134,8 @@
__annotate__(__##a##__)
#define CUDARTAPI \
__stdcall
#define CUDARTAPI_CDECL \
__cdecl
#else /* __GNUC__ || __CUDA_LIBDEVICE__ || __CUDACC_RTC__ */

View File

@@ -1,5 +1,5 @@
/*
* Copyright 1993-2018 NVIDIA Corporation. All rights reserved.
* Copyright 1993-2021 NVIDIA Corporation. All rights reserved.
*
* NOTICE TO LICENSEE:
*
@@ -66,43 +66,37 @@ extern "C" {
struct cudaFuncAttributes;
#if defined(_WIN32)
#define __NV_WEAK__ __declspec(nv_weak)
#else
#define __NV_WEAK__ __attribute__((nv_weak))
#endif
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaMalloc(void **p, size_t s)
inline __device__ cudaError_t CUDARTAPI cudaMalloc(void **p, size_t s)
{
return cudaErrorUnknown;
}
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaFuncGetAttributes(struct cudaFuncAttributes *p, const void *c)
inline __device__ cudaError_t CUDARTAPI cudaFuncGetAttributes(struct cudaFuncAttributes *p, const void *c)
{
return cudaErrorUnknown;
}
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaDeviceGetAttribute(int *value, enum cudaDeviceAttr attr, int device)
inline __device__ cudaError_t CUDARTAPI cudaDeviceGetAttribute(int *value, enum cudaDeviceAttr attr, int device)
{
return cudaErrorUnknown;
}
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaGetDevice(int *device)
inline __device__ cudaError_t CUDARTAPI cudaGetDevice(int *device)
{
return cudaErrorUnknown;
}
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaOccupancyMaxActiveBlocksPerMultiprocessor(int *numBlocks, const void *func, int blockSize, size_t dynamicSmemSize)
inline __device__ cudaError_t CUDARTAPI cudaOccupancyMaxActiveBlocksPerMultiprocessor(int *numBlocks, const void *func, int blockSize, size_t dynamicSmemSize)
{
return cudaErrorUnknown;
}
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaOccupancyMaxActiveBlocksPerMultiprocessorWithFlags(int *numBlocks, const void *func, int blockSize, size_t dynamicSmemSize, unsigned int flags)
inline __device__ cudaError_t CUDARTAPI cudaOccupancyMaxActiveBlocksPerMultiprocessorWithFlags(int *numBlocks, const void *func, int blockSize, size_t dynamicSmemSize, unsigned int flags)
{
return cudaErrorUnknown;
}
#undef __NV_WEAK__
#if defined(__cplusplus)
}

View File

@@ -188,7 +188,8 @@ struct __device_builtin__ __nv_lambda_preheader_injection { };
* ::cudaErrorInvalidPtx,
* ::cudaErrorUnsupportedPtxVersion,
* ::cudaErrorNoKernelImageForDevice,
* ::cudaErrorJitCompilerNotFound
* ::cudaErrorJitCompilerNotFound,
* ::cudaErrorJitCompilationDisabled
* \notefnerr
* \note_async
* \note_null_stream
@@ -628,6 +629,57 @@ static __inline__ __host__ cudaError_t cudaMallocPitch(
return ::cudaMallocPitch((void**)(void*)devPtr, pitch, width, height);
}
/**
* \brief Allocate from a pool
*
* This is an alternate spelling for cudaMallocFromPoolAsync
* made available through operator overloading.
*
* \sa ::cudaMallocFromPoolAsync,
* \ref ::cudaMallocAsync(void** ptr, size_t size, cudaStream_t hStream) "cudaMallocAsync (C API)"
*/
static __inline__ __host__ cudaError_t cudaMallocAsync(
void **ptr,
size_t size,
cudaMemPool_t memPool,
cudaStream_t stream
)
{
return ::cudaMallocFromPoolAsync(ptr, size, memPool, stream);
}
template<class T>
static __inline__ __host__ cudaError_t cudaMallocAsync(
T **ptr,
size_t size,
cudaMemPool_t memPool,
cudaStream_t stream
)
{
return ::cudaMallocFromPoolAsync((void**)(void*)ptr, size, memPool, stream);
}
template<class T>
static __inline__ __host__ cudaError_t cudaMallocAsync(
T **ptr,
size_t size,
cudaStream_t stream
)
{
return ::cudaMallocAsync((void**)(void*)ptr, size, stream);
}
template<class T>
static __inline__ __host__ cudaError_t cudaMallocFromPoolAsync(
T **ptr,
size_t size,
cudaMemPool_t memPool,
cudaStream_t stream
)
{
return ::cudaMallocFromPoolAsync((void**)(void*)ptr, size, memPool, stream);
}
#if defined(__CUDACC__)
/**
@@ -1190,6 +1242,59 @@ static __inline__ __host__ cudaError_t cudaGraphExecMemcpyNodeSetParamsFromSymbo
return ::cudaGraphExecMemcpyNodeSetParamsFromSymbol(hGraphExec, node, dst, (const void*)&symbol, count, offset, kind);
}
#if __cplusplus >= 201103
/**
* \brief Creates a user object by wrapping a C++ object
*
* TODO detail
*
* \param object_out - Location to return the user object handle
* \param objectToWrap - This becomes the \ptr argument to ::cudaUserObjectCreate. A
* lambda will be passed for the \p destroy argument, which calls
* delete on this object pointer.
* \param initialRefcount - The initial refcount to create the object with, typically 1. The
* initial references are owned by the calling thread.
* \param flags - Currently it is required to pass cudaUserObjectNoDestructorSync,
* which is the only defined flag. This indicates that the destroy
* callback cannot be waited on by any CUDA API. Users requiring
* synchronization of the callback should signal its completion
* manually.
*
* \return
* ::cudaSuccess,
* ::cudaErrorInvalidValue
*
* \sa
* ::cudaUserObjectCreate
*/
template<class T>
static __inline__ __host__ cudaError_t cudaUserObjectCreate(
cudaUserObject_t *object_out,
T *objectToWrap,
unsigned int initialRefcount,
unsigned int flags)
{
return ::cudaUserObjectCreate(
object_out,
objectToWrap,
[](void *vpObj) { delete reinterpret_cast<T *>(vpObj); },
initialRefcount,
flags);
}
template<class T>
static __inline__ __host__ cudaError_t cudaUserObjectCreate(
cudaUserObject_t *object_out,
T *objectToWrap,
unsigned int initialRefcount,
cudaUserObjectFlags flags)
{
return cudaUserObjectCreate(object_out, objectToWrap, initialRefcount, (unsigned int)flags);
}
#endif
/**
* \brief \hl Finds the address associated with a CUDA symbol
*

File diff suppressed because it is too large Load Diff

View File

@@ -393,6 +393,13 @@ enum __device_builtin__ cudaError
*/
cudaErrorMemoryValueTooLarge = 32,
/**
* This indicates that the CUDA driver that the application has loaded is a
* stub library. Applications that run with the stub rather than a real
* driver loaded will result in CUDA API returning this error.
*/
cudaErrorStubLibrary = 34,
/**
* This indicates that the installed NVIDIA CUDA driver is older than the
* CUDA runtime library. This is not a supported configuration. Users should
@@ -538,10 +545,19 @@ enum __device_builtin__ cudaError
cudaErrorInvalidDevice = 101,
/**
* This indicates that the device doesn't have valid Grid License.
* This indicates that the device doesn't have a valid Grid License.
*/
cudaErrorDeviceNotLicensed = 102,
/**
* By default, the CUDA runtime may perform a minimal set of self-tests,
* as well as CUDA driver tests, to establish the validity of both.
* Introduced in CUDA 11.2, this error return indicates that at least one
* of these tests has failed and the validity of either the runtime
* or the driver could not be established.
*/
cudaErrorSoftwareValidityNotEstablished = 103,
/**
* This indicates an internal startup failure in the CUDA runtime.
*/
@@ -668,6 +684,13 @@ enum __device_builtin__ cudaError
*/
cudaErrorUnsupportedPtxVersion = 222,
/**
* This indicates that the JIT compilation was disabled. The JIT compilation compiles
* PTX. The runtime may fall back to compiling PTX if an application does not contain
* a suitable binary for the current device.
*/
cudaErrorJitCompilationDisabled = 223,
/**
* This indicates that the device kernel source is invalid.
*/
@@ -708,7 +731,8 @@ enum __device_builtin__ cudaError
/**
* This indicates that a named symbol was not found. Examples of symbols
* are global/constant variable names, texture names, and surface names.
* are global/constant variable names, driver function names, texture names,
* and surface names.
*/
cudaErrorSymbolNotFound = 500,
@@ -1002,7 +1026,8 @@ enum __device_builtin__ cudaChannelFormatKind
cudaChannelFormatKindSigned = 0, /**< Signed channel format */
cudaChannelFormatKindUnsigned = 1, /**< Unsigned channel format */
cudaChannelFormatKindFloat = 2, /**< Float channel format */
cudaChannelFormatKindNone = 3 /**< No channel format */
cudaChannelFormatKindNone = 3, /**< No channel format */
cudaChannelFormatKindNV12 = 4 /**< Unsigned 8-bit integers, planar 4:2:0 YUV format */
};
/**
@@ -1058,6 +1083,7 @@ struct __device_builtin__ cudaArraySparseProperties {
unsigned int miptailFirstLevel; /**< First mip level at which the mip tail begins */
unsigned long long miptailSize; /**< Total size of the mip tail. */
unsigned int flags; /**< Flags will either be zero or ::cudaArraySparsePropertiesSingleMipTail */
unsigned int reserved[4];
};
/**
@@ -1163,7 +1189,7 @@ struct __device_builtin__ cudaMemsetParams {
size_t pitch; /**< Pitch of destination device pointer. Unused if height is 1 */
unsigned int value; /**< Value to be set */
unsigned int elementSize; /**< Size of each element in bytes. Must be 1, 2, or 4. */
size_t width; /**< Width in bytes, of the row */
size_t width; /**< Width of the row in elements */
size_t height; /**< Number of rows */
};
@@ -1258,6 +1284,28 @@ union __device_builtin__ cudaStreamAttrValue {
enum cudaSynchronizationPolicy syncPolicy;
};
/**
* Flags for ::cudaStreamUpdateCaptureDependencies
*/
enum __device_builtin__ cudaStreamUpdateCaptureDependenciesFlags {
cudaStreamAddCaptureDependencies = 0x0, /**< Add new nodes to the dependency set */
cudaStreamSetCaptureDependencies = 0x1 /**< Replace the dependency set with the new nodes */
};
/**
* Flags for user objects for graphs
*/
enum __device_builtin__ cudaUserObjectFlags {
cudaUserObjectNoDestructorSync = 0x1 /**< Indicates the destructor execution is not synchronized by any CUDA handle. */
};
/**
* Flags for retaining user object references for graphs
*/
enum __device_builtin__ cudaUserObjectRetainFlags {
cudaGraphUserObjectMove = 0x1 /**< Transfer references from the caller rather than creating new references. */
};
/**
* CUDA graphics interop resource
*/
@@ -1307,7 +1355,7 @@ enum __device_builtin__ cudaKernelNodeAttrID {
};
/**
* Graph kernel node attributes union, used with ::cudaKernelNodeSetAttribute/::cudaKernelNodeGetAttribute
* Graph kernel node attributes union, used with ::cudaGraphKernelNodeSetAttribute/::cudaGraphKernelNodeGetAttribute
*/
union __device_builtin__ cudaKernelNodeAttrValue {
struct cudaAccessPolicyWindow accessPolicyWindow; /**< Attribute ::CUaccessPolicyWindow. */
@@ -1619,6 +1667,39 @@ enum __device_builtin__ cudaOutputMode
cudaCSV = 0x01 /**< Output mode Comma separated values format. */
};
/**
* CUDA GPUDirect RDMA flush writes APIs supported on the device
*/
enum __device_builtin__ cudaFlushGPUDirectRDMAWritesOptions {
cudaFlushGPUDirectRDMAWritesOptionHost = 1<<0, /**< ::cudaDeviceFlushGPUDirectRDMAWrites() and its CUDA Driver API counterpart are supported on the device. */
cudaFlushGPUDirectRDMAWritesOptionMemOps = 1<<1 /**< The ::CU_STREAM_WAIT_VALUE_FLUSH flag and the ::CU_STREAM_MEM_OP_FLUSH_REMOTE_WRITES MemOp are supported on the CUDA device. */
};
/**
* CUDA GPUDirect RDMA flush writes ordering features of the device
*/
enum __device_builtin__ cudaGPUDirectRDMAWritesOrdering {
cudaGPUDirectRDMAWritesOrderingNone = 0, /**< The device does not natively support ordering of GPUDirect RDMA writes. ::cudaFlushGPUDirectRDMAWrites() can be leveraged if supported. */
cudaGPUDirectRDMAWritesOrderingOwner = 100, /**< Natively, the device can consistently consume GPUDirect RDMA writes, although other CUDA devices may not. */
cudaGPUDirectRDMAWritesOrderingAllDevices = 200 /**< Any CUDA device in the system can consistently consume GPUDirect RDMA writes to this device. */
};
/**
* CUDA GPUDirect RDMA flush writes scopes
*/
enum __device_builtin__ cudaFlushGPUDirectRDMAWritesScope {
cudaFlushGPUDirectRDMAWritesToOwner = 100, /**< Blocks until remote writes are visible to the CUDA device context owning the data. */
cudaFlushGPUDirectRDMAWritesToAllDevices = 200 /**< Blocks until remote writes are visible to all CUDA device contexts. */
};
/**
* CUDA GPUDirect RDMA flush writes targets
*/
enum __device_builtin__ cudaFlushGPUDirectRDMAWritesTarget {
cudaFlushGPUDirectRDMAWritesTargetCurrentDevice /**< Sets the target for ::cudaDeviceFlushGPUDirectRDMAWrites() to the currently active CUDA device context. */
};
/**
* CUDA device attributes
*/
@@ -1725,9 +1806,166 @@ enum __device_builtin__ cudaDeviceAttr
cudaDevAttrPageableMemoryAccessUsesHostPageTables = 100, /**< Device accesses pageable memory via the host's page tables. */
cudaDevAttrDirectManagedMemAccessFromHost = 101, /**< Host can directly access managed memory on the device without migration. */
cudaDevAttrMaxBlocksPerMultiprocessor = 106, /**< Maximum number of blocks per multiprocessor */
cudaDevAttrMaxPersistingL2CacheSize = 108, /**< Maximum L2 persisting lines capacity setting in bytes. */
cudaDevAttrMaxAccessPolicyWindowSize = 109, /**< Maximum value of cudaAccessPolicyWindow::num_bytes. */
cudaDevAttrReservedSharedMemoryPerBlock = 111, /**< Shared memory reserved by CUDA driver per block in bytes */
cudaDevAttrSparseCudaArraySupported = 112, /**< Device supports sparse CUDA arrays and sparse CUDA mipmapped arrays */
cudaDevAttrHostRegisterReadOnlySupported = 113 /**< Device supports using the ::cuMemHostRegister flag CU_MEMHOSTERGISTER_READ_ONLY to register memory that must be mapped as read-only to the GPU */
cudaDevAttrHostRegisterReadOnlySupported = 113, /**< Device supports using the ::cudaHostRegister flag cudaHostRegisterReadOnly to register memory that must be mapped as read-only to the GPU */
cudaDevAttrMaxTimelineSemaphoreInteropSupported = 114, /**< External timeline semaphore interop is supported on the device */
cudaDevAttrMemoryPoolsSupported = 115, /**< Device supports using the ::cudaMallocAsync and ::cudaMemPool family of APIs */
cudaDevAttrGPUDirectRDMASupported = 116, /**< Device supports GPUDirect RDMA APIs, like nvidia_p2p_get_pages (see https://docs.nvidia.com/cuda/gpudirect-rdma for more information) */
cudaDevAttrGPUDirectRDMAFlushWritesOptions = 117, /**< The returned attribute shall be interpreted as a bitmask, where the individual bits are listed in the ::cudaFlushGPUDirectRDMAWritesOptions enum */
cudaDevAttrGPUDirectRDMAWritesOrdering = 118, /**< GPUDirect RDMA writes to the device do not need to be flushed for consumers within the scope indicated by the returned attribute. See ::cudaGPUDirectRDMAWritesOrdering for the numerical values returned here. */
cudaDevAttrMemoryPoolSupportedHandleTypes = 119 /**< Handle types supported with mempool based IPC */
};
/**
* CUDA memory pool attributes
*/
enum __device_builtin__ cudaMemPoolAttr
{
/**
* (value type = int)
* Allow cuMemAllocAsync to use memory asynchronously freed
* in another streams as long as a stream ordering dependency
* of the allocating stream on the free action exists.
* Cuda events and null stream interactions can create the required
* stream ordered dependencies. (default enabled)
*/
cudaMemPoolReuseFollowEventDependencies = 0x1,
/**
* (value type = int)
* Allow reuse of already completed frees when there is no dependency
* between the free and allocation. (default enabled)
*/
cudaMemPoolReuseAllowOpportunistic = 0x2,
/**
* (value type = int)
* Allow cuMemAllocAsync to insert new stream dependencies
* in order to establish the stream ordering required to reuse
* a piece of memory released by cuFreeAsync (default enabled).
*/
cudaMemPoolReuseAllowInternalDependencies = 0x3,
/**
* (value type = cuuint64_t)
* Amount of reserved memory in bytes to hold onto before trying
* to release memory back to the OS. When more than the release
* threshold bytes of memory are held by the memory pool, the
* allocator will try to release memory back to the OS on the
* next call to stream, event or context synchronize. (default 0)
*/
cudaMemPoolAttrReleaseThreshold = 0x4,
/**
* (value type = cuuint64_t)
* Amount of backing memory currently allocated for the mempool.
*/
cudaMemPoolAttrReservedMemCurrent = 0x5,
/**
* (value type = cuuint64_t)
* High watermark of backing memory allocated for the mempool since the
* last time it was reset. High watermark can only be reset to zero.
*/
cudaMemPoolAttrReservedMemHigh = 0x6,
/**
* (value type = cuuint64_t)
* Amount of memory from the pool that is currently in use by the application.
*/
cudaMemPoolAttrUsedMemCurrent = 0x7,
/**
* (value type = cuuint64_t)
* High watermark of the amount of memory from the pool that was in use by the application since
* the last time it was reset. High watermark can only be reset to zero.
*/
cudaMemPoolAttrUsedMemHigh = 0x8
};
/**
* Specifies the type of location
*/
enum __device_builtin__ cudaMemLocationType {
cudaMemLocationTypeInvalid = 0,
cudaMemLocationTypeDevice = 1 /**< Location is a device location, thus id is a device ordinal */
};
/**
* Specifies a memory location.
*
* To specify a gpu, set type = ::cudaMemLocationTypeDevice and set id = the gpu's device ordinal.
*/
struct __device_builtin__ cudaMemLocation {
enum cudaMemLocationType type; /**< Specifies the location type, which modifies the meaning of id. */
int id; /**< identifier for a given this location's ::CUmemLocationType. */
};
/**
* Specifies the memory protection flags for mapping.
*/
enum __device_builtin__ cudaMemAccessFlags {
cudaMemAccessFlagsProtNone = 0, /**< Default, make the address range not accessible */
cudaMemAccessFlagsProtRead = 1, /**< Make the address range read accessible */
cudaMemAccessFlagsProtReadWrite = 3 /**< Make the address range read-write accessible */
};
/**
* Memory access descriptor
*/
struct __device_builtin__ cudaMemAccessDesc {
struct cudaMemLocation location; /**< Location on which the request is to change it's accessibility */
enum cudaMemAccessFlags flags; /**< ::CUmemProt accessibility flags to set on the request */
};
/**
* Defines the allocation types available
*/
enum __device_builtin__ cudaMemAllocationType {
cudaMemAllocationTypeInvalid = 0x0,
/** This allocation type is 'pinned', i.e. cannot migrate from its current
* location while the application is actively using it
*/
cudaMemAllocationTypePinned = 0x1,
cudaMemAllocationTypeMax = 0x7FFFFFFF
};
/**
* Flags for specifying particular handle types
*/
enum __device_builtin__ cudaMemAllocationHandleType {
cudaMemHandleTypeNone = 0x0, /**< Does not allow any export mechanism. > */
cudaMemHandleTypePosixFileDescriptor = 0x1, /**< Allows a file descriptor to be used for exporting. Permitted only on POSIX systems. (int) */
cudaMemHandleTypeWin32 = 0x2, /**< Allows a Win32 NT handle to be used for exporting. (HANDLE) */
cudaMemHandleTypeWin32Kmt = 0x4 /**< Allows a Win32 KMT handle to be used for exporting. (D3DKMT_HANDLE) */
};
/**
* Specifies the properties of allocations made from the pool.
*/
struct __device_builtin__ cudaMemPoolProps {
enum cudaMemAllocationType allocType; /**< Allocation type. Currently must be specified as cudaMemAllocationTypePinned */
enum cudaMemAllocationHandleType handleTypes; /**< Handle types that will be supported by allocations from the pool. */
struct cudaMemLocation location; /**< Location allocations should reside. */
/**
* Windows-specific LPSECURITYATTRIBUTES required when
* ::cudaMemHandleTypeWin32 is specified. This security attribute defines
* the scope of which exported allocations may be tranferred to other
* processes. In all other cases, this field is required to be zero.
*/
void *win32SecurityAttributes;
unsigned char reserved[64]; /**< reserved for future use, must be 0 */
};
/**
* Opaque data for exporting a pool allocation
*/
struct __device_builtin__ cudaMemPoolPtrExportData {
unsigned char reserved[64];
};
/**
@@ -1784,7 +2022,7 @@ struct __device_builtin__ cudaDeviceProp
int computeMode; /**< Compute mode (See ::cudaComputeMode) */
int maxTexture1D; /**< Maximum 1D texture size */
int maxTexture1DMipmap; /**< Maximum 1D mipmapped texture size */
int maxTexture1DLinear; /**< Maximum size for 1D textures bound to linear memory */
int maxTexture1DLinear; /**< Deprecated, do not use. Use cudaDeviceGetTexture1DLinearMaxWidth() or cuDeviceGetTexture1DLinearMaxWidth() instead. */
int maxTexture2D[2]; /**< Maximum 2D texture dimensions */
int maxTexture2DMipmap[2]; /**< Maximum 2D mipmapped texture dimensions */
int maxTexture2DLinear[3]; /**< Maximum dimensions (width, height, pitch) for 2D textures bound to pitched memory */
@@ -1831,7 +2069,7 @@ struct __device_builtin__ cudaDeviceProp
int computePreemptionSupported; /**< Device supports Compute Preemption */
int canUseHostPointerForRegisteredMem; /**< Device can access host registered memory at the same virtual address as the CPU */
int cooperativeLaunch; /**< Device supports launching cooperative kernels via ::cudaLaunchCooperativeKernel */
int cooperativeMultiDeviceLaunch; /**< Device can participate in cooperative kernels launched via ::cudaLaunchCooperativeKernelMultiDevice */
int cooperativeMultiDeviceLaunch; /**< Deprecated, cudaLaunchCooperativeKernelMultiDevice is deprecated. */
size_t sharedMemPerBlockOptin; /**< Per device maximum shared memory per block usable by special opt in */
int pageableMemoryAccessUsesHostPageTables; /**< Device accesses pageable memory via the host's page tables */
int directManagedMemAccessFromHost; /**< Host can directly access managed memory on the device without migration. */
@@ -2134,7 +2372,7 @@ enum __device_builtin__ cudaExternalSemaphoreHandleType {
* Handle is an opaque shared NT handle
*/
cudaExternalSemaphoreHandleTypeOpaqueWin32 = 2,
/**
/**
* Handle is an opaque, globally shared handle
*/
cudaExternalSemaphoreHandleTypeOpaqueWin32Kmt = 3,
@@ -2157,7 +2395,15 @@ enum __device_builtin__ cudaExternalSemaphoreHandleType {
/**
* Handle is a shared KMT handle referencing a D3D11 keyed mutex object
*/
cudaExternalSemaphoreHandleTypeKeyedMutexKmt = 8
cudaExternalSemaphoreHandleTypeKeyedMutexKmt = 8,
/**
* Handle is an opaque handle file descriptor referencing a timeline semaphore
*/
cudaExternalSemaphoreHandleTypeTimelineSemaphoreFd = 9,
/**
* Handle is an opaque handle file descriptor referencing a timeline semaphore
*/
cudaExternalSemaphoreHandleTypeTimelineSemaphoreWin32 = 10
};
/**
@@ -2170,8 +2416,10 @@ struct __device_builtin__ cudaExternalSemaphoreHandleDesc {
enum cudaExternalSemaphoreHandleType type;
union {
/**
* File descriptor referencing the semaphore object. Valid
* when type is ::cudaExternalSemaphoreHandleTypeOpaqueFd
* File descriptor referencing the semaphore object. Valid when
* type is one of the following:
* - ::cudaExternalSemaphoreHandleTypeOpaqueFd
* - ::cudaExternalSemaphoreHandleTypeTimelineSemaphoreFd
*/
int fd;
/**
@@ -2182,6 +2430,7 @@ struct __device_builtin__ cudaExternalSemaphoreHandleDesc {
* - ::cudaExternalSemaphoreHandleTypeD3D12Fence
* - ::cudaExternalSemaphoreHandleTypeD3D11Fence
* - ::cudaExternalSemaphoreHandleTypeKeyedMutex
* - ::cudaExternalSemaphoreHandleTypeTimelineSemaphoreWin32
* Exactly one of 'handle' and 'name' must be non-NULL. If
* type is one of the following:
* ::cudaExternalSemaphoreHandleTypeOpaqueWin32Kmt
@@ -2210,10 +2459,11 @@ struct __device_builtin__ cudaExternalSemaphoreHandleDesc {
unsigned int flags;
};
#if defined(__CUDA_API_VERSION_INTERNAL)
/**
* External semaphore signal parameters
* External semaphore signal parameters(deprecated)
*/
struct __device_builtin__ cudaExternalSemaphoreSignalParams {
struct __device_builtin__ cudaExternalSemaphoreSignalParams_v1 {
struct {
/**
* Parameters for fence objects
@@ -2256,9 +2506,9 @@ struct __device_builtin__ cudaExternalSemaphoreSignalParams {
};
/**
* External semaphore wait parameters
* External semaphore wait parameters(deprecated)
*/
struct __device_builtin__ cudaExternalSemaphoreWaitParams {
struct __device_builtin__ cudaExternalSemaphoreWaitParams_v1 {
struct {
/**
* Parameters for fence objects
@@ -2303,6 +2553,105 @@ struct __device_builtin__ cudaExternalSemaphoreWaitParams {
*/
unsigned int flags;
};
#endif
/**
* External semaphore signal parameters, compatible with driver type
*/
struct __device_builtin__ cudaExternalSemaphoreSignalParams{
struct {
/**
* Parameters for fence objects
*/
struct {
/**
* Value of fence to be signaled
*/
unsigned long long value;
} fence;
union {
/**
* Pointer to NvSciSyncFence. Valid if ::cudaExternalSemaphoreHandleType
* is of type ::cudaExternalSemaphoreHandleTypeNvSciSync.
*/
void *fence;
unsigned long long reserved;
} nvSciSync;
/**
* Parameters for keyed mutex objects
*/
struct {
/*
* Value of key to release the mutex with
*/
unsigned long long key;
} keyedMutex;
unsigned int reserved[12];
} params;
/**
* Only when ::cudaExternalSemaphoreSignalParams is used to
* signal a ::cudaExternalSemaphore_t of type
* ::cudaExternalSemaphoreHandleTypeNvSciSync, the valid flag is
* ::cudaExternalSemaphoreSignalSkipNvSciBufMemSync: which indicates
* that while signaling the ::cudaExternalSemaphore_t, no memory
* synchronization operations should be performed for any external memory
* object imported as ::cudaExternalMemoryHandleTypeNvSciBuf.
* For all other types of ::cudaExternalSemaphore_t, flags must be zero.
*/
unsigned int flags;
unsigned int reserved[16];
};
/**
* External semaphore wait parameters, compatible with driver type
*/
struct __device_builtin__ cudaExternalSemaphoreWaitParams {
struct {
/**
* Parameters for fence objects
*/
struct {
/**
* Value of fence to be waited on
*/
unsigned long long value;
} fence;
union {
/**
* Pointer to NvSciSyncFence. Valid if ::cudaExternalSemaphoreHandleType
* is of type ::cudaExternalSemaphoreHandleTypeNvSciSync.
*/
void *fence;
unsigned long long reserved;
} nvSciSync;
/**
* Parameters for keyed mutex objects
*/
struct {
/**
* Value of key to acquire the mutex with
*/
unsigned long long key;
/**
* Timeout in milliseconds to wait to acquire the mutex
*/
unsigned int timeoutMs;
} keyedMutex;
unsigned int reserved[10];
} params;
/**
* Only when ::cudaExternalSemaphoreSignalParams is used to
* signal a ::cudaExternalSemaphore_t of type
* ::cudaExternalSemaphoreHandleTypeNvSciSync, the valid flag is
* ::cudaExternalSemaphoreSignalSkipNvSciBufMemSync: which indicates
* that while waiting for the ::cudaExternalSemaphore_t, no memory
* synchronization operations should be performed for any external memory
* object imported as ::cudaExternalMemoryHandleTypeNvSciBuf.
* For all other types of ::cudaExternalSemaphore_t, flags must be zero.
*/
unsigned int flags;
unsigned int reserved[16];
};
/*******************************************************************************
@@ -2356,11 +2705,21 @@ typedef __device_builtin__ struct CUgraph_st *cudaGraph_t;
*/
typedef __device_builtin__ struct CUgraphNode_st *cudaGraphNode_t;
/**
* CUDA user object for graphs
*/
typedef __device_builtin__ struct CUuserObject_st *cudaUserObject_t;
/**
* CUDA function
*/
typedef __device_builtin__ struct CUfunc_st *cudaFunction_t;
/**
* CUDA memory pool
*/
typedef __device_builtin__ struct CUmemPoolHandle_st *cudaMemPool_t;
/**
* CUDA cooperative group scope
*/
@@ -2395,16 +2754,36 @@ struct __device_builtin__ cudaKernelNodeParams {
void **extra; /**< Pointer to kernel arguments in the "extra" format */
};
/**
* External semaphore signal node parameters
*/
struct __device_builtin__ cudaExternalSemaphoreSignalNodeParams {
cudaExternalSemaphore_t* extSemArray; /**< Array of external semaphore handles. */
const struct cudaExternalSemaphoreSignalParams* paramsArray; /**< Array of external semaphore signal parameters. */
unsigned int numExtSems; /**< Number of handles and parameters supplied in extSemArray and paramsArray. */
};
/**
* External semaphore wait node parameters
*/
struct __device_builtin__ cudaExternalSemaphoreWaitNodeParams {
cudaExternalSemaphore_t* extSemArray; /**< Array of external semaphore handles. */
const struct cudaExternalSemaphoreWaitParams* paramsArray; /**< Array of external semaphore wait parameters. */
unsigned int numExtSems; /**< Number of handles and parameters supplied in extSemArray and paramsArray. */
};
/**
* CUDA Graph node types
*/
enum __device_builtin__ cudaGraphNodeType {
cudaGraphNodeTypeKernel = 0x00, /**< GPU kernel node */
cudaGraphNodeTypeMemcpy = 0x01, /**< Memcpy node */
cudaGraphNodeTypeMemset = 0x02, /**< Memset node */
cudaGraphNodeTypeHost = 0x03, /**< Host (executable) node */
cudaGraphNodeTypeGraph = 0x04, /**< Node which executes an embedded graph */
cudaGraphNodeTypeEmpty = 0x05, /**< Empty (no-op) node */
cudaGraphNodeTypeKernel = 0x00, /**< GPU kernel node */
cudaGraphNodeTypeMemcpy = 0x01, /**< Memcpy node */
cudaGraphNodeTypeMemset = 0x02, /**< Memset node */
cudaGraphNodeTypeHost = 0x03, /**< Host (executable) node */
cudaGraphNodeTypeGraph = 0x04, /**< Node which executes an embedded graph */
cudaGraphNodeTypeEmpty = 0x05, /**< Empty (no-op) node */
cudaGraphNodeTypeWaitEvent = 0x06, /**< External event wait node */
cudaGraphNodeTypeEventRecord = 0x07, /**< External event record node */
cudaGraphNodeTypeCount
};
@@ -2421,9 +2800,36 @@ enum __device_builtin__ cudaGraphExecUpdateResult {
cudaGraphExecUpdateError = 0x1, /**< The update failed for an unexpected reason which is described in the return value of the function */
cudaGraphExecUpdateErrorTopologyChanged = 0x2, /**< The update failed because the topology changed */
cudaGraphExecUpdateErrorNodeTypeChanged = 0x3, /**< The update failed because a node type changed */
cudaGraphExecUpdateErrorFunctionChanged = 0x4, /**< The update failed because the function of a kernel node changed */
cudaGraphExecUpdateErrorFunctionChanged = 0x4, /**< The update failed because the function of a kernel node changed (CUDA driver < 11.2) */
cudaGraphExecUpdateErrorParametersChanged = 0x5, /**< The update failed because the parameters changed in a way that is not supported */
cudaGraphExecUpdateErrorNotSupported = 0x6 /**< The update failed because something about the node is not supported */
cudaGraphExecUpdateErrorNotSupported = 0x6, /**< The update failed because something about the node is not supported */
cudaGraphExecUpdateErrorUnsupportedFunctionChange = 0x7 /**< The update failed because the function of a kernel node changed in an unsupported way */
};
/**
* Flags to specify search options to be used with ::cudaGetDriverEntryPoint
* For more details see ::cuGetProcAddress
*/
enum __device_builtin__ cudaGetDriverEntryPointFlags {
cudaEnableDefault = 0x0, /**< Default search mode for driver symbols. */
cudaEnableLegacyStream = 0x1, /**< Search for legacy versions of driver symbols. */
cudaEnablePerThreadDefaultStream = 0x2 /**< Search for per-thread versions of driver symbols. */
};
/**
* CUDA Graph debug write options
*/
enum __device_builtin__ cudaGraphDebugDotFlags {
cudaGraphDebugDotFlagsVerbose = 1<<0, /** Output all debug data as if every debug flag is enabled */
cudaGraphDebugDotFlagsKernelNodeParams = 1<<2, /** Adds cudaKernelNodeParams to output */
cudaGraphDebugDotFlagsMemcpyNodeParams = 1<<3, /** Adds cudaMemcpy3DParms to output */
cudaGraphDebugDotFlagsMemsetNodeParams = 1<<4, /** Adds cudaMemsetParams to output */
cudaGraphDebugDotFlagsHostNodeParams = 1<<5, /** Adds cudaHostNodeParams to output */
cudaGraphDebugDotFlagsEventNodeParams = 1<<6, /** Adds cudaEvent_t handle from record and wait nodes to output */
cudaGraphDebugDotFlagsExtSemasSignalNodeParams = 1<<7, /** Adds cudaExternalSemaphoreSignalNodeParams values to output */
cudaGraphDebugDotFlagsExtSemasWaitNodeParams = 1<<8, /** Adds cudaExternalSemaphoreWaitNodeParams to output */
cudaGraphDebugDotFlagsKernelNodeAttributes = 1<<9, /** Adds cudaKernelNodeAttrID values to output */
cudaGraphDebugDotFlagsHandles = 1<<10 /** Adds node handles and every kernel function handle to output */
};
/** @} */

Binary file not shown.

View File

@@ -58,21 +58,22 @@ inline void CVWrapperForNvCVImage(const NvCVImage *nvcvIm, cv::Mat *cvIm) {
// Wrap a cv::Mat in an NvCVImage.
inline void NVWrapperForCVMat(const cv::Mat *cvIm, NvCVImage *nvcvIm) {
static const NvCVImage_PixelFormat nvFormat[] = { NVCV_FORMAT_UNKNOWN, NVCV_Y, NVCV_YA, NVCV_BGR, NVCV_BGRA };
static const NvCVImage_ComponentType nvType[] = { NVCV_U8, NVCV_TYPE_UNKNOWN, NVCV_U16, NVCV_S16, NVCV_S32, NVCV_F32, NVCV_F64 };
static const NvCVImage_ComponentType nvType[] = {NVCV_U8, NVCV_TYPE_UNKNOWN, NVCV_U16, NVCV_S16,
NVCV_S32, NVCV_F32, NVCV_F64, NVCV_TYPE_UNKNOWN};
nvcvIm->pixels = cvIm->data;
nvcvIm->width = cvIm->cols;
nvcvIm->height = cvIm->rows;
nvcvIm->pitch = (unsigned)cvIm->step1();
nvcvIm->pixelFormat = nvFormat[cvIm->channels()];
nvcvIm->componentType = nvType[cvIm->depth()];
nvcvIm->pitch = (int)cvIm->step[0];
nvcvIm->pixelFormat = nvFormat[cvIm->channels() <= 4 ? cvIm->channels() : 0];
nvcvIm->componentType = nvType[cvIm->depth() & 7];
nvcvIm->bufferBytes = 0;
nvcvIm->deletePtr = nullptr;
nvcvIm->deleteProc = nullptr;
nvcvIm->pixelBytes = (unsigned char)cvIm->elemSize();
nvcvIm->pixelBytes = (unsigned char)cvIm->step[1];
nvcvIm->componentBytes = (unsigned char)cvIm->elemSize1();
nvcvIm->numComponents = (unsigned char)cvIm->channels();
nvcvIm->planar = 0;
nvcvIm->gpuMem = 0;
nvcvIm->planar = NVCV_CHUNKY;
nvcvIm->gpuMem = NVCV_CPU;
nvcvIm->reserved[0] = 0;
nvcvIm->reserved[1] = 0;
}