v0.6.5.2 Release
v0.6.5.2
This commit is contained in:
@@ -63,9 +63,9 @@
|
||||
|
||||
bool FLAG_progress = false;
|
||||
bool FLAG_show = false;
|
||||
bool FLAG_useOTAU = false;
|
||||
bool FLAG_verbose = false;
|
||||
bool FLAG_webcam = false;
|
||||
bool FLAG_cudaGraph = false;
|
||||
int FLAG_compMode = 3 /*compWhite*/;
|
||||
int FLAG_mode = 0;
|
||||
float FLAG_blurStrength = 0.5;
|
||||
@@ -75,6 +75,7 @@ std::string FLAG_inFile;
|
||||
std::string FLAG_modelDir;
|
||||
std::string FLAG_outDir;
|
||||
std::string FLAG_outFile;
|
||||
std::string FLAG_bgFile;
|
||||
|
||||
static bool GetFlagArgVal(const char *flag, const char *arg, const char **val) {
|
||||
if (*arg != '-') return false;
|
||||
@@ -135,6 +136,7 @@ static void Usage() {
|
||||
" where args is:\n"
|
||||
" --in_file=<path> input file to be processed\n"
|
||||
" --out_file=<path> output file to be written\n"
|
||||
" --bg_file=<path> background file for composition\n"
|
||||
" --webcam use a webcam as input\n"
|
||||
" --cam_res=[WWWx]HHH specify resolution as height or width x height\n"
|
||||
" --model_dir=<path> the path to the directory that contains the models\n"
|
||||
@@ -144,8 +146,15 @@ static void Usage() {
|
||||
" --mode=(0|1) pick one of the green screen modes\n"
|
||||
" 0 - Best quality\n"
|
||||
" 1 - Best performance\n"
|
||||
" --comp_mode choose the composition mode - { compMatte = 0, compLight = 1, compGreen = 2, compWhite = 3, compNone = 4, compBG = 5, compBlur = 6}\n"
|
||||
" --blur_strength change the blur strength, range is [0, 1]"
|
||||
" --comp_mode choose the composition mode - {\n"
|
||||
" 0 (show matte - compMatte),\n"
|
||||
" 1 (overlay mask on foreground - compLight),\n"
|
||||
" 2 (composite over green - compGreen),\n"
|
||||
" 3 (composite over white - compWhite),\n"
|
||||
" 4 (show input - compNone),\n"
|
||||
" 5 (composite over a specified background image - compBG),\n"
|
||||
" 6 (blur the background of the image - compBlur) }\n"
|
||||
" --cuda_graph Enable cuda graph.\n"
|
||||
);
|
||||
}
|
||||
|
||||
@@ -160,10 +169,12 @@ static int ParseMyArgs(int argc, char **argv) {
|
||||
(GetFlagArgVal("verbose", arg, &FLAG_verbose) || GetFlagArgVal("in", arg, &FLAG_inFile) ||
|
||||
GetFlagArgVal("in_file", arg, &FLAG_inFile) || GetFlagArgVal("out", arg, &FLAG_outFile) ||
|
||||
GetFlagArgVal("out_file", arg, &FLAG_outFile) || GetFlagArgVal("model_dir", arg, &FLAG_modelDir) ||
|
||||
GetFlagArgVal("bg_file", arg, &FLAG_bgFile) ||
|
||||
GetFlagArgVal("codec", arg, &FLAG_codec) || GetFlagArgVal("webcam", arg, &FLAG_webcam) ||
|
||||
GetFlagArgVal("cam_res", arg, &FLAG_camRes) || GetFlagArgVal("mode", arg, &FLAG_mode) ||
|
||||
GetFlagArgVal("progress", arg, &FLAG_progress) || GetFlagArgVal("show", arg, &FLAG_show) ||
|
||||
GetFlagArgVal("comp_mode", arg, &FLAG_compMode) || GetFlagArgVal("blur_strength", arg, &FLAG_blurStrength))) {
|
||||
GetFlagArgVal("comp_mode", arg, &FLAG_compMode) || GetFlagArgVal("blur_strength", arg, &FLAG_blurStrength) ||
|
||||
GetFlagArgVal("cuda_graph", arg, &FLAG_cudaGraph) )) {
|
||||
continue;
|
||||
} else if (GetFlagArgVal("help", arg, &help)) {
|
||||
return NVCV_ERR_HELP;
|
||||
@@ -208,6 +219,10 @@ static bool IsImageFile(const char *str) {
|
||||
return HasOneOfTheseSuffixes(str, ".bmp", ".jpg", ".jpeg", ".png", nullptr);
|
||||
}
|
||||
|
||||
static bool IsLossyImageFile(const char *str) {
|
||||
return HasOneOfTheseSuffixes(str, ".jpg", ".jpeg", nullptr);
|
||||
}
|
||||
|
||||
static const char *DurationString(double sc) {
|
||||
static char buf[16];
|
||||
int hr, mn;
|
||||
@@ -246,9 +261,9 @@ static void GetVideoInfo(cv::VideoCapture &reader, const char *fileName, VideoIn
|
||||
info->height = (int)reader.get(cv::CAP_PROP_FRAME_HEIGHT);
|
||||
info->frameRate = (double)reader.get(cv::CAP_PROP_FPS);
|
||||
if(!strcmp(fileName,"webcam"))
|
||||
info->frameCount = 0;
|
||||
info->frameCount = 0;
|
||||
else
|
||||
info->frameCount = (long long)reader.get(cv::CAP_PROP_FRAME_COUNT);
|
||||
info->frameCount = (long long)reader.get(cv::CAP_PROP_FRAME_COUNT);
|
||||
if (FLAG_verbose) PrintVideoInfo(info, fileName);
|
||||
}
|
||||
|
||||
@@ -282,8 +297,8 @@ struct FXApp {
|
||||
errLibrary = NVCV_ERR_LIBRARY,
|
||||
errInitialization = NVCV_ERR_INITIALIZATION,
|
||||
errFileNotFound = NVCV_ERR_FILE,
|
||||
errFeatureNotFound = NVCV_ERR_FEATURENOTFOUND,
|
||||
errMissingInput = NVCV_ERR_MISSINGINPUT,
|
||||
errFeatureNotFound = NVCV_ERR_FEATURENOTFOUND,
|
||||
errMissingInput = NVCV_ERR_MISSINGINPUT,
|
||||
errResolution = NVCV_ERR_RESOLUTION,
|
||||
errUnsupportedGPU = NVCV_ERR_UNSUPPORTEDGPU,
|
||||
errWrongGPU = NVCV_ERR_WRONGGPU,
|
||||
@@ -341,6 +356,8 @@ struct FXApp {
|
||||
NvVFX_Handle _eff, _bgblurEff;
|
||||
cv::Mat _srcImg;
|
||||
cv::Mat _dstImg;
|
||||
cv::Mat _bgImg;
|
||||
cv::Mat _resizedCroppedBgImg;
|
||||
NvCVImage _srcVFX;
|
||||
NvCVImage _dstVFX;
|
||||
bool _show;
|
||||
@@ -401,24 +418,26 @@ void FXApp::drawFrameRate(cv::Mat &img) {
|
||||
void FXApp::nextCompMode() {
|
||||
switch (_compMode) {
|
||||
default:
|
||||
case compBG:
|
||||
case compMatte:
|
||||
_compMode = compLight;
|
||||
break;
|
||||
case compLight:
|
||||
_compMode = compGreen;
|
||||
break;
|
||||
case compMatte:
|
||||
_compMode = compNone;
|
||||
break;
|
||||
case compGreen:
|
||||
_compMode = compWhite;
|
||||
break;
|
||||
case compWhite:
|
||||
_compMode = compMatte;
|
||||
_compMode = compNone;
|
||||
break;
|
||||
case compNone:
|
||||
_compMode = compBG;
|
||||
break;
|
||||
case compBG:
|
||||
_compMode = compBlur;
|
||||
break;
|
||||
case compBlur:
|
||||
_compMode = compLight;
|
||||
_compMode = compMatte;
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -473,7 +492,7 @@ NvCV_Status FXApp::createAigsEffect() {
|
||||
|
||||
if (!FLAG_modelDir.empty()) {
|
||||
vfxErr = NvVFX_SetString(_eff, NVVFX_MODEL_DIRECTORY, FLAG_modelDir.c_str());
|
||||
}
|
||||
}
|
||||
if (vfxErr != NVCV_SUCCESS) {
|
||||
std::cerr << "Error setting the model path to \"" << FLAG_modelDir << "\"\n";
|
||||
return vfxErr;
|
||||
@@ -493,6 +512,12 @@ NvCV_Status FXApp::createAigsEffect() {
|
||||
return vfxErr;
|
||||
}
|
||||
|
||||
vfxErr = NvVFX_SetU32(_eff, NVVFX_CUDA_GRAPH, FLAG_cudaGraph?1u:0u);
|
||||
if (vfxErr != NVCV_SUCCESS) {
|
||||
std::cerr << "Error enabling cuda graph \n";
|
||||
return vfxErr;
|
||||
}
|
||||
|
||||
vfxErr = NvVFX_CudaStreamCreate(&_stream);
|
||||
if (vfxErr != NVCV_SUCCESS) {
|
||||
std::cerr << "Error creating CUDA stream " << std::endl;
|
||||
@@ -570,6 +595,8 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
|
||||
|
||||
overlay(_srcImg, _dstImg, 0.5, result);
|
||||
if (!std::string(outFile).empty()) {
|
||||
if(IsLossyImageFile(outFile))
|
||||
fprintf(stderr, "WARNING: JPEG output file format will reduce image quality\n");
|
||||
ok = cv::imwrite(outFile, result);
|
||||
if (!ok) {
|
||||
printf("Error writing: \"%s\"\n", outFile);
|
||||
@@ -647,6 +674,30 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
unsigned int width = (unsigned int)reader.get(cv::CAP_PROP_FRAME_WIDTH);
|
||||
unsigned int height = (unsigned int)reader.get(cv::CAP_PROP_FRAME_HEIGHT);
|
||||
|
||||
if (!FLAG_bgFile.empty())
|
||||
{
|
||||
_bgImg = cv::imread(FLAG_bgFile);
|
||||
if (!_bgImg.data)
|
||||
{
|
||||
return errRead;
|
||||
}
|
||||
else
|
||||
{
|
||||
// Find the scale to resize background such that image can fit into background
|
||||
float scale = float(height) / float(_bgImg.rows);
|
||||
if ((scale * _bgImg.cols) < float(width))
|
||||
{
|
||||
scale = float(width) / float(_bgImg.cols);
|
||||
}
|
||||
cv::Mat resizedBg;
|
||||
cv::resize(_bgImg, resizedBg, cv::Size(), scale, scale, cv::INTER_AREA);
|
||||
|
||||
// Always crop from top left of background.
|
||||
cv::Rect rect(0, 0, width, height);
|
||||
_resizedCroppedBgImg = resizedBg(rect);
|
||||
}
|
||||
}
|
||||
|
||||
// allocate src for GPU
|
||||
if (!_srcNvVFXImage.pixels)
|
||||
BAIL_IF_ERR(vfxErr =
|
||||
@@ -695,6 +746,25 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
case compNone:
|
||||
_srcImg.copyTo(result);
|
||||
break;
|
||||
case compBG: {
|
||||
if (FLAG_bgFile.empty())
|
||||
{
|
||||
_resizedCroppedBgImg = cv::Mat(_srcImg.rows, _srcImg.cols, CV_8UC3, cv::Scalar(118, 185, 0));
|
||||
size_t startX = _resizedCroppedBgImg.cols/20;
|
||||
size_t offsetY = _resizedCroppedBgImg.rows/20;
|
||||
std::string text = "No Background Image!";
|
||||
for (size_t startY = offsetY; startY < _resizedCroppedBgImg.rows; startY += offsetY)
|
||||
{
|
||||
cv::putText(_resizedCroppedBgImg, text, cv::Point(startX, startY),
|
||||
cv::FONT_HERSHEY_DUPLEX, 1.0, CV_RGB(0, 0, 0), 1);
|
||||
}
|
||||
}
|
||||
NvCVImage bgVFX;
|
||||
(void)NVWrapperForCVMat(&_resizedCroppedBgImg, &bgVFX);
|
||||
NvCVImage matVFX;
|
||||
(void)NVWrapperForCVMat(&result, &matVFX);
|
||||
NvCVImage_Composite(&_srcVFX, &bgVFX, &_dstVFX, &matVFX, _stream);
|
||||
} break;
|
||||
case compLight:
|
||||
if (inFile) {
|
||||
overlay(_srcImg, _dstImg, 0.5, result);
|
||||
@@ -708,13 +778,13 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
const unsigned char bgColor[3] = {0, 255, 0};
|
||||
NvCVImage matVFX;
|
||||
(void)NVWrapperForCVMat(&result, &matVFX);
|
||||
NvCVImage_CompositeOverConstant(&_srcVFX, &_dstVFX, bgColor, &matVFX);
|
||||
NvCVImage_CompositeOverConstant(&_srcVFX, &_dstVFX, bgColor, &matVFX, _stream);
|
||||
} break;
|
||||
case compWhite: {
|
||||
const unsigned char bgColor[3] = {255, 255, 255};
|
||||
NvCVImage matVFX;
|
||||
(void)NVWrapperForCVMat(&result, &matVFX);
|
||||
NvCVImage_CompositeOverConstant(&_srcVFX, &_dstVFX, bgColor, &matVFX);
|
||||
NvCVImage_CompositeOverConstant(&_srcVFX, &_dstVFX, bgColor, &matVFX, _stream);
|
||||
} break;
|
||||
case compMatte:
|
||||
cv::cvtColor(_dstImg, result, cv::COLOR_GRAY2BGR);
|
||||
@@ -726,7 +796,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_bgblurEff, NVVFX_OUTPUT_IMAGE, &_blurNvVFXImage));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_Load(_bgblurEff));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_Run(_bgblurEff, 0));
|
||||
|
||||
|
||||
NvCVImage matVFX;
|
||||
(void)NVWrapperForCVMat(&result, &matVFX);
|
||||
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_blurNvVFXImage, &matVFX, 1.0f, _stream, NULL));
|
||||
@@ -750,12 +820,11 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
if (errQuit == appErr) break;
|
||||
}
|
||||
}
|
||||
if (_progress) {
|
||||
if(info.frameCount == 0) // no progress for a webcam
|
||||
fprintf(stderr, "\b\b\b\b???%%");
|
||||
else
|
||||
fprintf(stderr, "\b\b\b\b%3.0f%%", 100.f * frameNum / info.frameCount);
|
||||
}
|
||||
if (_progress)
|
||||
if(info.frameCount == 0) // no progress for a webcam
|
||||
fprintf(stderr, "\b\b\b\b???%%");
|
||||
else
|
||||
fprintf(stderr, "\b\b\b\b%3.0f%%", 100.f * frameNum / info.frameCount);
|
||||
}
|
||||
|
||||
if (_progress) fprintf(stderr, "\n");
|
||||
@@ -769,17 +838,44 @@ bail:
|
||||
return appErrFromVfxStatus(vfxErr);
|
||||
}
|
||||
|
||||
// This path is used by nvVideoEffectsProxy.cpp to load the SDK dll
|
||||
char *g_nvVFXSDKPath = NULL;
|
||||
|
||||
int chooseGPU() {
|
||||
// If the system has multiple supported GPUs then the application
|
||||
// should use CUDA driver APIs or CUDA runtime APIs to enumerate
|
||||
// the GPUs and select one based on the application's requirements
|
||||
|
||||
// Cuda device 0
|
||||
return 0;
|
||||
}
|
||||
|
||||
bool isCompModeEnumValid(const FXApp::CompMode& mode)
|
||||
{
|
||||
if (mode != FXApp::CompMode::compMatte &&
|
||||
mode != FXApp::CompMode::compLight &&
|
||||
mode != FXApp::CompMode::compGreen &&
|
||||
mode != FXApp::CompMode::compWhite &&
|
||||
mode != FXApp::CompMode::compNone &&
|
||||
mode != FXApp::CompMode::compBG &&
|
||||
mode != FXApp::CompMode::compBlur)
|
||||
{
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
int main(int argc, char **argv) {
|
||||
int nErrs = 0;
|
||||
FXApp::Err fxErr = FXApp::errNone;
|
||||
FXApp app;
|
||||
|
||||
nErrs = ParseMyArgs(argc, argv);
|
||||
if (nErrs) {
|
||||
Usage();
|
||||
return nErrs;
|
||||
}
|
||||
|
||||
FXApp::Err fxErr = FXApp::errNone;
|
||||
FXApp app;
|
||||
|
||||
if (FLAG_inFile.empty() && !FLAG_webcam) {
|
||||
std::cerr << "Please specify --in_file=XXX or --webcam\n";
|
||||
++nErrs;
|
||||
@@ -793,6 +889,12 @@ int main(int argc, char **argv) {
|
||||
app.setShow(FLAG_show);
|
||||
|
||||
app._compMode = static_cast<FXApp::CompMode>(FLAG_compMode);
|
||||
if (!isCompModeEnumValid(app._compMode))
|
||||
{
|
||||
std::cerr << "Please specify a valid --comp_mode=XXX, valid range is [0,6] check help section\n";
|
||||
++nErrs;
|
||||
}
|
||||
|
||||
app._blurStrength = FLAG_blurStrength;
|
||||
if (app._blurStrength < 0) {
|
||||
app._blurStrength = 0;
|
||||
@@ -823,4 +925,4 @@ int main(int argc, char **argv) {
|
||||
|
||||
if (fxErr) std::cerr << "Error: " << app.errorStringFromCode(fxErr) << std::endl;
|
||||
return (int)fxErr;
|
||||
}
|
||||
}
|
||||
Binary file not shown.
@@ -22,11 +22,10 @@ target_link_libraries(AigsEffectApp PUBLIC
|
||||
)
|
||||
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
|
||||
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "--show --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input_003054.jpg\"")
|
||||
set(PATH_STR "PATH=%PATH%" ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "--show --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input_003054.jpg\" ")
|
||||
set_target_properties(AigsEffectApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
)
|
||||
Binary file not shown.
@@ -21,6 +21,7 @@
|
||||
#
|
||||
###############################################################################*/
|
||||
|
||||
#include <stdarg.h>
|
||||
#include <stdio.h>
|
||||
#include <string.h>
|
||||
|
||||
@@ -127,8 +128,7 @@ static void Usage() {
|
||||
" where flags is:\n"
|
||||
" --out_file=<path> output image files to be written, default \"BatchOut_%%02u.png\"\n"
|
||||
" --effect=<effect> the effect to apply\n"
|
||||
" --strength=<value> strength of an effect, 0 or 1 for super res and artifact reduction,\n"
|
||||
" and [0.0, 1.0] for upscaling\n"
|
||||
" --strength=<value> strength of the upscaling effect, [0.0, 1.0]\n"
|
||||
" --scale=<scale> scale factor to be applied: 1.5, 2, 3, maybe 1.3333333\n"
|
||||
" --resolution=<height> the desired height (either --scale or --resolution may be used)\n"
|
||||
" --mode=<mode> mode 0 or 1\n"
|
||||
@@ -183,6 +183,32 @@ static int ParseMyArgs(int argc, char **argv) {
|
||||
return errs;
|
||||
}
|
||||
|
||||
static bool HasSuffix(const char *str, const char *suf) {
|
||||
size_t strSize = strlen(str),
|
||||
sufSize = strlen(suf);
|
||||
if (strSize < sufSize)
|
||||
return false;
|
||||
return (0 == strcasecmp(suf, str + strSize - sufSize));
|
||||
}
|
||||
|
||||
static bool HasOneOfTheseSuffixes(const char *str, ...) {
|
||||
bool matches = false;
|
||||
const char *suf;
|
||||
va_list ap;
|
||||
va_start(ap, str);
|
||||
while (nullptr != (suf = va_arg(ap, const char*))) {
|
||||
if (HasSuffix(str, suf)) {
|
||||
matches = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
va_end(ap);
|
||||
return matches;
|
||||
}
|
||||
|
||||
static bool IsLossyImageFile(const char *str) {
|
||||
return HasOneOfTheseSuffixes(str, ".jpg", ".jpeg", nullptr);
|
||||
}
|
||||
|
||||
class App {
|
||||
public:
|
||||
@@ -233,7 +259,7 @@ public:
|
||||
BAIL_IF_ERR(err = AllocateBatchBuffer(&_src, _batchSize, src->width, src->height, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_CUDA, 1));
|
||||
BAIL_IF_ERR(err = AllocateBatchBuffer(&_dst, _batchSize, src->width, src->height, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_CUDA, 1));
|
||||
BAIL_IF_ERR(err = NvVFX_SetString(_eff, NVVFX_MODEL_DIRECTORY, FLAG_modelDir.c_str()));
|
||||
BAIL_IF_ERR(err = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned)FLAG_strength));
|
||||
BAIL_IF_ERR(err = NvVFX_SetU32(_eff, NVVFX_MODE, FLAG_mode));
|
||||
}
|
||||
#endif // NVVFX_FX_ARTIFACT_REDUCTION
|
||||
#ifdef NVVFX_FX_SUPER_RES
|
||||
@@ -241,7 +267,7 @@ public:
|
||||
BAIL_IF_ERR(err = AllocateBatchBuffer(&_src, _batchSize, src->width, src->height, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_CUDA, 1));
|
||||
BAIL_IF_ERR(err = AllocateBatchBuffer(&_dst, _batchSize, dw, dh, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_CUDA, 1));
|
||||
BAIL_IF_ERR(err = NvVFX_SetString(_eff, NVVFX_MODEL_DIRECTORY, FLAG_modelDir.c_str()));
|
||||
BAIL_IF_ERR(err = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned)FLAG_strength));
|
||||
BAIL_IF_ERR(err = NvVFX_SetU32(_eff, NVVFX_MODE, FLAG_mode));
|
||||
}
|
||||
#endif // NVVFX_FX_SUPER_RES
|
||||
else {
|
||||
@@ -339,6 +365,8 @@ NvCV_Status BatchProcessImages(const char* effectName, const std::vector<const c
|
||||
dstHeight = app._dst.height / batchSize;
|
||||
BAIL_IF_ERR(err = NvCVImage_Alloc(&nvx, app._dst.width, dstHeight, ((app._dst.numComponents == 1) ? NVCV_Y : NVCV_BGR), NVCV_U8, NVCV_CHUNKY, NVCV_CPU, 0));
|
||||
CVWrapperForNvCVImage(&nvx, &ocv);
|
||||
if(IsLossyImageFile(outfilePattern))
|
||||
fprintf(stderr, "WARNING: JPEG output file format will reduce image quality\n");
|
||||
for (i = 0; i < batchSize; ++i) {
|
||||
char fileName[1024];
|
||||
snprintf(fileName, sizeof(fileName), outfilePattern, i);
|
||||
|
||||
Binary file not shown.
@@ -4,6 +4,7 @@ set(SOURCE_FILES
|
||||
../../nvvfx/src/nvVideoEffectsProxy.cpp
|
||||
../../nvvfx/src/nvCVImageProxy.cpp)
|
||||
|
||||
|
||||
# Set Visual Studio source filters
|
||||
source_group("Source Files" FILES ${SOURCE_FILES})
|
||||
|
||||
@@ -16,32 +17,20 @@ target_include_directories(BatchEffectApp PUBLIC
|
||||
${SDK_INCLUDES_PATH}
|
||||
)
|
||||
|
||||
if(MSVC)
|
||||
target_link_libraries(BatchEffectApp PUBLIC
|
||||
opencv346
|
||||
NVVideoEffects
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
|
||||
)
|
||||
target_link_libraries(BatchEffectApp PUBLIC
|
||||
opencv346
|
||||
NVVideoEffects
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
|
||||
)
|
||||
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
|
||||
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "--show --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input_003054.jpg\"")
|
||||
set_target_properties(BatchEffectApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
else()
|
||||
|
||||
target_link_libraries(BatchEffectApp PUBLIC
|
||||
NVVideoEffects
|
||||
NVCVImage
|
||||
OpenCV
|
||||
TensorRT
|
||||
CUDA
|
||||
)
|
||||
endif()
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(PATH_STR "PATH=%PATH%" ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "--show --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input_003054.jpg\" ")
|
||||
set_target_properties(BatchEffectApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
|
||||
#Batch denoise effect
|
||||
set(SOURCE_FILES
|
||||
@@ -62,30 +51,19 @@ target_include_directories(BatchDenoiseEffectApp PUBLIC
|
||||
${SDK_INCLUDES_PATH}
|
||||
)
|
||||
|
||||
if(MSVC)
|
||||
target_link_libraries(BatchDenoiseEffectApp PUBLIC
|
||||
opencv346
|
||||
NVVideoEffects
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
|
||||
)
|
||||
target_include_directories(BatchDenoiseEffectApp PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/include)
|
||||
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
|
||||
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "video1.mp4 video2.mp4")
|
||||
set_target_properties(BatchDenoiseEffectApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
else()
|
||||
|
||||
target_link_libraries(BatchDenoiseEffectApp PUBLIC
|
||||
NVVideoEffects
|
||||
NVCVImage
|
||||
OpenCV
|
||||
TensorRT
|
||||
CUDA
|
||||
)
|
||||
endif()
|
||||
target_link_libraries(BatchDenoiseEffectApp PUBLIC
|
||||
opencv346
|
||||
NVVideoEffects
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
|
||||
)
|
||||
target_include_directories(BatchDenoiseEffectApp PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/include)
|
||||
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(PATH_STR "PATH=%PATH%" ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "video1.mp4 video2.mp4 ")
|
||||
set_target_properties(BatchDenoiseEffectApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
|
||||
@@ -7,29 +7,20 @@ add_executable(DenoiseEffectApp ${SOURCE_FILES})
|
||||
target_include_directories(DenoiseEffectApp PRIVATE ${CMAKE_CURRENT_SOURCE_DIR} ${CMAKE_CURRENT_SOURCE_DIR}/../utils)
|
||||
target_include_directories(DenoiseEffectApp PUBLIC ${SDK_INCLUDES_PATH})
|
||||
|
||||
if(MSVC)
|
||||
target_link_libraries(DenoiseEffectApp PUBLIC
|
||||
opencv346
|
||||
NVVideoEffects
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
|
||||
)
|
||||
target_include_directories(DenoiseEffectApp PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/include)
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
|
||||
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "--model_dir=\"${CMAKE_CURRENT_SOURCE_DIR}/../../bin/models\" --show --webcam")
|
||||
set_target_properties(DenoiseEffectApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
else()
|
||||
|
||||
target_link_libraries(DenoiseEffectApp PUBLIC
|
||||
NVVideoEffects
|
||||
NVCVImage
|
||||
OpenCV
|
||||
TensorRT
|
||||
CUDA
|
||||
)
|
||||
endif()
|
||||
target_link_libraries(DenoiseEffectApp PUBLIC
|
||||
opencv346
|
||||
NVVideoEffects
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
|
||||
)
|
||||
target_include_directories(DenoiseEffectApp PUBLIC ${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/include)
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
|
||||
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "--show --webcam")
|
||||
set_target_properties(DenoiseEffectApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
|
||||
|
||||
@@ -220,6 +220,10 @@ static bool IsImageFile(const char *str) {
|
||||
return HasOneOfTheseSuffixes(str, ".bmp", ".jpg", ".jpeg", ".png", nullptr);
|
||||
}
|
||||
|
||||
static bool IsLossyImageFile(const char *str) {
|
||||
return HasOneOfTheseSuffixes(str, ".jpg", ".jpeg", nullptr);
|
||||
}
|
||||
|
||||
static const char* DurationString(double sc) {
|
||||
static char buf[16];
|
||||
int hr, mn;
|
||||
@@ -388,7 +392,7 @@ FXApp::Err FXApp::processKey(int key) {
|
||||
case 'p': case 'P': case '%':
|
||||
_progress = !_progress;
|
||||
case 'e': case 'E':
|
||||
_enableEffect = !_enableEffect;
|
||||
_enableEffect = !_enableEffect;
|
||||
break;
|
||||
case 'd': case'D':
|
||||
if (FLAG_webcam)
|
||||
@@ -474,10 +478,10 @@ NvCV_Status FXApp::allocBuffers(unsigned width, unsigned height) {
|
||||
_srcImg.create(height, width, CV_8UC3); // src CPU
|
||||
BAIL_IF_NULL(_srcImg.data, vfxErr, NVCV_ERR_MEMORY);
|
||||
}
|
||||
|
||||
|
||||
_dstImg.create(_srcImg.rows, _srcImg.cols, _srcImg.type()); //
|
||||
BAIL_IF_NULL(_dstImg.data, vfxErr, NVCV_ERR_MEMORY); //
|
||||
BAIL_IF_ERR(vfxErr = NvCVImage_Alloc(&_srcGpuBuf, _srcImg.cols, _srcImg.rows, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_GPU, 1)); // src GPU
|
||||
BAIL_IF_ERR(vfxErr = NvCVImage_Alloc(&_srcGpuBuf, _srcImg.cols, _srcImg.rows, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_GPU, 1)); // src GPU
|
||||
BAIL_IF_ERR(vfxErr = NvCVImage_Alloc(&_dstGpuBuf, _srcImg.cols, _srcImg.rows, NVCV_BGR, NVCV_F32, NVCV_PLANAR, NVCV_GPU, 1)); //dst GPU
|
||||
|
||||
NVWrapperForCVMat(&_srcImg, &_srcVFX); // _srcVFX is an alias for _srcImg
|
||||
@@ -510,9 +514,9 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
|
||||
BAIL_IF_ERR(vfxErr = allocBuffers(_srcImg.cols, _srcImg.rows));
|
||||
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_srcVFX, &_srcGpuBuf, 1.f / 255.f, stream, &_tmpVFX)); // _srcVFX--> _tmpVFX --> _srcGpuBuf
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_INPUT_IMAGE, &_srcGpuBuf));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetF32(_eff, NVVFX_STRENGTH, FLAG_strength));
|
||||
|
||||
|
||||
unsigned int stateSizeInBytes;
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_GetU32(_eff, NVVFX_STATE_SIZE, &stateSizeInBytes));
|
||||
cudaMalloc(&state, stateSizeInBytes);
|
||||
@@ -525,6 +529,8 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
|
||||
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_dstGpuBuf, &_dstVFX, 255.f, stream, &_tmpVFX));
|
||||
|
||||
if (outFile && outFile[0]) {
|
||||
if(IsLossyImageFile(outFile))
|
||||
fprintf(stderr, "WARNING: JPEG output file format will reduce image quality\n");
|
||||
if (!cv::imwrite(outFile, _dstImg)) {
|
||||
printf("Error writing: \"%s\"\n", outFile);
|
||||
return errWrite;
|
||||
@@ -552,7 +558,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
|
||||
void* state = nullptr;
|
||||
void* stateArray[1];
|
||||
|
||||
|
||||
if (inFile && !inFile[0]) inFile = nullptr; // Set file paths to NULL if zero length
|
||||
|
||||
if (!FLAG_webcam && inFile) {
|
||||
@@ -574,7 +580,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
printf("Filters only target H264 videos, not %.4s\n", (char*)&info.codec);
|
||||
|
||||
BAIL_IF_ERR(vfxErr = allocBuffers(info.width, info.height));
|
||||
|
||||
|
||||
if (outFile && !outFile[0]) outFile = nullptr;
|
||||
if (outFile) {
|
||||
ok = writer.open(outFile, StringToFourcc(FLAG_codec), info.frameRate, cv::Size(_dstVFX.width, _dstVFX.height));
|
||||
@@ -586,7 +592,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
}
|
||||
}
|
||||
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_INPUT_IMAGE, &_srcGpuBuf));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_INPUT_IMAGE, &_srcGpuBuf));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetF32(_eff, NVVFX_STRENGTH, FLAG_strength));
|
||||
|
||||
@@ -598,7 +604,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetObject(_eff, NVVFX_STATE, (void*)stateArray));
|
||||
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_Load(_eff));
|
||||
|
||||
|
||||
for (frameNum = 0; reader.read(_srcImg); frameNum++) {
|
||||
if (_enableEffect) {
|
||||
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_srcVFX, &_srcGpuBuf, 1.f / 255.f, stream, &_tmpVFX));
|
||||
@@ -613,7 +619,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
writer.write(_dstImg);
|
||||
|
||||
if (_show) {
|
||||
if (_drawVisualization) drawEffectStatus(_dstImg);
|
||||
if (_drawVisualization) drawEffectStatus(_dstImg);
|
||||
drawFrameRate(_dstImg);
|
||||
cv::imshow("Output", _dstImg);
|
||||
int key= cv::waitKey(1);
|
||||
@@ -630,7 +636,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
if (_progress) fprintf(stderr, "\n");
|
||||
reader.release();
|
||||
if (outFile)
|
||||
writer.release();
|
||||
writer.release();
|
||||
bail:
|
||||
if (state) cudaFree(state); // release state memory
|
||||
return appErrFromVfxStatus(vfxErr);
|
||||
|
||||
Binary file not shown.
@@ -12,29 +12,19 @@ target_include_directories(UpscalePipelineApp PUBLIC
|
||||
${SDK_INCLUDES_PATH}
|
||||
)
|
||||
|
||||
if(MSVC)
|
||||
target_link_libraries(UpscalePipelineApp PUBLIC
|
||||
opencv346
|
||||
NVVideoEffects
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
|
||||
)
|
||||
target_link_libraries(UpscalePipelineApp PUBLIC
|
||||
opencv346
|
||||
NVVideoEffects
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
|
||||
)
|
||||
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
|
||||
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "--model_dir=\"${CMAKE_CURRENT_SOURCE_DIR}/../../bin/models\" --show --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input1.jpg\"")
|
||||
set_target_properties(UpscalePipelineApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
else()
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
|
||||
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "--show --resolution=1440 --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input1.jpg\"")
|
||||
set_target_properties(UpscalePipelineApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
|
||||
target_link_libraries(UpscalePipelineApp PUBLIC
|
||||
NVVideoEffects
|
||||
NVCVImage
|
||||
OpenCV
|
||||
TensorRT
|
||||
CUDA
|
||||
)
|
||||
endif()
|
||||
|
||||
@@ -66,7 +66,7 @@ bool FLAG_debug = false,
|
||||
FLAG_show = false,
|
||||
FLAG_progress = false;
|
||||
int FLAG_resolution = 0,
|
||||
FLAG_arStrength = 0;
|
||||
FLAG_arMode = 0;
|
||||
float FLAG_upscaleStrength = 0.2f;
|
||||
std::string FLAG_codec = DEFAULT_CODEC,
|
||||
FLAG_inFile,
|
||||
@@ -74,6 +74,11 @@ std::string FLAG_codec = DEFAULT_CODEC,
|
||||
FLAG_outDir,
|
||||
FLAG_modelDir;
|
||||
|
||||
// Set this when using OTA Updates
|
||||
// This path is used by nvVideoEffectsProxy.cpp to load the SDK dll
|
||||
// when using OTA Updates
|
||||
char *g_nvVFXSDKPath = NULL;
|
||||
|
||||
static bool GetFlagArgVal(const char *flag, const char *arg, const char **val) {
|
||||
if (*arg != '-')
|
||||
return false;
|
||||
@@ -146,7 +151,7 @@ static void Usage() {
|
||||
" --in_file=<path> input file to be processed\n"
|
||||
" --out_file=<path> output file to be written\n"
|
||||
" --show display the results in a window\n"
|
||||
" --ar_strength=(0|1) strength of artifact reduction filter (0: conservative, 1: aggressive, default 0)\n"
|
||||
" --ar_mode=(0|1) mode of artifact reduction filter (0: conservative, 1: aggressive, default 0)\n"
|
||||
" --upscale_strength=(0 to 1) strength of upscale filter (float value between 0 to 1)\n"
|
||||
" --resolution=<height> the desired height of the output\n"
|
||||
" --out_height=<height> the desired height of the output\n"
|
||||
@@ -172,7 +177,7 @@ static int ParseMyArgs(int argc, char **argv) {
|
||||
GetFlagArgVal("out", arg, &FLAG_outFile) ||
|
||||
GetFlagArgVal("out_file", arg, &FLAG_outFile) ||
|
||||
GetFlagArgVal("show", arg, &FLAG_show) ||
|
||||
GetFlagArgVal("ar_strength", arg, &FLAG_arStrength) ||
|
||||
GetFlagArgVal("ar_mode", arg, &FLAG_arMode) ||
|
||||
GetFlagArgVal("upscale_strength", arg, &FLAG_upscaleStrength) ||
|
||||
GetFlagArgVal("resolution", arg, &FLAG_resolution) ||
|
||||
GetFlagArgVal("model_dir", arg, &FLAG_modelDir) ||
|
||||
@@ -226,6 +231,10 @@ static bool IsImageFile(const char *str) {
|
||||
return HasOneOfTheseSuffixes(str, ".bmp", ".jpg", ".jpeg", ".png", nullptr);
|
||||
}
|
||||
|
||||
static bool IsLossyImageFile(const char *str) {
|
||||
return HasOneOfTheseSuffixes(str, ".jpg", ".jpeg", nullptr);
|
||||
}
|
||||
|
||||
static const char* DurationString(double sc) {
|
||||
static char buf[16];
|
||||
int hr, mn;
|
||||
@@ -488,7 +497,7 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_arEff, NVVFX_INPUT_IMAGE, &_srcGpuBuf));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_arEff, NVVFX_OUTPUT_IMAGE, &_interGpuBGRf32pl));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetCudaStream(_arEff, NVVFX_CUDA_STREAM, stream));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_arEff, NVVFX_STRENGTH, FLAG_arStrength));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_arEff, NVVFX_MODE, FLAG_arMode));
|
||||
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_upscaleEff, NVVFX_INPUT_IMAGE, &_interGpuRGBAu8));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_upscaleEff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
|
||||
@@ -503,6 +512,8 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
|
||||
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_dstGpuBuf, &_dstVFX, 1.f, stream, &_tmpVFX)); // _dstGpuBuf --> _dstTmpVFX --> _dstVFX
|
||||
|
||||
if (outFile && outFile[0]) {
|
||||
if(IsLossyImageFile(outFile))
|
||||
fprintf(stderr, "WARNING: JPEG output file format will reduce image quality\n");
|
||||
if (!cv::imwrite(outFile, _dstImg)) {
|
||||
printf("Error writing: \"%s\"\n", outFile);
|
||||
return errWrite;
|
||||
@@ -552,7 +563,7 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_arEff, NVVFX_INPUT_IMAGE, &_srcGpuBuf));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_arEff, NVVFX_OUTPUT_IMAGE, &_interGpuBGRf32pl));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetCudaStream(_arEff, NVVFX_CUDA_STREAM, stream));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_arEff, NVVFX_STRENGTH, FLAG_arStrength));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_arEff, NVVFX_MODE, FLAG_arMode));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_Load(_arEff));
|
||||
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_upscaleEff, NVVFX_INPUT_IMAGE, &_interGpuRGBAu8));
|
||||
|
||||
Binary file not shown.
@@ -1,6 +1,6 @@
|
||||
SETLOCAL
|
||||
SET PATH=%PATH%;..\external\opencv\bin;
|
||||
REM Use --show to show the output in a window or use --out_file=<filename> to write output to file
|
||||
UpscalePipelineApp.exe --in_file=..\input\input1.jpg --ar_strength=0 --upscale_strength=0 --resolution=1080 --show --out_file=ar_sr_0.png
|
||||
UpscalePipelineApp.exe --in_file=..\input\input1.jpg --ar_strength=0 --upscale_strength=1 --resolution=1080 --show --out_file=ar_sr_1.png
|
||||
UpscalePipelineApp.exe --in_file=..\input\input1.jpg --ar_mode=0 --upscale_strength=0 --resolution=1080 --show --out_file=ar_sr_0.png
|
||||
UpscalePipelineApp.exe --in_file=..\input\input1.jpg --ar_mode=0 --upscale_strength=1 --resolution=1080 --show --out_file=ar_sr_1.png
|
||||
|
||||
|
||||
@@ -7,29 +7,19 @@ add_executable(VideoEffectsApp ${SOURCE_FILES})
|
||||
target_include_directories(VideoEffectsApp PRIVATE ${CMAKE_CURRENT_SOURCE_DIR} ${CMAKE_CURRENT_SOURCE_DIR}/../utils)
|
||||
target_include_directories(VideoEffectsApp PUBLIC ${SDK_INCLUDES_PATH})
|
||||
|
||||
if(MSVC)
|
||||
target_link_libraries(VideoEffectsApp PUBLIC
|
||||
opencv346
|
||||
NVVideoEffects
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
|
||||
)
|
||||
target_link_libraries(VideoEffectsApp PUBLIC
|
||||
opencv346
|
||||
NVVideoEffects
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/../external/cuda/lib/x64/cudart.lib
|
||||
)
|
||||
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
|
||||
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "--model_dir=\"${CMAKE_CURRENT_SOURCE_DIR}/../../bin/models\" --show --effect=SuperRes --resolution=1080 --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input2.png\"")
|
||||
set_target_properties(VideoEffectsApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
else()
|
||||
set(OPENCV_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../external/opencv/bin)
|
||||
set(VFXSDK_PATH_STR ${CMAKE_CURRENT_SOURCE_DIR}/../../bin) # Also the location for CUDA/NVTRT/libcrypto
|
||||
set(PATH_STR "PATH=%PATH%" ${VFXSDK_PATH_STR} ${OPENCV_PATH_STR})
|
||||
set(CMD_ARG_STR "--show --effect=SuperRes --resolution=1080 --in_file=\"${CMAKE_CURRENT_SOURCE_DIR}/../input/input1.jpg\"")
|
||||
set_target_properties(VideoEffectsApp PROPERTIES
|
||||
FOLDER SampleApps
|
||||
VS_DEBUGGER_ENVIRONMENT "${PATH_STR}"
|
||||
VS_DEBUGGER_COMMAND_ARGUMENTS "${CMD_ARG_STR}"
|
||||
)
|
||||
|
||||
target_link_libraries(VideoEffectsApp PUBLIC
|
||||
NVVideoEffects
|
||||
NVCVImage
|
||||
OpenCV
|
||||
TensorRT
|
||||
CUDA
|
||||
)
|
||||
endif()
|
||||
|
||||
@@ -57,6 +57,7 @@ bool FLAG_debug = false,
|
||||
FLAG_progress = false,
|
||||
FLAG_webcam = false;
|
||||
float FLAG_strength = 0.f;
|
||||
int FLAG_mode = 0;
|
||||
int FLAG_resolution = 0;
|
||||
std::string FLAG_codec = DEFAULT_CODEC,
|
||||
FLAG_camRes = "1280x720",
|
||||
@@ -66,6 +67,11 @@ std::string FLAG_codec = DEFAULT_CODEC,
|
||||
FLAG_modelDir,
|
||||
FLAG_effect;
|
||||
|
||||
// Set this when using OTA Updates
|
||||
// This path is used by nvVideoEffectsProxy.cpp to load the SDK dll
|
||||
// when using OTA Updates
|
||||
char *g_nvVFXSDKPath = NULL;
|
||||
|
||||
static bool GetFlagArgVal(const char *flag, const char *arg, const char **val) {
|
||||
if (*arg != '-')
|
||||
return false;
|
||||
@@ -140,8 +146,9 @@ static void Usage() {
|
||||
" --out_file=<path> output file to be written\n"
|
||||
" --effect=<effect> the effect to apply\n"
|
||||
" --show display the results in a window (for webcam, it is always true)\n"
|
||||
" --strength=<value> strength of an effect, 0 or 1 for super res and artifact reduction,\n"
|
||||
" and [0.0, 1.0] for upscaling\n"
|
||||
" --strength=<value> strength of the upscaling effect, [0.0, 1.0]\n"
|
||||
" --mode=<value> mode of the super res or artifact reduction effect, 0 or 1, \n"
|
||||
" where 0 - conservative and 1 - aggressive\n"
|
||||
" --cam_res=[WWWx]HHH specify camera resolution as height or width x height\n"
|
||||
" supports 720 and 1080 resolutions (default \"720\") \n"
|
||||
" --resolution=<height> the desired height of the output\n"
|
||||
@@ -176,6 +183,7 @@ static int ParseMyArgs(int argc, char **argv) {
|
||||
GetFlagArgVal("webcam", arg, &FLAG_webcam) ||
|
||||
GetFlagArgVal("cam_res", arg, &FLAG_camRes) ||
|
||||
GetFlagArgVal("strength", arg, &FLAG_strength) ||
|
||||
GetFlagArgVal("mode", arg, &FLAG_mode) ||
|
||||
GetFlagArgVal("resolution", arg, &FLAG_resolution) ||
|
||||
GetFlagArgVal("model_dir", arg, &FLAG_modelDir) ||
|
||||
GetFlagArgVal("codec", arg, &FLAG_codec) ||
|
||||
@@ -228,6 +236,11 @@ static bool IsImageFile(const char *str) {
|
||||
return HasOneOfTheseSuffixes(str, ".bmp", ".jpg", ".jpeg", ".png", nullptr);
|
||||
}
|
||||
|
||||
static bool IsLossyImageFile(const char *str) {
|
||||
return HasOneOfTheseSuffixes(str, ".jpg", ".jpeg", nullptr);
|
||||
}
|
||||
|
||||
|
||||
static const char* DurationString(double sc) {
|
||||
static char buf[16];
|
||||
int hr, mn;
|
||||
@@ -561,9 +574,9 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetCudaStream(_eff, NVVFX_CUDA_STREAM, stream));
|
||||
if (!strcmp(_effectName, NVVFX_FX_ARTIFACT_REDUCTION)) {
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned int)FLAG_strength));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_MODE, (unsigned int)FLAG_mode));
|
||||
} else if (!strcmp(_effectName, NVVFX_FX_SUPER_RES)) {
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned int)FLAG_strength));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_MODE, (unsigned int)FLAG_mode));
|
||||
}
|
||||
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_Load(_eff));
|
||||
@@ -571,6 +584,8 @@ FXApp::Err FXApp::processImage(const char *inFile, const char *outFile) {
|
||||
BAIL_IF_ERR(vfxErr = NvCVImage_Transfer(&_dstGpuBuf, &_dstVFX, 255.f, stream, &_tmpVFX)); // _dstGpuBuf --> _tmpVFX --> _dstVFX
|
||||
|
||||
if (outFile && outFile[0]) {
|
||||
if(IsLossyImageFile(outFile))
|
||||
fprintf(stderr, "WARNING: JPEG output file format will reduce image quality\n");
|
||||
if (!cv::imwrite(outFile, _dstImg)) {
|
||||
printf("Error writing: \"%s\"\n", outFile);
|
||||
return errWrite;
|
||||
@@ -632,9 +647,9 @@ FXApp::Err FXApp::processMovie(const char *inFile, const char *outFile) {
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetImage(_eff, NVVFX_OUTPUT_IMAGE, &_dstGpuBuf));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetCudaStream(_eff, NVVFX_CUDA_STREAM, stream));
|
||||
if (!strcmp(_effectName, NVVFX_FX_ARTIFACT_REDUCTION)) {
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned int)FLAG_strength));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_MODE, (unsigned int)FLAG_mode));
|
||||
} else if (!strcmp(_effectName, NVVFX_FX_SUPER_RES)) {
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_STRENGTH, (unsigned int)FLAG_strength));
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_SetU32(_eff, NVVFX_MODE, (unsigned int)FLAG_mode));
|
||||
}
|
||||
BAIL_IF_ERR(vfxErr = NvVFX_Load(_eff));
|
||||
|
||||
|
||||
Binary file not shown.
@@ -1,7 +1,7 @@
|
||||
SETLOCAL
|
||||
SET PATH=%PATH%;..\external\opencv\bin;
|
||||
REM Use --show to show the output in a window or use --out_file=<filename> to write output to file
|
||||
VideoEffectsApp.exe --in_file=..\input\input1.jpg --out_file=ar_1.png --effect=ArtifactReduction --strength=1 --show
|
||||
VideoEffectsApp.exe --in_file=..\input\input1.jpg --out_file=ar_0.png --effect=ArtifactReduction --strength=0 --show
|
||||
VideoEffectsApp.exe --in_file=..\input\input2.jpg --out_file=sr_0.png --effect=SuperRes --resolution=2160 --strength=0 --show
|
||||
VideoEffectsApp.exe --in_file=..\input\input2.jpg --out_file=sr_1.png --effect=SuperRes --resolution=2160 --strength=1 --show
|
||||
VideoEffectsApp.exe --in_file=..\input\input1.jpg --out_file=ar_1.png --effect=ArtifactReduction --mode=1 --show
|
||||
VideoEffectsApp.exe --in_file=..\input\input1.jpg --out_file=ar_0.png --effect=ArtifactReduction --mode=0 --show
|
||||
VideoEffectsApp.exe --in_file=..\input\input2.jpg --out_file=sr_0.png --effect=SuperRes --resolution=2160 --mode=0 --show
|
||||
VideoEffectsApp.exe --in_file=..\input\input2.jpg --out_file=sr_1.png --effect=SuperRes --resolution=2160 --mode=1 --show
|
||||
4
samples/external/CMakeLists.txt
vendored
4
samples/external/CMakeLists.txt
vendored
@@ -40,7 +40,7 @@ else()
|
||||
message("OpenCV_LIBRARIES ${OpenCV_LIBRARIES}")
|
||||
message("OpenCV_LIBS ${OpenCV_LIBS}")
|
||||
|
||||
find_package(CUDA 11.1 REQUIRED)
|
||||
find_package(CUDA 11.3 REQUIRED)
|
||||
add_library(CUDA INTERFACE)
|
||||
target_include_directories(CUDA INTERFACE ${CUDA_INCLUDE_DIRS})
|
||||
target_link_libraries(CUDA INTERFACE "${CUDA_LIBRARIES};cuda")
|
||||
@@ -48,7 +48,7 @@ else()
|
||||
message("CUDA_INCLUDE_DIRS ${CUDA_INCLUDE_DIRS}")
|
||||
message("CUDA_LIBRARIES ${CUDA_LIBRARIES}")
|
||||
|
||||
find_package(TensorRT 7 REQUIRED)
|
||||
find_package(TensorRT 8 REQUIRED)
|
||||
add_library(TensorRT INTERFACE)
|
||||
target_include_directories(TensorRT INTERFACE ${TensorRT_INCLUDE_DIRS})
|
||||
target_link_libraries(TensorRT INTERFACE ${TensorRT_LIBRARIES})
|
||||
|
||||
@@ -86,7 +86,7 @@
|
||||
* \endcode
|
||||
*
|
||||
* where ::cudaChannelFormatKind is one of ::cudaChannelFormatKindSigned,
|
||||
* ::cudaChannelFormatKindUnsigned, or ::cudaChannelFormatKindFloat.
|
||||
* ::cudaChannelFormatKindUnsigned, cudaChannelFormatKindFloat or ::cudaChannelFormatKindNV12.
|
||||
*
|
||||
* \return
|
||||
* Channel descriptor with format \p f
|
||||
@@ -401,6 +401,12 @@ template<> __inline__ __host__ cudaChannelFormatDesc cudaCreateChannelDesc<float
|
||||
return cudaCreateChannelDesc(e, e, e, e, cudaChannelFormatKindFloat);
|
||||
}
|
||||
|
||||
static __inline__ __host__ cudaChannelFormatDesc cudaCreateChannelDescNV12(void)
|
||||
{
|
||||
int e = (int)sizeof(char) * 8;
|
||||
|
||||
return cudaCreateChannelDesc(e, e, e, 0, cudaChannelFormatKindNV12);
|
||||
}
|
||||
#endif /* __cplusplus */
|
||||
|
||||
/** @} */
|
||||
|
||||
20
samples/external/cuda/include/crt/host_config.h
vendored
20
samples/external/cuda/include/crt/host_config.h
vendored
@@ -114,8 +114,8 @@
|
||||
#endif /* __ICC */
|
||||
|
||||
#if defined(__PGIC__)
|
||||
#if ((__PGIC__ != 18) && (__PGIC__ != 19) && (__PGIC__ != 20) && !(__PGIC__ == 99 && __PGIC_MINOR__ == 99))
|
||||
#error -- unsupported pgc++ configuration! Only pgc++ 18, 19 and 20 are supported! The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
|
||||
#if ((__PGIC__ != 18) && (__PGIC__ != 19) && (__PGIC__ != 20) && (__PGIC__ != 21) && !(__PGIC__ == 99 && __PGIC_MINOR__ == 99))
|
||||
#error -- unsupported pgc++ configuration! Only pgc++ 18, 19, 20 and 21 are supported! The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
|
||||
|
||||
#endif
|
||||
#endif /* __PGIC__ */
|
||||
@@ -143,10 +143,10 @@
|
||||
|
||||
#if defined(__clang__) && !defined(__ibmxl_vrm__) && !defined(__ICC) && !defined(__HORIZON__) && !defined(__APPLE__)
|
||||
|
||||
#if (__clang_major__ >= 11) || (__clang_major__ < 3) || ((__clang_major__ == 3) && (__clang_minor__ < 3))
|
||||
#error -- unsupported clang version! clang version must be less than 11 and greater than 3.2 . The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
|
||||
#if (__clang_major__ >= 12) || (__clang_major__ < 3) || ((__clang_major__ == 3) && (__clang_minor__ < 3))
|
||||
#error -- unsupported clang version! clang version must be less than 12 and greater than 3.2 . The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
|
||||
|
||||
#endif /* (__clang_major__ >= 11) || (__clang_major__ < 3) || ((__clang_major__ == 3) && (__clang_minor__ < 3)) */
|
||||
#endif /* (__clang_major__ >= 12) || (__clang_major__ < 3) || ((__clang_major__ == 3) && (__clang_minor__ < 3)) */
|
||||
|
||||
#endif /* defined(__clang__) && !defined(__ibmxl_vrm__) && !defined(__ICC) && !defined(__HORIZON__) && !defined(__APPLE__) */
|
||||
|
||||
@@ -155,15 +155,15 @@
|
||||
|
||||
#if defined(_WIN32)
|
||||
|
||||
#if _MSC_VER < 1700 || _MSC_VER >= 1930
|
||||
#if _MSC_VER < 1910 || _MSC_VER >= 1930
|
||||
|
||||
#error -- unsupported Microsoft Visual Studio version! Only the versions between 2015 and 2019 (inclusive) are supported! The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
|
||||
#error -- unsupported Microsoft Visual Studio version! Only the versions between 2017 and 2019 (inclusive) are supported! The nvcc flag '-allow-unsupported-compiler' can be used to override this version check; however, using an unsupported host compiler may cause compilation failure or incorrect run time execution. Use at your own risk.
|
||||
|
||||
#elif _MSC_VER >= 1700 && _MSC_VER < 1900
|
||||
#elif _MSC_VER >= 1910 && _MSC_VER < 1910
|
||||
|
||||
#pragma message("support for this version of Microsoft Visual Studio has been deprecated! Only the versions between 2015 and 2019 (inclusive) are supported!")
|
||||
#pragma message("support for this version of Microsoft Visual Studio has been deprecated! Only the versions between 2017 and 2019 (inclusive) are supported!")
|
||||
|
||||
#endif /* (_MSC_VER < 1700 || _MSC_VER >= 1930) || (_MSC_VER >= 1700 && _MSC_VER < 1900) */
|
||||
#endif /* (_MSC_VER < 1910 || _MSC_VER >= 1930) || (_MSC_VER >= 1910 && _MSC_VER < 1910) */
|
||||
|
||||
#endif /* _WIN32 */
|
||||
#endif /* !__NV_NO_HOST_COMPILER_CHECK */
|
||||
|
||||
@@ -97,6 +97,7 @@
|
||||
#define __location__(a) \
|
||||
__annotate__(a)
|
||||
#define CUDARTAPI
|
||||
#define CUDARTAPI_CDECL
|
||||
|
||||
#elif defined(_MSC_VER)
|
||||
|
||||
@@ -133,6 +134,8 @@
|
||||
__annotate__(__##a##__)
|
||||
#define CUDARTAPI \
|
||||
__stdcall
|
||||
#define CUDARTAPI_CDECL \
|
||||
__cdecl
|
||||
|
||||
#else /* __GNUC__ || __CUDA_LIBDEVICE__ || __CUDACC_RTC__ */
|
||||
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
/*
|
||||
* Copyright 1993-2018 NVIDIA Corporation. All rights reserved.
|
||||
* Copyright 1993-2021 NVIDIA Corporation. All rights reserved.
|
||||
*
|
||||
* NOTICE TO LICENSEE:
|
||||
*
|
||||
@@ -66,43 +66,37 @@ extern "C" {
|
||||
|
||||
struct cudaFuncAttributes;
|
||||
|
||||
#if defined(_WIN32)
|
||||
#define __NV_WEAK__ __declspec(nv_weak)
|
||||
#else
|
||||
#define __NV_WEAK__ __attribute__((nv_weak))
|
||||
#endif
|
||||
|
||||
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaMalloc(void **p, size_t s)
|
||||
inline __device__ cudaError_t CUDARTAPI cudaMalloc(void **p, size_t s)
|
||||
{
|
||||
return cudaErrorUnknown;
|
||||
}
|
||||
|
||||
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaFuncGetAttributes(struct cudaFuncAttributes *p, const void *c)
|
||||
inline __device__ cudaError_t CUDARTAPI cudaFuncGetAttributes(struct cudaFuncAttributes *p, const void *c)
|
||||
{
|
||||
return cudaErrorUnknown;
|
||||
}
|
||||
|
||||
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaDeviceGetAttribute(int *value, enum cudaDeviceAttr attr, int device)
|
||||
inline __device__ cudaError_t CUDARTAPI cudaDeviceGetAttribute(int *value, enum cudaDeviceAttr attr, int device)
|
||||
{
|
||||
return cudaErrorUnknown;
|
||||
}
|
||||
|
||||
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaGetDevice(int *device)
|
||||
inline __device__ cudaError_t CUDARTAPI cudaGetDevice(int *device)
|
||||
{
|
||||
return cudaErrorUnknown;
|
||||
}
|
||||
|
||||
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaOccupancyMaxActiveBlocksPerMultiprocessor(int *numBlocks, const void *func, int blockSize, size_t dynamicSmemSize)
|
||||
inline __device__ cudaError_t CUDARTAPI cudaOccupancyMaxActiveBlocksPerMultiprocessor(int *numBlocks, const void *func, int blockSize, size_t dynamicSmemSize)
|
||||
{
|
||||
return cudaErrorUnknown;
|
||||
}
|
||||
|
||||
__device__ __NV_WEAK__ cudaError_t CUDARTAPI cudaOccupancyMaxActiveBlocksPerMultiprocessorWithFlags(int *numBlocks, const void *func, int blockSize, size_t dynamicSmemSize, unsigned int flags)
|
||||
inline __device__ cudaError_t CUDARTAPI cudaOccupancyMaxActiveBlocksPerMultiprocessorWithFlags(int *numBlocks, const void *func, int blockSize, size_t dynamicSmemSize, unsigned int flags)
|
||||
{
|
||||
return cudaErrorUnknown;
|
||||
}
|
||||
|
||||
#undef __NV_WEAK__
|
||||
|
||||
#if defined(__cplusplus)
|
||||
}
|
||||
|
||||
107
samples/external/cuda/include/cuda_runtime.h
vendored
107
samples/external/cuda/include/cuda_runtime.h
vendored
@@ -188,7 +188,8 @@ struct __device_builtin__ __nv_lambda_preheader_injection { };
|
||||
* ::cudaErrorInvalidPtx,
|
||||
* ::cudaErrorUnsupportedPtxVersion,
|
||||
* ::cudaErrorNoKernelImageForDevice,
|
||||
* ::cudaErrorJitCompilerNotFound
|
||||
* ::cudaErrorJitCompilerNotFound,
|
||||
* ::cudaErrorJitCompilationDisabled
|
||||
* \notefnerr
|
||||
* \note_async
|
||||
* \note_null_stream
|
||||
@@ -628,6 +629,57 @@ static __inline__ __host__ cudaError_t cudaMallocPitch(
|
||||
return ::cudaMallocPitch((void**)(void*)devPtr, pitch, width, height);
|
||||
}
|
||||
|
||||
/**
|
||||
* \brief Allocate from a pool
|
||||
*
|
||||
* This is an alternate spelling for cudaMallocFromPoolAsync
|
||||
* made available through operator overloading.
|
||||
*
|
||||
* \sa ::cudaMallocFromPoolAsync,
|
||||
* \ref ::cudaMallocAsync(void** ptr, size_t size, cudaStream_t hStream) "cudaMallocAsync (C API)"
|
||||
*/
|
||||
static __inline__ __host__ cudaError_t cudaMallocAsync(
|
||||
void **ptr,
|
||||
size_t size,
|
||||
cudaMemPool_t memPool,
|
||||
cudaStream_t stream
|
||||
)
|
||||
{
|
||||
return ::cudaMallocFromPoolAsync(ptr, size, memPool, stream);
|
||||
}
|
||||
|
||||
template<class T>
|
||||
static __inline__ __host__ cudaError_t cudaMallocAsync(
|
||||
T **ptr,
|
||||
size_t size,
|
||||
cudaMemPool_t memPool,
|
||||
cudaStream_t stream
|
||||
)
|
||||
{
|
||||
return ::cudaMallocFromPoolAsync((void**)(void*)ptr, size, memPool, stream);
|
||||
}
|
||||
|
||||
template<class T>
|
||||
static __inline__ __host__ cudaError_t cudaMallocAsync(
|
||||
T **ptr,
|
||||
size_t size,
|
||||
cudaStream_t stream
|
||||
)
|
||||
{
|
||||
return ::cudaMallocAsync((void**)(void*)ptr, size, stream);
|
||||
}
|
||||
|
||||
template<class T>
|
||||
static __inline__ __host__ cudaError_t cudaMallocFromPoolAsync(
|
||||
T **ptr,
|
||||
size_t size,
|
||||
cudaMemPool_t memPool,
|
||||
cudaStream_t stream
|
||||
)
|
||||
{
|
||||
return ::cudaMallocFromPoolAsync((void**)(void*)ptr, size, memPool, stream);
|
||||
}
|
||||
|
||||
#if defined(__CUDACC__)
|
||||
|
||||
/**
|
||||
@@ -1190,6 +1242,59 @@ static __inline__ __host__ cudaError_t cudaGraphExecMemcpyNodeSetParamsFromSymbo
|
||||
return ::cudaGraphExecMemcpyNodeSetParamsFromSymbol(hGraphExec, node, dst, (const void*)&symbol, count, offset, kind);
|
||||
}
|
||||
|
||||
#if __cplusplus >= 201103
|
||||
|
||||
/**
|
||||
* \brief Creates a user object by wrapping a C++ object
|
||||
*
|
||||
* TODO detail
|
||||
*
|
||||
* \param object_out - Location to return the user object handle
|
||||
* \param objectToWrap - This becomes the \ptr argument to ::cudaUserObjectCreate. A
|
||||
* lambda will be passed for the \p destroy argument, which calls
|
||||
* delete on this object pointer.
|
||||
* \param initialRefcount - The initial refcount to create the object with, typically 1. The
|
||||
* initial references are owned by the calling thread.
|
||||
* \param flags - Currently it is required to pass cudaUserObjectNoDestructorSync,
|
||||
* which is the only defined flag. This indicates that the destroy
|
||||
* callback cannot be waited on by any CUDA API. Users requiring
|
||||
* synchronization of the callback should signal its completion
|
||||
* manually.
|
||||
*
|
||||
* \return
|
||||
* ::cudaSuccess,
|
||||
* ::cudaErrorInvalidValue
|
||||
*
|
||||
* \sa
|
||||
* ::cudaUserObjectCreate
|
||||
*/
|
||||
template<class T>
|
||||
static __inline__ __host__ cudaError_t cudaUserObjectCreate(
|
||||
cudaUserObject_t *object_out,
|
||||
T *objectToWrap,
|
||||
unsigned int initialRefcount,
|
||||
unsigned int flags)
|
||||
{
|
||||
return ::cudaUserObjectCreate(
|
||||
object_out,
|
||||
objectToWrap,
|
||||
[](void *vpObj) { delete reinterpret_cast<T *>(vpObj); },
|
||||
initialRefcount,
|
||||
flags);
|
||||
}
|
||||
|
||||
template<class T>
|
||||
static __inline__ __host__ cudaError_t cudaUserObjectCreate(
|
||||
cudaUserObject_t *object_out,
|
||||
T *objectToWrap,
|
||||
unsigned int initialRefcount,
|
||||
cudaUserObjectFlags flags)
|
||||
{
|
||||
return cudaUserObjectCreate(object_out, objectToWrap, initialRefcount, (unsigned int)flags);
|
||||
}
|
||||
|
||||
#endif
|
||||
|
||||
/**
|
||||
* \brief \hl Finds the address associated with a CUDA symbol
|
||||
*
|
||||
|
||||
1938
samples/external/cuda/include/cuda_runtime_api.h
vendored
1938
samples/external/cuda/include/cuda_runtime_api.h
vendored
File diff suppressed because it is too large
Load Diff
454
samples/external/cuda/include/driver_types.h
vendored
454
samples/external/cuda/include/driver_types.h
vendored
@@ -393,6 +393,13 @@ enum __device_builtin__ cudaError
|
||||
*/
|
||||
cudaErrorMemoryValueTooLarge = 32,
|
||||
|
||||
/**
|
||||
* This indicates that the CUDA driver that the application has loaded is a
|
||||
* stub library. Applications that run with the stub rather than a real
|
||||
* driver loaded will result in CUDA API returning this error.
|
||||
*/
|
||||
cudaErrorStubLibrary = 34,
|
||||
|
||||
/**
|
||||
* This indicates that the installed NVIDIA CUDA driver is older than the
|
||||
* CUDA runtime library. This is not a supported configuration. Users should
|
||||
@@ -538,10 +545,19 @@ enum __device_builtin__ cudaError
|
||||
cudaErrorInvalidDevice = 101,
|
||||
|
||||
/**
|
||||
* This indicates that the device doesn't have valid Grid License.
|
||||
* This indicates that the device doesn't have a valid Grid License.
|
||||
*/
|
||||
cudaErrorDeviceNotLicensed = 102,
|
||||
|
||||
/**
|
||||
* By default, the CUDA runtime may perform a minimal set of self-tests,
|
||||
* as well as CUDA driver tests, to establish the validity of both.
|
||||
* Introduced in CUDA 11.2, this error return indicates that at least one
|
||||
* of these tests has failed and the validity of either the runtime
|
||||
* or the driver could not be established.
|
||||
*/
|
||||
cudaErrorSoftwareValidityNotEstablished = 103,
|
||||
|
||||
/**
|
||||
* This indicates an internal startup failure in the CUDA runtime.
|
||||
*/
|
||||
@@ -668,6 +684,13 @@ enum __device_builtin__ cudaError
|
||||
*/
|
||||
cudaErrorUnsupportedPtxVersion = 222,
|
||||
|
||||
/**
|
||||
* This indicates that the JIT compilation was disabled. The JIT compilation compiles
|
||||
* PTX. The runtime may fall back to compiling PTX if an application does not contain
|
||||
* a suitable binary for the current device.
|
||||
*/
|
||||
cudaErrorJitCompilationDisabled = 223,
|
||||
|
||||
/**
|
||||
* This indicates that the device kernel source is invalid.
|
||||
*/
|
||||
@@ -708,7 +731,8 @@ enum __device_builtin__ cudaError
|
||||
|
||||
/**
|
||||
* This indicates that a named symbol was not found. Examples of symbols
|
||||
* are global/constant variable names, texture names, and surface names.
|
||||
* are global/constant variable names, driver function names, texture names,
|
||||
* and surface names.
|
||||
*/
|
||||
cudaErrorSymbolNotFound = 500,
|
||||
|
||||
@@ -1002,7 +1026,8 @@ enum __device_builtin__ cudaChannelFormatKind
|
||||
cudaChannelFormatKindSigned = 0, /**< Signed channel format */
|
||||
cudaChannelFormatKindUnsigned = 1, /**< Unsigned channel format */
|
||||
cudaChannelFormatKindFloat = 2, /**< Float channel format */
|
||||
cudaChannelFormatKindNone = 3 /**< No channel format */
|
||||
cudaChannelFormatKindNone = 3, /**< No channel format */
|
||||
cudaChannelFormatKindNV12 = 4 /**< Unsigned 8-bit integers, planar 4:2:0 YUV format */
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -1058,6 +1083,7 @@ struct __device_builtin__ cudaArraySparseProperties {
|
||||
unsigned int miptailFirstLevel; /**< First mip level at which the mip tail begins */
|
||||
unsigned long long miptailSize; /**< Total size of the mip tail. */
|
||||
unsigned int flags; /**< Flags will either be zero or ::cudaArraySparsePropertiesSingleMipTail */
|
||||
unsigned int reserved[4];
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -1163,7 +1189,7 @@ struct __device_builtin__ cudaMemsetParams {
|
||||
size_t pitch; /**< Pitch of destination device pointer. Unused if height is 1 */
|
||||
unsigned int value; /**< Value to be set */
|
||||
unsigned int elementSize; /**< Size of each element in bytes. Must be 1, 2, or 4. */
|
||||
size_t width; /**< Width in bytes, of the row */
|
||||
size_t width; /**< Width of the row in elements */
|
||||
size_t height; /**< Number of rows */
|
||||
};
|
||||
|
||||
@@ -1258,6 +1284,28 @@ union __device_builtin__ cudaStreamAttrValue {
|
||||
enum cudaSynchronizationPolicy syncPolicy;
|
||||
};
|
||||
|
||||
/**
|
||||
* Flags for ::cudaStreamUpdateCaptureDependencies
|
||||
*/
|
||||
enum __device_builtin__ cudaStreamUpdateCaptureDependenciesFlags {
|
||||
cudaStreamAddCaptureDependencies = 0x0, /**< Add new nodes to the dependency set */
|
||||
cudaStreamSetCaptureDependencies = 0x1 /**< Replace the dependency set with the new nodes */
|
||||
};
|
||||
|
||||
/**
|
||||
* Flags for user objects for graphs
|
||||
*/
|
||||
enum __device_builtin__ cudaUserObjectFlags {
|
||||
cudaUserObjectNoDestructorSync = 0x1 /**< Indicates the destructor execution is not synchronized by any CUDA handle. */
|
||||
};
|
||||
|
||||
/**
|
||||
* Flags for retaining user object references for graphs
|
||||
*/
|
||||
enum __device_builtin__ cudaUserObjectRetainFlags {
|
||||
cudaGraphUserObjectMove = 0x1 /**< Transfer references from the caller rather than creating new references. */
|
||||
};
|
||||
|
||||
/**
|
||||
* CUDA graphics interop resource
|
||||
*/
|
||||
@@ -1307,7 +1355,7 @@ enum __device_builtin__ cudaKernelNodeAttrID {
|
||||
};
|
||||
|
||||
/**
|
||||
* Graph kernel node attributes union, used with ::cudaKernelNodeSetAttribute/::cudaKernelNodeGetAttribute
|
||||
* Graph kernel node attributes union, used with ::cudaGraphKernelNodeSetAttribute/::cudaGraphKernelNodeGetAttribute
|
||||
*/
|
||||
union __device_builtin__ cudaKernelNodeAttrValue {
|
||||
struct cudaAccessPolicyWindow accessPolicyWindow; /**< Attribute ::CUaccessPolicyWindow. */
|
||||
@@ -1619,6 +1667,39 @@ enum __device_builtin__ cudaOutputMode
|
||||
cudaCSV = 0x01 /**< Output mode Comma separated values format. */
|
||||
};
|
||||
|
||||
/**
|
||||
* CUDA GPUDirect RDMA flush writes APIs supported on the device
|
||||
*/
|
||||
enum __device_builtin__ cudaFlushGPUDirectRDMAWritesOptions {
|
||||
cudaFlushGPUDirectRDMAWritesOptionHost = 1<<0, /**< ::cudaDeviceFlushGPUDirectRDMAWrites() and its CUDA Driver API counterpart are supported on the device. */
|
||||
cudaFlushGPUDirectRDMAWritesOptionMemOps = 1<<1 /**< The ::CU_STREAM_WAIT_VALUE_FLUSH flag and the ::CU_STREAM_MEM_OP_FLUSH_REMOTE_WRITES MemOp are supported on the CUDA device. */
|
||||
};
|
||||
|
||||
/**
|
||||
* CUDA GPUDirect RDMA flush writes ordering features of the device
|
||||
*/
|
||||
enum __device_builtin__ cudaGPUDirectRDMAWritesOrdering {
|
||||
cudaGPUDirectRDMAWritesOrderingNone = 0, /**< The device does not natively support ordering of GPUDirect RDMA writes. ::cudaFlushGPUDirectRDMAWrites() can be leveraged if supported. */
|
||||
cudaGPUDirectRDMAWritesOrderingOwner = 100, /**< Natively, the device can consistently consume GPUDirect RDMA writes, although other CUDA devices may not. */
|
||||
cudaGPUDirectRDMAWritesOrderingAllDevices = 200 /**< Any CUDA device in the system can consistently consume GPUDirect RDMA writes to this device. */
|
||||
};
|
||||
|
||||
/**
|
||||
* CUDA GPUDirect RDMA flush writes scopes
|
||||
*/
|
||||
enum __device_builtin__ cudaFlushGPUDirectRDMAWritesScope {
|
||||
cudaFlushGPUDirectRDMAWritesToOwner = 100, /**< Blocks until remote writes are visible to the CUDA device context owning the data. */
|
||||
cudaFlushGPUDirectRDMAWritesToAllDevices = 200 /**< Blocks until remote writes are visible to all CUDA device contexts. */
|
||||
};
|
||||
|
||||
/**
|
||||
* CUDA GPUDirect RDMA flush writes targets
|
||||
*/
|
||||
enum __device_builtin__ cudaFlushGPUDirectRDMAWritesTarget {
|
||||
cudaFlushGPUDirectRDMAWritesTargetCurrentDevice /**< Sets the target for ::cudaDeviceFlushGPUDirectRDMAWrites() to the currently active CUDA device context. */
|
||||
};
|
||||
|
||||
|
||||
/**
|
||||
* CUDA device attributes
|
||||
*/
|
||||
@@ -1725,9 +1806,166 @@ enum __device_builtin__ cudaDeviceAttr
|
||||
cudaDevAttrPageableMemoryAccessUsesHostPageTables = 100, /**< Device accesses pageable memory via the host's page tables. */
|
||||
cudaDevAttrDirectManagedMemAccessFromHost = 101, /**< Host can directly access managed memory on the device without migration. */
|
||||
cudaDevAttrMaxBlocksPerMultiprocessor = 106, /**< Maximum number of blocks per multiprocessor */
|
||||
cudaDevAttrMaxPersistingL2CacheSize = 108, /**< Maximum L2 persisting lines capacity setting in bytes. */
|
||||
cudaDevAttrMaxAccessPolicyWindowSize = 109, /**< Maximum value of cudaAccessPolicyWindow::num_bytes. */
|
||||
cudaDevAttrReservedSharedMemoryPerBlock = 111, /**< Shared memory reserved by CUDA driver per block in bytes */
|
||||
cudaDevAttrSparseCudaArraySupported = 112, /**< Device supports sparse CUDA arrays and sparse CUDA mipmapped arrays */
|
||||
cudaDevAttrHostRegisterReadOnlySupported = 113 /**< Device supports using the ::cuMemHostRegister flag CU_MEMHOSTERGISTER_READ_ONLY to register memory that must be mapped as read-only to the GPU */
|
||||
cudaDevAttrHostRegisterReadOnlySupported = 113, /**< Device supports using the ::cudaHostRegister flag cudaHostRegisterReadOnly to register memory that must be mapped as read-only to the GPU */
|
||||
cudaDevAttrMaxTimelineSemaphoreInteropSupported = 114, /**< External timeline semaphore interop is supported on the device */
|
||||
cudaDevAttrMemoryPoolsSupported = 115, /**< Device supports using the ::cudaMallocAsync and ::cudaMemPool family of APIs */
|
||||
cudaDevAttrGPUDirectRDMASupported = 116, /**< Device supports GPUDirect RDMA APIs, like nvidia_p2p_get_pages (see https://docs.nvidia.com/cuda/gpudirect-rdma for more information) */
|
||||
cudaDevAttrGPUDirectRDMAFlushWritesOptions = 117, /**< The returned attribute shall be interpreted as a bitmask, where the individual bits are listed in the ::cudaFlushGPUDirectRDMAWritesOptions enum */
|
||||
cudaDevAttrGPUDirectRDMAWritesOrdering = 118, /**< GPUDirect RDMA writes to the device do not need to be flushed for consumers within the scope indicated by the returned attribute. See ::cudaGPUDirectRDMAWritesOrdering for the numerical values returned here. */
|
||||
cudaDevAttrMemoryPoolSupportedHandleTypes = 119 /**< Handle types supported with mempool based IPC */
|
||||
};
|
||||
|
||||
/**
|
||||
* CUDA memory pool attributes
|
||||
*/
|
||||
enum __device_builtin__ cudaMemPoolAttr
|
||||
{
|
||||
/**
|
||||
* (value type = int)
|
||||
* Allow cuMemAllocAsync to use memory asynchronously freed
|
||||
* in another streams as long as a stream ordering dependency
|
||||
* of the allocating stream on the free action exists.
|
||||
* Cuda events and null stream interactions can create the required
|
||||
* stream ordered dependencies. (default enabled)
|
||||
*/
|
||||
cudaMemPoolReuseFollowEventDependencies = 0x1,
|
||||
|
||||
/**
|
||||
* (value type = int)
|
||||
* Allow reuse of already completed frees when there is no dependency
|
||||
* between the free and allocation. (default enabled)
|
||||
*/
|
||||
cudaMemPoolReuseAllowOpportunistic = 0x2,
|
||||
|
||||
/**
|
||||
* (value type = int)
|
||||
* Allow cuMemAllocAsync to insert new stream dependencies
|
||||
* in order to establish the stream ordering required to reuse
|
||||
* a piece of memory released by cuFreeAsync (default enabled).
|
||||
*/
|
||||
cudaMemPoolReuseAllowInternalDependencies = 0x3,
|
||||
|
||||
|
||||
/**
|
||||
* (value type = cuuint64_t)
|
||||
* Amount of reserved memory in bytes to hold onto before trying
|
||||
* to release memory back to the OS. When more than the release
|
||||
* threshold bytes of memory are held by the memory pool, the
|
||||
* allocator will try to release memory back to the OS on the
|
||||
* next call to stream, event or context synchronize. (default 0)
|
||||
*/
|
||||
cudaMemPoolAttrReleaseThreshold = 0x4,
|
||||
|
||||
/**
|
||||
* (value type = cuuint64_t)
|
||||
* Amount of backing memory currently allocated for the mempool.
|
||||
*/
|
||||
cudaMemPoolAttrReservedMemCurrent = 0x5,
|
||||
|
||||
/**
|
||||
* (value type = cuuint64_t)
|
||||
* High watermark of backing memory allocated for the mempool since the
|
||||
* last time it was reset. High watermark can only be reset to zero.
|
||||
*/
|
||||
cudaMemPoolAttrReservedMemHigh = 0x6,
|
||||
|
||||
/**
|
||||
* (value type = cuuint64_t)
|
||||
* Amount of memory from the pool that is currently in use by the application.
|
||||
*/
|
||||
cudaMemPoolAttrUsedMemCurrent = 0x7,
|
||||
|
||||
/**
|
||||
* (value type = cuuint64_t)
|
||||
* High watermark of the amount of memory from the pool that was in use by the application since
|
||||
* the last time it was reset. High watermark can only be reset to zero.
|
||||
*/
|
||||
cudaMemPoolAttrUsedMemHigh = 0x8
|
||||
};
|
||||
|
||||
/**
|
||||
* Specifies the type of location
|
||||
*/
|
||||
enum __device_builtin__ cudaMemLocationType {
|
||||
cudaMemLocationTypeInvalid = 0,
|
||||
cudaMemLocationTypeDevice = 1 /**< Location is a device location, thus id is a device ordinal */
|
||||
};
|
||||
|
||||
/**
|
||||
* Specifies a memory location.
|
||||
*
|
||||
* To specify a gpu, set type = ::cudaMemLocationTypeDevice and set id = the gpu's device ordinal.
|
||||
*/
|
||||
struct __device_builtin__ cudaMemLocation {
|
||||
enum cudaMemLocationType type; /**< Specifies the location type, which modifies the meaning of id. */
|
||||
int id; /**< identifier for a given this location's ::CUmemLocationType. */
|
||||
};
|
||||
|
||||
/**
|
||||
* Specifies the memory protection flags for mapping.
|
||||
*/
|
||||
enum __device_builtin__ cudaMemAccessFlags {
|
||||
cudaMemAccessFlagsProtNone = 0, /**< Default, make the address range not accessible */
|
||||
cudaMemAccessFlagsProtRead = 1, /**< Make the address range read accessible */
|
||||
cudaMemAccessFlagsProtReadWrite = 3 /**< Make the address range read-write accessible */
|
||||
};
|
||||
|
||||
/**
|
||||
* Memory access descriptor
|
||||
*/
|
||||
struct __device_builtin__ cudaMemAccessDesc {
|
||||
struct cudaMemLocation location; /**< Location on which the request is to change it's accessibility */
|
||||
enum cudaMemAccessFlags flags; /**< ::CUmemProt accessibility flags to set on the request */
|
||||
};
|
||||
|
||||
/**
|
||||
* Defines the allocation types available
|
||||
*/
|
||||
enum __device_builtin__ cudaMemAllocationType {
|
||||
cudaMemAllocationTypeInvalid = 0x0,
|
||||
/** This allocation type is 'pinned', i.e. cannot migrate from its current
|
||||
* location while the application is actively using it
|
||||
*/
|
||||
cudaMemAllocationTypePinned = 0x1,
|
||||
cudaMemAllocationTypeMax = 0x7FFFFFFF
|
||||
};
|
||||
|
||||
/**
|
||||
* Flags for specifying particular handle types
|
||||
*/
|
||||
enum __device_builtin__ cudaMemAllocationHandleType {
|
||||
cudaMemHandleTypeNone = 0x0, /**< Does not allow any export mechanism. > */
|
||||
cudaMemHandleTypePosixFileDescriptor = 0x1, /**< Allows a file descriptor to be used for exporting. Permitted only on POSIX systems. (int) */
|
||||
cudaMemHandleTypeWin32 = 0x2, /**< Allows a Win32 NT handle to be used for exporting. (HANDLE) */
|
||||
cudaMemHandleTypeWin32Kmt = 0x4 /**< Allows a Win32 KMT handle to be used for exporting. (D3DKMT_HANDLE) */
|
||||
};
|
||||
|
||||
/**
|
||||
* Specifies the properties of allocations made from the pool.
|
||||
*/
|
||||
struct __device_builtin__ cudaMemPoolProps {
|
||||
enum cudaMemAllocationType allocType; /**< Allocation type. Currently must be specified as cudaMemAllocationTypePinned */
|
||||
enum cudaMemAllocationHandleType handleTypes; /**< Handle types that will be supported by allocations from the pool. */
|
||||
struct cudaMemLocation location; /**< Location allocations should reside. */
|
||||
/**
|
||||
* Windows-specific LPSECURITYATTRIBUTES required when
|
||||
* ::cudaMemHandleTypeWin32 is specified. This security attribute defines
|
||||
* the scope of which exported allocations may be tranferred to other
|
||||
* processes. In all other cases, this field is required to be zero.
|
||||
*/
|
||||
void *win32SecurityAttributes;
|
||||
unsigned char reserved[64]; /**< reserved for future use, must be 0 */
|
||||
};
|
||||
|
||||
/**
|
||||
* Opaque data for exporting a pool allocation
|
||||
*/
|
||||
struct __device_builtin__ cudaMemPoolPtrExportData {
|
||||
unsigned char reserved[64];
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -1784,7 +2022,7 @@ struct __device_builtin__ cudaDeviceProp
|
||||
int computeMode; /**< Compute mode (See ::cudaComputeMode) */
|
||||
int maxTexture1D; /**< Maximum 1D texture size */
|
||||
int maxTexture1DMipmap; /**< Maximum 1D mipmapped texture size */
|
||||
int maxTexture1DLinear; /**< Maximum size for 1D textures bound to linear memory */
|
||||
int maxTexture1DLinear; /**< Deprecated, do not use. Use cudaDeviceGetTexture1DLinearMaxWidth() or cuDeviceGetTexture1DLinearMaxWidth() instead. */
|
||||
int maxTexture2D[2]; /**< Maximum 2D texture dimensions */
|
||||
int maxTexture2DMipmap[2]; /**< Maximum 2D mipmapped texture dimensions */
|
||||
int maxTexture2DLinear[3]; /**< Maximum dimensions (width, height, pitch) for 2D textures bound to pitched memory */
|
||||
@@ -1831,7 +2069,7 @@ struct __device_builtin__ cudaDeviceProp
|
||||
int computePreemptionSupported; /**< Device supports Compute Preemption */
|
||||
int canUseHostPointerForRegisteredMem; /**< Device can access host registered memory at the same virtual address as the CPU */
|
||||
int cooperativeLaunch; /**< Device supports launching cooperative kernels via ::cudaLaunchCooperativeKernel */
|
||||
int cooperativeMultiDeviceLaunch; /**< Device can participate in cooperative kernels launched via ::cudaLaunchCooperativeKernelMultiDevice */
|
||||
int cooperativeMultiDeviceLaunch; /**< Deprecated, cudaLaunchCooperativeKernelMultiDevice is deprecated. */
|
||||
size_t sharedMemPerBlockOptin; /**< Per device maximum shared memory per block usable by special opt in */
|
||||
int pageableMemoryAccessUsesHostPageTables; /**< Device accesses pageable memory via the host's page tables */
|
||||
int directManagedMemAccessFromHost; /**< Host can directly access managed memory on the device without migration. */
|
||||
@@ -2134,7 +2372,7 @@ enum __device_builtin__ cudaExternalSemaphoreHandleType {
|
||||
* Handle is an opaque shared NT handle
|
||||
*/
|
||||
cudaExternalSemaphoreHandleTypeOpaqueWin32 = 2,
|
||||
/**
|
||||
/**
|
||||
* Handle is an opaque, globally shared handle
|
||||
*/
|
||||
cudaExternalSemaphoreHandleTypeOpaqueWin32Kmt = 3,
|
||||
@@ -2157,7 +2395,15 @@ enum __device_builtin__ cudaExternalSemaphoreHandleType {
|
||||
/**
|
||||
* Handle is a shared KMT handle referencing a D3D11 keyed mutex object
|
||||
*/
|
||||
cudaExternalSemaphoreHandleTypeKeyedMutexKmt = 8
|
||||
cudaExternalSemaphoreHandleTypeKeyedMutexKmt = 8,
|
||||
/**
|
||||
* Handle is an opaque handle file descriptor referencing a timeline semaphore
|
||||
*/
|
||||
cudaExternalSemaphoreHandleTypeTimelineSemaphoreFd = 9,
|
||||
/**
|
||||
* Handle is an opaque handle file descriptor referencing a timeline semaphore
|
||||
*/
|
||||
cudaExternalSemaphoreHandleTypeTimelineSemaphoreWin32 = 10
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -2170,8 +2416,10 @@ struct __device_builtin__ cudaExternalSemaphoreHandleDesc {
|
||||
enum cudaExternalSemaphoreHandleType type;
|
||||
union {
|
||||
/**
|
||||
* File descriptor referencing the semaphore object. Valid
|
||||
* when type is ::cudaExternalSemaphoreHandleTypeOpaqueFd
|
||||
* File descriptor referencing the semaphore object. Valid when
|
||||
* type is one of the following:
|
||||
* - ::cudaExternalSemaphoreHandleTypeOpaqueFd
|
||||
* - ::cudaExternalSemaphoreHandleTypeTimelineSemaphoreFd
|
||||
*/
|
||||
int fd;
|
||||
/**
|
||||
@@ -2182,6 +2430,7 @@ struct __device_builtin__ cudaExternalSemaphoreHandleDesc {
|
||||
* - ::cudaExternalSemaphoreHandleTypeD3D12Fence
|
||||
* - ::cudaExternalSemaphoreHandleTypeD3D11Fence
|
||||
* - ::cudaExternalSemaphoreHandleTypeKeyedMutex
|
||||
* - ::cudaExternalSemaphoreHandleTypeTimelineSemaphoreWin32
|
||||
* Exactly one of 'handle' and 'name' must be non-NULL. If
|
||||
* type is one of the following:
|
||||
* ::cudaExternalSemaphoreHandleTypeOpaqueWin32Kmt
|
||||
@@ -2210,10 +2459,11 @@ struct __device_builtin__ cudaExternalSemaphoreHandleDesc {
|
||||
unsigned int flags;
|
||||
};
|
||||
|
||||
#if defined(__CUDA_API_VERSION_INTERNAL)
|
||||
/**
|
||||
* External semaphore signal parameters
|
||||
* External semaphore signal parameters(deprecated)
|
||||
*/
|
||||
struct __device_builtin__ cudaExternalSemaphoreSignalParams {
|
||||
struct __device_builtin__ cudaExternalSemaphoreSignalParams_v1 {
|
||||
struct {
|
||||
/**
|
||||
* Parameters for fence objects
|
||||
@@ -2256,9 +2506,9 @@ struct __device_builtin__ cudaExternalSemaphoreSignalParams {
|
||||
};
|
||||
|
||||
/**
|
||||
* External semaphore wait parameters
|
||||
* External semaphore wait parameters(deprecated)
|
||||
*/
|
||||
struct __device_builtin__ cudaExternalSemaphoreWaitParams {
|
||||
struct __device_builtin__ cudaExternalSemaphoreWaitParams_v1 {
|
||||
struct {
|
||||
/**
|
||||
* Parameters for fence objects
|
||||
@@ -2303,6 +2553,105 @@ struct __device_builtin__ cudaExternalSemaphoreWaitParams {
|
||||
*/
|
||||
unsigned int flags;
|
||||
};
|
||||
#endif
|
||||
|
||||
/**
|
||||
* External semaphore signal parameters, compatible with driver type
|
||||
*/
|
||||
struct __device_builtin__ cudaExternalSemaphoreSignalParams{
|
||||
struct {
|
||||
/**
|
||||
* Parameters for fence objects
|
||||
*/
|
||||
struct {
|
||||
/**
|
||||
* Value of fence to be signaled
|
||||
*/
|
||||
unsigned long long value;
|
||||
} fence;
|
||||
union {
|
||||
/**
|
||||
* Pointer to NvSciSyncFence. Valid if ::cudaExternalSemaphoreHandleType
|
||||
* is of type ::cudaExternalSemaphoreHandleTypeNvSciSync.
|
||||
*/
|
||||
void *fence;
|
||||
unsigned long long reserved;
|
||||
} nvSciSync;
|
||||
/**
|
||||
* Parameters for keyed mutex objects
|
||||
*/
|
||||
struct {
|
||||
/*
|
||||
* Value of key to release the mutex with
|
||||
*/
|
||||
unsigned long long key;
|
||||
} keyedMutex;
|
||||
unsigned int reserved[12];
|
||||
} params;
|
||||
/**
|
||||
* Only when ::cudaExternalSemaphoreSignalParams is used to
|
||||
* signal a ::cudaExternalSemaphore_t of type
|
||||
* ::cudaExternalSemaphoreHandleTypeNvSciSync, the valid flag is
|
||||
* ::cudaExternalSemaphoreSignalSkipNvSciBufMemSync: which indicates
|
||||
* that while signaling the ::cudaExternalSemaphore_t, no memory
|
||||
* synchronization operations should be performed for any external memory
|
||||
* object imported as ::cudaExternalMemoryHandleTypeNvSciBuf.
|
||||
* For all other types of ::cudaExternalSemaphore_t, flags must be zero.
|
||||
*/
|
||||
unsigned int flags;
|
||||
unsigned int reserved[16];
|
||||
};
|
||||
|
||||
/**
|
||||
* External semaphore wait parameters, compatible with driver type
|
||||
*/
|
||||
struct __device_builtin__ cudaExternalSemaphoreWaitParams {
|
||||
struct {
|
||||
/**
|
||||
* Parameters for fence objects
|
||||
*/
|
||||
struct {
|
||||
/**
|
||||
* Value of fence to be waited on
|
||||
*/
|
||||
unsigned long long value;
|
||||
} fence;
|
||||
union {
|
||||
/**
|
||||
* Pointer to NvSciSyncFence. Valid if ::cudaExternalSemaphoreHandleType
|
||||
* is of type ::cudaExternalSemaphoreHandleTypeNvSciSync.
|
||||
*/
|
||||
void *fence;
|
||||
unsigned long long reserved;
|
||||
} nvSciSync;
|
||||
/**
|
||||
* Parameters for keyed mutex objects
|
||||
*/
|
||||
struct {
|
||||
/**
|
||||
* Value of key to acquire the mutex with
|
||||
*/
|
||||
unsigned long long key;
|
||||
/**
|
||||
* Timeout in milliseconds to wait to acquire the mutex
|
||||
*/
|
||||
unsigned int timeoutMs;
|
||||
} keyedMutex;
|
||||
unsigned int reserved[10];
|
||||
} params;
|
||||
/**
|
||||
* Only when ::cudaExternalSemaphoreSignalParams is used to
|
||||
* signal a ::cudaExternalSemaphore_t of type
|
||||
* ::cudaExternalSemaphoreHandleTypeNvSciSync, the valid flag is
|
||||
* ::cudaExternalSemaphoreSignalSkipNvSciBufMemSync: which indicates
|
||||
* that while waiting for the ::cudaExternalSemaphore_t, no memory
|
||||
* synchronization operations should be performed for any external memory
|
||||
* object imported as ::cudaExternalMemoryHandleTypeNvSciBuf.
|
||||
* For all other types of ::cudaExternalSemaphore_t, flags must be zero.
|
||||
*/
|
||||
unsigned int flags;
|
||||
unsigned int reserved[16];
|
||||
};
|
||||
|
||||
|
||||
/*******************************************************************************
|
||||
@@ -2356,11 +2705,21 @@ typedef __device_builtin__ struct CUgraph_st *cudaGraph_t;
|
||||
*/
|
||||
typedef __device_builtin__ struct CUgraphNode_st *cudaGraphNode_t;
|
||||
|
||||
/**
|
||||
* CUDA user object for graphs
|
||||
*/
|
||||
typedef __device_builtin__ struct CUuserObject_st *cudaUserObject_t;
|
||||
|
||||
/**
|
||||
* CUDA function
|
||||
*/
|
||||
typedef __device_builtin__ struct CUfunc_st *cudaFunction_t;
|
||||
|
||||
/**
|
||||
* CUDA memory pool
|
||||
*/
|
||||
typedef __device_builtin__ struct CUmemPoolHandle_st *cudaMemPool_t;
|
||||
|
||||
/**
|
||||
* CUDA cooperative group scope
|
||||
*/
|
||||
@@ -2395,16 +2754,36 @@ struct __device_builtin__ cudaKernelNodeParams {
|
||||
void **extra; /**< Pointer to kernel arguments in the "extra" format */
|
||||
};
|
||||
|
||||
/**
|
||||
* External semaphore signal node parameters
|
||||
*/
|
||||
struct __device_builtin__ cudaExternalSemaphoreSignalNodeParams {
|
||||
cudaExternalSemaphore_t* extSemArray; /**< Array of external semaphore handles. */
|
||||
const struct cudaExternalSemaphoreSignalParams* paramsArray; /**< Array of external semaphore signal parameters. */
|
||||
unsigned int numExtSems; /**< Number of handles and parameters supplied in extSemArray and paramsArray. */
|
||||
};
|
||||
|
||||
/**
|
||||
* External semaphore wait node parameters
|
||||
*/
|
||||
struct __device_builtin__ cudaExternalSemaphoreWaitNodeParams {
|
||||
cudaExternalSemaphore_t* extSemArray; /**< Array of external semaphore handles. */
|
||||
const struct cudaExternalSemaphoreWaitParams* paramsArray; /**< Array of external semaphore wait parameters. */
|
||||
unsigned int numExtSems; /**< Number of handles and parameters supplied in extSemArray and paramsArray. */
|
||||
};
|
||||
|
||||
/**
|
||||
* CUDA Graph node types
|
||||
*/
|
||||
enum __device_builtin__ cudaGraphNodeType {
|
||||
cudaGraphNodeTypeKernel = 0x00, /**< GPU kernel node */
|
||||
cudaGraphNodeTypeMemcpy = 0x01, /**< Memcpy node */
|
||||
cudaGraphNodeTypeMemset = 0x02, /**< Memset node */
|
||||
cudaGraphNodeTypeHost = 0x03, /**< Host (executable) node */
|
||||
cudaGraphNodeTypeGraph = 0x04, /**< Node which executes an embedded graph */
|
||||
cudaGraphNodeTypeEmpty = 0x05, /**< Empty (no-op) node */
|
||||
cudaGraphNodeTypeKernel = 0x00, /**< GPU kernel node */
|
||||
cudaGraphNodeTypeMemcpy = 0x01, /**< Memcpy node */
|
||||
cudaGraphNodeTypeMemset = 0x02, /**< Memset node */
|
||||
cudaGraphNodeTypeHost = 0x03, /**< Host (executable) node */
|
||||
cudaGraphNodeTypeGraph = 0x04, /**< Node which executes an embedded graph */
|
||||
cudaGraphNodeTypeEmpty = 0x05, /**< Empty (no-op) node */
|
||||
cudaGraphNodeTypeWaitEvent = 0x06, /**< External event wait node */
|
||||
cudaGraphNodeTypeEventRecord = 0x07, /**< External event record node */
|
||||
cudaGraphNodeTypeCount
|
||||
};
|
||||
|
||||
@@ -2421,9 +2800,36 @@ enum __device_builtin__ cudaGraphExecUpdateResult {
|
||||
cudaGraphExecUpdateError = 0x1, /**< The update failed for an unexpected reason which is described in the return value of the function */
|
||||
cudaGraphExecUpdateErrorTopologyChanged = 0x2, /**< The update failed because the topology changed */
|
||||
cudaGraphExecUpdateErrorNodeTypeChanged = 0x3, /**< The update failed because a node type changed */
|
||||
cudaGraphExecUpdateErrorFunctionChanged = 0x4, /**< The update failed because the function of a kernel node changed */
|
||||
cudaGraphExecUpdateErrorFunctionChanged = 0x4, /**< The update failed because the function of a kernel node changed (CUDA driver < 11.2) */
|
||||
cudaGraphExecUpdateErrorParametersChanged = 0x5, /**< The update failed because the parameters changed in a way that is not supported */
|
||||
cudaGraphExecUpdateErrorNotSupported = 0x6 /**< The update failed because something about the node is not supported */
|
||||
cudaGraphExecUpdateErrorNotSupported = 0x6, /**< The update failed because something about the node is not supported */
|
||||
cudaGraphExecUpdateErrorUnsupportedFunctionChange = 0x7 /**< The update failed because the function of a kernel node changed in an unsupported way */
|
||||
};
|
||||
|
||||
/**
|
||||
* Flags to specify search options to be used with ::cudaGetDriverEntryPoint
|
||||
* For more details see ::cuGetProcAddress
|
||||
*/
|
||||
enum __device_builtin__ cudaGetDriverEntryPointFlags {
|
||||
cudaEnableDefault = 0x0, /**< Default search mode for driver symbols. */
|
||||
cudaEnableLegacyStream = 0x1, /**< Search for legacy versions of driver symbols. */
|
||||
cudaEnablePerThreadDefaultStream = 0x2 /**< Search for per-thread versions of driver symbols. */
|
||||
};
|
||||
|
||||
/**
|
||||
* CUDA Graph debug write options
|
||||
*/
|
||||
enum __device_builtin__ cudaGraphDebugDotFlags {
|
||||
cudaGraphDebugDotFlagsVerbose = 1<<0, /** Output all debug data as if every debug flag is enabled */
|
||||
cudaGraphDebugDotFlagsKernelNodeParams = 1<<2, /** Adds cudaKernelNodeParams to output */
|
||||
cudaGraphDebugDotFlagsMemcpyNodeParams = 1<<3, /** Adds cudaMemcpy3DParms to output */
|
||||
cudaGraphDebugDotFlagsMemsetNodeParams = 1<<4, /** Adds cudaMemsetParams to output */
|
||||
cudaGraphDebugDotFlagsHostNodeParams = 1<<5, /** Adds cudaHostNodeParams to output */
|
||||
cudaGraphDebugDotFlagsEventNodeParams = 1<<6, /** Adds cudaEvent_t handle from record and wait nodes to output */
|
||||
cudaGraphDebugDotFlagsExtSemasSignalNodeParams = 1<<7, /** Adds cudaExternalSemaphoreSignalNodeParams values to output */
|
||||
cudaGraphDebugDotFlagsExtSemasWaitNodeParams = 1<<8, /** Adds cudaExternalSemaphoreWaitNodeParams to output */
|
||||
cudaGraphDebugDotFlagsKernelNodeAttributes = 1<<9, /** Adds cudaKernelNodeAttrID values to output */
|
||||
cudaGraphDebugDotFlagsHandles = 1<<10 /** Adds node handles and every kernel function handle to output */
|
||||
};
|
||||
|
||||
/** @} */
|
||||
|
||||
BIN
samples/external/cuda/lib/x64/cudart.lib
vendored
BIN
samples/external/cuda/lib/x64/cudart.lib
vendored
Binary file not shown.
@@ -58,21 +58,22 @@ inline void CVWrapperForNvCVImage(const NvCVImage *nvcvIm, cv::Mat *cvIm) {
|
||||
// Wrap a cv::Mat in an NvCVImage.
|
||||
inline void NVWrapperForCVMat(const cv::Mat *cvIm, NvCVImage *nvcvIm) {
|
||||
static const NvCVImage_PixelFormat nvFormat[] = { NVCV_FORMAT_UNKNOWN, NVCV_Y, NVCV_YA, NVCV_BGR, NVCV_BGRA };
|
||||
static const NvCVImage_ComponentType nvType[] = { NVCV_U8, NVCV_TYPE_UNKNOWN, NVCV_U16, NVCV_S16, NVCV_S32, NVCV_F32, NVCV_F64 };
|
||||
static const NvCVImage_ComponentType nvType[] = {NVCV_U8, NVCV_TYPE_UNKNOWN, NVCV_U16, NVCV_S16,
|
||||
NVCV_S32, NVCV_F32, NVCV_F64, NVCV_TYPE_UNKNOWN};
|
||||
nvcvIm->pixels = cvIm->data;
|
||||
nvcvIm->width = cvIm->cols;
|
||||
nvcvIm->height = cvIm->rows;
|
||||
nvcvIm->pitch = (unsigned)cvIm->step1();
|
||||
nvcvIm->pixelFormat = nvFormat[cvIm->channels()];
|
||||
nvcvIm->componentType = nvType[cvIm->depth()];
|
||||
nvcvIm->pitch = (int)cvIm->step[0];
|
||||
nvcvIm->pixelFormat = nvFormat[cvIm->channels() <= 4 ? cvIm->channels() : 0];
|
||||
nvcvIm->componentType = nvType[cvIm->depth() & 7];
|
||||
nvcvIm->bufferBytes = 0;
|
||||
nvcvIm->deletePtr = nullptr;
|
||||
nvcvIm->deleteProc = nullptr;
|
||||
nvcvIm->pixelBytes = (unsigned char)cvIm->elemSize();
|
||||
nvcvIm->pixelBytes = (unsigned char)cvIm->step[1];
|
||||
nvcvIm->componentBytes = (unsigned char)cvIm->elemSize1();
|
||||
nvcvIm->numComponents = (unsigned char)cvIm->channels();
|
||||
nvcvIm->planar = 0;
|
||||
nvcvIm->gpuMem = 0;
|
||||
nvcvIm->planar = NVCV_CHUNKY;
|
||||
nvcvIm->gpuMem = NVCV_CPU;
|
||||
nvcvIm->reserved[0] = 0;
|
||||
nvcvIm->reserved[1] = 0;
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user