diff --git a/docs/guide/advanced/video-codecs.md b/docs/guide/advanced/video-codecs.md index 2d759ea5..be5070d7 100644 --- a/docs/guide/advanced/video-codecs.md +++ b/docs/guide/advanced/video-codecs.md @@ -52,14 +52,16 @@ PeerConnectionFactory factory = PeerConnectionFactory.builder() | Platform | Hardware decoding | |---|---| | Windows | H.264, AV1 and VP9 (profile 0), on GPUs that decode them, through the Media Foundation decoders of Windows on Direct3D 11 (DXVA) | -| Linux | Not yet; decoding is in software | +| Linux | H.264 and VP9 (profile 0) on NVIDIA GPUs, through NVDEC; VA-API, for Intel and AMD GPUs, is not there yet | | macOS | H.264 through VideoToolbox, as with `DefaultVideoDecoderFactory`, and VP9 (profile 0) through VideoToolbox's VP9 decoder, on Macs that have one | +On Linux, NVDEC needs an NVIDIA driver, which provides `libcuda.so.1` and `libnvcuvid.so.1`; both are loaded when the factory is made, so a machine without the driver is not affected. Streams with spatial layers are decoded in software. + AV1 on Windows needs the *AV1 Video Extension*, which Windows 11 includes, and VP9 the *VP9 Video Extensions*, a free component of the Microsoft Store that a system may not have; without it, VP9 is decoded in software. Streams with spatial layers (SVC) are decoded in software too, since their layers reach a decoder without the index that says where each ends. Decoded frames are copied from GPU memory back to system memory, where WebRTC's frames are, so hardware decoding pays off mostly at high resolutions and with many streams; at low resolutions WebRTC's software decoders are about as cheap. A hardware decoder that fails, or turns out to decode in software, is replaced by the software decoder, which starts with the next key frame. On macOS, VP9 decoding in hardware saves CPU: in a measurement on an Apple M2 it took a fraction of the processor time libvpx needs, but multi-threaded libvpx was about as fast on the clock. VideoToolbox decodes VP9 in hardware only where the Mac has a decoder for it, which Apple silicon does; a Mac without one decodes VP9 with libvpx as before. Profile 2 (10 bit), sizes below 64x64 or above 4096x4096, and streams with spatial layers (SVC) are decoded with libvpx too. -Which decoder a stream uses shows in the `decoderImplementation` statistic of its `inbound-rtp` stats, e.g. `MediaFoundation (Microsoft H264 Video Decoder MFT)`, `MediaFoundation (AV1VideoExtension)` or `VideoToolbox (VP9)`. +Which decoder a stream uses shows in the `decoderImplementation` statistic of its `inbound-rtp` stats, e.g. `MediaFoundation (Microsoft H264 Video Decoder MFT)`, `MediaFoundation (AV1VideoExtension)`, `VideoToolbox (VP9)` or `NVDEC (NVIDIA GeForce RTX 4070)`. ### Native Codecs diff --git a/webrtc-jni/src/main/cpp/CMakeLists.txt b/webrtc-jni/src/main/cpp/CMakeLists.txt index e23cb2ed..c4134cdc 100644 --- a/webrtc-jni/src/main/cpp/CMakeLists.txt +++ b/webrtc-jni/src/main/cpp/CMakeLists.txt @@ -63,6 +63,11 @@ file(GLOB SOURCES_MEDIA_VIDEO_CODEC_OS "src/media/video/codec/${SOURCE_TARGET}/* if(WIN32 OR LINUX) file(GLOB SOURCES_MEDIA_VIDEO_CODEC_NVENC "src/media/video/codec/nvenc/*.cpp") endif() +# NVDEC the same, for decoding. Only Linux has a decoder built on it so far; +# Windows decodes through Media Foundation. +if(LINUX) + file(GLOB SOURCES_MEDIA_VIDEO_CODEC_NVDEC "src/media/video/codec/nvdec/*.cpp") +endif() file(GLOB SOURCES_MEDIA_VIDEO_DESKTOP "src/media/video/desktop/*.cpp") file(GLOB SOURCES_MEDIA_VIDEO_DESKTOP_OS "src/media/video/desktop/${SOURCE_TARGET}/*.cpp") file(GLOB SOURCES_MEDIA_VIDEO_OS "src/media/video/${SOURCE_TARGET}/*.cpp") @@ -80,6 +85,7 @@ list(APPEND SOURCES ${SOURCES_MEDIA_VIDEO_CODEC} ${SOURCES_MEDIA_VIDEO_CODEC_OS} ${SOURCES_MEDIA_VIDEO_CODEC_NVENC} + ${SOURCES_MEDIA_VIDEO_CODEC_NVDEC} ${SOURCES_MEDIA_VIDEO_DESKTOP} ${SOURCES_MEDIA_VIDEO_DESKTOP_OS} ${SOURCES_MEDIA_VIDEO_OS} @@ -113,6 +119,9 @@ endif() if(LINUX) # libva is loaded at run time, so building needs its headers only. target_include_directories(${PROJECT_NAME} PRIVATE dependencies/libva/include) + + # So is NVDEC, and the CUDA driver it runs on. + target_include_directories(${PROJECT_NAME} PRIVATE dependencies/nvdec/include) endif() set_target_properties(${PROJECT_NAME} PROPERTIES @@ -180,4 +189,9 @@ if(LINUX) DESTINATION "${CMAKE_INSTALL_PREFIX}/META-INF/licenses/libva" COMPONENT Runtime ) + install(FILES + "${CMAKE_CURRENT_SOURCE_DIR}/dependencies/nvdec/LICENSE" + DESTINATION "${CMAKE_INSTALL_PREFIX}/META-INF/licenses/nvdec" + COMPONENT Runtime + ) endif() diff --git a/webrtc-jni/src/main/cpp/dependencies/nvdec/LICENSE b/webrtc-jni/src/main/cpp/dependencies/nvdec/LICENSE new file mode 100644 index 00000000..6def9d02 --- /dev/null +++ b/webrtc-jni/src/main/cpp/dependencies/nvdec/LICENSE @@ -0,0 +1,88 @@ +The headers in include/ffnvcodec come from FFmpeg's nv-codec-headers, tag n12.0.16.1, and are +compile-time only: the library loads the NVIDIA driver's libraries at run time. Each file carries its +own license, which applies to that file only, and they are reproduced here as they stand at the head +of each file. + +============================================================================== +dynlink_cuda.h +============================================================================== +This copyright notice applies to this header file only: + +Copyright (c) 2016 + +Permission is hereby granted, free of charge, to any person +obtaining a copy of this software and associated documentation +files (the "Software"), to deal in the Software without +restriction, including without limitation the rights to use, +copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the software, and to permit persons to whom the +software is furnished to do so, subject to the following +conditions: + +The above copyright notice and this permission notice shall be +included in all copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES +OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND +NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT +HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR +OTHER DEALINGS IN THE SOFTWARE. + +============================================================================== +dynlink_cuviddec.h +============================================================================== +This copyright notice applies to this header file only: + +Copyright (c) 2010-2022 NVIDIA Corporation + +Permission is hereby granted, free of charge, to any person +obtaining a copy of this software and associated documentation +files (the "Software"), to deal in the Software without +restriction, including without limitation the rights to use, +copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the software, and to permit persons to whom the +software is furnished to do so, subject to the following +conditions: + +The above copyright notice and this permission notice shall be +included in all copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES +OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND +NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT +HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR +OTHER DEALINGS IN THE SOFTWARE. + +============================================================================== +dynlink_nvcuvid.h +============================================================================== +This copyright notice applies to this header file only: + +Copyright (c) 2010-2022 NVIDIA Corporation + +Permission is hereby granted, free of charge, to any person +obtaining a copy of this software and associated documentation +files (the "Software"), to deal in the Software without +restriction, including without limitation the rights to use, +copy, modify, merge, publish, distribute, sublicense, and/or sell +copies of the software, and to permit persons to whom the +software is furnished to do so, subject to the following +conditions: + +The above copyright notice and this permission notice shall be +included in all copies or substantial portions of the Software. + +THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, +EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES +OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND +NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT +HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, +WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING +FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR +OTHER DEALINGS IN THE SOFTWARE. diff --git a/webrtc-jni/src/main/cpp/dependencies/nvdec/README.md b/webrtc-jni/src/main/cpp/dependencies/nvdec/README.md new file mode 100644 index 00000000..f7fc2222 --- /dev/null +++ b/webrtc-jni/src/main/cpp/dependencies/nvdec/README.md @@ -0,0 +1,18 @@ +# NVDEC headers + +`include/ffnvcodec` holds three headers of FFmpeg's `nv-codec-headers` at tag `n12.0.16.1`, taken +unchanged: + +https://github.com/FFmpeg/nv-codec-headers/tree/n12.0.16.1/include/ffnvcodec + +- `dynlink_cuda.h`: the types and function signatures of the CUDA driver API that decoding needs. +- `dynlink_cuviddec.h` and `dynlink_nvcuvid.h`: the decoder and the bitstream parser of NVDEC + (`libnvcuvid`). + +The library loads the CUDA driver (`libcuda.so.1`) and `libnvcuvid.so.1` at run time, so the headers +are all the build needs, and a machine without an NVIDIA driver is unaffected. Version 12.0.16.1 is +the one the NVENC header (`../nvenc`) comes from. `webrtc-java-media` carries the same files for +FFmpeg's NVDEC hwaccels. + +`LICENSE` gathers the notices at the head of the files. It is installed into the Linux platform jars +under `META-INF/licenses/nvdec`. diff --git a/webrtc-jni/src/main/cpp/dependencies/nvdec/include/ffnvcodec/dynlink_cuda.h b/webrtc-jni/src/main/cpp/dependencies/nvdec/include/ffnvcodec/dynlink_cuda.h new file mode 100644 index 00000000..68e73320 --- /dev/null +++ b/webrtc-jni/src/main/cpp/dependencies/nvdec/include/ffnvcodec/dynlink_cuda.h @@ -0,0 +1,518 @@ +/* + * This copyright notice applies to this header file only: + * + * Copyright (c) 2016 + * + * Permission is hereby granted, free of charge, to any person + * obtaining a copy of this software and associated documentation + * files (the "Software"), to deal in the Software without + * restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the software, and to permit persons to whom the + * software is furnished to do so, subject to the following + * conditions: + * + * The above copyright notice and this permission notice shall be + * included in all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES + * OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND + * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT + * HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, + * WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING + * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR + * OTHER DEALINGS IN THE SOFTWARE. + */ + +#if !defined(FFNV_DYNLINK_CUDA_H) && !defined(CUDA_VERSION) +#define FFNV_DYNLINK_CUDA_H + +#include +#include + +#define CUDA_VERSION 7050 + +#if defined(_WIN32) || defined(__CYGWIN__) +#define CUDAAPI __stdcall +#else +#define CUDAAPI +#endif + +#define CU_CTX_SCHED_BLOCKING_SYNC 4 + +typedef int CUdevice; +#if defined(__x86_64) || defined(AMD64) || defined(_M_AMD64) || defined(__LP64__) || defined(__aarch64__) +typedef unsigned long long CUdeviceptr; +#else +typedef unsigned int CUdeviceptr; +#endif +typedef unsigned long long CUtexObject; + +typedef struct CUarray_st *CUarray; +typedef struct CUctx_st *CUcontext; +typedef struct CUstream_st *CUstream; +typedef struct CUevent_st *CUevent; +typedef struct CUfunc_st *CUfunction; +typedef struct CUmod_st *CUmodule; +typedef struct CUmipmappedArray_st *CUmipmappedArray; +typedef struct CUgraphicsResource_st *CUgraphicsResource; +typedef struct CUextMemory_st *CUexternalMemory; +typedef struct CUextSemaphore_st *CUexternalSemaphore; +typedef struct CUeglStreamConnection_st *CUeglStreamConnection; + +typedef struct CUlinkState_st *CUlinkState; + +typedef enum cudaError_enum { + CUDA_SUCCESS = 0, + CUDA_ERROR_NOT_READY = 600, + CUDA_ERROR_LAUNCH_TIMEOUT = 702, + CUDA_ERROR_UNKNOWN = 999 +} CUresult; + +/** + * Device properties (subset) + */ +typedef enum CUdevice_attribute_enum { + CU_DEVICE_ATTRIBUTE_CLOCK_RATE = 13, + CU_DEVICE_ATTRIBUTE_TEXTURE_ALIGNMENT = 14, + CU_DEVICE_ATTRIBUTE_MULTIPROCESSOR_COUNT = 16, + CU_DEVICE_ATTRIBUTE_INTEGRATED = 18, + CU_DEVICE_ATTRIBUTE_CAN_MAP_HOST_MEMORY = 19, + CU_DEVICE_ATTRIBUTE_COMPUTE_MODE = 20, + CU_DEVICE_ATTRIBUTE_CONCURRENT_KERNELS = 31, + CU_DEVICE_ATTRIBUTE_PCI_BUS_ID = 33, + CU_DEVICE_ATTRIBUTE_PCI_DEVICE_ID = 34, + CU_DEVICE_ATTRIBUTE_TCC_DRIVER = 35, + CU_DEVICE_ATTRIBUTE_MEMORY_CLOCK_RATE = 36, + CU_DEVICE_ATTRIBUTE_GLOBAL_MEMORY_BUS_WIDTH = 37, + CU_DEVICE_ATTRIBUTE_ASYNC_ENGINE_COUNT = 40, + CU_DEVICE_ATTRIBUTE_UNIFIED_ADDRESSING = 41, + CU_DEVICE_ATTRIBUTE_PCI_DOMAIN_ID = 50, + CU_DEVICE_ATTRIBUTE_TEXTURE_PITCH_ALIGNMENT = 51, + CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MAJOR = 75, + CU_DEVICE_ATTRIBUTE_COMPUTE_CAPABILITY_MINOR = 76, + CU_DEVICE_ATTRIBUTE_MANAGED_MEMORY = 83, + CU_DEVICE_ATTRIBUTE_MULTI_GPU_BOARD = 84, + CU_DEVICE_ATTRIBUTE_MULTI_GPU_BOARD_GROUP_ID = 85, +} CUdevice_attribute; + +typedef enum CUarray_format_enum { + CU_AD_FORMAT_UNSIGNED_INT8 = 0x01, + CU_AD_FORMAT_UNSIGNED_INT16 = 0x02, + CU_AD_FORMAT_UNSIGNED_INT32 = 0x03, + CU_AD_FORMAT_SIGNED_INT8 = 0x08, + CU_AD_FORMAT_SIGNED_INT16 = 0x09, + CU_AD_FORMAT_SIGNED_INT32 = 0x0a, + CU_AD_FORMAT_HALF = 0x10, + CU_AD_FORMAT_FLOAT = 0x20 +} CUarray_format; + +typedef enum CUmemorytype_enum { + CU_MEMORYTYPE_HOST = 1, + CU_MEMORYTYPE_DEVICE = 2, + CU_MEMORYTYPE_ARRAY = 3 +} CUmemorytype; + +typedef enum CUlimit_enum { + CU_LIMIT_STACK_SIZE = 0, + CU_LIMIT_PRINTF_FIFO_SIZE = 1, + CU_LIMIT_MALLOC_HEAP_SIZE = 2, + CU_LIMIT_DEV_RUNTIME_SYNC_DEPTH = 3, + CU_LIMIT_DEV_RUNTIME_PENDING_LAUNCH_COUNT = 4 +} CUlimit; + +typedef enum CUresourcetype_enum { + CU_RESOURCE_TYPE_ARRAY = 0x00, + CU_RESOURCE_TYPE_MIPMAPPED_ARRAY = 0x01, + CU_RESOURCE_TYPE_LINEAR = 0x02, + CU_RESOURCE_TYPE_PITCH2D = 0x03 +} CUresourcetype; + +typedef enum CUaddress_mode_enum { + CU_TR_ADDRESS_MODE_WRAP = 0, + CU_TR_ADDRESS_MODE_CLAMP = 1, + CU_TR_ADDRESS_MODE_MIRROR = 2, + CU_TR_ADDRESS_MODE_BORDER = 3 +} CUaddress_mode; + +typedef enum CUfilter_mode_enum { + CU_TR_FILTER_MODE_POINT = 0, + CU_TR_FILTER_MODE_LINEAR = 1 +} CUfilter_mode; + +typedef enum CUgraphicsRegisterFlags_enum { + CU_GRAPHICS_REGISTER_FLAGS_NONE = 0, + CU_GRAPHICS_REGISTER_FLAGS_READ_ONLY = 1, + CU_GRAPHICS_REGISTER_FLAGS_WRITE_DISCARD = 2, + CU_GRAPHICS_REGISTER_FLAGS_SURFACE_LDST = 4, + CU_GRAPHICS_REGISTER_FLAGS_TEXTURE_GATHER = 8 +} CUgraphicsRegisterFlags; + +typedef enum CUexternalMemoryHandleType_enum { + CU_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_FD = 1, + CU_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_WIN32 = 2, + CU_EXTERNAL_MEMORY_HANDLE_TYPE_OPAQUE_WIN32_KMT = 3, + CU_EXTERNAL_MEMORY_HANDLE_TYPE_D3D12_HEAP = 4, + CU_EXTERNAL_MEMORY_HANDLE_TYPE_D3D12_RESOURCE = 5, +} CUexternalMemoryHandleType; + +typedef enum CUexternalSemaphoreHandleType_enum { + CU_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_FD = 1, + CU_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_WIN32 = 2, + CU_EXTERNAL_SEMAPHORE_HANDLE_TYPE_OPAQUE_WIN32_KMT = 3, + CU_EXTERNAL_SEMAPHORE_HANDLE_TYPE_D3D12_FENCE = 4, + CU_EXTERNAL_SEMAPHORE_HANDLE_TYPE_TIMELINE_SEMAPHORE_FD = 9, + CU_EXTERNAL_SEMAPHORE_HANDLE_TYPE_TIMELINE_SEMAPHORE_WIN32 = 10, +} CUexternalSemaphoreHandleType; + +typedef enum CUjit_option_enum +{ + CU_JIT_MAX_REGISTERS = 0, + CU_JIT_THREADS_PER_BLOCK = 1, + CU_JIT_WALL_TIME = 2, + CU_JIT_INFO_LOG_BUFFER = 3, + CU_JIT_INFO_LOG_BUFFER_SIZE_BYTES = 4, + CU_JIT_ERROR_LOG_BUFFER = 5, + CU_JIT_ERROR_LOG_BUFFER_SIZE_BYTES = 6, + CU_JIT_OPTIMIZATION_LEVEL = 7, + CU_JIT_TARGET_FROM_CUCONTEXT = 8, + CU_JIT_TARGET = 9, + CU_JIT_FALLBACK_STRATEGY = 10, + CU_JIT_GENERATE_DEBUG_INFO = 11, + CU_JIT_LOG_VERBOSE = 12, + CU_JIT_GENERATE_LINE_INFO = 13, + CU_JIT_CACHE_MODE = 14, + CU_JIT_NEW_SM3X_OPT = 15, + CU_JIT_FAST_COMPILE = 16, + CU_JIT_GLOBAL_SYMBOL_NAMES = 17, + CU_JIT_GLOBAL_SYMBOL_ADDRESSES = 18, + CU_JIT_GLOBAL_SYMBOL_COUNT = 19, + CU_JIT_NUM_OPTIONS +} CUjit_option; + +typedef enum CUjitInputType_enum +{ + CU_JIT_INPUT_CUBIN = 0, + CU_JIT_INPUT_PTX = 1, + CU_JIT_INPUT_FATBINARY = 2, + CU_JIT_INPUT_OBJECT = 3, + CU_JIT_INPUT_LIBRARY = 4, + CU_JIT_NUM_INPUT_TYPES +} CUjitInputType; + +typedef enum CUeglFrameType +{ + CU_EGL_FRAME_TYPE_ARRAY = 0, + CU_EGL_FRAME_TYPE_PITCH = 1, +} CUeglFrameType; + +typedef enum CUeglColorFormat +{ + CU_EGL_COLOR_FORMAT_YUV420_PLANAR = 0x00, + CU_EGL_COLOR_FORMAT_YUV420_SEMIPLANAR = 0x01, + CU_EGL_COLOR_FORMAT_YVU420_SEMIPLANAR = 0x15, + CU_EGL_COLOR_FORMAT_Y10V10U10_420_SEMIPLANAR = 0x17, + CU_EGL_COLOR_FORMAT_Y12V12U12_420_SEMIPLANAR = 0x19, +} CUeglColorFormat; + +typedef enum CUd3d11DeviceList_enum +{ + CU_D3D11_DEVICE_LIST_ALL = 1, + CU_D3D11_DEVICE_LIST_CURRENT_FRAME = 2, + CU_D3D11_DEVICE_LIST_NEXT_FRAME = 3, +} CUd3d11DeviceList; + +#ifndef CU_UUID_HAS_BEEN_DEFINED +#define CU_UUID_HAS_BEEN_DEFINED +typedef struct CUuuid_st { + char bytes[16]; +} CUuuid; +#endif + +typedef struct CUDA_MEMCPY2D_st { + size_t srcXInBytes; + size_t srcY; + CUmemorytype srcMemoryType; + const void *srcHost; + CUdeviceptr srcDevice; + CUarray srcArray; + size_t srcPitch; + + size_t dstXInBytes; + size_t dstY; + CUmemorytype dstMemoryType; + void *dstHost; + CUdeviceptr dstDevice; + CUarray dstArray; + size_t dstPitch; + + size_t WidthInBytes; + size_t Height; +} CUDA_MEMCPY2D; + +typedef struct CUDA_RESOURCE_DESC_st { + CUresourcetype resType; + union { + struct { + CUarray hArray; + } array; + struct { + CUmipmappedArray hMipmappedArray; + } mipmap; + struct { + CUdeviceptr devPtr; + CUarray_format format; + unsigned int numChannels; + size_t sizeInBytes; + } linear; + struct { + CUdeviceptr devPtr; + CUarray_format format; + unsigned int numChannels; + size_t width; + size_t height; + size_t pitchInBytes; + } pitch2D; + struct { + int reserved[32]; + } reserved; + } res; + unsigned int flags; +} CUDA_RESOURCE_DESC; + +typedef struct CUDA_TEXTURE_DESC_st { + CUaddress_mode addressMode[3]; + CUfilter_mode filterMode; + unsigned int flags; + unsigned int maxAnisotropy; + CUfilter_mode mipmapFilterMode; + float mipmapLevelBias; + float minMipmapLevelClamp; + float maxMipmapLevelClamp; + float borderColor[4]; + int reserved[12]; +} CUDA_TEXTURE_DESC; + +/* Unused type */ +typedef struct CUDA_RESOURCE_VIEW_DESC_st CUDA_RESOURCE_VIEW_DESC; + +typedef unsigned int GLenum; +typedef unsigned int GLuint; +/* + * Prefix type name to avoid collisions. Clients using these types + * will include the real headers with real definitions. + */ +typedef int32_t ffnv_EGLint; +typedef void *ffnv_EGLStreamKHR; + +typedef enum CUGLDeviceList_enum { + CU_GL_DEVICE_LIST_ALL = 1, + CU_GL_DEVICE_LIST_CURRENT_FRAME = 2, + CU_GL_DEVICE_LIST_NEXT_FRAME = 3, +} CUGLDeviceList; + +typedef struct CUDA_EXTERNAL_MEMORY_HANDLE_DESC_st { + CUexternalMemoryHandleType type; + union { + int fd; + struct { + void *handle; + const void *name; + } win32; + } handle; + unsigned long long size; + unsigned int flags; + unsigned int reserved[16]; +} CUDA_EXTERNAL_MEMORY_HANDLE_DESC; + +typedef struct CUDA_EXTERNAL_MEMORY_BUFFER_DESC_st { + unsigned long long offset; + unsigned long long size; + unsigned int flags; + unsigned int reserved[16]; +} CUDA_EXTERNAL_MEMORY_BUFFER_DESC; + +typedef struct CUDA_EXTERNAL_SEMAPHORE_HANDLE_DESC_st { + CUexternalSemaphoreHandleType type; + union { + int fd; + struct { + void *handle; + const void *name; + } win32; + } handle; + unsigned int flags; + unsigned int reserved[16]; +} CUDA_EXTERNAL_SEMAPHORE_HANDLE_DESC; + +typedef struct CUDA_EXTERNAL_SEMAPHORE_SIGNAL_PARAMS_st { + struct { + struct { + unsigned long long value; + } fence; + unsigned int reserved[16]; + } params; + unsigned int flags; + unsigned int reserved[16]; +} CUDA_EXTERNAL_SEMAPHORE_SIGNAL_PARAMS; + +typedef CUDA_EXTERNAL_SEMAPHORE_SIGNAL_PARAMS CUDA_EXTERNAL_SEMAPHORE_WAIT_PARAMS; + +typedef struct CUDA_ARRAY_DESCRIPTOR_st { + size_t Width; + size_t Height; + + CUarray_format Format; + unsigned int NumChannels; +} CUDA_ARRAY_DESCRIPTOR; + +typedef struct CUDA_ARRAY3D_DESCRIPTOR_st { + size_t Width; + size_t Height; + size_t Depth; + + CUarray_format Format; + unsigned int NumChannels; + unsigned int Flags; +} CUDA_ARRAY3D_DESCRIPTOR; + +typedef struct CUDA_EXTERNAL_MEMORY_MIPMAPPED_ARRAY_DESC_st { + unsigned long long offset; + CUDA_ARRAY3D_DESCRIPTOR arrayDesc; + unsigned int numLevels; + unsigned int reserved[16]; +} CUDA_EXTERNAL_MEMORY_MIPMAPPED_ARRAY_DESC; + +#define CU_EGL_FRAME_MAX_PLANES 3 +typedef struct CUeglFrame_st { + union { + CUarray pArray[CU_EGL_FRAME_MAX_PLANES]; + void* pPitch[CU_EGL_FRAME_MAX_PLANES]; + } frame; + unsigned int width; + unsigned int height; + unsigned int depth; + unsigned int pitch; + unsigned int planeCount; + unsigned int numChannels; + CUeglFrameType frameType; + CUeglColorFormat eglColorFormat; + CUarray_format cuFormat; +} CUeglFrame; + +#define CU_STREAM_DEFAULT 0 +#define CU_STREAM_NON_BLOCKING 1 + +#define CU_EVENT_DEFAULT 0 +#define CU_EVENT_BLOCKING_SYNC 1 +#define CU_EVENT_DISABLE_TIMING 2 + +#define CU_EVENT_WAIT_DEFAULT 0 +#define CU_EVENT_WAIT_EXTERNAL 1 + +#define CU_TRSF_READ_AS_INTEGER 1 + +typedef void CUDAAPI CUstreamCallback(CUstream hStream, CUresult status, void *userdata); + +typedef CUresult CUDAAPI tcuInit(unsigned int Flags); +typedef CUresult CUDAAPI tcuDriverGetVersion(int *driverVersion); +typedef CUresult CUDAAPI tcuDeviceGetCount(int *count); +typedef CUresult CUDAAPI tcuDeviceGet(CUdevice *device, int ordinal); +typedef CUresult CUDAAPI tcuDeviceGetAttribute(int *pi, CUdevice_attribute attrib, CUdevice dev); +typedef CUresult CUDAAPI tcuDeviceGetName(char *name, int len, CUdevice dev); +typedef CUresult CUDAAPI tcuDeviceGetUuid(CUuuid *uuid, CUdevice dev); +typedef CUresult CUDAAPI tcuDeviceGetUuid_v2(CUuuid *uuid, CUdevice dev); +typedef CUresult CUDAAPI tcuDeviceGetLuid(char* luid, unsigned int* deviceNodeMask, CUdevice dev); +typedef CUresult CUDAAPI tcuDeviceGetByPCIBusId(CUdevice* dev, const char* pciBusId); +typedef CUresult CUDAAPI tcuDeviceGetPCIBusId(char* pciBusId, int len, CUdevice dev); +typedef CUresult CUDAAPI tcuDeviceComputeCapability(int *major, int *minor, CUdevice dev); +typedef CUresult CUDAAPI tcuCtxCreate_v2(CUcontext *pctx, unsigned int flags, CUdevice dev); +typedef CUresult CUDAAPI tcuCtxGetCurrent(CUcontext *pctx); +typedef CUresult CUDAAPI tcuCtxSetLimit(CUlimit limit, size_t value); +typedef CUresult CUDAAPI tcuCtxPushCurrent_v2(CUcontext pctx); +typedef CUresult CUDAAPI tcuCtxPopCurrent_v2(CUcontext *pctx); +typedef CUresult CUDAAPI tcuCtxDestroy_v2(CUcontext ctx); +typedef CUresult CUDAAPI tcuMemAlloc_v2(CUdeviceptr *dptr, size_t bytesize); +typedef CUresult CUDAAPI tcuMemAllocPitch_v2(CUdeviceptr *dptr, size_t *pPitch, size_t WidthInBytes, size_t Height, unsigned int ElementSizeBytes); +typedef CUresult CUDAAPI tcuMemAllocManaged(CUdeviceptr *dptr, size_t bytesize, unsigned int flags); +typedef CUresult CUDAAPI tcuMemsetD8Async(CUdeviceptr dstDevice, unsigned char uc, size_t N, CUstream hStream); +typedef CUresult CUDAAPI tcuMemFree_v2(CUdeviceptr dptr); +typedef CUresult CUDAAPI tcuMemcpy(CUdeviceptr dst, CUdeviceptr src, size_t bytesize); +typedef CUresult CUDAAPI tcuMemcpyAsync(CUdeviceptr dst, CUdeviceptr src, size_t bytesize, CUstream hStream); +typedef CUresult CUDAAPI tcuMemcpy2D_v2(const CUDA_MEMCPY2D *pcopy); +typedef CUresult CUDAAPI tcuMemcpy2DAsync_v2(const CUDA_MEMCPY2D *pcopy, CUstream hStream); +typedef CUresult CUDAAPI tcuMemcpyHtoD_v2(CUdeviceptr dstDevice, const void *srcHost, size_t ByteCount); +typedef CUresult CUDAAPI tcuMemcpyHtoDAsync_v2(CUdeviceptr dstDevice, const void *srcHost, size_t ByteCount, CUstream hStream); +typedef CUresult CUDAAPI tcuMemcpyDtoH_v2(void *dstHost, CUdeviceptr srcDevice, size_t ByteCount); +typedef CUresult CUDAAPI tcuMemcpyDtoHAsync_v2(void *dstHost, CUdeviceptr srcDevice, size_t ByteCount, CUstream hStream); +typedef CUresult CUDAAPI tcuMemcpyDtoD_v2(CUdeviceptr dstDevice, CUdeviceptr srcDevice, size_t ByteCount); +typedef CUresult CUDAAPI tcuMemcpyDtoDAsync_v2(CUdeviceptr dstDevice, CUdeviceptr srcDevice, size_t ByteCount, CUstream hStream); +typedef CUresult CUDAAPI tcuGetErrorName(CUresult error, const char** pstr); +typedef CUresult CUDAAPI tcuGetErrorString(CUresult error, const char** pstr); +typedef CUresult CUDAAPI tcuCtxGetDevice(CUdevice *device); + +typedef CUresult CUDAAPI tcuDevicePrimaryCtxRetain(CUcontext *pctx, CUdevice dev); +typedef CUresult CUDAAPI tcuDevicePrimaryCtxRelease(CUdevice dev); +typedef CUresult CUDAAPI tcuDevicePrimaryCtxSetFlags(CUdevice dev, unsigned int flags); +typedef CUresult CUDAAPI tcuDevicePrimaryCtxGetState(CUdevice dev, unsigned int *flags, int *active); +typedef CUresult CUDAAPI tcuDevicePrimaryCtxReset(CUdevice dev); + +typedef CUresult CUDAAPI tcuStreamCreate(CUstream *phStream, unsigned int flags); +typedef CUresult CUDAAPI tcuStreamQuery(CUstream hStream); +typedef CUresult CUDAAPI tcuStreamSynchronize(CUstream hStream); +typedef CUresult CUDAAPI tcuStreamDestroy_v2(CUstream hStream); +typedef CUresult CUDAAPI tcuStreamAddCallback(CUstream hStream, CUstreamCallback *callback, void *userdata, unsigned int flags); +typedef CUresult CUDAAPI tcuStreamWaitEvent(CUstream hStream, CUevent hEvent, unsigned int flags); +typedef CUresult CUDAAPI tcuEventCreate(CUevent *phEvent, unsigned int flags); +typedef CUresult CUDAAPI tcuEventDestroy_v2(CUevent hEvent); +typedef CUresult CUDAAPI tcuEventSynchronize(CUevent hEvent); +typedef CUresult CUDAAPI tcuEventQuery(CUevent hEvent); +typedef CUresult CUDAAPI tcuEventRecord(CUevent hEvent, CUstream hStream); + +typedef CUresult CUDAAPI tcuLaunchKernel(CUfunction f, unsigned int gridDimX, unsigned int gridDimY, unsigned int gridDimZ, unsigned int blockDimX, unsigned int blockDimY, unsigned int blockDimZ, unsigned int sharedMemBytes, CUstream hStream, void** kernelParams, void** extra); +typedef CUresult CUDAAPI tcuLinkCreate(unsigned int numOptions, CUjit_option* options, void** optionValues, CUlinkState* stateOut); +typedef CUresult CUDAAPI tcuLinkAddData(CUlinkState state, CUjitInputType type, void* data, size_t size, const char* name, unsigned int numOptions, CUjit_option* options, void** optionValues); +typedef CUresult CUDAAPI tcuLinkComplete(CUlinkState state, void** cubinOut, size_t* sizeOut); +typedef CUresult CUDAAPI tcuLinkDestroy(CUlinkState state); +typedef CUresult CUDAAPI tcuModuleLoadData(CUmodule* module, const void* image); +typedef CUresult CUDAAPI tcuModuleUnload(CUmodule hmod); +typedef CUresult CUDAAPI tcuModuleGetFunction(CUfunction* hfunc, CUmodule hmod, const char* name); +typedef CUresult CUDAAPI tcuModuleGetGlobal(CUdeviceptr *dptr, size_t *bytes, CUmodule hmod, const char* name); +typedef CUresult CUDAAPI tcuTexObjectCreate(CUtexObject* pTexObject, const CUDA_RESOURCE_DESC* pResDesc, const CUDA_TEXTURE_DESC* pTexDesc, const CUDA_RESOURCE_VIEW_DESC* pResViewDesc); +typedef CUresult CUDAAPI tcuTexObjectDestroy(CUtexObject texObject); + +typedef CUresult CUDAAPI tcuGLGetDevices_v2(unsigned int* pCudaDeviceCount, CUdevice* pCudaDevices, unsigned int cudaDeviceCount, CUGLDeviceList deviceList); +typedef CUresult CUDAAPI tcuGraphicsGLRegisterImage(CUgraphicsResource* pCudaResource, GLuint image, GLenum target, unsigned int Flags); +typedef CUresult CUDAAPI tcuGraphicsUnregisterResource(CUgraphicsResource resource); +typedef CUresult CUDAAPI tcuGraphicsMapResources(unsigned int count, CUgraphicsResource* resources, CUstream hStream); +typedef CUresult CUDAAPI tcuGraphicsUnmapResources(unsigned int count, CUgraphicsResource* resources, CUstream hStream); +typedef CUresult CUDAAPI tcuGraphicsSubResourceGetMappedArray(CUarray* pArray, CUgraphicsResource resource, unsigned int arrayIndex, unsigned int mipLevel); +typedef CUresult CUDAAPI tcuGraphicsResourceGetMappedPointer(CUdeviceptr *devPtrOut, size_t *sizeOut, CUgraphicsResource resource); + +typedef CUresult CUDAAPI tcuImportExternalMemory(CUexternalMemory* extMem_out, const CUDA_EXTERNAL_MEMORY_HANDLE_DESC* memHandleDesc); +typedef CUresult CUDAAPI tcuDestroyExternalMemory(CUexternalMemory extMem); +typedef CUresult CUDAAPI tcuExternalMemoryGetMappedBuffer(CUdeviceptr* devPtr, CUexternalMemory extMem, const CUDA_EXTERNAL_MEMORY_BUFFER_DESC* bufferDesc); +typedef CUresult CUDAAPI tcuExternalMemoryGetMappedMipmappedArray(CUmipmappedArray* mipmap, CUexternalMemory extMem, const CUDA_EXTERNAL_MEMORY_MIPMAPPED_ARRAY_DESC* mipmapDesc); +typedef CUresult CUDAAPI tcuMipmappedArrayGetLevel(CUarray* pLevelArray, CUmipmappedArray hMipmappedArray, unsigned int level); +typedef CUresult CUDAAPI tcuMipmappedArrayDestroy(CUmipmappedArray hMipmappedArray); + +typedef CUresult CUDAAPI tcuImportExternalSemaphore(CUexternalSemaphore* extSem_out, const CUDA_EXTERNAL_SEMAPHORE_HANDLE_DESC* semHandleDesc); +typedef CUresult CUDAAPI tcuDestroyExternalSemaphore(CUexternalSemaphore extSem); +typedef CUresult CUDAAPI tcuSignalExternalSemaphoresAsync(const CUexternalSemaphore* extSemArray, const CUDA_EXTERNAL_SEMAPHORE_SIGNAL_PARAMS* paramsArray, unsigned int numExtSems, CUstream stream); +typedef CUresult CUDAAPI tcuWaitExternalSemaphoresAsync(const CUexternalSemaphore* extSemArray, const CUDA_EXTERNAL_SEMAPHORE_WAIT_PARAMS* paramsArray, unsigned int numExtSems, CUstream stream); + +typedef CUresult CUDAAPI tcuArrayCreate(CUarray *pHandle, const CUDA_ARRAY_DESCRIPTOR* pAllocateArray); +typedef CUresult CUDAAPI tcuArray3DCreate(CUarray *pHandle, const CUDA_ARRAY3D_DESCRIPTOR* pAllocateArray); +typedef CUresult CUDAAPI tcuArrayDestroy(CUarray hArray); + +typedef CUresult CUDAAPI tcuEGLStreamProducerConnect(CUeglStreamConnection* conn, ffnv_EGLStreamKHR stream, ffnv_EGLint width, ffnv_EGLint height); +typedef CUresult CUDAAPI tcuEGLStreamProducerDisconnect(CUeglStreamConnection* conn); +typedef CUresult CUDAAPI tcuEGLStreamConsumerDisconnect(CUeglStreamConnection* conn); +typedef CUresult CUDAAPI tcuEGLStreamProducerPresentFrame(CUeglStreamConnection* conn, CUeglFrame eglframe, CUstream* pStream); +typedef CUresult CUDAAPI tcuEGLStreamProducerReturnFrame(CUeglStreamConnection* conn, CUeglFrame* eglframe, CUstream* pStream); + +typedef CUresult CUDAAPI tcuD3D11GetDevice(CUdevice *device, void *dxgiAdapter); +typedef CUresult CUDAAPI tcuD3D11GetDevices(unsigned int *deviceCountOut, CUdevice *devices, unsigned int deviceCount, void *d3d11device, CUd3d11DeviceList listType); +typedef CUresult CUDAAPI tcuGraphicsD3D11RegisterResource(CUgraphicsResource *cudaResourceOut, void *d3d11Resource, unsigned int flags); +#endif diff --git a/webrtc-jni/src/main/cpp/dependencies/nvdec/include/ffnvcodec/dynlink_cuviddec.h b/webrtc-jni/src/main/cpp/dependencies/nvdec/include/ffnvcodec/dynlink_cuviddec.h new file mode 100644 index 00000000..d60423e3 --- /dev/null +++ b/webrtc-jni/src/main/cpp/dependencies/nvdec/include/ffnvcodec/dynlink_cuviddec.h @@ -0,0 +1,1184 @@ +/* + * This copyright notice applies to this header file only: + * + * Copyright (c) 2010-2022 NVIDIA Corporation + * + * Permission is hereby granted, free of charge, to any person + * obtaining a copy of this software and associated documentation + * files (the "Software"), to deal in the Software without + * restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the software, and to permit persons to whom the + * software is furnished to do so, subject to the following + * conditions: + * + * The above copyright notice and this permission notice shall be + * included in all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES + * OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND + * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT + * HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, + * WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING + * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR + * OTHER DEALINGS IN THE SOFTWARE. + */ + +/*****************************************************************************************************/ +//! \file cuviddec.h +//! NVDECODE API provides video decoding interface to NVIDIA GPU devices. +//! This file contains constants, structure definitions and function prototypes used for decoding. +/*****************************************************************************************************/ + +#if !defined(__CUDA_VIDEO_H__) +#define __CUDA_VIDEO_H__ + +#if defined(_WIN64) || defined(__LP64__) || defined(__x86_64) || defined(AMD64) || defined(_M_AMD64) +#if (CUDA_VERSION >= 3020) && (!defined(CUDA_FORCE_API_VERSION) || (CUDA_FORCE_API_VERSION >= 3020)) +#define __CUVID_DEVPTR64 +#endif +#endif + +#define NVDECAPI_MAJOR_VERSION 12 +#define NVDECAPI_MINOR_VERSION 0 + +#define NVDECAPI_VERSION (NVDECAPI_MAJOR_VERSION | (NVDECAPI_MINOR_VERSION << 24)) + +#if defined(__cplusplus) +extern "C" { +#endif /* __cplusplus */ + +#if defined(__CYGWIN__) +typedef unsigned int tcu_ulong; +#else +typedef unsigned long tcu_ulong; +#endif + +typedef void *CUvideodecoder; +typedef struct _CUcontextlock_st *CUvideoctxlock; + +/*********************************************************************************/ +//! \enum cudaVideoCodec +//! Video codec enums +//! These enums are used in CUVIDDECODECREATEINFO and CUVIDDECODECAPS structures +/*********************************************************************************/ +typedef enum cudaVideoCodec_enum { + cudaVideoCodec_MPEG1=0, /**< MPEG1 */ + cudaVideoCodec_MPEG2, /**< MPEG2 */ + cudaVideoCodec_MPEG4, /**< MPEG4 */ + cudaVideoCodec_VC1, /**< VC1 */ + cudaVideoCodec_H264, /**< H264 */ + cudaVideoCodec_JPEG, /**< JPEG */ + cudaVideoCodec_H264_SVC, /**< H264-SVC */ + cudaVideoCodec_H264_MVC, /**< H264-MVC */ + cudaVideoCodec_HEVC, /**< HEVC */ + cudaVideoCodec_VP8, /**< VP8 */ + cudaVideoCodec_VP9, /**< VP9 */ + cudaVideoCodec_AV1, /**< AV1 */ + cudaVideoCodec_NumCodecs, /**< Max codecs */ + // Uncompressed YUV + cudaVideoCodec_YUV420 = (('I'<<24)|('Y'<<16)|('U'<<8)|('V')), /**< Y,U,V (4:2:0) */ + cudaVideoCodec_YV12 = (('Y'<<24)|('V'<<16)|('1'<<8)|('2')), /**< Y,V,U (4:2:0) */ + cudaVideoCodec_NV12 = (('N'<<24)|('V'<<16)|('1'<<8)|('2')), /**< Y,UV (4:2:0) */ + cudaVideoCodec_YUYV = (('Y'<<24)|('U'<<16)|('Y'<<8)|('V')), /**< YUYV/YUY2 (4:2:2) */ + cudaVideoCodec_UYVY = (('U'<<24)|('Y'<<16)|('V'<<8)|('Y')) /**< UYVY (4:2:2) */ +} cudaVideoCodec; + +/*********************************************************************************/ +//! \enum cudaVideoSurfaceFormat +//! Video surface format enums used for output format of decoded output +//! These enums are used in CUVIDDECODECREATEINFO structure +/*********************************************************************************/ +typedef enum cudaVideoSurfaceFormat_enum { + cudaVideoSurfaceFormat_NV12=0, /**< Semi-Planar YUV [Y plane followed by interleaved UV plane] */ + cudaVideoSurfaceFormat_P016=1, /**< 16 bit Semi-Planar YUV [Y plane followed by interleaved UV plane]. + Can be used for 10 bit(6LSB bits 0), 12 bit (4LSB bits 0) */ + cudaVideoSurfaceFormat_YUV444=2, /**< Planar YUV [Y plane followed by U and V planes] */ + cudaVideoSurfaceFormat_YUV444_16Bit=3, /**< 16 bit Planar YUV [Y plane followed by U and V planes]. + Can be used for 10 bit(6LSB bits 0), 12 bit (4LSB bits 0) */ +} cudaVideoSurfaceFormat; + +/******************************************************************************************************************/ +//! \enum cudaVideoDeinterlaceMode +//! Deinterlacing mode enums +//! These enums are used in CUVIDDECODECREATEINFO structure +//! Use cudaVideoDeinterlaceMode_Weave for progressive content and for content that doesn't need deinterlacing +//! cudaVideoDeinterlaceMode_Adaptive needs more video memory than other DImodes +/******************************************************************************************************************/ +typedef enum cudaVideoDeinterlaceMode_enum { + cudaVideoDeinterlaceMode_Weave=0, /**< Weave both fields (no deinterlacing) */ + cudaVideoDeinterlaceMode_Bob, /**< Drop one field */ + cudaVideoDeinterlaceMode_Adaptive /**< Adaptive deinterlacing */ +} cudaVideoDeinterlaceMode; + +/**************************************************************************************************************/ +//! \enum cudaVideoChromaFormat +//! Chroma format enums +//! These enums are used in CUVIDDECODECREATEINFO and CUVIDDECODECAPS structures +/**************************************************************************************************************/ +typedef enum cudaVideoChromaFormat_enum { + cudaVideoChromaFormat_Monochrome=0, /**< MonoChrome */ + cudaVideoChromaFormat_420, /**< YUV 4:2:0 */ + cudaVideoChromaFormat_422, /**< YUV 4:2:2 */ + cudaVideoChromaFormat_444 /**< YUV 4:4:4 */ +} cudaVideoChromaFormat; + +/*************************************************************************************************************/ +//! \enum cudaVideoCreateFlags +//! Decoder flag enums to select preferred decode path +//! cudaVideoCreate_Default and cudaVideoCreate_PreferCUVID are most optimized, use these whenever possible +/*************************************************************************************************************/ +typedef enum cudaVideoCreateFlags_enum { + cudaVideoCreate_Default = 0x00, /**< Default operation mode: use dedicated video engines */ + cudaVideoCreate_PreferCUDA = 0x01, /**< Use CUDA-based decoder (requires valid vidLock object for multi-threading) */ + cudaVideoCreate_PreferDXVA = 0x02, /**< Go through DXVA internally if possible (requires D3D9 interop) */ + cudaVideoCreate_PreferCUVID = 0x04 /**< Use dedicated video engines directly */ +} cudaVideoCreateFlags; + + +/*************************************************************************/ +//! \enum cuvidDecodeStatus +//! Decode status enums +//! These enums are used in CUVIDGETDECODESTATUS structure +/*************************************************************************/ +typedef enum cuvidDecodeStatus_enum +{ + cuvidDecodeStatus_Invalid = 0, // Decode status is not valid + cuvidDecodeStatus_InProgress = 1, // Decode is in progress + cuvidDecodeStatus_Success = 2, // Decode is completed without any errors + // 3 to 7 enums are reserved for future use + cuvidDecodeStatus_Error = 8, // Decode is completed with an error (error is not concealed) + cuvidDecodeStatus_Error_Concealed = 9, // Decode is completed with an error and error is concealed +} cuvidDecodeStatus; + +/**************************************************************************************************************/ +//! \struct CUVIDDECODECAPS; +//! This structure is used in cuvidGetDecoderCaps API +/**************************************************************************************************************/ +typedef struct _CUVIDDECODECAPS +{ + cudaVideoCodec eCodecType; /**< IN: cudaVideoCodec_XXX */ + cudaVideoChromaFormat eChromaFormat; /**< IN: cudaVideoChromaFormat_XXX */ + unsigned int nBitDepthMinus8; /**< IN: The Value "BitDepth minus 8" */ + unsigned int reserved1[3]; /**< Reserved for future use - set to zero */ + + unsigned char bIsSupported; /**< OUT: 1 if codec supported, 0 if not supported */ + unsigned char nNumNVDECs; /**< OUT: Number of NVDECs that can support IN params */ + unsigned short nOutputFormatMask; /**< OUT: each bit represents corresponding cudaVideoSurfaceFormat enum */ + unsigned int nMaxWidth; /**< OUT: Max supported coded width in pixels */ + unsigned int nMaxHeight; /**< OUT: Max supported coded height in pixels */ + unsigned int nMaxMBCount; /**< OUT: Max supported macroblock count + CodedWidth*CodedHeight/256 must be <= nMaxMBCount */ + unsigned short nMinWidth; /**< OUT: Min supported coded width in pixels */ + unsigned short nMinHeight; /**< OUT: Min supported coded height in pixels */ + unsigned char bIsHistogramSupported; /**< OUT: 1 if Y component histogram output is supported, 0 if not + Note: histogram is computed on original picture data before + any post-processing like scaling, cropping, etc. is applied */ + unsigned char nCounterBitDepth; /**< OUT: histogram counter bit depth */ + unsigned short nMaxHistogramBins; /**< OUT: Max number of histogram bins */ + unsigned int reserved3[10]; /**< Reserved for future use - set to zero */ +} CUVIDDECODECAPS; + +/**************************************************************************************************************/ +//! \struct CUVIDDECODECREATEINFO +//! This structure is used in cuvidCreateDecoder API +/**************************************************************************************************************/ +typedef struct _CUVIDDECODECREATEINFO +{ + tcu_ulong ulWidth; /**< IN: Coded sequence width in pixels */ + tcu_ulong ulHeight; /**< IN: Coded sequence height in pixels */ + tcu_ulong ulNumDecodeSurfaces; /**< IN: Maximum number of internal decode surfaces */ + cudaVideoCodec CodecType; /**< IN: cudaVideoCodec_XXX */ + cudaVideoChromaFormat ChromaFormat; /**< IN: cudaVideoChromaFormat_XXX */ + tcu_ulong ulCreationFlags; /**< IN: Decoder creation flags (cudaVideoCreateFlags_XXX) */ + tcu_ulong bitDepthMinus8; /**< IN: The value "BitDepth minus 8" */ + tcu_ulong ulIntraDecodeOnly; /**< IN: Set 1 only if video has all intra frames (default value is 0). This will + optimize video memory for Intra frames only decoding. The support is limited + to specific codecs - H264, HEVC, VP9, the flag will be ignored for codecs which + are not supported. However decoding might fail if the flag is enabled in case + of supported codecs for regular bit streams having P and/or B frames. */ + tcu_ulong ulMaxWidth; /**< IN: Coded sequence max width in pixels used with reconfigure Decoder */ + tcu_ulong ulMaxHeight; /**< IN: Coded sequence max height in pixels used with reconfigure Decoder */ + tcu_ulong Reserved1; /**< Reserved for future use - set to zero */ + /** + * IN: area of the frame that should be displayed + */ + struct { + short left; + short top; + short right; + short bottom; + } display_area; + + cudaVideoSurfaceFormat OutputFormat; /**< IN: cudaVideoSurfaceFormat_XXX */ + cudaVideoDeinterlaceMode DeinterlaceMode; /**< IN: cudaVideoDeinterlaceMode_XXX */ + tcu_ulong ulTargetWidth; /**< IN: Post-processed output width (Should be aligned to 2) */ + tcu_ulong ulTargetHeight; /**< IN: Post-processed output height (Should be aligned to 2) */ + tcu_ulong ulNumOutputSurfaces; /**< IN: Maximum number of output surfaces simultaneously mapped */ + CUvideoctxlock vidLock; /**< IN: If non-NULL, context lock used for synchronizing ownership of + the cuda context. Needed for cudaVideoCreate_PreferCUDA decode */ + /** + * IN: target rectangle in the output frame (for aspect ratio conversion) + * if a null rectangle is specified, {0,0,ulTargetWidth,ulTargetHeight} will be used + */ + struct { + short left; + short top; + short right; + short bottom; + } target_rect; + + tcu_ulong enableHistogram; /**< IN: enable histogram output, if supported */ + tcu_ulong Reserved2[4]; /**< Reserved for future use - set to zero */ +} CUVIDDECODECREATEINFO; + +/*********************************************************/ +//! \struct CUVIDH264DPBENTRY +//! H.264 DPB entry +//! This structure is used in CUVIDH264PICPARAMS structure +/*********************************************************/ +typedef struct _CUVIDH264DPBENTRY +{ + int PicIdx; /**< picture index of reference frame */ + int FrameIdx; /**< frame_num(short-term) or LongTermFrameIdx(long-term) */ + int is_long_term; /**< 0=short term reference, 1=long term reference */ + int not_existing; /**< non-existing reference frame (corresponding PicIdx should be set to -1) */ + int used_for_reference; /**< 0=unused, 1=top_field, 2=bottom_field, 3=both_fields */ + int FieldOrderCnt[2]; /**< field order count of top and bottom fields */ +} CUVIDH264DPBENTRY; + +/************************************************************/ +//! \struct CUVIDH264MVCEXT +//! H.264 MVC picture parameters ext +//! This structure is used in CUVIDH264PICPARAMS structure +/************************************************************/ +typedef struct _CUVIDH264MVCEXT +{ + int num_views_minus1; /**< Max number of coded views minus 1 in video : Range - 0 to 1023 */ + int view_id; /**< view identifier */ + unsigned char inter_view_flag; /**< 1 if used for inter-view prediction, 0 if not */ + unsigned char num_inter_view_refs_l0; /**< number of inter-view ref pics in RefPicList0 */ + unsigned char num_inter_view_refs_l1; /**< number of inter-view ref pics in RefPicList1 */ + unsigned char MVCReserved8Bits; /**< Reserved bits */ + int InterViewRefsL0[16]; /**< view id of the i-th view component for inter-view prediction in RefPicList0 */ + int InterViewRefsL1[16]; /**< view id of the i-th view component for inter-view prediction in RefPicList1 */ +} CUVIDH264MVCEXT; + +/*********************************************************/ +//! \struct CUVIDH264SVCEXT +//! H.264 SVC picture parameters ext +//! This structure is used in CUVIDH264PICPARAMS structure +/*********************************************************/ +typedef struct _CUVIDH264SVCEXT +{ + unsigned char profile_idc; + unsigned char level_idc; + unsigned char DQId; + unsigned char DQIdMax; + unsigned char disable_inter_layer_deblocking_filter_idc; + unsigned char ref_layer_chroma_phase_y_plus1; + signed char inter_layer_slice_alpha_c0_offset_div2; + signed char inter_layer_slice_beta_offset_div2; + + unsigned short DPBEntryValidFlag; + unsigned char inter_layer_deblocking_filter_control_present_flag; + unsigned char extended_spatial_scalability_idc; + unsigned char adaptive_tcoeff_level_prediction_flag; + unsigned char slice_header_restriction_flag; + unsigned char chroma_phase_x_plus1_flag; + unsigned char chroma_phase_y_plus1; + + unsigned char tcoeff_level_prediction_flag; + unsigned char constrained_intra_resampling_flag; + unsigned char ref_layer_chroma_phase_x_plus1_flag; + unsigned char store_ref_base_pic_flag; + unsigned char Reserved8BitsA; + unsigned char Reserved8BitsB; + + short scaled_ref_layer_left_offset; + short scaled_ref_layer_top_offset; + short scaled_ref_layer_right_offset; + short scaled_ref_layer_bottom_offset; + unsigned short Reserved16Bits; + struct _CUVIDPICPARAMS *pNextLayer; /**< Points to the picparams for the next layer to be decoded. + Linked list ends at the target layer. */ + int bRefBaseLayer; /**< whether to store ref base pic */ +} CUVIDH264SVCEXT; + +/******************************************************/ +//! \struct CUVIDH264PICPARAMS +//! H.264 picture parameters +//! This structure is used in CUVIDPICPARAMS structure +/******************************************************/ +typedef struct _CUVIDH264PICPARAMS +{ + // SPS + int log2_max_frame_num_minus4; + int pic_order_cnt_type; + int log2_max_pic_order_cnt_lsb_minus4; + int delta_pic_order_always_zero_flag; + int frame_mbs_only_flag; + int direct_8x8_inference_flag; + int num_ref_frames; // NOTE: shall meet level 4.1 restrictions + unsigned char residual_colour_transform_flag; + unsigned char bit_depth_luma_minus8; // Must be 0 (only 8-bit supported) + unsigned char bit_depth_chroma_minus8; // Must be 0 (only 8-bit supported) + unsigned char qpprime_y_zero_transform_bypass_flag; + // PPS + int entropy_coding_mode_flag; + int pic_order_present_flag; + int num_ref_idx_l0_active_minus1; + int num_ref_idx_l1_active_minus1; + int weighted_pred_flag; + int weighted_bipred_idc; + int pic_init_qp_minus26; + int deblocking_filter_control_present_flag; + int redundant_pic_cnt_present_flag; + int transform_8x8_mode_flag; + int MbaffFrameFlag; + int constrained_intra_pred_flag; + int chroma_qp_index_offset; + int second_chroma_qp_index_offset; + int ref_pic_flag; + int frame_num; + int CurrFieldOrderCnt[2]; + // DPB + CUVIDH264DPBENTRY dpb[16]; // List of reference frames within the DPB + // Quantization Matrices (raster-order) + unsigned char WeightScale4x4[6][16]; + unsigned char WeightScale8x8[2][64]; + // FMO/ASO + unsigned char fmo_aso_enable; + unsigned char num_slice_groups_minus1; + unsigned char slice_group_map_type; + signed char pic_init_qs_minus26; + unsigned int slice_group_change_rate_minus1; + union + { + unsigned long long slice_group_map_addr; + const unsigned char *pMb2SliceGroupMap; + } fmo; + unsigned int Reserved[12]; + // SVC/MVC + union + { + CUVIDH264MVCEXT mvcext; + CUVIDH264SVCEXT svcext; + }; +} CUVIDH264PICPARAMS; + + +/********************************************************/ +//! \struct CUVIDMPEG2PICPARAMS +//! MPEG-2 picture parameters +//! This structure is used in CUVIDPICPARAMS structure +/********************************************************/ +typedef struct _CUVIDMPEG2PICPARAMS +{ + int ForwardRefIdx; // Picture index of forward reference (P/B-frames) + int BackwardRefIdx; // Picture index of backward reference (B-frames) + int picture_coding_type; + int full_pel_forward_vector; + int full_pel_backward_vector; + int f_code[2][2]; + int intra_dc_precision; + int frame_pred_frame_dct; + int concealment_motion_vectors; + int q_scale_type; + int intra_vlc_format; + int alternate_scan; + int top_field_first; + // Quantization matrices (raster order) + unsigned char QuantMatrixIntra[64]; + unsigned char QuantMatrixInter[64]; +} CUVIDMPEG2PICPARAMS; + +// MPEG-4 has VOP types instead of Picture types +#define I_VOP 0 +#define P_VOP 1 +#define B_VOP 2 +#define S_VOP 3 + +/*******************************************************/ +//! \struct CUVIDMPEG4PICPARAMS +//! MPEG-4 picture parameters +//! This structure is used in CUVIDPICPARAMS structure +/*******************************************************/ +typedef struct _CUVIDMPEG4PICPARAMS +{ + int ForwardRefIdx; // Picture index of forward reference (P/B-frames) + int BackwardRefIdx; // Picture index of backward reference (B-frames) + // VOL + int video_object_layer_width; + int video_object_layer_height; + int vop_time_increment_bitcount; + int top_field_first; + int resync_marker_disable; + int quant_type; + int quarter_sample; + int short_video_header; + int divx_flags; + // VOP + int vop_coding_type; + int vop_coded; + int vop_rounding_type; + int alternate_vertical_scan_flag; + int interlaced; + int vop_fcode_forward; + int vop_fcode_backward; + int trd[2]; + int trb[2]; + // Quantization matrices (raster order) + unsigned char QuantMatrixIntra[64]; + unsigned char QuantMatrixInter[64]; + int gmc_enabled; +} CUVIDMPEG4PICPARAMS; + +/********************************************************/ +//! \struct CUVIDVC1PICPARAMS +//! VC1 picture parameters +//! This structure is used in CUVIDPICPARAMS structure +/********************************************************/ +typedef struct _CUVIDVC1PICPARAMS +{ + int ForwardRefIdx; /**< Picture index of forward reference (P/B-frames) */ + int BackwardRefIdx; /**< Picture index of backward reference (B-frames) */ + int FrameWidth; /**< Actual frame width */ + int FrameHeight; /**< Actual frame height */ + // PICTURE + int intra_pic_flag; /**< Set to 1 for I,BI frames */ + int ref_pic_flag; /**< Set to 1 for I,P frames */ + int progressive_fcm; /**< Progressive frame */ + // SEQUENCE + int profile; + int postprocflag; + int pulldown; + int interlace; + int tfcntrflag; + int finterpflag; + int psf; + int multires; + int syncmarker; + int rangered; + int maxbframes; + // ENTRYPOINT + int panscan_flag; + int refdist_flag; + int extended_mv; + int dquant; + int vstransform; + int loopfilter; + int fastuvmc; + int overlap; + int quantizer; + int extended_dmv; + int range_mapy_flag; + int range_mapy; + int range_mapuv_flag; + int range_mapuv; + int rangeredfrm; // range reduction state +} CUVIDVC1PICPARAMS; + +/***********************************************************/ +//! \struct CUVIDJPEGPICPARAMS +//! JPEG picture parameters +//! This structure is used in CUVIDPICPARAMS structure +/***********************************************************/ +typedef struct _CUVIDJPEGPICPARAMS +{ + int Reserved; +} CUVIDJPEGPICPARAMS; + + +/*******************************************************/ +//! \struct CUVIDHEVCPICPARAMS +//! HEVC picture parameters +//! This structure is used in CUVIDPICPARAMS structure +/*******************************************************/ +typedef struct _CUVIDHEVCPICPARAMS +{ + // sps + int pic_width_in_luma_samples; + int pic_height_in_luma_samples; + unsigned char log2_min_luma_coding_block_size_minus3; + unsigned char log2_diff_max_min_luma_coding_block_size; + unsigned char log2_min_transform_block_size_minus2; + unsigned char log2_diff_max_min_transform_block_size; + unsigned char pcm_enabled_flag; + unsigned char log2_min_pcm_luma_coding_block_size_minus3; + unsigned char log2_diff_max_min_pcm_luma_coding_block_size; + unsigned char pcm_sample_bit_depth_luma_minus1; + + unsigned char pcm_sample_bit_depth_chroma_minus1; + unsigned char pcm_loop_filter_disabled_flag; + unsigned char strong_intra_smoothing_enabled_flag; + unsigned char max_transform_hierarchy_depth_intra; + unsigned char max_transform_hierarchy_depth_inter; + unsigned char amp_enabled_flag; + unsigned char separate_colour_plane_flag; + unsigned char log2_max_pic_order_cnt_lsb_minus4; + + unsigned char num_short_term_ref_pic_sets; + unsigned char long_term_ref_pics_present_flag; + unsigned char num_long_term_ref_pics_sps; + unsigned char sps_temporal_mvp_enabled_flag; + unsigned char sample_adaptive_offset_enabled_flag; + unsigned char scaling_list_enable_flag; + unsigned char IrapPicFlag; + unsigned char IdrPicFlag; + + unsigned char bit_depth_luma_minus8; + unsigned char bit_depth_chroma_minus8; + //sps/pps extension fields + unsigned char log2_max_transform_skip_block_size_minus2; + unsigned char log2_sao_offset_scale_luma; + unsigned char log2_sao_offset_scale_chroma; + unsigned char high_precision_offsets_enabled_flag; + unsigned char reserved1[10]; + + // pps + unsigned char dependent_slice_segments_enabled_flag; + unsigned char slice_segment_header_extension_present_flag; + unsigned char sign_data_hiding_enabled_flag; + unsigned char cu_qp_delta_enabled_flag; + unsigned char diff_cu_qp_delta_depth; + signed char init_qp_minus26; + signed char pps_cb_qp_offset; + signed char pps_cr_qp_offset; + + unsigned char constrained_intra_pred_flag; + unsigned char weighted_pred_flag; + unsigned char weighted_bipred_flag; + unsigned char transform_skip_enabled_flag; + unsigned char transquant_bypass_enabled_flag; + unsigned char entropy_coding_sync_enabled_flag; + unsigned char log2_parallel_merge_level_minus2; + unsigned char num_extra_slice_header_bits; + + unsigned char loop_filter_across_tiles_enabled_flag; + unsigned char loop_filter_across_slices_enabled_flag; + unsigned char output_flag_present_flag; + unsigned char num_ref_idx_l0_default_active_minus1; + unsigned char num_ref_idx_l1_default_active_minus1; + unsigned char lists_modification_present_flag; + unsigned char cabac_init_present_flag; + unsigned char pps_slice_chroma_qp_offsets_present_flag; + + unsigned char deblocking_filter_override_enabled_flag; + unsigned char pps_deblocking_filter_disabled_flag; + signed char pps_beta_offset_div2; + signed char pps_tc_offset_div2; + unsigned char tiles_enabled_flag; + unsigned char uniform_spacing_flag; + unsigned char num_tile_columns_minus1; + unsigned char num_tile_rows_minus1; + + unsigned short column_width_minus1[21]; + unsigned short row_height_minus1[21]; + + // sps and pps extension HEVC-main 444 + unsigned char sps_range_extension_flag; + unsigned char transform_skip_rotation_enabled_flag; + unsigned char transform_skip_context_enabled_flag; + unsigned char implicit_rdpcm_enabled_flag; + + unsigned char explicit_rdpcm_enabled_flag; + unsigned char extended_precision_processing_flag; + unsigned char intra_smoothing_disabled_flag; + unsigned char persistent_rice_adaptation_enabled_flag; + + unsigned char cabac_bypass_alignment_enabled_flag; + unsigned char pps_range_extension_flag; + unsigned char cross_component_prediction_enabled_flag; + unsigned char chroma_qp_offset_list_enabled_flag; + + unsigned char diff_cu_chroma_qp_offset_depth; + unsigned char chroma_qp_offset_list_len_minus1; + signed char cb_qp_offset_list[6]; + + signed char cr_qp_offset_list[6]; + unsigned char reserved2[2]; + + unsigned int reserved3[8]; + + // RefPicSets + int NumBitsForShortTermRPSInSlice; + int NumDeltaPocsOfRefRpsIdx; + int NumPocTotalCurr; + int NumPocStCurrBefore; + int NumPocStCurrAfter; + int NumPocLtCurr; + int CurrPicOrderCntVal; + int RefPicIdx[16]; // [refpic] Indices of valid reference pictures (-1 if unused for reference) + int PicOrderCntVal[16]; // [refpic] + unsigned char IsLongTerm[16]; // [refpic] 0=not a long-term reference, 1=long-term reference + unsigned char RefPicSetStCurrBefore[8]; // [0..NumPocStCurrBefore-1] -> refpic (0..15) + unsigned char RefPicSetStCurrAfter[8]; // [0..NumPocStCurrAfter-1] -> refpic (0..15) + unsigned char RefPicSetLtCurr[8]; // [0..NumPocLtCurr-1] -> refpic (0..15) + unsigned char RefPicSetInterLayer0[8]; + unsigned char RefPicSetInterLayer1[8]; + unsigned int reserved4[12]; + + // scaling lists (diag order) + unsigned char ScalingList4x4[6][16]; // [matrixId][i] + unsigned char ScalingList8x8[6][64]; // [matrixId][i] + unsigned char ScalingList16x16[6][64]; // [matrixId][i] + unsigned char ScalingList32x32[2][64]; // [matrixId][i] + unsigned char ScalingListDCCoeff16x16[6]; // [matrixId] + unsigned char ScalingListDCCoeff32x32[2]; // [matrixId] +} CUVIDHEVCPICPARAMS; + + +/***********************************************************/ +//! \struct CUVIDVP8PICPARAMS +//! VP8 picture parameters +//! This structure is used in CUVIDPICPARAMS structure +/***********************************************************/ +typedef struct _CUVIDVP8PICPARAMS +{ + int width; + int height; + unsigned int first_partition_size; + //Frame Indexes + unsigned char LastRefIdx; + unsigned char GoldenRefIdx; + unsigned char AltRefIdx; + union { + struct { + unsigned char frame_type : 1; /**< 0 = KEYFRAME, 1 = INTERFRAME */ + unsigned char version : 3; + unsigned char show_frame : 1; + unsigned char update_mb_segmentation_data : 1; /**< Must be 0 if segmentation is not enabled */ + unsigned char Reserved2Bits : 2; + }vp8_frame_tag; + unsigned char wFrameTagFlags; + }; + unsigned char Reserved1[4]; + unsigned int Reserved2[3]; +} CUVIDVP8PICPARAMS; + +/***********************************************************/ +//! \struct CUVIDVP9PICPARAMS +//! VP9 picture parameters +//! This structure is used in CUVIDPICPARAMS structure +/***********************************************************/ +typedef struct _CUVIDVP9PICPARAMS +{ + unsigned int width; + unsigned int height; + + //Frame Indices + unsigned char LastRefIdx; + unsigned char GoldenRefIdx; + unsigned char AltRefIdx; + unsigned char colorSpace; + + unsigned short profile : 3; + unsigned short frameContextIdx : 2; + unsigned short frameType : 1; + unsigned short showFrame : 1; + unsigned short errorResilient : 1; + unsigned short frameParallelDecoding : 1; + unsigned short subSamplingX : 1; + unsigned short subSamplingY : 1; + unsigned short intraOnly : 1; + unsigned short allow_high_precision_mv : 1; + unsigned short refreshEntropyProbs : 1; + unsigned short reserved2Bits : 2; + + unsigned short reserved16Bits; + + unsigned char refFrameSignBias[4]; + + unsigned char bitDepthMinus8Luma; + unsigned char bitDepthMinus8Chroma; + unsigned char loopFilterLevel; + unsigned char loopFilterSharpness; + + unsigned char modeRefLfEnabled; + unsigned char log2_tile_columns; + unsigned char log2_tile_rows; + + unsigned char segmentEnabled : 1; + unsigned char segmentMapUpdate : 1; + unsigned char segmentMapTemporalUpdate : 1; + unsigned char segmentFeatureMode : 1; + unsigned char reserved4Bits : 4; + + + unsigned char segmentFeatureEnable[8][4]; + short segmentFeatureData[8][4]; + unsigned char mb_segment_tree_probs[7]; + unsigned char segment_pred_probs[3]; + unsigned char reservedSegment16Bits[2]; + + int qpYAc; + int qpYDc; + int qpChDc; + int qpChAc; + + unsigned int activeRefIdx[3]; + unsigned int resetFrameContext; + unsigned int mcomp_filter_type; + unsigned int mbRefLfDelta[4]; + unsigned int mbModeLfDelta[2]; + unsigned int frameTagSize; + unsigned int offsetToDctParts; + unsigned int reserved128Bits[4]; + +} CUVIDVP9PICPARAMS; + +/***********************************************************/ +//! \struct CUVIDAV1PICPARAMS +//! AV1 picture parameters +//! This structure is used in CUVIDPICPARAMS structure +/***********************************************************/ +typedef struct _CUVIDAV1PICPARAMS +{ + unsigned int width; // coded width, if superres enabled then it is upscaled width + unsigned int height; // coded height + unsigned int frame_offset; // defined as order_hint in AV1 specification + int decodePicIdx; // decoded output pic index, if film grain enabled, it will keep decoded (without film grain) output + // It can be used as reference frame for future frames + + // sequence header + unsigned int profile : 3; // 0 = profile0, 1 = profile1, 2 = profile2 + unsigned int use_128x128_superblock : 1; // superblock size 0:64x64, 1: 128x128 + unsigned int subsampling_x : 1; // (subsampling_x, _y) 1,1 = 420, 1,0 = 422, 0,0 = 444 + unsigned int subsampling_y : 1; + unsigned int mono_chrome : 1; // for monochrome content, mono_chrome = 1 and (subsampling_x, _y) should be 1,1 + unsigned int bit_depth_minus8 : 4; // bit depth minus 8 + unsigned int enable_filter_intra : 1; // tool enable in seq level, 0 : disable 1: frame header control + unsigned int enable_intra_edge_filter : 1; // intra edge filtering process, 0 : disable 1: enabled + unsigned int enable_interintra_compound : 1; // interintra, 0 : not present 1: present + unsigned int enable_masked_compound : 1; // 1: mode info for inter blocks may contain the syntax element compound_type. + // 0: syntax element compound_type will not be present + unsigned int enable_dual_filter : 1; // vertical and horiz filter selection, 1: enable and 0: disable + unsigned int enable_order_hint : 1; // order hint, and related tools, 1: enable and 0: disable + unsigned int order_hint_bits_minus1 : 3; // is used to compute OrderHintBits + unsigned int enable_jnt_comp : 1; // joint compound modes, 1: enable and 0: disable + unsigned int enable_superres : 1; // superres in seq level, 0 : disable 1: frame level control + unsigned int enable_cdef : 1; // cdef filtering in seq level, 0 : disable 1: frame level control + unsigned int enable_restoration : 1; // loop restoration filtering in seq level, 0 : disable 1: frame level control + unsigned int enable_fgs : 1; // defined as film_grain_params_present in AV1 specification + unsigned int reserved0_7bits : 7; // reserved bits; must be set to 0 + + // frame header + unsigned int frame_type : 2 ; // 0:Key frame, 1:Inter frame, 2:intra only, 3:s-frame + unsigned int show_frame : 1 ; // show_frame = 1 implies that frame should be immediately output once decoded + unsigned int disable_cdf_update : 1; // CDF update during symbol decoding, 1: disabled, 0: enabled + unsigned int allow_screen_content_tools : 1; // 1: intra blocks may use palette encoding, 0: palette encoding is never used + unsigned int force_integer_mv : 1; // 1: motion vectors will always be integers, 0: can contain fractional bits + unsigned int coded_denom : 3; // coded_denom of the superres scale as specified in AV1 specification + unsigned int allow_intrabc : 1; // 1: intra block copy may be used, 0: intra block copy is not allowed + unsigned int allow_high_precision_mv : 1; // 1/8 precision mv enable + unsigned int interp_filter : 3; // interpolation filter. Refer to section 6.8.9 of the AV1 specification Version 1.0.0 with Errata 1 + unsigned int switchable_motion_mode : 1; // defined as is_motion_mode_switchable in AV1 specification + unsigned int use_ref_frame_mvs : 1; // 1: current frame can use the previous frame mv information, 0: will not use. + unsigned int disable_frame_end_update_cdf : 1; // 1: indicates that the end of frame CDF update is disabled + unsigned int delta_q_present : 1; // quantizer index delta values are present in the block level + unsigned int delta_q_res : 2; // left shift which should be applied to decoded quantizer index delta values + unsigned int using_qmatrix : 1; // 1: quantizer matrix will be used to compute quantizers + unsigned int coded_lossless : 1; // 1: all segments use lossless coding + unsigned int use_superres : 1; // 1: superres enabled for frame + unsigned int tx_mode : 2; // 0: ONLY4x4,1:LARGEST,2:SELECT + unsigned int reference_mode : 1; // 0: SINGLE, 1: SELECT + unsigned int allow_warped_motion : 1; // 1: allow_warped_motion may be present, 0: allow_warped_motion will not be present + unsigned int reduced_tx_set : 1; // 1: frame is restricted to subset of the full set of transform types, 0: no such restriction + unsigned int skip_mode : 1; // 1: most of the mode info is skipped, 0: mode info is not skipped + unsigned int reserved1_3bits : 3; // reserved bits; must be set to 0 + + // tiling info + unsigned int num_tile_cols : 8; // number of tiles across the frame., max is 64 + unsigned int num_tile_rows : 8; // number of tiles down the frame., max is 64 + unsigned int context_update_tile_id : 16; // specifies which tile to use for the CDF update + unsigned short tile_widths[64]; // Width of each column in superblocks + unsigned short tile_heights[64]; // height of each row in superblocks + + // CDEF - refer to section 6.10.14 of the AV1 specification Version 1.0.0 with Errata 1 + unsigned char cdef_damping_minus_3 : 2; // controls the amount of damping in the deringing filter + unsigned char cdef_bits : 2; // the number of bits needed to specify which CDEF filter to apply + unsigned char reserved2_4bits : 4; // reserved bits; must be set to 0 + unsigned char cdef_y_strength[8]; // 0-3 bits: y_pri_strength, 4-7 bits y_sec_strength + unsigned char cdef_uv_strength[8]; // 0-3 bits: uv_pri_strength, 4-7 bits uv_sec_strength + + // SkipModeFrames + unsigned char SkipModeFrame0 : 4; // specifies the frames to use for compound prediction when skip_mode is equal to 1. + unsigned char SkipModeFrame1 : 4; + + // qp information - refer to section 6.8.11 of the AV1 specification Version 1.0.0 with Errata 1 + unsigned char base_qindex; // indicates the base frame qindex. Defined as base_q_idx in AV1 specification + char qp_y_dc_delta_q; // indicates the Y DC quantizer relative to base_q_idx. Defined as DeltaQYDc in AV1 specification + char qp_u_dc_delta_q; // indicates the U DC quantizer relative to base_q_idx. Defined as DeltaQUDc in AV1 specification + char qp_v_dc_delta_q; // indicates the V DC quantizer relative to base_q_idx. Defined as DeltaQVDc in AV1 specification + char qp_u_ac_delta_q; // indicates the U AC quantizer relative to base_q_idx. Defined as DeltaQUAc in AV1 specification + char qp_v_ac_delta_q; // indicates the V AC quantizer relative to base_q_idx. Defined as DeltaQVAc in AV1 specification + unsigned char qm_y; // specifies the level in the quantizer matrix that should be used for luma plane decoding + unsigned char qm_u; // specifies the level in the quantizer matrix that should be used for chroma U plane decoding + unsigned char qm_v; // specifies the level in the quantizer matrix that should be used for chroma V plane decoding + + // segmentation - refer to section 6.8.13 of the AV1 specification Version 1.0.0 with Errata 1 + unsigned char segmentation_enabled : 1; // 1 indicates that this frame makes use of the segmentation tool + unsigned char segmentation_update_map : 1; // 1 indicates that the segmentation map are updated during the decoding of this frame + unsigned char segmentation_update_data : 1; // 1 indicates that new parameters are about to be specified for each segment + unsigned char segmentation_temporal_update : 1; // 1 indicates that the updates to the segmentation map are coded relative to the existing segmentation map + unsigned char reserved3_4bits : 4; // reserved bits; must be set to 0 + short segmentation_feature_data[8][8]; // specifies the feature data for a segment feature + unsigned char segmentation_feature_mask[8]; // indicates that the corresponding feature is unused or feature value is coded + + // loopfilter - refer to section 6.8.10 of the AV1 specification Version 1.0.0 with Errata 1 + unsigned char loop_filter_level[2]; // contains loop filter strength values + unsigned char loop_filter_level_u; // loop filter strength value of U plane + unsigned char loop_filter_level_v; // loop filter strength value of V plane + unsigned char loop_filter_sharpness; // indicates the sharpness level + char loop_filter_ref_deltas[8]; // contains the adjustment needed for the filter level based on the chosen reference frame + char loop_filter_mode_deltas[2]; // contains the adjustment needed for the filter level based on the chosen mode + unsigned char loop_filter_delta_enabled : 1; // indicates that the filter level depends on the mode and reference frame used to predict a block + unsigned char loop_filter_delta_update : 1; // indicates that additional syntax elements are present that specify which mode and + // reference frame deltas are to be updated + unsigned char delta_lf_present : 1; // specifies whether loop filter delta values are present in the block level + unsigned char delta_lf_res : 2; // specifies the left shift to apply to the decoded loop filter values + unsigned char delta_lf_multi : 1; // separate loop filter deltas for Hy,Vy,U,V edges + unsigned char reserved4_2bits : 2; // reserved bits; must be set to 0 + + // restoration - refer to section 6.10.15 of the AV1 specification Version 1.0.0 with Errata 1 + unsigned char lr_unit_size[3]; // specifies the size of loop restoration units: 0: 32, 1: 64, 2: 128, 3: 256 + unsigned char lr_type[3] ; // used to compute FrameRestorationType + + // reference frames + unsigned char primary_ref_frame; // specifies which reference frame contains the CDF values and other state that should be + // loaded at the start of the frame + unsigned char ref_frame_map[8]; // frames in dpb that can be used as reference for current or future frames + + unsigned char temporal_layer_id : 4; // temporal layer id + unsigned char spatial_layer_id : 4; // spatial layer id + + unsigned char reserved5_32bits[4]; // reserved bits; must be set to 0 + + // ref frame list + struct + { + unsigned int width; + unsigned int height; + unsigned char index; + unsigned char reserved24Bits[3]; // reserved bits; must be set to 0 + } ref_frame[7]; // frames used as reference frame for current frame. + + // global motion + struct { + unsigned char invalid : 1; + unsigned char wmtype : 2; // defined as GmType in AV1 specification + unsigned char reserved5Bits : 5; // reserved bits; must be set to 0 + char reserved24Bits[3]; // reserved bits; must be set to 0 + int wmmat[6]; // defined as gm_params[] in AV1 specification + } global_motion[7]; // global motion params for reference frames + + // film grain params - refer to section 6.8.20 of the AV1 specification Version 1.0.0 with Errata 1 + unsigned short apply_grain : 1; + unsigned short overlap_flag : 1; + unsigned short scaling_shift_minus8 : 2; + unsigned short chroma_scaling_from_luma : 1; + unsigned short ar_coeff_lag : 2; + unsigned short ar_coeff_shift_minus6 : 2; + unsigned short grain_scale_shift : 2; + unsigned short clip_to_restricted_range : 1; + unsigned short reserved6_4bits : 4; // reserved bits; must be set to 0 + unsigned char num_y_points; + unsigned char scaling_points_y[14][2]; + unsigned char num_cb_points; + unsigned char scaling_points_cb[10][2]; + unsigned char num_cr_points; + unsigned char scaling_points_cr[10][2]; + unsigned char reserved7_8bits; // reserved bits; must be set to 0 + unsigned short random_seed; + short ar_coeffs_y[24]; + short ar_coeffs_cb[25]; + short ar_coeffs_cr[25]; + unsigned char cb_mult; + unsigned char cb_luma_mult; + short cb_offset; + unsigned char cr_mult; + unsigned char cr_luma_mult; + short cr_offset; + + int reserved[7]; // reserved bits; must be set to 0 +} CUVIDAV1PICPARAMS; + +/******************************************************************************************/ +//! \struct CUVIDPICPARAMS +//! Picture parameters for decoding +//! This structure is used in cuvidDecodePicture API +//! IN for cuvidDecodePicture +/******************************************************************************************/ +typedef struct _CUVIDPICPARAMS +{ + int PicWidthInMbs; /**< IN: Coded frame size in macroblocks */ + int FrameHeightInMbs; /**< IN: Coded frame height in macroblocks */ + int CurrPicIdx; /**< IN: Output index of the current picture */ + int field_pic_flag; /**< IN: 0=frame picture, 1=field picture */ + int bottom_field_flag; /**< IN: 0=top field, 1=bottom field (ignored if field_pic_flag=0) */ + int second_field; /**< IN: Second field of a complementary field pair */ + // Bitstream data + unsigned int nBitstreamDataLen; /**< IN: Number of bytes in bitstream data buffer */ + const unsigned char *pBitstreamData; /**< IN: Ptr to bitstream data for this picture (slice-layer) */ + unsigned int nNumSlices; /**< IN: Number of slices in this picture */ + const unsigned int *pSliceDataOffsets; /**< IN: nNumSlices entries, contains offset of each slice within + the bitstream data buffer */ + int ref_pic_flag; /**< IN: This picture is a reference picture */ + int intra_pic_flag; /**< IN: This picture is entirely intra coded */ + unsigned int Reserved[30]; /**< Reserved for future use */ + // IN: Codec-specific data + union { + CUVIDMPEG2PICPARAMS mpeg2; /**< Also used for MPEG-1 */ + CUVIDH264PICPARAMS h264; + CUVIDVC1PICPARAMS vc1; + CUVIDMPEG4PICPARAMS mpeg4; + CUVIDJPEGPICPARAMS jpeg; + CUVIDHEVCPICPARAMS hevc; + CUVIDVP8PICPARAMS vp8; + CUVIDVP9PICPARAMS vp9; + CUVIDAV1PICPARAMS av1; + unsigned int CodecReserved[1024]; + } CodecSpecific; +} CUVIDPICPARAMS; + + +/******************************************************/ +//! \struct CUVIDPROCPARAMS +//! Picture parameters for postprocessing +//! This structure is used in cuvidMapVideoFrame API +/******************************************************/ +typedef struct _CUVIDPROCPARAMS +{ + int progressive_frame; /**< IN: Input is progressive (deinterlace_mode will be ignored) */ + int second_field; /**< IN: Output the second field (ignored if deinterlace mode is Weave) */ + int top_field_first; /**< IN: Input frame is top field first (1st field is top, 2nd field is bottom) */ + int unpaired_field; /**< IN: Input only contains one field (2nd field is invalid) */ + // The fields below are used for raw YUV input + unsigned int reserved_flags; /**< Reserved for future use (set to zero) */ + unsigned int reserved_zero; /**< Reserved (set to zero) */ + unsigned long long raw_input_dptr; /**< IN: Input CUdeviceptr for raw YUV extensions */ + unsigned int raw_input_pitch; /**< IN: pitch in bytes of raw YUV input (should be aligned appropriately) */ + unsigned int raw_input_format; /**< IN: Input YUV format (cudaVideoCodec_enum) */ + unsigned long long raw_output_dptr; /**< IN: Output CUdeviceptr for raw YUV extensions */ + unsigned int raw_output_pitch; /**< IN: pitch in bytes of raw YUV output (should be aligned appropriately) */ + unsigned int Reserved1; /**< Reserved for future use (set to zero) */ + CUstream output_stream; /**< IN: stream object used by cuvidMapVideoFrame */ + unsigned int Reserved[46]; /**< Reserved for future use (set to zero) */ + unsigned long long *histogram_dptr; /**< OUT: Output CUdeviceptr for histogram extensions */ + void *Reserved2[1]; /**< Reserved for future use (set to zero) */ +} CUVIDPROCPARAMS; + +/*********************************************************************************************************/ +//! \struct CUVIDGETDECODESTATUS +//! Struct for reporting decode status. +//! This structure is used in cuvidGetDecodeStatus API. +/*********************************************************************************************************/ +typedef struct _CUVIDGETDECODESTATUS +{ + cuvidDecodeStatus decodeStatus; + unsigned int reserved[31]; + void *pReserved[8]; +} CUVIDGETDECODESTATUS; + +/****************************************************/ +//! \struct CUVIDRECONFIGUREDECODERINFO +//! Struct for decoder reset +//! This structure is used in cuvidReconfigureDecoder() API +/****************************************************/ +typedef struct _CUVIDRECONFIGUREDECODERINFO +{ + unsigned int ulWidth; /**< IN: Coded sequence width in pixels, MUST be < = ulMaxWidth defined at CUVIDDECODECREATEINFO */ + unsigned int ulHeight; /**< IN: Coded sequence height in pixels, MUST be < = ulMaxHeight defined at CUVIDDECODECREATEINFO */ + unsigned int ulTargetWidth; /**< IN: Post processed output width */ + unsigned int ulTargetHeight; /**< IN: Post Processed output height */ + unsigned int ulNumDecodeSurfaces; /**< IN: Maximum number of internal decode surfaces */ + unsigned int reserved1[12]; /**< Reserved for future use. Set to Zero */ + /** + * IN: Area of frame to be displayed. Use-case : Source Cropping + */ + struct { + short left; + short top; + short right; + short bottom; + } display_area; + /** + * IN: Target Rectangle in the OutputFrame. Use-case : Aspect ratio Conversion + */ + struct { + short left; + short top; + short right; + short bottom; + } target_rect; + unsigned int reserved2[11]; /**< Reserved for future use. Set to Zero */ +} CUVIDRECONFIGUREDECODERINFO; + + +/***********************************************************************************************************/ +//! VIDEO_DECODER +//! +//! In order to minimize decode latencies, there should be always at least 2 pictures in the decode +//! queue at any time, in order to make sure that all decode engines are always busy. +//! +//! Overall data flow: +//! - cuvidGetDecoderCaps(...) +//! - cuvidCreateDecoder(...) +//! - For each picture: +//! + cuvidDecodePicture(N) +//! + cuvidMapVideoFrame(N-4) +//! + do some processing in cuda +//! + cuvidUnmapVideoFrame(N-4) +//! + cuvidDecodePicture(N+1) +//! + cuvidMapVideoFrame(N-3) +//! + ... +//! - cuvidDestroyDecoder(...) +//! +//! NOTE: +//! - When the cuda context is created from a D3D device, the D3D device must also be created +//! with the D3DCREATE_MULTITHREADED flag. +//! - There is a limit to how many pictures can be mapped simultaneously (ulNumOutputSurfaces) +//! - cuvidDecodePicture may block the calling thread if there are too many pictures pending +//! in the decode queue +/***********************************************************************************************************/ + + +/**********************************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidGetDecoderCaps(CUVIDDECODECAPS *pdc) +//! Queries decode capabilities of NVDEC-HW based on CodecType, ChromaFormat and BitDepthMinus8 parameters. +//! 1. Application fills IN parameters CodecType, ChromaFormat and BitDepthMinus8 of CUVIDDECODECAPS structure +//! 2. On calling cuvidGetDecoderCaps, driver fills OUT parameters if the IN parameters are supported +//! If IN parameters passed to the driver are not supported by NVDEC-HW, then all OUT params are set to 0. +//! E.g. on Geforce GTX 960: +//! App fills - eCodecType = cudaVideoCodec_H264; eChromaFormat = cudaVideoChromaFormat_420; nBitDepthMinus8 = 0; +//! Given IN parameters are supported, hence driver fills: bIsSupported = 1; nMinWidth = 48; nMinHeight = 16; +//! nMaxWidth = 4096; nMaxHeight = 4096; nMaxMBCount = 65536; +//! CodedWidth*CodedHeight/256 must be less than or equal to nMaxMBCount +/**********************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidGetDecoderCaps(CUVIDDECODECAPS *pdc); + +/*****************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidCreateDecoder(CUvideodecoder *phDecoder, CUVIDDECODECREATEINFO *pdci) +//! Create the decoder object based on pdci. A handle to the created decoder is returned +/*****************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidCreateDecoder(CUvideodecoder *phDecoder, CUVIDDECODECREATEINFO *pdci); + +/*****************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidDestroyDecoder(CUvideodecoder hDecoder) +//! Destroy the decoder object +/*****************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidDestroyDecoder(CUvideodecoder hDecoder); + +/*****************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidDecodePicture(CUvideodecoder hDecoder, CUVIDPICPARAMS *pPicParams) +//! Decode a single picture (field or frame) +//! Kicks off HW decoding +/*****************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidDecodePicture(CUvideodecoder hDecoder, CUVIDPICPARAMS *pPicParams); + +/************************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidGetDecodeStatus(CUvideodecoder hDecoder, int nPicIdx); +//! Get the decode status for frame corresponding to nPicIdx +//! API is supported for Maxwell and above generation GPUs. +//! API is currently supported for HEVC, H264 and JPEG codecs. +//! API returns CUDA_ERROR_NOT_SUPPORTED error code for unsupported GPU or codec. +/************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidGetDecodeStatus(CUvideodecoder hDecoder, int nPicIdx, CUVIDGETDECODESTATUS* pDecodeStatus); + +/*********************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidReconfigureDecoder(CUvideodecoder hDecoder, CUVIDRECONFIGUREDECODERINFO *pDecReconfigParams) +//! Used to reuse single decoder for multiple clips. Currently supports resolution change, resize params, display area +//! params, target area params change for same codec. Must be called during CUVIDPARSERPARAMS::pfnSequenceCallback +/*********************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidReconfigureDecoder(CUvideodecoder hDecoder, CUVIDRECONFIGUREDECODERINFO *pDecReconfigParams); + + +#if !defined(__CUVID_DEVPTR64) || defined(__CUVID_INTERNAL) +/************************************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidMapVideoFrame(CUvideodecoder hDecoder, int nPicIdx, unsigned int *pDevPtr, +//! unsigned int *pPitch, CUVIDPROCPARAMS *pVPP); +//! Post-process and map video frame corresponding to nPicIdx for use in cuda. Returns cuda device pointer and associated +//! pitch of the video frame +/************************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidMapVideoFrame(CUvideodecoder hDecoder, int nPicIdx, + unsigned int *pDevPtr, unsigned int *pPitch, + CUVIDPROCPARAMS *pVPP); + +/*****************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidUnmapVideoFrame(CUvideodecoder hDecoder, unsigned int DevPtr) +//! Unmap a previously mapped video frame +/*****************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidUnmapVideoFrame(CUvideodecoder hDecoder, unsigned int DevPtr); +#endif + +/****************************************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidMapVideoFrame64(CUvideodecoder hDecoder, int nPicIdx, unsigned long long *pDevPtr, +//! unsigned int * pPitch, CUVIDPROCPARAMS *pVPP); +//! Post-process and map video frame corresponding to nPicIdx for use in cuda. Returns cuda device pointer and associated +//! pitch of the video frame +/****************************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidMapVideoFrame64(CUvideodecoder hDecoder, int nPicIdx, unsigned long long *pDevPtr, + unsigned int *pPitch, CUVIDPROCPARAMS *pVPP); + +/**************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidUnmapVideoFrame64(CUvideodecoder hDecoder, unsigned long long DevPtr); +//! Unmap a previously mapped video frame +/**************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidUnmapVideoFrame64(CUvideodecoder hDecoder, unsigned long long DevPtr); + +#if defined(__CUVID_DEVPTR64) && !defined(__CUVID_INTERNAL) +#define tcuvidMapVideoFrame tcuvidMapVideoFrame64 +#define tcuvidUnmapVideoFrame tcuvidUnmapVideoFrame64 +#endif + + + +/********************************************************************************************************************/ +//! +//! Context-locking: to facilitate multi-threaded implementations, the following 4 functions +//! provide a simple mutex-style host synchronization. If a non-NULL context is specified +//! in CUVIDDECODECREATEINFO, the codec library will acquire the mutex associated with the given +//! context before making any cuda calls. +//! A multi-threaded application could create a lock associated with a context handle so that +//! multiple threads can safely share the same cuda context: +//! - use cuCtxPopCurrent immediately after context creation in order to create a 'floating' context +//! that can be passed to cuvidCtxLockCreate. +//! - When using a floating context, all cuda calls should only be made within a cuvidCtxLock/cuvidCtxUnlock section. +//! +//! NOTE: This is a safer alternative to cuCtxPushCurrent and cuCtxPopCurrent, and is not related to video +//! decoder in any way (implemented as a critical section associated with cuCtx{Push|Pop}Current calls). +/********************************************************************************************************************/ + +/********************************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidCtxLockCreate(CUvideoctxlock *pLock, CUcontext ctx) +//! This API is used to create CtxLock object +/********************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidCtxLockCreate(CUvideoctxlock *pLock, CUcontext ctx); + +/********************************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidCtxLockDestroy(CUvideoctxlock lck) +//! This API is used to free CtxLock object +/********************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidCtxLockDestroy(CUvideoctxlock lck); + +/********************************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidCtxLock(CUvideoctxlock lck, unsigned int reserved_flags) +//! This API is used to acquire ctxlock +/********************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidCtxLock(CUvideoctxlock lck, unsigned int reserved_flags); + +/********************************************************************************************************************/ +//! \fn CUresult CUDAAPI cuvidCtxUnlock(CUvideoctxlock lck, unsigned int reserved_flags) +//! This API is used to release ctxlock +/********************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidCtxUnlock(CUvideoctxlock lck, unsigned int reserved_flags); + +/**********************************************************************************************/ + +#if defined(__cplusplus) +} +#endif /* __cplusplus */ + +#endif // __CUDA_VIDEO_H__ diff --git a/webrtc-jni/src/main/cpp/dependencies/nvdec/include/ffnvcodec/dynlink_nvcuvid.h b/webrtc-jni/src/main/cpp/dependencies/nvdec/include/ffnvcodec/dynlink_nvcuvid.h new file mode 100644 index 00000000..62dbca07 --- /dev/null +++ b/webrtc-jni/src/main/cpp/dependencies/nvdec/include/ffnvcodec/dynlink_nvcuvid.h @@ -0,0 +1,499 @@ +/* + * This copyright notice applies to this header file only: + * + * Copyright (c) 2010-2022 NVIDIA Corporation + * + * Permission is hereby granted, free of charge, to any person + * obtaining a copy of this software and associated documentation + * files (the "Software"), to deal in the Software without + * restriction, including without limitation the rights to use, + * copy, modify, merge, publish, distribute, sublicense, and/or sell + * copies of the software, and to permit persons to whom the + * software is furnished to do so, subject to the following + * conditions: + * + * The above copyright notice and this permission notice shall be + * included in all copies or substantial portions of the Software. + * + * THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, + * EXPRESS OR IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES + * OF MERCHANTABILITY, FITNESS FOR A PARTICULAR PURPOSE AND + * NONINFRINGEMENT. IN NO EVENT SHALL THE AUTHORS OR COPYRIGHT + * HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER LIABILITY, + * WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING + * FROM, OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR + * OTHER DEALINGS IN THE SOFTWARE. + */ + +/********************************************************************************************************************/ +//! \file nvcuvid.h +//! NVDECODE API provides video decoding interface to NVIDIA GPU devices. +//! \date 2015-2022 +//! This file contains the interface constants, structure definitions and function prototypes. +/********************************************************************************************************************/ + +#if !defined(__NVCUVID_H__) +#define __NVCUVID_H__ + +#include "dynlink_cuviddec.h" + +#if defined(__cplusplus) +extern "C" { +#endif /* __cplusplus */ + +#define MAX_CLOCK_TS 3 + +/***********************************************/ +//! +//! High-level helper APIs for video sources +//! +/***********************************************/ + +typedef void *CUvideosource; +typedef void *CUvideoparser; +typedef long long CUvideotimestamp; + + +/************************************************************************/ +//! \enum cudaVideoState +//! Video source state enums +//! Used in cuvidSetVideoSourceState and cuvidGetVideoSourceState APIs +/************************************************************************/ +typedef enum { + cudaVideoState_Error = -1, /**< Error state (invalid source) */ + cudaVideoState_Stopped = 0, /**< Source is stopped (or reached end-of-stream) */ + cudaVideoState_Started = 1 /**< Source is running and delivering data */ +} cudaVideoState; + +/************************************************************************/ +//! \enum cudaAudioCodec +//! Audio compression enums +//! Used in CUAUDIOFORMAT structure +/************************************************************************/ +typedef enum { + cudaAudioCodec_MPEG1=0, /**< MPEG-1 Audio */ + cudaAudioCodec_MPEG2, /**< MPEG-2 Audio */ + cudaAudioCodec_MP3, /**< MPEG-1 Layer III Audio */ + cudaAudioCodec_AC3, /**< Dolby Digital (AC3) Audio */ + cudaAudioCodec_LPCM, /**< PCM Audio */ + cudaAudioCodec_AAC, /**< AAC Audio */ +} cudaAudioCodec; + +/************************************************************************/ +//! \ingroup STRUCTS +//! \struct HEVCTIMECODESET +//! Used to store Time code extracted from Time code SEI in HEVC codec +/************************************************************************/ +typedef struct _HEVCTIMECODESET +{ + unsigned int time_offset_value; + unsigned short n_frames; + unsigned char clock_timestamp_flag; + unsigned char units_field_based_flag; + unsigned char counting_type; + unsigned char full_timestamp_flag; + unsigned char discontinuity_flag; + unsigned char cnt_dropped_flag; + unsigned char seconds_value; + unsigned char minutes_value; + unsigned char hours_value; + unsigned char seconds_flag; + unsigned char minutes_flag; + unsigned char hours_flag; + unsigned char time_offset_length; + unsigned char reserved; +} HEVCTIMECODESET; + +/************************************************************************/ +//! \ingroup STRUCTS +//! \struct HEVCSEITIMECODE +//! Used to extract Time code SEI in HEVC codec +/************************************************************************/ +typedef struct _HEVCSEITIMECODE +{ + HEVCTIMECODESET time_code_set[MAX_CLOCK_TS]; + unsigned char num_clock_ts; +} HEVCSEITIMECODE; + +/**********************************************************************************/ +//! \ingroup STRUCTS +//! \struct CUSEIMESSAGE; +//! Used in CUVIDSEIMESSAGEINFO structure +/**********************************************************************************/ +typedef struct _CUSEIMESSAGE +{ + unsigned char sei_message_type; /**< OUT: SEI Message Type */ + unsigned char reserved[3]; + unsigned int sei_message_size; /**< OUT: SEI Message Size */ +} CUSEIMESSAGE; + +/************************************************************************************************/ +//! \ingroup STRUCTS +//! \struct CUVIDEOFORMAT +//! Video format +//! Used in cuvidGetSourceVideoFormat API +/************************************************************************************************/ +typedef struct +{ + cudaVideoCodec codec; /**< OUT: Compression format */ + /** + * OUT: frame rate = numerator / denominator (for example: 30000/1001) + */ + struct { + /**< OUT: frame rate numerator (0 = unspecified or variable frame rate) */ + unsigned int numerator; + /**< OUT: frame rate denominator (0 = unspecified or variable frame rate) */ + unsigned int denominator; + } frame_rate; + unsigned char progressive_sequence; /**< OUT: 0=interlaced, 1=progressive */ + unsigned char bit_depth_luma_minus8; /**< OUT: high bit depth luma. E.g, 2 for 10-bitdepth, 4 for 12-bitdepth */ + unsigned char bit_depth_chroma_minus8; /**< OUT: high bit depth chroma. E.g, 2 for 10-bitdepth, 4 for 12-bitdepth */ + unsigned char min_num_decode_surfaces; /**< OUT: Minimum number of decode surfaces to be allocated for correct + decoding. The client can send this value in ulNumDecodeSurfaces + (in CUVIDDECODECREATEINFO structure). + This guarantees correct functionality and optimal video memory + usage but not necessarily the best performance, which depends on + the design of the overall application. The optimal number of + decode surfaces (in terms of performance and memory utilization) + should be decided by experimentation for each application, but it + cannot go below min_num_decode_surfaces. + If this value is used for ulNumDecodeSurfaces then it must be + returned to parser during sequence callback. */ + unsigned int coded_width; /**< OUT: coded frame width in pixels */ + unsigned int coded_height; /**< OUT: coded frame height in pixels */ + /** + * area of the frame that should be displayed + * typical example: + * coded_width = 1920, coded_height = 1088 + * display_area = { 0,0,1920,1080 } + */ + struct { + int left; /**< OUT: left position of display rect */ + int top; /**< OUT: top position of display rect */ + int right; /**< OUT: right position of display rect */ + int bottom; /**< OUT: bottom position of display rect */ + } display_area; + cudaVideoChromaFormat chroma_format; /**< OUT: Chroma format */ + unsigned int bitrate; /**< OUT: video bitrate (bps, 0=unknown) */ + /** + * OUT: Display Aspect Ratio = x:y (4:3, 16:9, etc) + */ + struct { + int x; + int y; + } display_aspect_ratio; + /** + * Video Signal Description + * Refer section E.2.1 (VUI parameters semantics) of H264 spec file + */ + struct { + unsigned char video_format : 3; /**< OUT: 0-Component, 1-PAL, 2-NTSC, 3-SECAM, 4-MAC, 5-Unspecified */ + unsigned char video_full_range_flag : 1; /**< OUT: indicates the black level and luma and chroma range */ + unsigned char reserved_zero_bits : 4; /**< Reserved bits */ + unsigned char color_primaries; /**< OUT: chromaticity coordinates of source primaries */ + unsigned char transfer_characteristics; /**< OUT: opto-electronic transfer characteristic of the source picture */ + unsigned char matrix_coefficients; /**< OUT: used in deriving luma and chroma signals from RGB primaries */ + } video_signal_description; + unsigned int seqhdr_data_length; /**< OUT: Additional bytes following (CUVIDEOFORMATEX) */ +} CUVIDEOFORMAT; + +/****************************************************************/ +//! \ingroup STRUCTS +//! \struct CUVIDOPERATINGPOINTINFO +//! Operating point information of scalable bitstream +/****************************************************************/ +typedef struct +{ + cudaVideoCodec codec; + union + { + struct + { + unsigned char operating_points_cnt; + unsigned char reserved24_bits[3]; + unsigned short operating_points_idc[32]; + } av1; + unsigned char CodecReserved[1024]; + }; +} CUVIDOPERATINGPOINTINFO; + +/**********************************************************************************/ +//! \ingroup STRUCTS +//! \struct CUVIDSEIMESSAGEINFO +//! Used in cuvidParseVideoData API with PFNVIDSEIMSGCALLBACK pfnGetSEIMsg +/**********************************************************************************/ +typedef struct _CUVIDSEIMESSAGEINFO +{ + void *pSEIData; /**< OUT: SEI Message Data */ + CUSEIMESSAGE *pSEIMessage; /**< OUT: SEI Message Info */ + unsigned int sei_message_count; /**< OUT: SEI Message Count */ + unsigned int picIdx; /**< OUT: SEI Message Pic Index */ +} CUVIDSEIMESSAGEINFO; + +/****************************************************************/ +//! \ingroup STRUCTS +//! \struct CUVIDAV1SEQHDR +//! AV1 specific sequence header information +/****************************************************************/ +typedef struct { + unsigned int max_width; + unsigned int max_height; + unsigned char reserved[1016]; +} CUVIDAV1SEQHDR; + +/****************************************************************/ +//! \ingroup STRUCTS +//! \struct CUVIDEOFORMATEX +//! Video format including raw sequence header information +//! Used in cuvidGetSourceVideoFormat API +/****************************************************************/ +typedef struct +{ + CUVIDEOFORMAT format; /**< OUT: CUVIDEOFORMAT structure */ + union { + CUVIDAV1SEQHDR av1; + unsigned char raw_seqhdr_data[1024]; /**< OUT: Sequence header data */ + }; +} CUVIDEOFORMATEX; + +/****************************************************************/ +//! \ingroup STRUCTS +//! \struct CUAUDIOFORMAT +//! Audio formats +//! Used in cuvidGetSourceAudioFormat API +/****************************************************************/ +typedef struct +{ + cudaAudioCodec codec; /**< OUT: Compression format */ + unsigned int channels; /**< OUT: number of audio channels */ + unsigned int samplespersec; /**< OUT: sampling frequency */ + unsigned int bitrate; /**< OUT: For uncompressed, can also be used to determine bits per sample */ + unsigned int reserved1; /**< Reserved for future use */ + unsigned int reserved2; /**< Reserved for future use */ +} CUAUDIOFORMAT; + + +/***************************************************************/ +//! \enum CUvideopacketflags +//! Data packet flags +//! Used in CUVIDSOURCEDATAPACKET structure +/***************************************************************/ +typedef enum { + CUVID_PKT_ENDOFSTREAM = 0x01, /**< Set when this is the last packet for this stream */ + CUVID_PKT_TIMESTAMP = 0x02, /**< Timestamp is valid */ + CUVID_PKT_DISCONTINUITY = 0x04, /**< Set when a discontinuity has to be signalled */ + CUVID_PKT_ENDOFPICTURE = 0x08, /**< Set when the packet contains exactly one frame or one field */ + CUVID_PKT_NOTIFY_EOS = 0x10, /**< If this flag is set along with CUVID_PKT_ENDOFSTREAM, an additional (dummy) + display callback will be invoked with null value of CUVIDPARSERDISPINFO which + should be interpreted as end of the stream. */ +} CUvideopacketflags; + +/*****************************************************************************/ +//! \ingroup STRUCTS +//! \struct CUVIDSOURCEDATAPACKET +//! Data Packet +//! Used in cuvidParseVideoData API +//! IN for cuvidParseVideoData +/*****************************************************************************/ +typedef struct _CUVIDSOURCEDATAPACKET +{ + tcu_ulong flags; /**< IN: Combination of CUVID_PKT_XXX flags */ + tcu_ulong payload_size; /**< IN: number of bytes in the payload (may be zero if EOS flag is set) */ + const unsigned char *payload; /**< IN: Pointer to packet payload data (may be NULL if EOS flag is set) */ + CUvideotimestamp timestamp; /**< IN: Presentation time stamp (10MHz clock), only valid if + CUVID_PKT_TIMESTAMP flag is set */ +} CUVIDSOURCEDATAPACKET; + +// Callback for packet delivery +typedef int (CUDAAPI *PFNVIDSOURCECALLBACK)(void *, CUVIDSOURCEDATAPACKET *); + +/**************************************************************************************************************************/ +//! \ingroup STRUCTS +//! \struct CUVIDSOURCEPARAMS +//! Describes parameters needed in cuvidCreateVideoSource API +//! NVDECODE API is intended for HW accelerated video decoding so CUvideosource doesn't have audio demuxer for all supported +//! containers. It's recommended to clients to use their own or third party demuxer if audio support is needed. +/**************************************************************************************************************************/ +typedef struct _CUVIDSOURCEPARAMS +{ + unsigned int ulClockRate; /**< IN: Time stamp units in Hz (0=default=10000000Hz) */ + unsigned int bAnnexb : 1; /**< IN: AV1 annexB stream */ + unsigned int uReserved : 31; /**< Reserved for future use - set to zero */ + unsigned int uReserved1[6]; /**< Reserved for future use - set to zero */ + void *pUserData; /**< IN: User private data passed in to the data handlers */ + PFNVIDSOURCECALLBACK pfnVideoDataHandler; /**< IN: Called to deliver video packets */ + PFNVIDSOURCECALLBACK pfnAudioDataHandler; /**< IN: Called to deliver audio packets. */ + void *pvReserved2[8]; /**< Reserved for future use - set to NULL */ +} CUVIDSOURCEPARAMS; + + +/**********************************************/ +//! \ingroup ENUMS +//! \enum CUvideosourceformat_flags +//! CUvideosourceformat_flags +//! Used in cuvidGetSourceVideoFormat API +/**********************************************/ +typedef enum { + CUVID_FMT_EXTFORMATINFO = 0x100 /**< Return extended format structure (CUVIDEOFORMATEX) */ +} CUvideosourceformat_flags; + +#if !defined(__APPLE__) +/***************************************************************************************************************************/ +//! \ingroup FUNCTS +//! \fn CUresult CUDAAPI cuvidCreateVideoSource(CUvideosource *pObj, const char *pszFileName, CUVIDSOURCEPARAMS *pParams) +//! Create CUvideosource object. CUvideosource spawns demultiplexer thread that provides two callbacks: +//! pfnVideoDataHandler() and pfnAudioDataHandler() +//! NVDECODE API is intended for HW accelerated video decoding so CUvideosource doesn't have audio demuxer for all supported +//! containers. It's recommended to clients to use their own or third party demuxer if audio support is needed. +/***************************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidCreateVideoSource(CUvideosource *pObj, const char *pszFileName, CUVIDSOURCEPARAMS *pParams); + +/***************************************************************************************************************************/ +//! \ingroup FUNCTS +//! \fn CUresult CUDAAPI cuvidCreateVideoSourceW(CUvideosource *pObj, const wchar_t *pwszFileName, CUVIDSOURCEPARAMS *pParams) +//! Create video source +/***************************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidCreateVideoSourceW(CUvideosource *pObj, const wchar_t *pwszFileName, CUVIDSOURCEPARAMS *pParams); + +/********************************************************************/ +//! \ingroup FUNCTS +//! \fn CUresult CUDAAPI cuvidDestroyVideoSource(CUvideosource obj) +//! Destroy video source +/********************************************************************/ +typedef CUresult CUDAAPI tcuvidDestroyVideoSource(CUvideosource obj); + +/******************************************************************************************/ +//! \ingroup FUNCTS +//! \fn CUresult CUDAAPI cuvidSetVideoSourceState(CUvideosource obj, cudaVideoState state) +//! Set video source state to: +//! cudaVideoState_Started - to signal the source to run and deliver data +//! cudaVideoState_Stopped - to stop the source from delivering the data +//! cudaVideoState_Error - invalid source +/******************************************************************************************/ +typedef CUresult CUDAAPI tcuvidSetVideoSourceState(CUvideosource obj, cudaVideoState state); + +/******************************************************************************************/ +//! \ingroup FUNCTS +//! \fn cudaVideoState CUDAAPI cuvidGetVideoSourceState(CUvideosource obj) +//! Get video source state +//! Returns: +//! cudaVideoState_Started - if Source is running and delivering data +//! cudaVideoState_Stopped - if Source is stopped or reached end-of-stream +//! cudaVideoState_Error - if Source is in error state +/******************************************************************************************/ +typedef cudaVideoState CUDAAPI tcuvidGetVideoSourceState(CUvideosource obj); + +/******************************************************************************************************************/ +//! \ingroup FUNCTS +//! \fn CUresult CUDAAPI cuvidGetSourceVideoFormat(CUvideosource obj, CUVIDEOFORMAT *pvidfmt, unsigned int flags) +//! Gets video source format in pvidfmt, flags is set to combination of CUvideosourceformat_flags as per requirement +/******************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidGetSourceVideoFormat(CUvideosource obj, CUVIDEOFORMAT *pvidfmt, unsigned int flags); + +/**************************************************************************************************************************/ +//! \ingroup FUNCTS +//! \fn CUresult CUDAAPI cuvidGetSourceAudioFormat(CUvideosource obj, CUAUDIOFORMAT *paudfmt, unsigned int flags) +//! Get audio source format +//! NVDECODE API is intended for HW accelerated video decoding so CUvideosource doesn't have audio demuxer for all supported +//! containers. It's recommended to clients to use their own or third party demuxer if audio support is needed. +/**************************************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidGetSourceAudioFormat(CUvideosource obj, CUAUDIOFORMAT *paudfmt, unsigned int flags); + +#endif +/**********************************************************************************/ +//! \ingroup STRUCTS +//! \struct CUVIDPARSERDISPINFO +//! Used in cuvidParseVideoData API with PFNVIDDISPLAYCALLBACK pfnDisplayPicture +/**********************************************************************************/ +typedef struct _CUVIDPARSERDISPINFO +{ + int picture_index; /**< OUT: Index of the current picture */ + int progressive_frame; /**< OUT: 1 if progressive frame; 0 otherwise */ + int top_field_first; /**< OUT: 1 if top field is displayed first; 0 otherwise */ + int repeat_first_field; /**< OUT: Number of additional fields (1=ivtc, 2=frame doubling, 4=frame tripling, + -1=unpaired field) */ + CUvideotimestamp timestamp; /**< OUT: Presentation time stamp */ +} CUVIDPARSERDISPINFO; + +/***********************************************************************************************************************/ +//! Parser callbacks +//! The parser will call these synchronously from within cuvidParseVideoData(), whenever there is sequence change or a picture +//! is ready to be decoded and/or displayed. First argument in functions is "void *pUserData" member of structure CUVIDSOURCEPARAMS +//! Return values from these callbacks are interpreted as below. If the callbacks return failure, it will be propagated by +//! cuvidParseVideoData() to the application. +//! Parser picks default operating point as 0 and outputAllLayers flag as 0 if PFNVIDOPPOINTCALLBACK is not set or return value is +//! -1 or invalid operating point. +//! PFNVIDSEQUENCECALLBACK : 0: fail, 1: succeeded, > 1: override dpb size of parser (set by CUVIDPARSERPARAMS::ulMaxNumDecodeSurfaces +//! while creating parser) +//! PFNVIDDECODECALLBACK : 0: fail, >=1: succeeded +//! PFNVIDDISPLAYCALLBACK : 0: fail, >=1: succeeded +//! PFNVIDOPPOINTCALLBACK : <0: fail, >=0: succeeded (bit 0-9: OperatingPoint, bit 10-10: outputAllLayers, bit 11-30: reserved) +//! PFNVIDSEIMSGCALLBACK : 0: fail, >=1: succeeded +/***********************************************************************************************************************/ +typedef int (CUDAAPI *PFNVIDSEQUENCECALLBACK)(void *, CUVIDEOFORMAT *); +typedef int (CUDAAPI *PFNVIDDECODECALLBACK)(void *, CUVIDPICPARAMS *); +typedef int (CUDAAPI *PFNVIDDISPLAYCALLBACK)(void *, CUVIDPARSERDISPINFO *); +typedef int (CUDAAPI *PFNVIDOPPOINTCALLBACK)(void *, CUVIDOPERATINGPOINTINFO*); +typedef int (CUDAAPI *PFNVIDSEIMSGCALLBACK) (void *, CUVIDSEIMESSAGEINFO *); + +/**************************************/ +//! \ingroup STRUCTS +//! \struct CUVIDPARSERPARAMS +//! Used in cuvidCreateVideoParser API +/**************************************/ +typedef struct _CUVIDPARSERPARAMS +{ + cudaVideoCodec CodecType; /**< IN: cudaVideoCodec_XXX */ + unsigned int ulMaxNumDecodeSurfaces; /**< IN: Max # of decode surfaces (parser will cycle through these) */ + unsigned int ulClockRate; /**< IN: Timestamp units in Hz (0=default=10000000Hz) */ + unsigned int ulErrorThreshold; /**< IN: % Error threshold (0-100) for calling pfnDecodePicture (100=always + IN: call pfnDecodePicture even if picture bitstream is fully corrupted) */ + unsigned int ulMaxDisplayDelay; /**< IN: Max display queue delay (improves pipelining of decode with display) + 0=no delay (recommended values: 2..4) */ + unsigned int bAnnexb : 1; /**< IN: AV1 annexB stream */ + unsigned int uReserved : 31; /**< Reserved for future use - set to zero */ + unsigned int uReserved1[4]; /**< IN: Reserved for future use - set to 0 */ + void *pUserData; /**< IN: User data for callbacks */ + PFNVIDSEQUENCECALLBACK pfnSequenceCallback; /**< IN: Called before decoding frames and/or whenever there is a fmt change */ + PFNVIDDECODECALLBACK pfnDecodePicture; /**< IN: Called when a picture is ready to be decoded (decode order) */ + PFNVIDDISPLAYCALLBACK pfnDisplayPicture; /**< IN: Called whenever a picture is ready to be displayed (display order) */ + PFNVIDOPPOINTCALLBACK pfnGetOperatingPoint; /**< IN: Called from AV1 sequence header to get operating point of a AV1 + scalable bitstream */ + PFNVIDSEIMSGCALLBACK pfnGetSEIMsg; /**< IN: Called when all SEI messages are parsed for particular frame */ + void *pvReserved2[5]; /**< Reserved for future use - set to NULL */ + CUVIDEOFORMATEX *pExtVideoInfo; /**< IN: [Optional] sequence header data from system layer */ +} CUVIDPARSERPARAMS; + +/************************************************************************************************/ +//! \ingroup FUNCTS +//! \fn CUresult CUDAAPI cuvidCreateVideoParser(CUvideoparser *pObj, CUVIDPARSERPARAMS *pParams) +//! Create video parser object and initialize +/************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidCreateVideoParser(CUvideoparser *pObj, CUVIDPARSERPARAMS *pParams); + +/************************************************************************************************/ +//! \ingroup FUNCTS +//! \fn CUresult CUDAAPI cuvidParseVideoData(CUvideoparser obj, CUVIDSOURCEDATAPACKET *pPacket) +//! Parse the video data from source data packet in pPacket +//! Extracts parameter sets like SPS, PPS, bitstream etc. from pPacket and +//! calls back pfnDecodePicture with CUVIDPICPARAMS data for kicking of HW decoding +//! calls back pfnSequenceCallback with CUVIDEOFORMAT data for initial sequence header or when +//! the decoder encounters a video format change +//! calls back pfnDisplayPicture with CUVIDPARSERDISPINFO data to display a video frame +/************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidParseVideoData(CUvideoparser obj, CUVIDSOURCEDATAPACKET *pPacket); + +/************************************************************************************************/ +//! \ingroup FUNCTS +//! \fn CUresult CUDAAPI cuvidDestroyVideoParser(CUvideoparser obj) +//! Destroy the video parser +/************************************************************************************************/ +typedef CUresult CUDAAPI tcuvidDestroyVideoParser(CUvideoparser obj); + +/**********************************************************************************************/ + +#if defined(__cplusplus) +} +#endif /* __cplusplus */ + +#endif // __NVCUVID_H__ diff --git a/webrtc-jni/src/main/cpp/include/media/video/codec/nvdec/NvdecLibrary.h b/webrtc-jni/src/main/cpp/include/media/video/codec/nvdec/NvdecLibrary.h new file mode 100644 index 00000000..d91f34cf --- /dev/null +++ b/webrtc-jni/src/main/cpp/include/media/video/codec/nvdec/NvdecLibrary.h @@ -0,0 +1,125 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef JNI_WEBRTC_MEDIA_VIDEO_CODEC_NVDEC_LIBRARY_H_ +#define JNI_WEBRTC_MEDIA_VIDEO_CODEC_NVDEC_LIBRARY_H_ + +#include "media/video/codec/nvenc/DynamicLibrary.h" + +#include +#include +#include + +#include + +namespace jni +{ + // The functions of the CUDA driver and of libnvcuvid that decoding uses. + struct NvdecApi + { + tcuInit * cuInit = nullptr; + tcuDeviceGetCount * cuDeviceGetCount = nullptr; + tcuDeviceGet * cuDeviceGet = nullptr; + tcuDeviceGetName * cuDeviceGetName = nullptr; + tcuDevicePrimaryCtxRetain * cuDevicePrimaryCtxRetain = nullptr; + tcuDevicePrimaryCtxRelease * cuDevicePrimaryCtxRelease = nullptr; + tcuCtxPushCurrent_v2 * cuCtxPushCurrent = nullptr; + tcuCtxPopCurrent_v2 * cuCtxPopCurrent = nullptr; + tcuMemcpy2D_v2 * cuMemcpy2D = nullptr; + + tcuvidGetDecoderCaps * cuvidGetDecoderCaps = nullptr; + tcuvidCreateDecoder * cuvidCreateDecoder = nullptr; + tcuvidDestroyDecoder * cuvidDestroyDecoder = nullptr; + tcuvidDecodePicture * cuvidDecodePicture = nullptr; + tcuvidMapVideoFrame64 * cuvidMapVideoFrame = nullptr; + tcuvidUnmapVideoFrame64 * cuvidUnmapVideoFrame = nullptr; + tcuvidCreateVideoParser * cuvidCreateVideoParser = nullptr; + tcuvidParseVideoData * cuvidParseVideoData = nullptr; + tcuvidDestroyVideoParser * cuvidDestroyVideoParser = nullptr; + }; + + // The CUDA driver and the NVDEC library of the NVIDIA driver, loaded once, + // when first asked for, and kept for the life of the process. Decoders run + // on the primary CUDA context of the first device. + class NvdecLibrary + { + public: + // Returns the library, or null if there is no NVIDIA driver, no + // CUDA device, or a device that decodes neither H.264 nor VP9. + static NvdecLibrary * Get(); + + NvdecLibrary(const NvdecLibrary &) = delete; + NvdecLibrary & operator=(const NvdecLibrary &) = delete; + + const NvdecApi & Api() const; + + // The name of the device, e.g. "NVIDIA GeForce RTX 4070". + const std::string & DeviceName() const; + + bool SupportsH264() const; + bool SupportsVp9() const; + + // Whether the device decodes the codec, 8 bit 4:2:0, at this + // coded size. Needs the context current. + bool Supports(cudaVideoCodec codec, unsigned int width, unsigned int height) const; + + // Retains the primary context of the device. Every successful call + // has to be matched by ReleaseContext(). + bool RetainContext(CUcontext * context) const; + void ReleaseContext() const; + + bool PushContext(CUcontext context) const; + void PopContext() const; + + private: + NvdecLibrary() = default; + + bool Load(); + + // Asks the device whether it decodes the codec at all. + bool QueryCodec(cudaVideoCodec codec) const; + + private: + DynamicLibrary cuda; + DynamicLibrary cuvid; + + NvdecApi api; + CUdevice device = 0; + std::string deviceName; + bool h264 = false; + bool vp9 = false; + }; + + // Makes a CUDA context current on the calling thread for as long as it + // lives, which calls into NVDEC need. + class NvdecContextScope + { + public: + NvdecContextScope(const NvdecLibrary & library, CUcontext context); + ~NvdecContextScope(); + + NvdecContextScope(const NvdecContextScope &) = delete; + NvdecContextScope & operator=(const NvdecContextScope &) = delete; + + bool IsCurrent() const; + + private: + const NvdecLibrary & library; + const bool pushed; + }; +} + +#endif diff --git a/webrtc-jni/src/main/cpp/include/media/video/codec/nvdec/NvdecVideoDecoder.h b/webrtc-jni/src/main/cpp/include/media/video/codec/nvdec/NvdecVideoDecoder.h new file mode 100644 index 00000000..f7fc6fae --- /dev/null +++ b/webrtc-jni/src/main/cpp/include/media/video/codec/nvdec/NvdecVideoDecoder.h @@ -0,0 +1,121 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef JNI_WEBRTC_MEDIA_VIDEO_CODEC_NVDEC_VIDEO_DECODER_H_ +#define JNI_WEBRTC_MEDIA_VIDEO_CODEC_NVDEC_VIDEO_DECODER_H_ + +#include "media/video/codec/nvdec/NvdecLibrary.h" + +#include "api/video/encoded_image.h" +#include "api/video/video_codec_type.h" +#include "api/video/video_rotation.h" +#include "api/video_codecs/video_decoder.h" + +#include +#include +#include +#include + +namespace jni +{ + // Decodes H.264 or VP9 profile 0 on the NVDEC engine of an NVIDIA GPU. + // + // NVDEC's own parser takes the encoded image, finds the pictures in it and + // calls back in three steps: a sequence, which creates the decoder, a + // picture to decode, and a picture to show. Decoding is synchronous: with + // no display delay all of that happens inside Decode, on the thread that + // called it. A decoded picture is in NV12 in GPU memory; it is copied to + // system memory, cropped to the picture, and converted to I420, which is + // what WebRTC's frames hold. + // + // Whatever NVDEC cannot decode, such as another profile or a size the GPU + // does not take, and whatever fails, returns + // WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE, so that the software decoder takes + // over. + class NvdecVideoDecoder : public webrtc::VideoDecoder + { + public: + // The codec is H.264 or VP9. + NvdecVideoDecoder(NvdecLibrary & library, webrtc::VideoCodecType codec); + ~NvdecVideoDecoder() override; + + using webrtc::VideoDecoder::Decode; + + bool Configure(const Settings & settings) override; + int32_t Decode(const webrtc::EncodedImage & image, int64_t renderTimeMs) override; + int32_t RegisterDecodeCompleteCallback(webrtc::DecodedImageCallback * callback) override; + int32_t Release() override; + DecoderInfo GetDecoderInfo() const override; + const char * ImplementationName() const override; + + private: + // What becomes of an encoded image once it is decoded. + struct PendingFrame + { + uint32_t rtpTimestamp; + int64_t ntpTimeMs; + int64_t renderTimeMs; + webrtc::VideoRotation rotation; + }; + + static int CUDAAPI OnSequence(void * decoder, CUVIDEOFORMAT * format); + static int CUDAAPI OnDecode(void * decoder, CUVIDPICPARAMS * picture); + static int CUDAAPI OnDisplay(void * decoder, CUVIDPARSERDISPINFO * info); + + // Each returns 0 to stop the parser, and a positive number to go + // on; the sequence callback returns the number of decode surfaces. + int HandleSequence(CUVIDEOFORMAT * format); + int HandleDecode(CUVIDPICPARAMS * picture); + int HandleDisplay(CUVIDPARSERDISPINFO * info); + + void DestroyDecoder(); + + // Copies the picture at the device pointer into nv12, cropped. + bool Download(unsigned long long devicePointer, unsigned int pitch); + + private: + NvdecLibrary & library; + const webrtc::VideoCodecType codec; + const cudaVideoCodec cudaCodec; + const std::string implementationName; + + webrtc::DecodedImageCallback * callback; + + CUcontext context; + CUvideoparser parser; + CUvideodecoder decoder; + + // Whether a key frame has made the decoder, and whether anything + // since then failed for good. + bool started; + bool failed; + + // The picture within the decoded surface, which is larger. + unsigned int surfaceHeight; + int left; + int top; + int width; + int height; + + // The decoded picture in system memory, NV12. + std::vector nv12; + + int64_t counter; + std::map pendingFrames; + }; +} + +#endif diff --git a/webrtc-jni/src/main/cpp/include/media/video/codec/nvdec/NvdecVideoDecoderFactory.h b/webrtc-jni/src/main/cpp/include/media/video/codec/nvdec/NvdecVideoDecoderFactory.h new file mode 100644 index 00000000..92a543ac --- /dev/null +++ b/webrtc-jni/src/main/cpp/include/media/video/codec/nvdec/NvdecVideoDecoderFactory.h @@ -0,0 +1,57 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef JNI_WEBRTC_MEDIA_VIDEO_CODEC_NVDEC_VIDEO_DECODER_FACTORY_H_ +#define JNI_WEBRTC_MEDIA_VIDEO_CODEC_NVDEC_VIDEO_DECODER_FACTORY_H_ + +#include "media/video/codec/nvdec/NvdecLibrary.h" + +#include "api/environment/environment.h" +#include "api/video_codecs/sdp_video_format.h" +#include "api/video_codecs/video_decoder.h" +#include "api/video_codecs/video_decoder_factory.h" + +#include +#include + +namespace jni +{ + // Creates the NVDEC decoders that decode on an NVIDIA GPU, for H.264 and + // VP9, whichever the GPU decodes. It offers them in the formats WebRTC's + // software decoders offer too: H.264 in all its profiles, and VP9 in + // profile 0. + class NvdecVideoDecoderFactory : public webrtc::VideoDecoderFactory + { + public: + // Returns a factory, or null if there is no NVIDIA driver, no CUDA + // device, or a device that decodes neither codec. + static std::unique_ptr Create(); + + ~NvdecVideoDecoderFactory() override = default; + + std::vector GetSupportedFormats() const override; + std::unique_ptr Create(const webrtc::Environment & env, + const webrtc::SdpVideoFormat & format) override; + + private: + explicit NvdecVideoDecoderFactory(NvdecLibrary & library); + + private: + NvdecLibrary & library; + }; +} + +#endif diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/linux/LinuxHardwareVideoCodecFactories.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/linux/LinuxHardwareVideoCodecFactories.cpp index 6eaab4c1..549a703d 100644 --- a/webrtc-jni/src/main/cpp/src/media/video/codec/linux/LinuxHardwareVideoCodecFactories.cpp +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/linux/LinuxHardwareVideoCodecFactories.cpp @@ -17,6 +17,7 @@ #include "media/video/codec/HardwareVideoDecoderFactory.h" #include "media/video/codec/HardwareVideoEncoderFactory.h" #include "media/video/codec/linux/VaapiVideoEncoderFactory.h" +#include "media/video/codec/nvdec/NvdecVideoDecoderFactory.h" #include "media/video/codec/nvenc/NvencVideoEncoderFactory.h" namespace jni @@ -39,7 +40,14 @@ namespace jni std::vector> CreatePlatformHardwareVideoDecoderFactories() { - // No hardware decoders on Linux yet. - return {}; + std::vector> factories; + + // NVDEC, on NVIDIA GPUs. Decoding with VA-API, for the GPUs of Intel + // and AMD, is to come. + if (auto nvdec = NvdecVideoDecoderFactory::Create()) { + factories.push_back(std::move(nvdec)); + } + + return factories; } } diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/nvdec/NvdecLibrary.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/nvdec/NvdecLibrary.cpp new file mode 100644 index 00000000..5405a1ed --- /dev/null +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/nvdec/NvdecLibrary.cpp @@ -0,0 +1,209 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "media/video/codec/nvdec/NvdecLibrary.h" + +#include "rtc_base/logging.h" + +#include +#include + +namespace jni +{ + namespace + { + constexpr const char * kCudaLibrary = "libcuda.so.1"; + constexpr const char * kCuvidLibrary = "libnvcuvid.so.1"; + } + + NvdecLibrary * NvdecLibrary::Get() + { + static std::once_flag once; + static NvdecLibrary * instance = nullptr; + + std::call_once(once, [] { + std::unique_ptr library(new NvdecLibrary()); + + if (library->Load()) { + instance = library.release(); + } + }); + + return instance; + } + + bool NvdecLibrary::Load() + { + if (!cuda.Open(kCudaLibrary) || !cuvid.Open(kCuvidLibrary)) { + RTC_LOG(LS_INFO) << "NVDEC: no NVIDIA driver"; + return false; + } + + bool resolved = cuda.Resolve("cuInit", api.cuInit) + && cuda.Resolve("cuDeviceGetCount", api.cuDeviceGetCount) + && cuda.Resolve("cuDeviceGet", api.cuDeviceGet) + && cuda.Resolve("cuDeviceGetName", api.cuDeviceGetName) + && cuda.Resolve("cuDevicePrimaryCtxRetain", api.cuDevicePrimaryCtxRetain) + && (cuda.Resolve("cuDevicePrimaryCtxRelease_v2", api.cuDevicePrimaryCtxRelease) + || cuda.Resolve("cuDevicePrimaryCtxRelease", api.cuDevicePrimaryCtxRelease)) + && cuda.Resolve("cuCtxPushCurrent_v2", api.cuCtxPushCurrent) + && cuda.Resolve("cuCtxPopCurrent_v2", api.cuCtxPopCurrent) + && cuda.Resolve("cuMemcpy2D_v2", api.cuMemcpy2D); + + resolved = resolved + && cuvid.Resolve("cuvidGetDecoderCaps", api.cuvidGetDecoderCaps) + && cuvid.Resolve("cuvidCreateDecoder", api.cuvidCreateDecoder) + && cuvid.Resolve("cuvidDestroyDecoder", api.cuvidDestroyDecoder) + && cuvid.Resolve("cuvidDecodePicture", api.cuvidDecodePicture) + && cuvid.Resolve("cuvidMapVideoFrame64", api.cuvidMapVideoFrame) + && cuvid.Resolve("cuvidUnmapVideoFrame64", api.cuvidUnmapVideoFrame) + && cuvid.Resolve("cuvidCreateVideoParser", api.cuvidCreateVideoParser) + && cuvid.Resolve("cuvidParseVideoData", api.cuvidParseVideoData) + && cuvid.Resolve("cuvidDestroyVideoParser", api.cuvidDestroyVideoParser); + + if (!resolved) { + RTC_LOG(LS_WARNING) << "NVDEC: the driver lacks functions this library needs"; + return false; + } + + int count = 0; + + if (api.cuInit(0) != CUDA_SUCCESS || api.cuDeviceGetCount(&count) != CUDA_SUCCESS || count == 0 || + api.cuDeviceGet(&device, 0) != CUDA_SUCCESS) + { + RTC_LOG(LS_INFO) << "NVDEC: no CUDA device"; + return false; + } + + char name[256] = {}; + + if (api.cuDeviceGetName(name, sizeof(name), device) == CUDA_SUCCESS) { + deviceName = name; + } + + CUcontext context = nullptr; + + if (!RetainContext(&context)) { + RTC_LOG(LS_WARNING) << "NVDEC: no CUDA context on " << deviceName; + return false; + } + + if (PushContext(context)) { + h264 = QueryCodec(cudaVideoCodec_H264); + vp9 = QueryCodec(cudaVideoCodec_VP9); + + PopContext(); + } + + ReleaseContext(); + + if (!h264 && !vp9) { + RTC_LOG(LS_INFO) << "NVDEC: " << deviceName << " decodes neither H.264 nor VP9"; + return false; + } + + RTC_LOG(LS_INFO) << "NVDEC available on " << deviceName << ", H.264: " << h264 << ", VP9: " << vp9; + + return true; + } + + bool NvdecLibrary::QueryCodec(cudaVideoCodec codec) const + { + CUVIDDECODECAPS caps = {}; + caps.eCodecType = codec; + caps.eChromaFormat = cudaVideoChromaFormat_420; + caps.nBitDepthMinus8 = 0; + + return api.cuvidGetDecoderCaps(&caps) == CUDA_SUCCESS && caps.bIsSupported != 0 + && (caps.nOutputFormatMask & (1 << cudaVideoSurfaceFormat_NV12)) != 0; + } + + bool NvdecLibrary::Supports(cudaVideoCodec codec, unsigned int width, unsigned int height) const + { + CUVIDDECODECAPS caps = {}; + caps.eCodecType = codec; + caps.eChromaFormat = cudaVideoChromaFormat_420; + caps.nBitDepthMinus8 = 0; + + if (api.cuvidGetDecoderCaps(&caps) != CUDA_SUCCESS || !caps.bIsSupported) { + return false; + } + + return width >= caps.nMinWidth && height >= caps.nMinHeight + && width <= caps.nMaxWidth && height <= caps.nMaxHeight + && (width / 16) * (height / 16) <= caps.nMaxMBCount; + } + + const NvdecApi & NvdecLibrary::Api() const + { + return api; + } + + const std::string & NvdecLibrary::DeviceName() const + { + return deviceName; + } + + bool NvdecLibrary::SupportsH264() const + { + return h264; + } + + bool NvdecLibrary::SupportsVp9() const + { + return vp9; + } + + bool NvdecLibrary::RetainContext(CUcontext * context) const + { + return api.cuDevicePrimaryCtxRetain(context, device) == CUDA_SUCCESS; + } + + void NvdecLibrary::ReleaseContext() const + { + api.cuDevicePrimaryCtxRelease(device); + } + + bool NvdecLibrary::PushContext(CUcontext context) const + { + return api.cuCtxPushCurrent(context) == CUDA_SUCCESS; + } + + void NvdecLibrary::PopContext() const + { + CUcontext context = nullptr; + + api.cuCtxPopCurrent(&context); + } + + NvdecContextScope::NvdecContextScope(const NvdecLibrary & library, CUcontext context) : + library(library), + pushed(library.PushContext(context)) + { + } + + NvdecContextScope::~NvdecContextScope() + { + if (pushed) { + library.PopContext(); + } + } + + bool NvdecContextScope::IsCurrent() const + { + return pushed; + } +} diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/nvdec/NvdecVideoDecoder.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/nvdec/NvdecVideoDecoder.cpp new file mode 100644 index 00000000..cf6d7dfd --- /dev/null +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/nvdec/NvdecVideoDecoder.cpp @@ -0,0 +1,464 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "media/video/codec/nvdec/NvdecVideoDecoder.h" + +#include "api/video/i420_buffer.h" +#include "api/video/video_frame.h" +#include "modules/video_coding/include/video_error_codes.h" +#include "rtc_base/logging.h" +#include "third_party/libyuv/include/libyuv/convert.h" + +#include +#include + +namespace jni +{ + namespace + { + cudaVideoCodec CudaCodecOf(webrtc::VideoCodecType codec) + { + return codec == webrtc::kVideoCodecVP9 ? cudaVideoCodec_VP9 : cudaVideoCodec_H264; + } + + // How many entries of a frame that never came out are kept. + constexpr size_t kMaxPendingFrames = 64; + } + + NvdecVideoDecoder::NvdecVideoDecoder(NvdecLibrary & library, webrtc::VideoCodecType codec) : + library(library), + codec(codec), + cudaCodec(CudaCodecOf(codec)), + implementationName("NVDEC (" + library.DeviceName() + ")"), + callback(nullptr), + context(nullptr), + parser(nullptr), + decoder(nullptr), + started(false), + failed(false), + surfaceHeight(0), + left(0), + top(0), + width(0), + height(0), + counter(0) + { + } + + NvdecVideoDecoder::~NvdecVideoDecoder() + { + Release(); + } + + bool NvdecVideoDecoder::Configure(const Settings & settings) + { + if (settings.codec_type() != codec) { + return false; + } + + Release(); + + if (!library.RetainContext(&context)) { + RTC_LOG(LS_WARNING) << implementationName << " has no CUDA context"; + context = nullptr; + + return false; + } + + NvdecContextScope scope(library, context); + + if (!scope.IsCurrent()) { + Release(); + + return false; + } + + CUVIDPARSERPARAMS params = {}; + params.CodecType = cudaCodec; + // The sequence callback says how many surfaces the decoder needs. + params.ulMaxNumDecodeSurfaces = 1; + // Each picture is shown as soon as it is decoded. + params.ulMaxDisplayDelay = 0; + params.pUserData = this; + params.pfnSequenceCallback = &NvdecVideoDecoder::OnSequence; + params.pfnDecodePicture = &NvdecVideoDecoder::OnDecode; + params.pfnDisplayPicture = &NvdecVideoDecoder::OnDisplay; + + if (library.Api().cuvidCreateVideoParser(&parser, ¶ms) != CUDA_SUCCESS) { + RTC_LOG(LS_WARNING) << implementationName << " failed to create a parser"; + parser = nullptr; + + Release(); + + return false; + } + + RTC_LOG(LS_INFO) << implementationName << " configured"; + + return true; + } + + int32_t NvdecVideoDecoder::Decode(const webrtc::EncodedImage & image, int64_t renderTimeMs) + { + if (parser == nullptr || callback == nullptr) { + return WEBRTC_VIDEO_CODEC_UNINITIALIZED; + } + if (image.size() == 0) { + return WEBRTC_VIDEO_CODEC_ERR_PARAMETER; + } + if (failed) { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + + // The layers of a VP9 frame with spatial layers reach a decoder back to + // back, without the superframe index that tells where one ends. libvpx + // takes them that way; NVDEC is not known to. + if (codec == webrtc::kVideoCodecVP9 + && (image.SpatialIndex().value_or(0) > 0 || image.SpatialLayerFrameSize(1).has_value())) + { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + + // The parser needs the parameters of the stream, which a key frame + // carries. Until one has come, WebRTC is asked for it. + if (!started && image.FrameType() != webrtc::VideoFrameType::kVideoFrameKey) { + return WEBRTC_VIDEO_CODEC_ERROR; + } + + NvdecContextScope scope(library, context); + + if (!scope.IsCurrent()) { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + + // The timestamp identifies the frame when it is shown. + const int64_t timestamp = ++counter; + + pendingFrames[timestamp] = PendingFrame { + image.RtpTimestamp(), + image.ntp_time_ms_, + renderTimeMs, + image.rotation_ + }; + + if (pendingFrames.size() > kMaxPendingFrames) { + pendingFrames.erase(pendingFrames.begin()); + } + + CUVIDSOURCEDATAPACKET packet = {}; + packet.flags = CUVID_PKT_TIMESTAMP | CUVID_PKT_ENDOFPICTURE; + packet.payload = image.data(); + packet.payload_size = image.size(); + packet.timestamp = timestamp; + + const CUresult result = library.Api().cuvidParseVideoData(parser, &packet); + + if (result != CUDA_SUCCESS || failed) { + RTC_LOG(LS_WARNING) << implementationName << " failed to decode a frame, result=" << result; + + failed = true; + + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + + return WEBRTC_VIDEO_CODEC_OK; + } + + int32_t NvdecVideoDecoder::RegisterDecodeCompleteCallback(webrtc::DecodedImageCallback * decodeCallback) + { + callback = decodeCallback; + + return WEBRTC_VIDEO_CODEC_OK; + } + + int32_t NvdecVideoDecoder::Release() + { + if (context != nullptr) { + { + NvdecContextScope scope(library, context); + + if (parser != nullptr) { + library.Api().cuvidDestroyVideoParser(parser); + parser = nullptr; + } + + DestroyDecoder(); + } + + library.ReleaseContext(); + context = nullptr; + } + + started = false; + failed = false; + pendingFrames.clear(); + + return WEBRTC_VIDEO_CODEC_OK; + } + + webrtc::VideoDecoder::DecoderInfo NvdecVideoDecoder::GetDecoderInfo() const + { + DecoderInfo info; + info.implementation_name = implementationName; + info.is_hardware_accelerated = true; + + return info; + } + + const char * NvdecVideoDecoder::ImplementationName() const + { + return implementationName.c_str(); + } + + int CUDAAPI NvdecVideoDecoder::OnSequence(void * decoder, CUVIDEOFORMAT * format) + { + return static_cast(decoder)->HandleSequence(format); + } + + int CUDAAPI NvdecVideoDecoder::OnDecode(void * decoder, CUVIDPICPARAMS * picture) + { + return static_cast(decoder)->HandleDecode(picture); + } + + int CUDAAPI NvdecVideoDecoder::OnDisplay(void * decoder, CUVIDPARSERDISPINFO * info) + { + return static_cast(decoder)->HandleDisplay(info); + } + + int NvdecVideoDecoder::HandleSequence(CUVIDEOFORMAT * format) + { + const int pictureLeft = format->display_area.left; + const int pictureTop = format->display_area.top; + const int pictureWidth = format->display_area.right - format->display_area.left; + const int pictureHeight = format->display_area.bottom - format->display_area.top; + + // Check, rather than trust what was negotiated: a stream that is + // something else is not for this decoder. + if (format->codec != cudaCodec || format->chroma_format != cudaVideoChromaFormat_420 + || format->bit_depth_luma_minus8 != 0 || format->bit_depth_chroma_minus8 != 0) + { + RTC_LOG(LS_INFO) << implementationName << " does not decode this stream, 4:2:0 8 bit only"; + failed = true; + + return 0; + } + if (pictureWidth <= 0 || pictureHeight <= 0 || (pictureLeft % 2) != 0 || (pictureTop % 2) != 0) { + RTC_LOG(LS_INFO) << implementationName << " does not take a picture that does not start on an even row"; + failed = true; + + return 0; + } + if (!library.Supports(cudaCodec, format->coded_width, format->coded_height)) { + RTC_LOG(LS_INFO) << implementationName << " does not decode " << format->coded_width << "x" + << format->coded_height; + failed = true; + + return 0; + } + + // The stream changed its parameters, as at a key frame of another + // size: the decoder is made again. + DestroyDecoder(); + + // The surface is aligned to 2, which the decoder asks for. + const unsigned int alignedWidth = (format->coded_width + 1) & ~1u; + const unsigned int alignedHeight = (format->coded_height + 1) & ~1u; + const unsigned int surfaces = std::max(format->min_num_decode_surfaces, 2) + 1; + + CUVIDDECODECREATEINFO info = {}; + info.ulWidth = format->coded_width; + info.ulHeight = format->coded_height; + info.ulNumDecodeSurfaces = surfaces; + info.CodecType = cudaCodec; + info.ChromaFormat = cudaVideoChromaFormat_420; + info.ulCreationFlags = cudaVideoCreate_PreferCUVID; + info.bitDepthMinus8 = 0; + info.ulMaxWidth = format->coded_width; + info.ulMaxHeight = format->coded_height; + info.display_area.left = 0; + info.display_area.top = 0; + info.display_area.right = static_cast(alignedWidth); + info.display_area.bottom = static_cast(alignedHeight); + info.OutputFormat = cudaVideoSurfaceFormat_NV12; + info.DeinterlaceMode = cudaVideoDeinterlaceMode_Weave; + info.ulTargetWidth = alignedWidth; + info.ulTargetHeight = alignedHeight; + info.ulNumOutputSurfaces = 2; + + if (library.Api().cuvidCreateDecoder(&decoder, &info) != CUDA_SUCCESS) { + RTC_LOG(LS_WARNING) << implementationName << " failed to create a decoder for " << format->coded_width + << "x" << format->coded_height; + decoder = nullptr; + failed = true; + + return 0; + } + + surfaceHeight = alignedHeight; + left = pictureLeft; + top = pictureTop; + width = pictureWidth; + height = pictureHeight; + started = true; + + return static_cast(surfaces); + } + + int NvdecVideoDecoder::HandleDecode(CUVIDPICPARAMS * picture) + { + if (decoder == nullptr) { + failed = true; + + return 0; + } + if (library.Api().cuvidDecodePicture(decoder, picture) != CUDA_SUCCESS) { + RTC_LOG(LS_WARNING) << implementationName << " failed to decode a picture"; + failed = true; + + return 0; + } + + return 1; + } + + int NvdecVideoDecoder::HandleDisplay(CUVIDPARSERDISPINFO * info) + { + if (info == nullptr) { + // The end of the stream. + return 1; + } + if (decoder == nullptr) { + failed = true; + + return 0; + } + + CUVIDPROCPARAMS process = {}; + process.progressive_frame = info->progressive_frame; + process.second_field = info->repeat_first_field + 1; + process.top_field_first = info->top_field_first; + process.unpaired_field = info->repeat_first_field < 0; + + unsigned long long devicePointer = 0; + unsigned int pitch = 0; + + if (library.Api().cuvidMapVideoFrame(decoder, info->picture_index, &devicePointer, &pitch, &process) + != CUDA_SUCCESS) + { + RTC_LOG(LS_WARNING) << implementationName << " failed to map a picture"; + failed = true; + + return 0; + } + + const bool downloaded = Download(devicePointer, pitch); + + library.Api().cuvidUnmapVideoFrame(decoder, devicePointer); + + if (!downloaded) { + RTC_LOG(LS_WARNING) << implementationName << " failed to copy a picture"; + failed = true; + + return 0; + } + + auto found = pendingFrames.find(info->timestamp); + + if (found == pendingFrames.end()) { + RTC_LOG(LS_WARNING) << implementationName << " showed a picture for no input, time " << info->timestamp; + + return 1; + } + + const PendingFrame pending = found->second; + + // Frames that were not shown come out never. + pendingFrames.erase(pendingFrames.begin(), std::next(found)); + + webrtc::scoped_refptr i420 = webrtc::I420Buffer::Create(width, height); + + const int yStride = width; + const int uvStride = (width + 1) / 2 * 2; + + if (libyuv::NV12ToI420(nv12.data(), yStride, nv12.data() + static_cast(yStride) * height, uvStride, + i420->MutableDataY(), i420->StrideY(), i420->MutableDataU(), i420->StrideU(), i420->MutableDataV(), + i420->StrideV(), width, height) != 0) + { + failed = true; + + return 0; + } + + webrtc::VideoFrame frame = webrtc::VideoFrame::Builder() + .set_video_frame_buffer(i420) + .set_rtp_timestamp(pending.rtpTimestamp) + .set_timestamp_ms(pending.renderTimeMs) + .set_ntp_time_ms(pending.ntpTimeMs) + .set_rotation(pending.rotation) + .build(); + + callback->Decoded(frame, std::nullopt, std::nullopt); + + return 1; + } + + bool NvdecVideoDecoder::Download(unsigned long long devicePointer, unsigned int pitch) + { + const size_t yStride = static_cast(width); + const size_t uvStride = static_cast((width + 1) / 2 * 2); + const size_t uvHeight = static_cast((height + 1) / 2); + + nv12.resize(yStride * height + uvStride * uvHeight); + + CUDA_MEMCPY2D copy = {}; + copy.srcMemoryType = CU_MEMORYTYPE_DEVICE; + copy.dstMemoryType = CU_MEMORYTYPE_HOST; + + // The luma, from the picture within the surface. + copy.srcDevice = devicePointer; + copy.srcPitch = pitch; + copy.srcXInBytes = static_cast(left); + copy.srcY = static_cast(top); + copy.dstHost = nv12.data(); + copy.dstPitch = yStride; + copy.WidthInBytes = yStride; + copy.Height = static_cast(height); + + if (library.Api().cuMemcpy2D(©) != CUDA_SUCCESS) { + return false; + } + + // The chroma follows the whole surface: interleaved U and V, half the + // rows. Its columns are in bytes, so they start where the luma does. + copy.srcDevice = devicePointer + static_cast(pitch) * surfaceHeight; + copy.srcXInBytes = static_cast(left); + copy.srcY = static_cast(top / 2); + copy.dstHost = nv12.data() + yStride * height; + copy.dstPitch = uvStride; + copy.WidthInBytes = uvStride; + copy.Height = uvHeight; + + return library.Api().cuMemcpy2D(©) == CUDA_SUCCESS; + } + + void NvdecVideoDecoder::DestroyDecoder() + { + if (decoder != nullptr) { + library.Api().cuvidDestroyDecoder(decoder); + decoder = nullptr; + } + } +} diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/nvdec/NvdecVideoDecoderFactory.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/nvdec/NvdecVideoDecoderFactory.cpp new file mode 100644 index 00000000..1be1d4ad --- /dev/null +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/nvdec/NvdecVideoDecoderFactory.cpp @@ -0,0 +1,60 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "media/video/codec/nvdec/NvdecVideoDecoderFactory.h" +#include "media/video/codec/nvdec/NvdecVideoDecoder.h" + +#include "api/video_codecs/video_codec.h" +#include "modules/video_coding/codecs/h264/include/h264.h" + +namespace jni +{ + std::unique_ptr NvdecVideoDecoderFactory::Create() + { + NvdecLibrary * library = NvdecLibrary::Get(); + + if (library == nullptr) { + return nullptr; + } + + return std::unique_ptr(new NvdecVideoDecoderFactory(*library)); + } + + NvdecVideoDecoderFactory::NvdecVideoDecoderFactory(NvdecLibrary & library) : + library(library) + { + } + + std::vector NvdecVideoDecoderFactory::GetSupportedFormats() const + { + std::vector formats; + + if (library.SupportsH264()) { + formats = webrtc::SupportedH264DecoderCodecs(); + } + if (library.SupportsVp9()) { + formats.push_back(webrtc::SdpVideoFormat::VP9Profile0()); + } + + return formats; + } + + std::unique_ptr NvdecVideoDecoderFactory::Create(const webrtc::Environment & env, + const webrtc::SdpVideoFormat & format) + { + return std::make_unique(library, webrtc::PayloadStringToCodecType(format.name)); + } +} diff --git a/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java b/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java index 458cb4cf..8f5e465d 100644 --- a/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java +++ b/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java @@ -48,7 +48,12 @@ * through the VP9 decoder of VideoToolbox, on Macs that have one; streams it * cannot decode, such as those with spatial layers, go to libvpx. Where there * is no hardware decoder, this factory decodes like a {@link - * DefaultVideoDecoderFactory}. Linux has no hardware decoders yet. + * DefaultVideoDecoderFactory}. + *

+ * On Linux, H.264 and VP9 (profile 0) are decoded on NVIDIA GPUs with NVDEC, + * loaded from the NVIDIA driver when the factory is made; a machine without + * the driver is not affected. Streams with spatial layers go to libvpx. + * Decoding with VA-API, on the GPUs of Intel and AMD, is not there yet. *

* The hardware decoders take over only codecs the software decoders have too, * so the supported codecs are the same as those of a {@link diff --git a/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java b/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java index 1774ff03..50238bcc 100644 --- a/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java +++ b/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java @@ -47,7 +47,7 @@ * runners do. Set the system property {@code webrtc.test.hardwareDecoder} to * {@code true} on a machine that has one, to make those tests fail instead, * and {@code webrtc.test.hardwareAv1Decoder} for a GPU that decodes AV1, and - * {@code webrtc.test.hardwareVp9Decoder} for one that decodes VP9 on Windows. + * {@code webrtc.test.hardwareVp9Decoder} for one that decodes VP9 on Windows or Linux. * On macOS the VP9 tests follow {@code webrtc.test.hardwareDecoder}. */ @Execution(ExecutionMode.SAME_THREAD) @@ -93,7 +93,8 @@ void hardwareDecodesAv1() throws Exception { } private void assertHardwareDecodes(Predicate codec, boolean required) throws Exception { - assumeTrue(OS.contains("win"), "hardware decoders are implemented on Windows only"); + assumeTrue(OS.contains("win") || OS.contains("linux"), + "hardware decoders are implemented on Windows and Linux only"); PeerConnectionFactory hardware = PeerConnectionFactory.builder() .setAudioDeviceModule(audioDevModule) @@ -109,7 +110,7 @@ private void assertHardwareDecodes(Predicate codec, boole hardware.dispose(); } - boolean hardwareUsed = implementation.contains("MediaFoundation"); + boolean hardwareUsed = isHardware(implementation); if (required) { assertTrue(hardwareUsed, implementation); @@ -181,6 +182,10 @@ void hardwareFollowsResolutionChange() throws Exception { CountDownLatch half = new CountDownLatch(10); try (TestMediaCall call = new TestMediaCall(hardware, true, false, VP9)) { + // Large enough that half the size is still one every hardware + // decoder takes: NVDEC does not decode VP9 below 128 pixels on the + // shorter side, and falls back to libvpx for such a stream. + call.setVideoSize(640, 480); call.negotiate(); RTCRtpReceiver receiver = call.getReceiver("video"); @@ -188,10 +193,10 @@ void hardwareFollowsResolutionChange() throws Exception { VideoTrackSink sink = frame -> { int width = frame.buffer.getWidth(); - if (width == 320 && frame.buffer.getHeight() == 240) { + if (width == 640 && frame.buffer.getHeight() == 480) { full.countDown(); } - else if (width == 160 && frame.buffer.getHeight() == 120) { + else if (width == 320 && frame.buffer.getHeight() == 240) { half.countDown(); } @@ -271,7 +276,7 @@ void vp9SpatialLayers() throws Exception { // The frames arrive, from libvpx: the layers of a frame come without a // superframe index, which VideoToolbox does not decode, and which a - // Media Foundation decoder is not known to. + // Media Foundation or NVDEC decoder is not known to. assertFalse(isHardware(result.decoder), result.decoder); } @@ -286,12 +291,13 @@ private PeerConnectionFactory hardwareFactory() { * Skips the test where VP9 is not decoded in hardware: macOS and Windows. */ private static void assumeVp9Platform() { - assumeTrue(OS.contains("mac") || OS.contains("win"), - "VP9 is decoded in hardware on macOS and Windows only"); + assumeTrue(OS.contains("mac") || OS.contains("win") || OS.contains("linux"), + "VP9 is decoded in hardware on macOS, Windows and Linux only"); } private static boolean isHardware(String implementation) { - return implementation.contains("VideoToolbox") || implementation.contains("MediaFoundation"); + return implementation.contains("VideoToolbox") || implementation.contains("MediaFoundation") + || implementation.contains("NVDEC"); } /**