diff --git a/docs/guide/advanced/video-codecs.md b/docs/guide/advanced/video-codecs.md index 7aad255b..c8cb8a55 100644 --- a/docs/guide/advanced/video-codecs.md +++ b/docs/guide/advanced/video-codecs.md @@ -53,11 +53,13 @@ PeerConnectionFactory factory = PeerConnectionFactory.builder() |---|---| | Windows | H.264 and AV1, on GPUs that decode them, through the Media Foundation decoders of Windows on Direct3D 11 (DXVA) | | Linux | Not yet; decoding is in software | -| macOS | VideoToolbox, as with `DefaultVideoDecoderFactory` | +| macOS | H.264 through VideoToolbox, as with `DefaultVideoDecoderFactory`, and VP9 (profile 0) through VideoToolbox's VP9 decoder, on Macs that have one | AV1 on Windows needs the *AV1 Video Extension*, which Windows 11 includes. Decoded frames are copied from GPU memory back to system memory, where WebRTC's frames are, so hardware decoding pays off mostly at high resolutions and with many streams; at low resolutions WebRTC's software decoders are about as cheap. A hardware decoder that fails, or turns out to decode in software, is replaced by the software decoder, which starts with the next key frame. -Which decoder a stream uses shows in the `decoderImplementation` statistic of its `inbound-rtp` stats, e.g. `MediaFoundation (Microsoft H264 Video Decoder MFT)` or `MediaFoundation (AV1VideoExtension)`. +On macOS, VP9 decoding in hardware saves CPU: in a measurement on an Apple M2 it took a fraction of the processor time libvpx needs, but multi-threaded libvpx was about as fast on the clock. VideoToolbox decodes VP9 in hardware only where the Mac has a decoder for it, which Apple silicon does; a Mac without one decodes VP9 with libvpx as before. Profile 2 (10 bit), sizes below 64x64 or above 4096x4096, and streams with spatial layers (SVC) are decoded with libvpx too. + +Which decoder a stream uses shows in the `decoderImplementation` statistic of its `inbound-rtp` stats, e.g. `MediaFoundation (Microsoft H264 Video Decoder MFT)`, `MediaFoundation (AV1VideoExtension)` or `VideoToolbox (VP9)`. ### Native Codecs diff --git a/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVideoDecoderFactory.h b/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVideoDecoderFactory.h new file mode 100644 index 00000000..3e8d18c6 --- /dev/null +++ b/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVideoDecoderFactory.h @@ -0,0 +1,53 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef JNI_WEBRTC_MEDIA_VIDEO_CODEC_VT_VIDEO_DECODER_FACTORY_H_ +#define JNI_WEBRTC_MEDIA_VIDEO_CODEC_VT_VIDEO_DECODER_FACTORY_H_ + +#include "api/environment/environment.h" +#include "api/video_codecs/sdp_video_format.h" +#include "api/video_codecs/video_decoder.h" +#include "api/video_codecs/video_decoder_factory.h" + +#include +#include + +namespace jni +{ + // Creates the VideoToolbox decoders that WebRTC's own decoders for macOS + // do not offer: VP9, in profile 0, the format the software factory + // offers first. H.264 is decoded through VideoToolbox by the default + // decoders already. + class VTVideoDecoderFactory : public webrtc::VideoDecoderFactory + { + public: + // Returns a factory, or null if this Mac has no hardware decoder + // for VP9. The decoder VideoToolbox has for VP9 has to be + // registered first; this does it, once for the process. + static std::unique_ptr Create(); + + ~VTVideoDecoderFactory() override = default; + + std::vector GetSupportedFormats() const override; + std::unique_ptr Create(const webrtc::Environment & env, + const webrtc::SdpVideoFormat & format) override; + + private: + VTVideoDecoderFactory() = default; + }; +} + +#endif diff --git a/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVp9Decoder.h b/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVp9Decoder.h new file mode 100644 index 00000000..d5361817 --- /dev/null +++ b/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVp9Decoder.h @@ -0,0 +1,117 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef JNI_WEBRTC_MEDIA_VIDEO_CODEC_VT_VP9_DECODER_H_ +#define JNI_WEBRTC_MEDIA_VIDEO_CODEC_VT_VP9_DECODER_H_ + +#include "api/video/encoded_image.h" +#include "api/video_codecs/video_decoder.h" +#include "modules/video_coding/utility/vp9_uncompressed_header_parser.h" + +#include +#include +#include + +#include + +namespace jni +{ + // Decodes VP9 profile 0 on the media engine or GPU of a Mac, through the + // VP9 decoder VideoToolbox offers once it is registered. + // + // The decompression session needs the properties of the stream, so it is + // created on the first key frame, from the header the key frame carries, + // and again whenever a key frame changes the size or the range. Decoding + // is synchronous: each frame comes out of VideoToolbox before Decode + // returns. A frame is handed on as the CVPixelBuffer VideoToolbox made, + // without a copy. + // + // Whatever VideoToolbox cannot decode in place goes to the software + // decoder, by returning WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE: a profile + // other than 0, a size outside what the hardware decodes, frames with + // spatial layers, and a decoder that keeps failing. + class VTVp9Decoder : public webrtc::VideoDecoder + { + public: + VTVp9Decoder(); + ~VTVp9Decoder() override; + + using webrtc::VideoDecoder::Decode; + + bool Configure(const Settings & settings) override; + int32_t Decode(const webrtc::EncodedImage & image, int64_t renderTimeMs) override; + int32_t RegisterDecodeCompleteCallback(webrtc::DecodedImageCallback * callback) override; + int32_t Release() override; + DecoderInfo GetDecoderInfo() const override; + const char * ImplementationName() const override; + + private: + // What a session is created for. A key frame that differs from it + // needs a new format description, and often a new session. + struct StreamConfig + { + int width = 0; + int height = 0; + bool fullRange = false; + webrtc::Vp9ColorSpace colorSpace = webrtc::Vp9ColorSpace::CS_UNKNOWN; + + bool operator==(const StreamConfig & other) const; + }; + + // The result of decoding one frame, filled in by the callback of + // the session. + struct Output + { + OSStatus status = noErr; + CVImageBufferRef image = nullptr; + }; + + static void OnOutput(void * decoder, void * frame, OSStatus status, VTDecodeInfoFlags flags, + CVImageBufferRef image, CMTime presentationTime, CMTime duration); + + // Reads the stream properties from a key frame header. Returns + // false for a stream the hardware decoder does not take. + static bool ReadConfig(const webrtc::Vp9UncompressedHeader & header, StreamConfig & config); + + // Makes sure there is a session for the stream, creating or + // replacing it where the stream changed. + bool EnsureSession(const StreamConfig & config); + bool CreateSession(const StreamConfig & config, CMVideoFormatDescriptionRef format); + void DestroySession(); + + // Decodes one encoded frame as a single sample. + OSStatus DecodeSample(const webrtc::EncodedImage & image, Output & output); + + // Counts a failure and tells WebRTC what to do about it. + int32_t Fail(OSStatus status); + + void Deliver(const webrtc::EncodedImage & image, int64_t renderTimeMs, CVImageBufferRef pixels); + + private: + webrtc::DecodedImageCallback * callback; + + VTDecompressionSessionRef session; + CMVideoFormatDescriptionRef format; + StreamConfig active; + + // Set after a failure, until the next key frame. + bool requireKeyFrame; + int consecutiveErrors; + int64_t sampleCount; + }; +} + +#endif diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/DefaultVideoCodecFactories.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/DefaultVideoCodecFactories.cpp index 6a82dcca..b0b6bce1 100644 --- a/webrtc-jni/src/main/cpp/src/media/video/codec/DefaultVideoCodecFactories.cpp +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/DefaultVideoCodecFactories.cpp @@ -54,9 +54,6 @@ namespace jni std::unique_ptr CreateHardwareVideoEncoderFactory() { -#ifdef __APPLE__ - return CreateDefaultVideoEncoderFactory(); -#else std::vector> hardware = CreatePlatformHardwareVideoEncoderFactories(); if (hardware.empty()) { @@ -64,7 +61,6 @@ namespace jni } return std::make_unique(std::move(hardware), CreateDefaultVideoEncoderFactory()); -#endif } std::unique_ptr CreateDefaultVideoDecoderFactory() @@ -82,9 +78,6 @@ namespace jni std::unique_ptr CreateHardwareVideoDecoderFactory() { -#ifdef __APPLE__ - return CreateDefaultVideoDecoderFactory(); -#else std::vector> hardware = CreatePlatformHardwareVideoDecoderFactories(); if (hardware.empty()) { @@ -92,6 +85,5 @@ namespace jni } return std::make_unique(std::move(hardware), CreateDefaultVideoDecoderFactory()); -#endif } } diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/macos/MacHardwareVideoCodecFactories.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/MacHardwareVideoCodecFactories.cpp new file mode 100644 index 00000000..08bc0ca7 --- /dev/null +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/MacHardwareVideoCodecFactories.cpp @@ -0,0 +1,42 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "media/video/codec/HardwareVideoDecoderFactory.h" +#include "media/video/codec/HardwareVideoEncoderFactory.h" +#include "media/video/codec/macos/VTVideoDecoderFactory.h" + +namespace jni +{ + std::vector> CreatePlatformHardwareVideoEncoderFactories() + { + // H.264 is encoded through VideoToolbox by the default encoders, and + // VideoToolbox has no encoder for the other codecs. + return {}; + } + + std::vector> CreatePlatformHardwareVideoDecoderFactories() + { + std::vector> factories; + + // H.264 is decoded through VideoToolbox by the default decoders; + // VP9 is the codec they decode in software. + if (auto videoToolbox = VTVideoDecoderFactory::Create()) { + factories.push_back(std::move(videoToolbox)); + } + + return factories; + } +} diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVideoDecoderFactory.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVideoDecoderFactory.cpp new file mode 100644 index 00000000..e7a2afed --- /dev/null +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVideoDecoderFactory.cpp @@ -0,0 +1,66 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "media/video/codec/macos/VTVideoDecoderFactory.h" +#include "media/video/codec/macos/VTVp9Decoder.h" + +#include "rtc_base/logging.h" + +#include + +namespace jni +{ + namespace + { + // Whether this Mac decodes VP9 in hardware. VideoToolbox has the VP9 + // decoder only after it is registered, so that comes first. The + // answer does not change while the process runs, and is asked once. + bool HasHardwareVp9Decoder() + { + static const bool available = [] { + VTRegisterSupplementalVideoDecoderIfAvailable(kCMVideoCodecType_VP9); + + const bool supported = VTIsHardwareDecodeSupported(kCMVideoCodecType_VP9); + + RTC_LOG(LS_INFO) << "VideoToolbox hardware decoder for VP9: " << supported; + + return supported; + }(); + + return available; + } + } + + std::unique_ptr VTVideoDecoderFactory::Create() + { + if (!HasHardwareVp9Decoder()) { + return nullptr; + } + + return std::unique_ptr(new VTVideoDecoderFactory()); + } + + std::vector VTVideoDecoderFactory::GetSupportedFormats() const + { + return { webrtc::SdpVideoFormat::VP9Profile0() }; + } + + std::unique_ptr VTVideoDecoderFactory::Create(const webrtc::Environment & env, + const webrtc::SdpVideoFormat & format) + { + return std::make_unique(); + } +} diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp new file mode 100644 index 00000000..e3eb1f29 --- /dev/null +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp @@ -0,0 +1,559 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "media/video/codec/macos/VTVp9Decoder.h" + +#include "api/make_ref_counted.h" +#include "api/video/video_frame.h" +#include "modules/video_coding/include/video_error_codes.h" +#include "rtc_base/logging.h" +#include "sdk/objc/components/video_frame_buffer/RTCCVPixelBuffer.h" +#include "sdk/objc/native/src/objc_frame_buffer.h" + +#include +#include + +namespace jni +{ + namespace + { + // The sizes the hardware decoder is used for. Outside them the + // software decoder takes over. Chromium uses the same limits for + // VP9 on VideoToolbox; the lower one is that of Apple silicon. + constexpr int kMinSize = 64; + constexpr int kMaxSize = 4096; + + // A decoder that fails this often in a row, each time on a frame + // WebRTC then replaced by a key frame, is given up on. + constexpr int kMaxConsecutiveErrors = 3; + + // How many sessions the process keeps at a time. The media engine + // serves a limited number of streams; a stream past the limit + // starts in software. The number is a cautious guess. + constexpr int kMaxSessions = 8; + + std::atomic sessionCount(0); + + // Releases a Core Foundation object when it goes out of scope. + template + class ScopedCF + { + public: + ScopedCF() : ref(nullptr) {} + explicit ScopedCF(T ref) : ref(ref) {} + ~ScopedCF() { if (ref) CFRelease(ref); } + + ScopedCF(const ScopedCF &) = delete; + ScopedCF & operator=(const ScopedCF &) = delete; + + T get() const { return ref; } + T * receive() { return &ref; } + explicit operator bool() const { return ref != nullptr; } + + // Gives up ownership. + T release() { T result = ref; ref = nullptr; return result; } + + private: + T ref; + }; + + ScopedCF CreateDictionary() + { + return ScopedCF(CFDictionaryCreateMutable(kCFAllocatorDefault, 0, + &kCFTypeDictionaryKeyCallBacks, &kCFTypeDictionaryValueCallBacks)); + } + + // How a colour space is written into the VP9 configuration box, as + // its code points, and into the format description. + struct ColorInfo + { + uint8_t primaries; + uint8_t transfer; + uint8_t matrix; + CFStringRef cmPrimaries; + CFStringRef cmTransfer; + CFStringRef cmMatrix; + }; + + ColorInfo GetColorInfo(webrtc::Vp9ColorSpace colorSpace) + { + switch (colorSpace) { + case webrtc::Vp9ColorSpace::CS_BT_601: + case webrtc::Vp9ColorSpace::CS_SMPTE_170: + return { 6, 6, 6, kCMFormatDescriptionColorPrimaries_SMPTE_C, + kCMFormatDescriptionTransferFunction_ITU_R_709_2, kCMFormatDescriptionYCbCrMatrix_ITU_R_601_4 }; + + case webrtc::Vp9ColorSpace::CS_BT_709: + return { 1, 1, 1, kCMFormatDescriptionColorPrimaries_ITU_R_709_2, + kCMFormatDescriptionTransferFunction_ITU_R_709_2, kCMFormatDescriptionYCbCrMatrix_ITU_R_709_2 }; + + case webrtc::Vp9ColorSpace::CS_SMPTE_240: + return { 7, 7, 7, kCMFormatDescriptionColorPrimaries_SMPTE_C, + kCMFormatDescriptionTransferFunction_SMPTE_240M_1995, + kCMFormatDescriptionYCbCrMatrix_SMPTE_240M_1995 }; + + case webrtc::Vp9ColorSpace::CS_BT_2020: + return { 9, 14, 9, kCMFormatDescriptionColorPrimaries_ITU_R_2020, + kCMFormatDescriptionTransferFunction_ITU_R_2020, kCMFormatDescriptionYCbCrMatrix_ITU_R_2020 }; + + default: + // Not signalled in the stream: unspecified. + return { 2, 2, 2, nullptr, nullptr, nullptr }; + } + } + + // The format description of a VP9 stream: its dimensions, the VP9 + // configuration box ("vpcC") and the colour properties. + CMVideoFormatDescriptionRef CreateFormat(int width, int height, bool fullRange, + webrtc::Vp9ColorSpace colorSpace) + { + const ColorInfo color = GetColorInfo(colorSpace); + + // Version 1 and flags 0 come first: VideoToolbox rejects the + // box without them. Then the profile, a level that covers the + // sizes in use, the bit depth, 4:2:0 co-located with luma (1) + // and the range, the colour code points, and no initialization + // data. + const uint8_t box[] = { + 1, 0, 0, 0, + 0, + 51, + static_cast((8 << 4) | (1 << 1) | (fullRange ? 1 : 0)), + color.primaries, color.transfer, color.matrix, + 0, 0 + }; + + ScopedCF boxData(CFDataCreate(kCFAllocatorDefault, box, sizeof(box))); + ScopedCF atoms = CreateDictionary(); + ScopedCF extensions = CreateDictionary(); + + if (!boxData || !atoms || !extensions) { + return nullptr; + } + + CFDictionarySetValue(atoms.get(), CFSTR("vpcC"), boxData.get()); + CFDictionarySetValue(extensions.get(), kCMFormatDescriptionExtension_SampleDescriptionExtensionAtoms, + atoms.get()); + CFDictionarySetValue(extensions.get(), kCMFormatDescriptionExtension_FullRangeVideo, + fullRange ? kCFBooleanTrue : kCFBooleanFalse); + + if (color.cmPrimaries != nullptr) { + CFDictionarySetValue(extensions.get(), kCMFormatDescriptionExtension_ColorPrimaries, color.cmPrimaries); + CFDictionarySetValue(extensions.get(), kCMFormatDescriptionExtension_TransferFunction, color.cmTransfer); + CFDictionarySetValue(extensions.get(), kCMFormatDescriptionExtension_YCbCrMatrix, color.cmMatrix); + } + + CMVideoFormatDescriptionRef format = nullptr; + OSStatus status = CMVideoFormatDescriptionCreate(kCFAllocatorDefault, kCMVideoCodecType_VP9, width, + height, extensions.get(), &format); + + if (status != noErr) { + RTC_LOG(LS_WARNING) << "VideoToolbox VP9 format description failed, status " << status; + return nullptr; + } + + return format; + } + } + + bool VTVp9Decoder::StreamConfig::operator==(const StreamConfig & other) const + { + return width == other.width && height == other.height && fullRange == other.fullRange + && colorSpace == other.colorSpace; + } + + VTVp9Decoder::VTVp9Decoder() : + callback(nullptr), + session(nullptr), + format(nullptr), + requireKeyFrame(false), + consecutiveErrors(0), + sampleCount(0) + { + } + + VTVp9Decoder::~VTVp9Decoder() + { + Release(); + } + + bool VTVp9Decoder::Configure(const Settings & settings) + { + if (settings.codec_type() != webrtc::kVideoCodecVP9) { + return false; + } + + Release(); + + // The session waits for the first key frame, which tells what to + // create it for. + return true; + } + + int32_t VTVp9Decoder::Decode(const webrtc::EncodedImage & image, int64_t renderTimeMs) + { + if (callback == nullptr) { + return WEBRTC_VIDEO_CODEC_UNINITIALIZED; + } + if (image.size() == 0) { + return WEBRTC_VIDEO_CODEC_ERR_PARAMETER; + } + + // A frame with spatial layers reaches the decoder with its layers + // back to back and no superframe index, which VideoToolbox does not + // decode. libvpx does. + if (image.SpatialIndex().value_or(0) > 0 || image.SpatialLayerFrameSize(1).has_value()) { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + + // The parser gives a header only for a frame that states its size: + // key frames, and inter frames that do not take it from a reference. + // Most inter frames take it, and have none. + const std::optional header = + webrtc::ParseUncompressedVp9Header(std::span(image.data(), image.size())); + + if (image.FrameType() == webrtc::VideoFrameType::kVideoFrameKey) { + StreamConfig config; + + // A key frame has to state what the session is made for. + if (!header || !ReadConfig(*header, config) || !EnsureSession(config)) { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + + requireKeyFrame = false; + } + else { + if (session == nullptr || requireKeyFrame) { + // Nothing to build on: WebRTC asks the sender for a key frame. + return WEBRTC_VIDEO_CODEC_ERROR; + } + + // A frame that states another size without being a key frame is + // reference scaling, which the session does not take. One that + // takes its size from a reference has no header to check. + if (header && !header->show_existing_frame + && (header->frame_width != active.width || header->frame_height != active.height)) + { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + } + + Output output; + OSStatus status = DecodeSample(image, output); + + if (status == noErr) { + status = output.status; + } + + ScopedCF pixels(output.image); + + if (status != noErr) { + return Fail(status); + } + + consecutiveErrors = 0; + + // A frame that is not shown, or that VideoToolbox dropped, has no + // picture. + if (pixels) { + Deliver(image, renderTimeMs, pixels.get()); + } + + return WEBRTC_VIDEO_CODEC_OK; + } + + int32_t VTVp9Decoder::RegisterDecodeCompleteCallback(webrtc::DecodedImageCallback * decodeCallback) + { + callback = decodeCallback; + + return WEBRTC_VIDEO_CODEC_OK; + } + + int32_t VTVp9Decoder::Release() + { + DestroySession(); + + requireKeyFrame = false; + consecutiveErrors = 0; + + return WEBRTC_VIDEO_CODEC_OK; + } + + webrtc::VideoDecoder::DecoderInfo VTVp9Decoder::GetDecoderInfo() const + { + DecoderInfo info; + info.implementation_name = ImplementationName(); + info.is_hardware_accelerated = true; + + return info; + } + + const char * VTVp9Decoder::ImplementationName() const + { + return "VideoToolbox (VP9)"; + } + + void VTVp9Decoder::OnOutput(void * decoder, void * frame, OSStatus status, VTDecodeInfoFlags flags, + CVImageBufferRef image, CMTime presentationTime, CMTime duration) + { + Output * output = static_cast(frame); + output->status = status; + + if (status == noErr && image != nullptr) { + output->image = static_cast(const_cast(CFRetain(image))); + } + } + + bool VTVp9Decoder::ReadConfig(const webrtc::Vp9UncompressedHeader & header, StreamConfig & config) + { + // The negotiated format is profile 0, 8 bit 4:2:0. Check, rather than + // trust it: a stream that is something else is not for this decoder. + if (header.profile != 0 || header.bit_detph != webrtc::Vp9BitDept::k8Bit) { + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: profile " << header.profile << " is not decoded in hardware"; + return false; + } + if (header.sub_sampling && *header.sub_sampling != webrtc::Vp9YuvSubsampling::k420) { + return false; + } + if (header.frame_width < kMinSize || header.frame_height < kMinSize + || header.frame_width > kMaxSize || header.frame_height > kMaxSize) + { + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: " << header.frame_width << "x" << header.frame_height + << " is not decoded in hardware"; + return false; + } + + config.width = header.frame_width; + config.height = header.frame_height; + // The range a stream does not state is the studio range, as in libvpx. + config.fullRange = header.color_range && *header.color_range == webrtc::Vp9ColorRange::kFull; + config.colorSpace = header.color_space.value_or(webrtc::Vp9ColorSpace::CS_UNKNOWN); + + return true; + } + + bool VTVp9Decoder::EnsureSession(const StreamConfig & config) + { + if (session != nullptr && config == active) { + return true; + } + + CMVideoFormatDescriptionRef newFormat = CreateFormat(config.width, config.height, config.fullRange, + config.colorSpace); + + if (newFormat == nullptr) { + return false; + } + + // The pixel format of the output follows the range, so a session does + // not outlast a change of it. Another size is not accepted in place + // either, but a change of the colour space may be. + if (session != nullptr && active.fullRange == config.fullRange + && VTDecompressionSessionCanAcceptFormatDescription(session, newFormat)) + { + CFRelease(format); + format = newFormat; + active = config; + + return true; + } + + // Whatever is still decoding finishes before the session goes. + DestroySession(); + + if (!CreateSession(config, newFormat)) { + CFRelease(newFormat); + + return false; + } + + format = newFormat; + active = config; + + return true; + } + + bool VTVp9Decoder::CreateSession(const StreamConfig & config, CMVideoFormatDescriptionRef newFormat) + { + if (sessionCount.load() >= kMaxSessions) { + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: too many sessions, using the software decoder"; + return false; + } + + ScopedCF specification = CreateDictionary(); + ScopedCF attributes = CreateDictionary(); + ScopedCF surfaceProperties(CFDictionaryCreate(kCFAllocatorDefault, nullptr, nullptr, 0, + &kCFTypeDictionaryKeyCallBacks, &kCFTypeDictionaryValueCallBacks)); + + // The output is 8 bit 4:2:0, in the range of the stream, so that the + // samples reach WebRTC as the software decoder would give them. + const int pixelFormat = config.fullRange + ? kCVPixelFormatType_420YpCbCr8BiPlanarFullRange + : kCVPixelFormatType_420YpCbCr8BiPlanarVideoRange; + ScopedCF pixelFormatNumber(CFNumberCreate(kCFAllocatorDefault, kCFNumberIntType, &pixelFormat)); + + if (!specification || !attributes || !surfaceProperties || !pixelFormatNumber) { + return false; + } + + // Only the hardware decoder: where there is none, or it is busy, the + // session fails to create and the software decoder takes over. + CFDictionarySetValue(specification.get(), kVTVideoDecoderSpecification_EnableHardwareAcceleratedVideoDecoder, + kCFBooleanTrue); + CFDictionarySetValue(specification.get(), kVTVideoDecoderSpecification_RequireHardwareAcceleratedVideoDecoder, + kCFBooleanTrue); + + CFDictionarySetValue(attributes.get(), kCVPixelBufferPixelFormatTypeKey, pixelFormatNumber.get()); + CFDictionarySetValue(attributes.get(), kCVPixelBufferIOSurfacePropertiesKey, surfaceProperties.get()); + + VTDecompressionOutputCallbackRecord record = { OnOutput, this }; + OSStatus status = VTDecompressionSessionCreate(kCFAllocatorDefault, newFormat, specification.get(), + attributes.get(), &record, &session); + + if (status != noErr) { + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: no session, status " << status; + session = nullptr; + + return false; + } + + // Do not rely on the requirement alone. + CFBooleanRef hardware = nullptr; + bool usingHardware = false; + + if (VTSessionCopyProperty(session, kVTDecompressionPropertyKey_UsingHardwareAcceleratedVideoDecoder, + kCFAllocatorDefault, &hardware) == noErr && hardware != nullptr) + { + usingHardware = CFBooleanGetValue(hardware); + CFRelease(hardware); + } + if (!usingHardware) { + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: the session does not decode in hardware"; + + VTDecompressionSessionInvalidate(session); + CFRelease(session); + session = nullptr; + + return false; + } + + sessionCount++; + + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: " << config.width << "x" << config.height + << (config.fullRange ? ", full range" : ", studio range"); + + return true; + } + + void VTVp9Decoder::DestroySession() + { + if (session != nullptr) { + VTDecompressionSessionWaitForAsynchronousFrames(session); + VTDecompressionSessionInvalidate(session); + CFRelease(session); + session = nullptr; + + sessionCount--; + } + if (format != nullptr) { + CFRelease(format); + format = nullptr; + } + + active = StreamConfig(); + } + + OSStatus VTVp9Decoder::DecodeSample(const webrtc::EncodedImage & image, Output & output) + { + // The whole encoded image is one sample, hidden frames and all: a + // frame that is not shown, given to VideoToolbox on its own, would + // come out as a picture. + ScopedCF block; + size_t size = image.size(); + + OSStatus status = CMBlockBufferCreateWithMemoryBlock(kCFAllocatorDefault, nullptr, size, + kCFAllocatorDefault, nullptr, 0, size, kCMBlockBufferAssureMemoryNowFlag, block.receive()); + + if (status == noErr) { + status = CMBlockBufferReplaceDataBytes(image.data(), block.get(), 0, size); + } + + ScopedCF sample; + + if (status == noErr) { + CMSampleTimingInfo timing = { CMTimeMake(1, 30), CMTimeMake(sampleCount++, 30), kCMTimeInvalid }; + + status = CMSampleBufferCreateReady(kCFAllocatorDefault, block.get(), format, 1, 1, &timing, 1, &size, + sample.receive()); + } + if (status != noErr) { + return status; + } + + VTDecodeInfoFlags info = 0; + + status = VTDecompressionSessionDecodeFrame(session, sample.get(), kVTDecodeFrame_1xRealTimePlayback, + &output, &info); + + if (status == noErr) { + // The frame is out before this returns, unless VideoToolbox + // decides to work asynchronously after all. + status = VTDecompressionSessionWaitForAsynchronousFrames(session); + } + + return status; + } + + int32_t VTVp9Decoder::Fail(OSStatus status) + { + RTC_LOG(LS_WARNING) << ImplementationName() << " failed, status " << status; + + requireKeyFrame = true; + + if (status == kVTInvalidSessionErr) { + DestroySession(); + } + if (++consecutiveErrors >= kMaxConsecutiveErrors) { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + + // WebRTC asks the sender for a key frame. + return WEBRTC_VIDEO_CODEC_ERROR; + } + + void VTVp9Decoder::Deliver(const webrtc::EncodedImage & image, int64_t renderTimeMs, CVImageBufferRef pixels) + { + // The buffer holds on to the pixels; WebRTC's frame holds on to the + // buffer, and converts to I420 only where a consumer needs it. + RTC_OBJC_TYPE(RTCCVPixelBuffer) * pixelBuffer = + [[RTC_OBJC_TYPE(RTCCVPixelBuffer) alloc] initWithPixelBuffer:pixels]; + + webrtc::scoped_refptr buffer = + webrtc::make_ref_counted(pixelBuffer); + + [pixelBuffer release]; + + webrtc::VideoFrame frame = webrtc::VideoFrame::Builder() + .set_video_frame_buffer(buffer) + .set_rtp_timestamp(image.RtpTimestamp()) + .set_timestamp_ms(renderTimeMs) + .set_ntp_time_ms(image.ntp_time_ms_) + .set_rotation(image.rotation_) + .build(); + + callback->Decoded(frame, std::nullopt, std::nullopt); + } +} diff --git a/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java b/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java index e42dcdd8..63cabaea 100644 --- a/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java +++ b/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java @@ -40,9 +40,12 @@ * in, so hardware decoding saves CPU mostly at high resolutions. A hardware * decoder that fails to start, or fails while decoding, is replaced by the * software decoder of the same codec, which starts with the next key frame. - * Where there is no hardware decoder, this factory decodes like a {@link - * DefaultVideoDecoderFactory}. On macOS, that already uses VideoToolbox. - * Linux has no hardware decoders yet. + * On macOS, H.264 is decoded through VideoToolbox by the {@link + * DefaultVideoDecoderFactory} already, and this factory adds VP9 (profile 0) + * through the VP9 decoder of VideoToolbox, on Macs that have one; streams it + * cannot decode, such as those with spatial layers, go to libvpx. Where there + * is no hardware decoder, this factory decodes like a {@link + * DefaultVideoDecoderFactory}. Linux has no hardware decoders yet. *

* The hardware decoders take over only codecs the software decoders have too, * so the supported codecs are the same as those of a {@link diff --git a/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java b/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java index b2b9e8c1..91523e76 100644 --- a/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java +++ b/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java @@ -17,15 +17,18 @@ package dev.onvoid.webrtc; import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; import static org.junit.jupiter.api.Assertions.assertNull; import static org.junit.jupiter.api.Assertions.assertTrue; import static org.junit.jupiter.api.Assumptions.assumeTrue; +import dev.onvoid.webrtc.media.video.I420Buffer; import dev.onvoid.webrtc.media.video.VideoTrack; import dev.onvoid.webrtc.media.video.VideoTrackSink; import dev.onvoid.webrtc.media.video.codec.DefaultVideoDecoderFactory; import dev.onvoid.webrtc.media.video.codec.HardwareVideoDecoderFactory; +import java.nio.ByteBuffer; import java.util.Locale; import java.util.Map; import java.util.concurrent.CountDownLatch; @@ -63,6 +66,10 @@ class HardwareVideoDecoderIntegrationTest extends TestBase { private static final Predicate AV1 = codec -> "AV1".equalsIgnoreCase(codec.getName()); + private static final Predicate VP9 = codec -> + "VP9".equalsIgnoreCase(codec.getName()) + && "0".equals(codec.getSDPFmtp().getOrDefault("profile-id", "0")); + @Test void hardwareKeepsCodecs() { @@ -115,8 +122,8 @@ void macDecodesH264WithVideoToolbox() throws Exception { // The default decoders use VideoToolbox on macOS. assertTrue(decoderImplementation(factory, H264).contains("VideoToolbox")); - // The hardware factory has nothing of its own there, and hands over to - // the default decoders. + // The hardware factory has nothing of its own for H.264 there, and + // hands over to the default decoders. PeerConnectionFactory hardware = PeerConnectionFactory.builder() .setAudioDeviceModule(audioDevModule) .setVideoDecoderFactory(new HardwareVideoDecoderFactory()) @@ -130,6 +137,230 @@ void macDecodesH264WithVideoToolbox() throws Exception { } } + @Test + void macDecodesVp9WithVideoToolbox() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + PeerConnectionFactory hardware = PeerConnectionFactory.builder() + .setAudioDeviceModule(audioDevModule) + .setVideoDecoderFactory(new HardwareVideoDecoderFactory()) + .build(); + + String implementation; + + try { + implementation = decoderImplementation(hardware, VP9); + } + finally { + hardware.dispose(); + } + + boolean hardwareUsed = implementation.contains("VideoToolbox"); + + if (HARDWARE_REQUIRED) { + assertTrue(hardwareUsed, implementation); + } + else { + assumeTrue(hardwareUsed, "no hardware decoder: " + implementation); + } + } + + @Test + void macDecodesVp9InSoftwareByDefault() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + // Hardware decoding is opt-in; the shared factory decodes VP9 with libvpx. + String implementation = decoderImplementation(factory, VP9); + + assertFalse(implementation.contains("VideoToolbox"), implementation); + } + + @Test + void macFollowsResolutionChange() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + PeerConnectionFactory hardware = PeerConnectionFactory.builder() + .setAudioDeviceModule(audioDevModule) + .setVideoDecoderFactory(new HardwareVideoDecoderFactory()) + .build(); + + CountDownLatch full = new CountDownLatch(10); + CountDownLatch half = new CountDownLatch(10); + + try (TestMediaCall call = new TestMediaCall(hardware, true, false, VP9)) { + call.negotiate(); + + RTCRtpReceiver receiver = call.getReceiver("video"); + VideoTrack track = (VideoTrack) receiver.getTrack(); + VideoTrackSink sink = frame -> { + int width = frame.buffer.getWidth(); + + if (width == 320 && frame.buffer.getHeight() == 240) { + full.countDown(); + } + else if (width == 160 && frame.buffer.getHeight() == 120) { + half.countDown(); + } + + frame.release(); + }; + track.addSink(sink); + + call.awaitConnected(); + call.startMedia(); + + assertTrue(full.await(TIMEOUT_SECONDS, TimeUnit.SECONDS), "too few frames received"); + + String implementation = decoderImplementationOf(call); + + assumeTrue(implementation.contains("VideoToolbox"), "no hardware decoder: " + implementation); + + // The sender restarts at half the size, with a key frame; the + // decoder has to start a session for it. + RTCRtpSender sender = call.getVideoSender(); + RTCRtpSendParameters parameters = sender.getParameters(); + + for (RTCRtpEncodingParameters encoding : parameters.encodings) { + encoding.scaleResolutionDownBy = 2.0; + } + + sender.setParameters(parameters); + + assertTrue(half.await(TIMEOUT_SECONDS, TimeUnit.SECONDS), "no frames at the new size"); + + // Still the hardware decoder, not a fallback. + assertTrue(decoderImplementationOf(call).contains("VideoToolbox")); + + track.removeSink(sink); + receiver.dispose(); + } + finally { + hardware.dispose(); + } + } + + @Test + void macVp9NeedsNoKeyFrames() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + CallResult result = receiveVideo(hardwareFactory(), VP9, 320, 240, null, 60); + + assumeTrue(result.decoder.contains("VideoToolbox"), "no hardware decoder: " + result.decoder); + + // An inter frame that fails to decode makes the receiver ask for a key + // frame, and every key frame is then the only frame that decodes. A + // stream that is decoded has the one key frame it started with. + assertTrue(result.pliCount <= 1, "key frames requested: " + result.pliCount); + assertTrue(result.keyFrames <= 2, "key frames decoded: " + result.keyFrames); + } + + @Test + void macVp9TemporalLayers() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + // Temporal layers are one spatial layer, which VideoToolbox decodes. + CallResult result = receiveVideo(hardwareFactory(), VP9, 320, 240, "L1T3", 60); + + assumeTrue(result.decoder.contains("VideoToolbox"), "no hardware decoder: " + result.decoder); + + assertTrue(result.pliCount <= 1, "key frames requested: " + result.pliCount); + } + + @Test + void macVp9SpatialLayers() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + // WebRTC encodes spatial layers only above a size, and drops them + // below it. + CallResult result = receiveVideo(hardwareFactory(), VP9, 640, 480, "L2T2", 60); + + assumeTrue(result.scalability.startsWith("L2"), "no spatial layers were encoded: " + result.scalability); + + // The frames arrive, from libvpx: VideoToolbox does not decode a + // frame whose layers come without a superframe index. + assertFalse(result.decoder.contains("VideoToolbox"), result.decoder); + } + + private PeerConnectionFactory hardwareFactory() { + return PeerConnectionFactory.builder() + .setAudioDeviceModule(audioDevModule) + .setVideoDecoderFactory(new HardwareVideoDecoderFactory()) + .build(); + } + + /** + * What a call that received video tells about its decoder. + */ + private static final class CallResult { + + String decoder = ""; + String scalability = ""; + long pliCount; + long keyFrames; + + } + + /** + * Receives the given number of frames through a call that sends video of + * the given size, and reads what the receiver reports. Disposes the + * factory. + */ + private static CallResult receiveVideo(PeerConnectionFactory hardware, Predicate codec, + int width, int height, String scalabilityMode, int frames) throws Exception { + CountDownLatch received = new CountDownLatch(frames); + CallResult result = new CallResult(); + + try (TestMediaCall call = new TestMediaCall(hardware, true, false, codec)) { + call.setVideoSize(width, height); + call.negotiate(); + + RTCRtpReceiver receiver = call.getReceiver("video"); + VideoTrack track = (VideoTrack) receiver.getTrack(); + VideoTrackSink sink = frame -> { + frame.release(); + received.countDown(); + }; + track.addSink(sink); + + call.awaitConnected(); + + if (scalabilityMode != null) { + RTCRtpSender sender = call.getVideoSender(); + RTCRtpSendParameters parameters = sender.getParameters(); + parameters.encodings.get(0).scalabilityMode = scalabilityMode; + + sender.setParameters(parameters); + } + + call.startMedia(); + + assertTrue(received.await(TIMEOUT_SECONDS, TimeUnit.SECONDS), "too few frames received"); + + result.decoder = decoderImplementationOf(call); + + Map inbound = call.getInboundVideoStats(); + Map outbound = call.getOutboundVideoStats(); + + result.pliCount = count(inbound, "pliCount"); + result.keyFrames = count(inbound, "keyFramesDecoded"); + result.scalability = outbound == null ? "" : String.valueOf(outbound.get("scalabilityMode")); + + track.removeSink(sink); + receiver.dispose(); + } + finally { + hardware.dispose(); + } + + return result; + } + + private static long count(Map stats, String name) { + Object value = stats == null ? null : stats.get(name); + + return value instanceof Number ? ((Number) value).longValue() : 0; + } + /** * Receives video in the preferred codec through a call, checks that the * decoded frames are the size that was sent, and returns what the receiver @@ -139,6 +370,7 @@ private static String decoderImplementation(PeerConnectionFactory factory, Predicate codec) throws Exception { CountDownLatch received = new CountDownLatch(10); AtomicReference wrongSize = new AtomicReference<>(); + AtomicReference blank = new AtomicReference<>(); String implementation; try (TestMediaCall call = new TestMediaCall(factory, true, false, codec)) { @@ -155,6 +387,9 @@ private static String decoderImplementation(PeerConnectionFactory factory, if (width != 320 || height != 240) { wrongSize.compareAndSet(null, width + "x" + height); } + else if (!hasPicture(frame.buffer.toI420())) { + blank.compareAndSet(null, "a frame without a picture"); + } frame.release(); received.countDown(); @@ -173,10 +408,32 @@ private static String decoderImplementation(PeerConnectionFactory factory, } assertNull(wrongSize.get(), "decoded frames of the wrong size: " + wrongSize.get()); + assertNull(blank.get(), "decoded " + blank.get()); return implementation; } + /** + * Whether the luma of the frame has the gradient the call sends, rather + * than a flat or empty picture. The buffer of a frame is I420 already, so + * it is read as it is and stays with the frame. + */ + private static boolean hasPicture(I420Buffer buffer) { + ByteBuffer y = buffer.getDataY(); + int min = 255; + int max = 0; + + // The first row has the whole ramp, and a wrap of it. + for (int x = 0; x < buffer.getWidth(); x++) { + int value = y.get(x) & 0xff; + + min = Math.min(min, value); + max = Math.max(max, value); + } + + return max - min > 150; + } + /** * Returns what the receiver reports its decoder to be, once the * statistics have caught up with it. diff --git a/webrtc/src/test/java/dev/onvoid/webrtc/TestMediaCall.java b/webrtc/src/test/java/dev/onvoid/webrtc/TestMediaCall.java index ebc4e243..118f09f6 100644 --- a/webrtc/src/test/java/dev/onvoid/webrtc/TestMediaCall.java +++ b/webrtc/src/test/java/dev/onvoid/webrtc/TestMediaCall.java @@ -46,8 +46,8 @@ */ class TestMediaCall implements AutoCloseable { - private static final int WIDTH = 320; - private static final int HEIGHT = 240; + private static final int DEFAULT_WIDTH = 320; + private static final int DEFAULT_HEIGHT = 240; private final CustomVideoSource videoSource; private final CustomAudioSource audioSource; @@ -63,6 +63,9 @@ class TestMediaCall implements AutoCloseable { private volatile boolean feeding; private Thread feeder; + private volatile int width = DEFAULT_WIDTH; + private volatile int height = DEFAULT_HEIGHT; + TestMediaCall(PeerConnectionFactory factory, boolean video, boolean audio) { this(factory, video, audio, "VP8"); @@ -126,6 +129,16 @@ void negotiate() throws Exception { caller.setRemoteDescription(callee.createAnswer()); } + /** + * Sets the size of the video the call sends, 320x240 unless changed. Call + * it before {@link #startMedia()}. Spatial layers need a larger picture: + * WebRTC encodes none below a certain size. + */ + void setVideoSize(int width, int height) { + this.width = width; + this.height = height; + } + /** * Waits for both ends to connect. */ @@ -309,7 +322,7 @@ private void feed() { } if (videoSource != null && videoFrames * 1000 / 30 <= elapsedMs) { - NativeI420Buffer buffer = NativeI420Buffer.allocate(WIDTH, HEIGHT); + NativeI420Buffer buffer = NativeI420Buffer.allocate(width, height); ByteBuffer y = buffer.getDataY(); // A moving gradient, so that frames differ from each other.