From 408c54477f1ea87dbbddcd9fa23d84fe48386797 Mon Sep 17 00:00:00 2001 From: Alex Andres Date: Fri, 2 Oct 2026 00:41:12 +0200 Subject: [PATCH 1/3] feat: decode VP9 with VideoToolbox on macOS HardwareVideoDecoderFactory now decodes VP9 profile 0 in hardware on macOS, through the VP9 decoder VideoToolbox offers once it is registered. H.264 stays with the default decoders, which use VideoToolbox already. The decoder creates its session on the first key frame, from the header parsed with WebRTC's VP9 header parser: the format description carries the vpcC box and the colour properties, and the output pixel format follows the range of the stream. The whole encoded image goes in as one sample, so a hidden frame is not decoded on its own. A key frame of another size or range replaces the session. Frames come out as the CVPixelBuffer VideoToolbox made, without a copy. Anything it cannot decode goes to libvpx through FallbackVideoDecoder: other profiles, sizes outside 64x64 to 4096x4096, frames with spatial layers, a session that cannot be created or is not in hardware, and a decoder that keeps failing. The macOS shortcuts in DefaultVideoCodecFactories are gone; macOS now defines the platform hooks like Windows and Linux do. Tests: VP9 through VideoToolbox, VP9 in software by default, and a resolution change in the middle of a call. A new check makes sure decoded frames carry a picture. --- docs/guide/advanced/video-codecs.md | 6 +- .../video/codec/macos/VTVideoDecoderFactory.h | 53 ++ .../media/video/codec/macos/VTVp9Decoder.h | 117 ++++ .../codec/DefaultVideoCodecFactories.cpp | 8 - .../macos/MacHardwareVideoCodecFactories.cpp | 42 ++ .../codec/macos/VTVideoDecoderFactory.cpp | 66 +++ .../media/video/codec/macos/VTVp9Decoder.cpp | 558 ++++++++++++++++++ .../codec/HardwareVideoDecoderFactory.java | 9 +- .../HardwareVideoDecoderIntegrationTest.java | 139 ++++- 9 files changed, 983 insertions(+), 15 deletions(-) create mode 100644 webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVideoDecoderFactory.h create mode 100644 webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVp9Decoder.h create mode 100644 webrtc-jni/src/main/cpp/src/media/video/codec/macos/MacHardwareVideoCodecFactories.cpp create mode 100644 webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVideoDecoderFactory.cpp create mode 100644 webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp diff --git a/docs/guide/advanced/video-codecs.md b/docs/guide/advanced/video-codecs.md index 7aad255b..c8cb8a55 100644 --- a/docs/guide/advanced/video-codecs.md +++ b/docs/guide/advanced/video-codecs.md @@ -53,11 +53,13 @@ PeerConnectionFactory factory = PeerConnectionFactory.builder() |---|---| | Windows | H.264 and AV1, on GPUs that decode them, through the Media Foundation decoders of Windows on Direct3D 11 (DXVA) | | Linux | Not yet; decoding is in software | -| macOS | VideoToolbox, as with `DefaultVideoDecoderFactory` | +| macOS | H.264 through VideoToolbox, as with `DefaultVideoDecoderFactory`, and VP9 (profile 0) through VideoToolbox's VP9 decoder, on Macs that have one | AV1 on Windows needs the *AV1 Video Extension*, which Windows 11 includes. Decoded frames are copied from GPU memory back to system memory, where WebRTC's frames are, so hardware decoding pays off mostly at high resolutions and with many streams; at low resolutions WebRTC's software decoders are about as cheap. A hardware decoder that fails, or turns out to decode in software, is replaced by the software decoder, which starts with the next key frame. -Which decoder a stream uses shows in the `decoderImplementation` statistic of its `inbound-rtp` stats, e.g. `MediaFoundation (Microsoft H264 Video Decoder MFT)` or `MediaFoundation (AV1VideoExtension)`. +On macOS, VP9 decoding in hardware saves CPU: in a measurement on an Apple M2 it took a fraction of the processor time libvpx needs, but multi-threaded libvpx was about as fast on the clock. VideoToolbox decodes VP9 in hardware only where the Mac has a decoder for it, which Apple silicon does; a Mac without one decodes VP9 with libvpx as before. Profile 2 (10 bit), sizes below 64x64 or above 4096x4096, and streams with spatial layers (SVC) are decoded with libvpx too. + +Which decoder a stream uses shows in the `decoderImplementation` statistic of its `inbound-rtp` stats, e.g. `MediaFoundation (Microsoft H264 Video Decoder MFT)`, `MediaFoundation (AV1VideoExtension)` or `VideoToolbox (VP9)`. ### Native Codecs diff --git a/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVideoDecoderFactory.h b/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVideoDecoderFactory.h new file mode 100644 index 00000000..3e8d18c6 --- /dev/null +++ b/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVideoDecoderFactory.h @@ -0,0 +1,53 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef JNI_WEBRTC_MEDIA_VIDEO_CODEC_VT_VIDEO_DECODER_FACTORY_H_ +#define JNI_WEBRTC_MEDIA_VIDEO_CODEC_VT_VIDEO_DECODER_FACTORY_H_ + +#include "api/environment/environment.h" +#include "api/video_codecs/sdp_video_format.h" +#include "api/video_codecs/video_decoder.h" +#include "api/video_codecs/video_decoder_factory.h" + +#include +#include + +namespace jni +{ + // Creates the VideoToolbox decoders that WebRTC's own decoders for macOS + // do not offer: VP9, in profile 0, the format the software factory + // offers first. H.264 is decoded through VideoToolbox by the default + // decoders already. + class VTVideoDecoderFactory : public webrtc::VideoDecoderFactory + { + public: + // Returns a factory, or null if this Mac has no hardware decoder + // for VP9. The decoder VideoToolbox has for VP9 has to be + // registered first; this does it, once for the process. + static std::unique_ptr Create(); + + ~VTVideoDecoderFactory() override = default; + + std::vector GetSupportedFormats() const override; + std::unique_ptr Create(const webrtc::Environment & env, + const webrtc::SdpVideoFormat & format) override; + + private: + VTVideoDecoderFactory() = default; + }; +} + +#endif diff --git a/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVp9Decoder.h b/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVp9Decoder.h new file mode 100644 index 00000000..d5361817 --- /dev/null +++ b/webrtc-jni/src/main/cpp/include/media/video/codec/macos/VTVp9Decoder.h @@ -0,0 +1,117 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#ifndef JNI_WEBRTC_MEDIA_VIDEO_CODEC_VT_VP9_DECODER_H_ +#define JNI_WEBRTC_MEDIA_VIDEO_CODEC_VT_VP9_DECODER_H_ + +#include "api/video/encoded_image.h" +#include "api/video_codecs/video_decoder.h" +#include "modules/video_coding/utility/vp9_uncompressed_header_parser.h" + +#include +#include +#include + +#include + +namespace jni +{ + // Decodes VP9 profile 0 on the media engine or GPU of a Mac, through the + // VP9 decoder VideoToolbox offers once it is registered. + // + // The decompression session needs the properties of the stream, so it is + // created on the first key frame, from the header the key frame carries, + // and again whenever a key frame changes the size or the range. Decoding + // is synchronous: each frame comes out of VideoToolbox before Decode + // returns. A frame is handed on as the CVPixelBuffer VideoToolbox made, + // without a copy. + // + // Whatever VideoToolbox cannot decode in place goes to the software + // decoder, by returning WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE: a profile + // other than 0, a size outside what the hardware decodes, frames with + // spatial layers, and a decoder that keeps failing. + class VTVp9Decoder : public webrtc::VideoDecoder + { + public: + VTVp9Decoder(); + ~VTVp9Decoder() override; + + using webrtc::VideoDecoder::Decode; + + bool Configure(const Settings & settings) override; + int32_t Decode(const webrtc::EncodedImage & image, int64_t renderTimeMs) override; + int32_t RegisterDecodeCompleteCallback(webrtc::DecodedImageCallback * callback) override; + int32_t Release() override; + DecoderInfo GetDecoderInfo() const override; + const char * ImplementationName() const override; + + private: + // What a session is created for. A key frame that differs from it + // needs a new format description, and often a new session. + struct StreamConfig + { + int width = 0; + int height = 0; + bool fullRange = false; + webrtc::Vp9ColorSpace colorSpace = webrtc::Vp9ColorSpace::CS_UNKNOWN; + + bool operator==(const StreamConfig & other) const; + }; + + // The result of decoding one frame, filled in by the callback of + // the session. + struct Output + { + OSStatus status = noErr; + CVImageBufferRef image = nullptr; + }; + + static void OnOutput(void * decoder, void * frame, OSStatus status, VTDecodeInfoFlags flags, + CVImageBufferRef image, CMTime presentationTime, CMTime duration); + + // Reads the stream properties from a key frame header. Returns + // false for a stream the hardware decoder does not take. + static bool ReadConfig(const webrtc::Vp9UncompressedHeader & header, StreamConfig & config); + + // Makes sure there is a session for the stream, creating or + // replacing it where the stream changed. + bool EnsureSession(const StreamConfig & config); + bool CreateSession(const StreamConfig & config, CMVideoFormatDescriptionRef format); + void DestroySession(); + + // Decodes one encoded frame as a single sample. + OSStatus DecodeSample(const webrtc::EncodedImage & image, Output & output); + + // Counts a failure and tells WebRTC what to do about it. + int32_t Fail(OSStatus status); + + void Deliver(const webrtc::EncodedImage & image, int64_t renderTimeMs, CVImageBufferRef pixels); + + private: + webrtc::DecodedImageCallback * callback; + + VTDecompressionSessionRef session; + CMVideoFormatDescriptionRef format; + StreamConfig active; + + // Set after a failure, until the next key frame. + bool requireKeyFrame; + int consecutiveErrors; + int64_t sampleCount; + }; +} + +#endif diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/DefaultVideoCodecFactories.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/DefaultVideoCodecFactories.cpp index 6a82dcca..b0b6bce1 100644 --- a/webrtc-jni/src/main/cpp/src/media/video/codec/DefaultVideoCodecFactories.cpp +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/DefaultVideoCodecFactories.cpp @@ -54,9 +54,6 @@ namespace jni std::unique_ptr CreateHardwareVideoEncoderFactory() { -#ifdef __APPLE__ - return CreateDefaultVideoEncoderFactory(); -#else std::vector> hardware = CreatePlatformHardwareVideoEncoderFactories(); if (hardware.empty()) { @@ -64,7 +61,6 @@ namespace jni } return std::make_unique(std::move(hardware), CreateDefaultVideoEncoderFactory()); -#endif } std::unique_ptr CreateDefaultVideoDecoderFactory() @@ -82,9 +78,6 @@ namespace jni std::unique_ptr CreateHardwareVideoDecoderFactory() { -#ifdef __APPLE__ - return CreateDefaultVideoDecoderFactory(); -#else std::vector> hardware = CreatePlatformHardwareVideoDecoderFactories(); if (hardware.empty()) { @@ -92,6 +85,5 @@ namespace jni } return std::make_unique(std::move(hardware), CreateDefaultVideoDecoderFactory()); -#endif } } diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/macos/MacHardwareVideoCodecFactories.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/MacHardwareVideoCodecFactories.cpp new file mode 100644 index 00000000..08bc0ca7 --- /dev/null +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/MacHardwareVideoCodecFactories.cpp @@ -0,0 +1,42 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "media/video/codec/HardwareVideoDecoderFactory.h" +#include "media/video/codec/HardwareVideoEncoderFactory.h" +#include "media/video/codec/macos/VTVideoDecoderFactory.h" + +namespace jni +{ + std::vector> CreatePlatformHardwareVideoEncoderFactories() + { + // H.264 is encoded through VideoToolbox by the default encoders, and + // VideoToolbox has no encoder for the other codecs. + return {}; + } + + std::vector> CreatePlatformHardwareVideoDecoderFactories() + { + std::vector> factories; + + // H.264 is decoded through VideoToolbox by the default decoders; + // VP9 is the codec they decode in software. + if (auto videoToolbox = VTVideoDecoderFactory::Create()) { + factories.push_back(std::move(videoToolbox)); + } + + return factories; + } +} diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVideoDecoderFactory.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVideoDecoderFactory.cpp new file mode 100644 index 00000000..e7a2afed --- /dev/null +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVideoDecoderFactory.cpp @@ -0,0 +1,66 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "media/video/codec/macos/VTVideoDecoderFactory.h" +#include "media/video/codec/macos/VTVp9Decoder.h" + +#include "rtc_base/logging.h" + +#include + +namespace jni +{ + namespace + { + // Whether this Mac decodes VP9 in hardware. VideoToolbox has the VP9 + // decoder only after it is registered, so that comes first. The + // answer does not change while the process runs, and is asked once. + bool HasHardwareVp9Decoder() + { + static const bool available = [] { + VTRegisterSupplementalVideoDecoderIfAvailable(kCMVideoCodecType_VP9); + + const bool supported = VTIsHardwareDecodeSupported(kCMVideoCodecType_VP9); + + RTC_LOG(LS_INFO) << "VideoToolbox hardware decoder for VP9: " << supported; + + return supported; + }(); + + return available; + } + } + + std::unique_ptr VTVideoDecoderFactory::Create() + { + if (!HasHardwareVp9Decoder()) { + return nullptr; + } + + return std::unique_ptr(new VTVideoDecoderFactory()); + } + + std::vector VTVideoDecoderFactory::GetSupportedFormats() const + { + return { webrtc::SdpVideoFormat::VP9Profile0() }; + } + + std::unique_ptr VTVideoDecoderFactory::Create(const webrtc::Environment & env, + const webrtc::SdpVideoFormat & format) + { + return std::make_unique(); + } +} diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp new file mode 100644 index 00000000..3437048a --- /dev/null +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp @@ -0,0 +1,558 @@ +/* + * Copyright 2026 Alex Andres + * + * Licensed under the Apache License, Version 2.0 (the "License"); + * you may not use this file except in compliance with the License. + * You may obtain a copy of the License at + * + * http://www.apache.org/licenses/LICENSE-2.0 + * + * Unless required by applicable law or agreed to in writing, software + * distributed under the License is distributed on an "AS IS" BASIS, + * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied. + * See the License for the specific language governing permissions and + * limitations under the License. + */ + +#include "media/video/codec/macos/VTVp9Decoder.h" + +#include "api/make_ref_counted.h" +#include "api/video/video_frame.h" +#include "modules/video_coding/include/video_error_codes.h" +#include "rtc_base/logging.h" +#include "sdk/objc/components/video_frame_buffer/RTCCVPixelBuffer.h" +#include "sdk/objc/native/src/objc_frame_buffer.h" + +#include +#include + +namespace jni +{ + namespace + { + // The sizes the hardware decoder is used for. Outside them the + // software decoder takes over. Chromium uses the same limits for + // VP9 on VideoToolbox; the lower one is that of Apple silicon. + constexpr int kMinSize = 64; + constexpr int kMaxSize = 4096; + + // A decoder that fails this often in a row, each time on a frame + // WebRTC then replaced by a key frame, is given up on. + constexpr int kMaxConsecutiveErrors = 3; + + // How many sessions the process keeps at a time. The media engine + // serves a limited number of streams; a stream past the limit + // starts in software. The number is a cautious guess. + constexpr int kMaxSessions = 8; + + std::atomic sessionCount(0); + + // Releases a Core Foundation object when it goes out of scope. + template + class ScopedCF + { + public: + ScopedCF() : ref(nullptr) {} + explicit ScopedCF(T ref) : ref(ref) {} + ~ScopedCF() { if (ref) CFRelease(ref); } + + ScopedCF(const ScopedCF &) = delete; + ScopedCF & operator=(const ScopedCF &) = delete; + + T get() const { return ref; } + T * receive() { return &ref; } + explicit operator bool() const { return ref != nullptr; } + + // Gives up ownership. + T release() { T result = ref; ref = nullptr; return result; } + + private: + T ref; + }; + + ScopedCF CreateDictionary() + { + return ScopedCF(CFDictionaryCreateMutable(kCFAllocatorDefault, 0, + &kCFTypeDictionaryKeyCallBacks, &kCFTypeDictionaryValueCallBacks)); + } + + // How a colour space is written into the VP9 configuration box, as + // its code points, and into the format description. + struct ColorInfo + { + uint8_t primaries; + uint8_t transfer; + uint8_t matrix; + CFStringRef cmPrimaries; + CFStringRef cmTransfer; + CFStringRef cmMatrix; + }; + + ColorInfo GetColorInfo(webrtc::Vp9ColorSpace colorSpace) + { + switch (colorSpace) { + case webrtc::Vp9ColorSpace::CS_BT_601: + case webrtc::Vp9ColorSpace::CS_SMPTE_170: + return { 6, 6, 6, kCMFormatDescriptionColorPrimaries_SMPTE_C, + kCMFormatDescriptionTransferFunction_ITU_R_709_2, kCMFormatDescriptionYCbCrMatrix_ITU_R_601_4 }; + + case webrtc::Vp9ColorSpace::CS_BT_709: + return { 1, 1, 1, kCMFormatDescriptionColorPrimaries_ITU_R_709_2, + kCMFormatDescriptionTransferFunction_ITU_R_709_2, kCMFormatDescriptionYCbCrMatrix_ITU_R_709_2 }; + + case webrtc::Vp9ColorSpace::CS_SMPTE_240: + return { 7, 7, 7, kCMFormatDescriptionColorPrimaries_SMPTE_C, + kCMFormatDescriptionTransferFunction_SMPTE_240M_1995, + kCMFormatDescriptionYCbCrMatrix_SMPTE_240M_1995 }; + + case webrtc::Vp9ColorSpace::CS_BT_2020: + return { 9, 14, 9, kCMFormatDescriptionColorPrimaries_ITU_R_2020, + kCMFormatDescriptionTransferFunction_ITU_R_2020, kCMFormatDescriptionYCbCrMatrix_ITU_R_2020 }; + + default: + // Not signalled in the stream: unspecified. + return { 2, 2, 2, nullptr, nullptr, nullptr }; + } + } + + // The format description of a VP9 stream: its dimensions, the VP9 + // configuration box ("vpcC") and the colour properties. + CMVideoFormatDescriptionRef CreateFormat(int width, int height, bool fullRange, + webrtc::Vp9ColorSpace colorSpace) + { + const ColorInfo color = GetColorInfo(colorSpace); + + // Version 1 and flags 0 come first: VideoToolbox rejects the + // box without them. Then the profile, a level that covers the + // sizes in use, the bit depth, 4:2:0 co-located with luma (1) + // and the range, the colour code points, and no initialization + // data. + const uint8_t box[] = { + 1, 0, 0, 0, + 0, + 51, + static_cast((8 << 4) | (1 << 1) | (fullRange ? 1 : 0)), + color.primaries, color.transfer, color.matrix, + 0, 0 + }; + + ScopedCF boxData(CFDataCreate(kCFAllocatorDefault, box, sizeof(box))); + ScopedCF atoms = CreateDictionary(); + ScopedCF extensions = CreateDictionary(); + + if (!boxData || !atoms || !extensions) { + return nullptr; + } + + CFDictionarySetValue(atoms.get(), CFSTR("vpcC"), boxData.get()); + CFDictionarySetValue(extensions.get(), kCMFormatDescriptionExtension_SampleDescriptionExtensionAtoms, + atoms.get()); + CFDictionarySetValue(extensions.get(), kCMFormatDescriptionExtension_FullRangeVideo, + fullRange ? kCFBooleanTrue : kCFBooleanFalse); + + if (color.cmPrimaries != nullptr) { + CFDictionarySetValue(extensions.get(), kCMFormatDescriptionExtension_ColorPrimaries, color.cmPrimaries); + CFDictionarySetValue(extensions.get(), kCMFormatDescriptionExtension_TransferFunction, color.cmTransfer); + CFDictionarySetValue(extensions.get(), kCMFormatDescriptionExtension_YCbCrMatrix, color.cmMatrix); + } + + CMVideoFormatDescriptionRef format = nullptr; + OSStatus status = CMVideoFormatDescriptionCreate(kCFAllocatorDefault, kCMVideoCodecType_VP9, width, + height, extensions.get(), &format); + + if (status != noErr) { + RTC_LOG(LS_WARNING) << "VideoToolbox VP9 format description failed, status " << status; + return nullptr; + } + + return format; + } + } + + bool VTVp9Decoder::StreamConfig::operator==(const StreamConfig & other) const + { + return width == other.width && height == other.height && fullRange == other.fullRange + && colorSpace == other.colorSpace; + } + + VTVp9Decoder::VTVp9Decoder() : + callback(nullptr), + session(nullptr), + format(nullptr), + requireKeyFrame(false), + consecutiveErrors(0), + sampleCount(0) + { + } + + VTVp9Decoder::~VTVp9Decoder() + { + Release(); + } + + bool VTVp9Decoder::Configure(const Settings & settings) + { + if (settings.codec_type() != webrtc::kVideoCodecVP9) { + return false; + } + + Release(); + + // The session waits for the first key frame, which tells what to + // create it for. + return true; + } + + int32_t VTVp9Decoder::Decode(const webrtc::EncodedImage & image, int64_t renderTimeMs) + { + if (callback == nullptr) { + return WEBRTC_VIDEO_CODEC_UNINITIALIZED; + } + if (image.size() == 0) { + return WEBRTC_VIDEO_CODEC_ERR_PARAMETER; + } + + // A frame with spatial layers reaches the decoder with its layers + // back to back and no superframe index, which VideoToolbox does not + // decode. libvpx does. + if (image.SpatialIndex().value_or(0) > 0 || image.SpatialLayerFrameSize(1).has_value()) { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + + const std::optional header = + webrtc::ParseUncompressedVp9Header(std::span(image.data(), image.size())); + + if (!header) { + return Fail(kVTVideoDecoderBadDataErr); + } + + if (image.FrameType() == webrtc::VideoFrameType::kVideoFrameKey) { + StreamConfig config; + + if (!ReadConfig(*header, config) || !EnsureSession(config)) { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + + requireKeyFrame = false; + } + else { + if (session == nullptr || requireKeyFrame) { + // Nothing to build on: WebRTC asks the sender for a key frame. + return WEBRTC_VIDEO_CODEC_ERROR; + } + + // Another size without a key frame, by reference scaling, is not + // something the session takes. + if (!header->show_existing_frame && header->frame_width != 0 + && (header->frame_width != active.width || header->frame_height != active.height)) + { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + } + + Output output; + OSStatus status = DecodeSample(image, output); + + if (status == noErr) { + status = output.status; + } + + ScopedCF pixels(output.image); + + if (status != noErr) { + return Fail(status); + } + + consecutiveErrors = 0; + + // A frame that is not shown, or that VideoToolbox dropped, has no + // picture. + if (pixels) { + Deliver(image, renderTimeMs, pixels.get()); + } + + return WEBRTC_VIDEO_CODEC_OK; + } + + int32_t VTVp9Decoder::RegisterDecodeCompleteCallback(webrtc::DecodedImageCallback * decodeCallback) + { + callback = decodeCallback; + + return WEBRTC_VIDEO_CODEC_OK; + } + + int32_t VTVp9Decoder::Release() + { + DestroySession(); + + requireKeyFrame = false; + consecutiveErrors = 0; + + return WEBRTC_VIDEO_CODEC_OK; + } + + webrtc::VideoDecoder::DecoderInfo VTVp9Decoder::GetDecoderInfo() const + { + DecoderInfo info; + info.implementation_name = ImplementationName(); + info.is_hardware_accelerated = true; + + return info; + } + + const char * VTVp9Decoder::ImplementationName() const + { + return "VideoToolbox (VP9)"; + } + + void VTVp9Decoder::OnOutput(void * decoder, void * frame, OSStatus status, VTDecodeInfoFlags flags, + CVImageBufferRef image, CMTime presentationTime, CMTime duration) + { + Output * output = static_cast(frame); + output->status = status; + + if (status == noErr && image != nullptr) { + output->image = static_cast(const_cast(CFRetain(image))); + } + } + + bool VTVp9Decoder::ReadConfig(const webrtc::Vp9UncompressedHeader & header, StreamConfig & config) + { + // The negotiated format is profile 0, 8 bit 4:2:0. Check, rather than + // trust it: a stream that is something else is not for this decoder. + if (header.profile != 0 || header.bit_detph != webrtc::Vp9BitDept::k8Bit) { + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: profile " << header.profile << " is not decoded in hardware"; + return false; + } + if (header.sub_sampling && *header.sub_sampling != webrtc::Vp9YuvSubsampling::k420) { + return false; + } + if (header.frame_width < kMinSize || header.frame_height < kMinSize + || header.frame_width > kMaxSize || header.frame_height > kMaxSize) + { + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: " << header.frame_width << "x" << header.frame_height + << " is not decoded in hardware"; + return false; + } + + config.width = header.frame_width; + config.height = header.frame_height; + // The range a stream does not state is the studio range, as in libvpx. + config.fullRange = header.color_range && *header.color_range == webrtc::Vp9ColorRange::kFull; + config.colorSpace = header.color_space.value_or(webrtc::Vp9ColorSpace::CS_UNKNOWN); + + return true; + } + + bool VTVp9Decoder::EnsureSession(const StreamConfig & config) + { + if (session != nullptr && config == active) { + return true; + } + + CMVideoFormatDescriptionRef newFormat = CreateFormat(config.width, config.height, config.fullRange, + config.colorSpace); + + if (newFormat == nullptr) { + return false; + } + + // The pixel format of the output follows the range, so a session does + // not outlast a change of it. Another size is not accepted in place + // either, but a change of the colour space may be. + if (session != nullptr && active.fullRange == config.fullRange + && VTDecompressionSessionCanAcceptFormatDescription(session, newFormat)) + { + CFRelease(format); + format = newFormat; + active = config; + + return true; + } + + // Whatever is still decoding finishes before the session goes. + DestroySession(); + + if (!CreateSession(config, newFormat)) { + CFRelease(newFormat); + + return false; + } + + format = newFormat; + active = config; + + return true; + } + + bool VTVp9Decoder::CreateSession(const StreamConfig & config, CMVideoFormatDescriptionRef newFormat) + { + if (sessionCount.load() >= kMaxSessions) { + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: too many sessions, using the software decoder"; + return false; + } + + ScopedCF specification = CreateDictionary(); + ScopedCF attributes = CreateDictionary(); + ScopedCF surfaceProperties(CFDictionaryCreate(kCFAllocatorDefault, nullptr, nullptr, 0, + &kCFTypeDictionaryKeyCallBacks, &kCFTypeDictionaryValueCallBacks)); + + // The output is 8 bit 4:2:0, in the range of the stream, so that the + // samples reach WebRTC as the software decoder would give them. + const int pixelFormat = config.fullRange + ? kCVPixelFormatType_420YpCbCr8BiPlanarFullRange + : kCVPixelFormatType_420YpCbCr8BiPlanarVideoRange; + ScopedCF pixelFormatNumber(CFNumberCreate(kCFAllocatorDefault, kCFNumberIntType, &pixelFormat)); + + if (!specification || !attributes || !surfaceProperties || !pixelFormatNumber) { + return false; + } + + // Only the hardware decoder: where there is none, or it is busy, the + // session fails to create and the software decoder takes over. + CFDictionarySetValue(specification.get(), kVTVideoDecoderSpecification_EnableHardwareAcceleratedVideoDecoder, + kCFBooleanTrue); + CFDictionarySetValue(specification.get(), kVTVideoDecoderSpecification_RequireHardwareAcceleratedVideoDecoder, + kCFBooleanTrue); + + CFDictionarySetValue(attributes.get(), kCVPixelBufferPixelFormatTypeKey, pixelFormatNumber.get()); + CFDictionarySetValue(attributes.get(), kCVPixelBufferIOSurfacePropertiesKey, surfaceProperties.get()); + + VTDecompressionOutputCallbackRecord record = { OnOutput, this }; + OSStatus status = VTDecompressionSessionCreate(kCFAllocatorDefault, newFormat, specification.get(), + attributes.get(), &record, &session); + + if (status != noErr) { + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: no session, status " << status; + session = nullptr; + + return false; + } + + // Do not rely on the requirement alone. + CFBooleanRef hardware = nullptr; + bool usingHardware = false; + + if (VTSessionCopyProperty(session, kVTDecompressionPropertyKey_UsingHardwareAcceleratedVideoDecoder, + kCFAllocatorDefault, &hardware) == noErr && hardware != nullptr) + { + usingHardware = CFBooleanGetValue(hardware); + CFRelease(hardware); + } + if (!usingHardware) { + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: the session does not decode in hardware"; + + VTDecompressionSessionInvalidate(session); + CFRelease(session); + session = nullptr; + + return false; + } + + sessionCount++; + + RTC_LOG(LS_INFO) << "VideoToolbox VP9 decoder: " << config.width << "x" << config.height + << (config.fullRange ? ", full range" : ", studio range"); + + return true; + } + + void VTVp9Decoder::DestroySession() + { + if (session != nullptr) { + VTDecompressionSessionWaitForAsynchronousFrames(session); + VTDecompressionSessionInvalidate(session); + CFRelease(session); + session = nullptr; + + sessionCount--; + } + if (format != nullptr) { + CFRelease(format); + format = nullptr; + } + + active = StreamConfig(); + } + + OSStatus VTVp9Decoder::DecodeSample(const webrtc::EncodedImage & image, Output & output) + { + // The whole encoded image is one sample, hidden frames and all: a + // frame that is not shown, given to VideoToolbox on its own, would + // come out as a picture. + ScopedCF block; + size_t size = image.size(); + + OSStatus status = CMBlockBufferCreateWithMemoryBlock(kCFAllocatorDefault, nullptr, size, + kCFAllocatorDefault, nullptr, 0, size, kCMBlockBufferAssureMemoryNowFlag, block.receive()); + + if (status == noErr) { + status = CMBlockBufferReplaceDataBytes(image.data(), block.get(), 0, size); + } + + ScopedCF sample; + + if (status == noErr) { + CMSampleTimingInfo timing = { CMTimeMake(1, 30), CMTimeMake(sampleCount++, 30), kCMTimeInvalid }; + + status = CMSampleBufferCreateReady(kCFAllocatorDefault, block.get(), format, 1, 1, &timing, 1, &size, + sample.receive()); + } + if (status != noErr) { + return status; + } + + VTDecodeInfoFlags info = 0; + + status = VTDecompressionSessionDecodeFrame(session, sample.get(), kVTDecodeFrame_1xRealTimePlayback, + &output, &info); + + if (status == noErr) { + // The frame is out before this returns, unless VideoToolbox + // decides to work asynchronously after all. + status = VTDecompressionSessionWaitForAsynchronousFrames(session); + } + + return status; + } + + int32_t VTVp9Decoder::Fail(OSStatus status) + { + RTC_LOG(LS_WARNING) << ImplementationName() << " failed, status " << status; + + requireKeyFrame = true; + + if (status == kVTInvalidSessionErr) { + DestroySession(); + } + if (++consecutiveErrors >= kMaxConsecutiveErrors) { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + + // WebRTC asks the sender for a key frame. + return WEBRTC_VIDEO_CODEC_ERROR; + } + + void VTVp9Decoder::Deliver(const webrtc::EncodedImage & image, int64_t renderTimeMs, CVImageBufferRef pixels) + { + // The buffer holds on to the pixels; WebRTC's frame holds on to the + // buffer, and converts to I420 only where a consumer needs it. + RTC_OBJC_TYPE(RTCCVPixelBuffer) * pixelBuffer = + [[RTC_OBJC_TYPE(RTCCVPixelBuffer) alloc] initWithPixelBuffer:pixels]; + + webrtc::scoped_refptr buffer = + webrtc::make_ref_counted(pixelBuffer); + + [pixelBuffer release]; + + webrtc::VideoFrame frame = webrtc::VideoFrame::Builder() + .set_video_frame_buffer(buffer) + .set_rtp_timestamp(image.RtpTimestamp()) + .set_timestamp_ms(renderTimeMs) + .set_ntp_time_ms(image.ntp_time_ms_) + .set_rotation(image.rotation_) + .build(); + + callback->Decoded(frame, std::nullopt, std::nullopt); + } +} diff --git a/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java b/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java index e42dcdd8..63cabaea 100644 --- a/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java +++ b/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java @@ -40,9 +40,12 @@ * in, so hardware decoding saves CPU mostly at high resolutions. A hardware * decoder that fails to start, or fails while decoding, is replaced by the * software decoder of the same codec, which starts with the next key frame. - * Where there is no hardware decoder, this factory decodes like a {@link - * DefaultVideoDecoderFactory}. On macOS, that already uses VideoToolbox. - * Linux has no hardware decoders yet. + * On macOS, H.264 is decoded through VideoToolbox by the {@link + * DefaultVideoDecoderFactory} already, and this factory adds VP9 (profile 0) + * through the VP9 decoder of VideoToolbox, on Macs that have one; streams it + * cannot decode, such as those with spatial layers, go to libvpx. Where there + * is no hardware decoder, this factory decodes like a {@link + * DefaultVideoDecoderFactory}. Linux has no hardware decoders yet. *

* The hardware decoders take over only codecs the software decoders have too, * so the supported codecs are the same as those of a {@link diff --git a/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java b/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java index b2b9e8c1..eb685fac 100644 --- a/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java +++ b/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java @@ -17,15 +17,18 @@ package dev.onvoid.webrtc; import static org.junit.jupiter.api.Assertions.assertEquals; +import static org.junit.jupiter.api.Assertions.assertFalse; import static org.junit.jupiter.api.Assertions.assertNull; import static org.junit.jupiter.api.Assertions.assertTrue; import static org.junit.jupiter.api.Assumptions.assumeTrue; +import dev.onvoid.webrtc.media.video.I420Buffer; import dev.onvoid.webrtc.media.video.VideoTrack; import dev.onvoid.webrtc.media.video.VideoTrackSink; import dev.onvoid.webrtc.media.video.codec.DefaultVideoDecoderFactory; import dev.onvoid.webrtc.media.video.codec.HardwareVideoDecoderFactory; +import java.nio.ByteBuffer; import java.util.Locale; import java.util.Map; import java.util.concurrent.CountDownLatch; @@ -63,6 +66,10 @@ class HardwareVideoDecoderIntegrationTest extends TestBase { private static final Predicate AV1 = codec -> "AV1".equalsIgnoreCase(codec.getName()); + private static final Predicate VP9 = codec -> + "VP9".equalsIgnoreCase(codec.getName()) + && "0".equals(codec.getSDPFmtp().getOrDefault("profile-id", "0")); + @Test void hardwareKeepsCodecs() { @@ -115,8 +122,8 @@ void macDecodesH264WithVideoToolbox() throws Exception { // The default decoders use VideoToolbox on macOS. assertTrue(decoderImplementation(factory, H264).contains("VideoToolbox")); - // The hardware factory has nothing of its own there, and hands over to - // the default decoders. + // The hardware factory has nothing of its own for H.264 there, and + // hands over to the default decoders. PeerConnectionFactory hardware = PeerConnectionFactory.builder() .setAudioDeviceModule(audioDevModule) .setVideoDecoderFactory(new HardwareVideoDecoderFactory()) @@ -130,6 +137,108 @@ void macDecodesH264WithVideoToolbox() throws Exception { } } + @Test + void macDecodesVp9WithVideoToolbox() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + PeerConnectionFactory hardware = PeerConnectionFactory.builder() + .setAudioDeviceModule(audioDevModule) + .setVideoDecoderFactory(new HardwareVideoDecoderFactory()) + .build(); + + String implementation; + + try { + implementation = decoderImplementation(hardware, VP9); + } + finally { + hardware.dispose(); + } + + boolean hardwareUsed = implementation.contains("VideoToolbox"); + + if (HARDWARE_REQUIRED) { + assertTrue(hardwareUsed, implementation); + } + else { + assumeTrue(hardwareUsed, "no hardware decoder: " + implementation); + } + } + + @Test + void macDecodesVp9InSoftwareByDefault() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + // Hardware decoding is opt-in; the shared factory decodes VP9 with libvpx. + String implementation = decoderImplementation(factory, VP9); + + assertFalse(implementation.contains("VideoToolbox"), implementation); + } + + @Test + void macFollowsResolutionChange() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + PeerConnectionFactory hardware = PeerConnectionFactory.builder() + .setAudioDeviceModule(audioDevModule) + .setVideoDecoderFactory(new HardwareVideoDecoderFactory()) + .build(); + + CountDownLatch full = new CountDownLatch(10); + CountDownLatch half = new CountDownLatch(10); + + try (TestMediaCall call = new TestMediaCall(hardware, true, false, VP9)) { + call.negotiate(); + + RTCRtpReceiver receiver = call.getReceiver("video"); + VideoTrack track = (VideoTrack) receiver.getTrack(); + VideoTrackSink sink = frame -> { + int width = frame.buffer.getWidth(); + + if (width == 320 && frame.buffer.getHeight() == 240) { + full.countDown(); + } + else if (width == 160 && frame.buffer.getHeight() == 120) { + half.countDown(); + } + + frame.release(); + }; + track.addSink(sink); + + call.awaitConnected(); + call.startMedia(); + + assertTrue(full.await(TIMEOUT_SECONDS, TimeUnit.SECONDS), "too few frames received"); + + String implementation = decoderImplementationOf(call); + + assumeTrue(implementation.contains("VideoToolbox"), "no hardware decoder: " + implementation); + + // The sender restarts at half the size, with a key frame; the + // decoder has to start a session for it. + RTCRtpSender sender = call.getVideoSender(); + RTCRtpSendParameters parameters = sender.getParameters(); + + for (RTCRtpEncodingParameters encoding : parameters.encodings) { + encoding.scaleResolutionDownBy = 2.0; + } + + sender.setParameters(parameters); + + assertTrue(half.await(TIMEOUT_SECONDS, TimeUnit.SECONDS), "no frames at the new size"); + + // Still the hardware decoder, not a fallback. + assertTrue(decoderImplementationOf(call).contains("VideoToolbox")); + + track.removeSink(sink); + receiver.dispose(); + } + finally { + hardware.dispose(); + } + } + /** * Receives video in the preferred codec through a call, checks that the * decoded frames are the size that was sent, and returns what the receiver @@ -139,6 +248,7 @@ private static String decoderImplementation(PeerConnectionFactory factory, Predicate codec) throws Exception { CountDownLatch received = new CountDownLatch(10); AtomicReference wrongSize = new AtomicReference<>(); + AtomicReference blank = new AtomicReference<>(); String implementation; try (TestMediaCall call = new TestMediaCall(factory, true, false, codec)) { @@ -155,6 +265,9 @@ private static String decoderImplementation(PeerConnectionFactory factory, if (width != 320 || height != 240) { wrongSize.compareAndSet(null, width + "x" + height); } + else if (!hasPicture(frame.buffer.toI420())) { + blank.compareAndSet(null, "a frame without a picture"); + } frame.release(); received.countDown(); @@ -173,10 +286,32 @@ private static String decoderImplementation(PeerConnectionFactory factory, } assertNull(wrongSize.get(), "decoded frames of the wrong size: " + wrongSize.get()); + assertNull(blank.get(), "decoded " + blank.get()); return implementation; } + /** + * Whether the luma of the frame has the gradient the call sends, rather + * than a flat or empty picture. The buffer of a frame is I420 already, so + * it is read as it is and stays with the frame. + */ + private static boolean hasPicture(I420Buffer buffer) { + ByteBuffer y = buffer.getDataY(); + int min = 255; + int max = 0; + + // The first row has the whole ramp, and a wrap of it. + for (int x = 0; x < buffer.getWidth(); x++) { + int value = y.get(x) & 0xff; + + min = Math.min(min, value); + max = Math.max(max, value); + } + + return max - min > 150; + } + /** * Returns what the receiver reports its decoder to be, once the * statistics have caught up with it. From 6479395a4e3d299aa62afb88621d2140393cb391 Mon Sep 17 00:00:00 2001 From: Alex Andres Date: Fri, 2 Oct 2026 08:29:24 +0200 Subject: [PATCH 2/3] fix: keep the VP9 decoder from failing inter frames the parser has no size for WebRTC's VP9 header parser gives a header only for a frame that states its size: key frames, and inter frames that do not take the size from a reference. Most inter frames take it, and the parser returns nothing for them. The decoder treated that as a failure, so in a stream without temporal layers every inter frame failed, the receiver asked for a key frame, and only that key frame decoded: a few frames a second, and a key frame request for nearly every frame. A frame without a header is decoded now. A key frame without one is handed to the software decoder, since there is nothing to create a session from. The tests did not notice because they waited for ten frames and checked the picture and the size. New ones read the statistics of the receiver, that a stream asks for no more than one key frame, and cover temporal layers and spatial layers, which a real libwebrtc stream now makes: L2T2 at 640x480 ends up in libvpx. The test call can send another size than 320x240 for that, since WebRTC encodes no spatial layers below a size. --- .../media/video/codec/macos/VTVp9Decoder.cpp | 17 +-- .../HardwareVideoDecoderIntegrationTest.java | 122 ++++++++++++++++++ .../java/dev/onvoid/webrtc/TestMediaCall.java | 19 ++- 3 files changed, 147 insertions(+), 11 deletions(-) diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp index 3437048a..e3eb1f29 100644 --- a/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/macos/VTVp9Decoder.cpp @@ -219,17 +219,17 @@ namespace jni return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; } + // The parser gives a header only for a frame that states its size: + // key frames, and inter frames that do not take it from a reference. + // Most inter frames take it, and have none. const std::optional header = webrtc::ParseUncompressedVp9Header(std::span(image.data(), image.size())); - if (!header) { - return Fail(kVTVideoDecoderBadDataErr); - } - if (image.FrameType() == webrtc::VideoFrameType::kVideoFrameKey) { StreamConfig config; - if (!ReadConfig(*header, config) || !EnsureSession(config)) { + // A key frame has to state what the session is made for. + if (!header || !ReadConfig(*header, config) || !EnsureSession(config)) { return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; } @@ -241,9 +241,10 @@ namespace jni return WEBRTC_VIDEO_CODEC_ERROR; } - // Another size without a key frame, by reference scaling, is not - // something the session takes. - if (!header->show_existing_frame && header->frame_width != 0 + // A frame that states another size without being a key frame is + // reference scaling, which the session does not take. One that + // takes its size from a reference has no header to check. + if (header && !header->show_existing_frame && (header->frame_width != active.width || header->frame_height != active.height)) { return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; diff --git a/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java b/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java index eb685fac..91523e76 100644 --- a/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java +++ b/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java @@ -239,6 +239,128 @@ else if (width == 160 && frame.buffer.getHeight() == 120) { } } + @Test + void macVp9NeedsNoKeyFrames() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + CallResult result = receiveVideo(hardwareFactory(), VP9, 320, 240, null, 60); + + assumeTrue(result.decoder.contains("VideoToolbox"), "no hardware decoder: " + result.decoder); + + // An inter frame that fails to decode makes the receiver ask for a key + // frame, and every key frame is then the only frame that decodes. A + // stream that is decoded has the one key frame it started with. + assertTrue(result.pliCount <= 1, "key frames requested: " + result.pliCount); + assertTrue(result.keyFrames <= 2, "key frames decoded: " + result.keyFrames); + } + + @Test + void macVp9TemporalLayers() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + // Temporal layers are one spatial layer, which VideoToolbox decodes. + CallResult result = receiveVideo(hardwareFactory(), VP9, 320, 240, "L1T3", 60); + + assumeTrue(result.decoder.contains("VideoToolbox"), "no hardware decoder: " + result.decoder); + + assertTrue(result.pliCount <= 1, "key frames requested: " + result.pliCount); + } + + @Test + void macVp9SpatialLayers() throws Exception { + assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + + // WebRTC encodes spatial layers only above a size, and drops them + // below it. + CallResult result = receiveVideo(hardwareFactory(), VP9, 640, 480, "L2T2", 60); + + assumeTrue(result.scalability.startsWith("L2"), "no spatial layers were encoded: " + result.scalability); + + // The frames arrive, from libvpx: VideoToolbox does not decode a + // frame whose layers come without a superframe index. + assertFalse(result.decoder.contains("VideoToolbox"), result.decoder); + } + + private PeerConnectionFactory hardwareFactory() { + return PeerConnectionFactory.builder() + .setAudioDeviceModule(audioDevModule) + .setVideoDecoderFactory(new HardwareVideoDecoderFactory()) + .build(); + } + + /** + * What a call that received video tells about its decoder. + */ + private static final class CallResult { + + String decoder = ""; + String scalability = ""; + long pliCount; + long keyFrames; + + } + + /** + * Receives the given number of frames through a call that sends video of + * the given size, and reads what the receiver reports. Disposes the + * factory. + */ + private static CallResult receiveVideo(PeerConnectionFactory hardware, Predicate codec, + int width, int height, String scalabilityMode, int frames) throws Exception { + CountDownLatch received = new CountDownLatch(frames); + CallResult result = new CallResult(); + + try (TestMediaCall call = new TestMediaCall(hardware, true, false, codec)) { + call.setVideoSize(width, height); + call.negotiate(); + + RTCRtpReceiver receiver = call.getReceiver("video"); + VideoTrack track = (VideoTrack) receiver.getTrack(); + VideoTrackSink sink = frame -> { + frame.release(); + received.countDown(); + }; + track.addSink(sink); + + call.awaitConnected(); + + if (scalabilityMode != null) { + RTCRtpSender sender = call.getVideoSender(); + RTCRtpSendParameters parameters = sender.getParameters(); + parameters.encodings.get(0).scalabilityMode = scalabilityMode; + + sender.setParameters(parameters); + } + + call.startMedia(); + + assertTrue(received.await(TIMEOUT_SECONDS, TimeUnit.SECONDS), "too few frames received"); + + result.decoder = decoderImplementationOf(call); + + Map inbound = call.getInboundVideoStats(); + Map outbound = call.getOutboundVideoStats(); + + result.pliCount = count(inbound, "pliCount"); + result.keyFrames = count(inbound, "keyFramesDecoded"); + result.scalability = outbound == null ? "" : String.valueOf(outbound.get("scalabilityMode")); + + track.removeSink(sink); + receiver.dispose(); + } + finally { + hardware.dispose(); + } + + return result; + } + + private static long count(Map stats, String name) { + Object value = stats == null ? null : stats.get(name); + + return value instanceof Number ? ((Number) value).longValue() : 0; + } + /** * Receives video in the preferred codec through a call, checks that the * decoded frames are the size that was sent, and returns what the receiver diff --git a/webrtc/src/test/java/dev/onvoid/webrtc/TestMediaCall.java b/webrtc/src/test/java/dev/onvoid/webrtc/TestMediaCall.java index ebc4e243..118f09f6 100644 --- a/webrtc/src/test/java/dev/onvoid/webrtc/TestMediaCall.java +++ b/webrtc/src/test/java/dev/onvoid/webrtc/TestMediaCall.java @@ -46,8 +46,8 @@ */ class TestMediaCall implements AutoCloseable { - private static final int WIDTH = 320; - private static final int HEIGHT = 240; + private static final int DEFAULT_WIDTH = 320; + private static final int DEFAULT_HEIGHT = 240; private final CustomVideoSource videoSource; private final CustomAudioSource audioSource; @@ -63,6 +63,9 @@ class TestMediaCall implements AutoCloseable { private volatile boolean feeding; private Thread feeder; + private volatile int width = DEFAULT_WIDTH; + private volatile int height = DEFAULT_HEIGHT; + TestMediaCall(PeerConnectionFactory factory, boolean video, boolean audio) { this(factory, video, audio, "VP8"); @@ -126,6 +129,16 @@ void negotiate() throws Exception { caller.setRemoteDescription(callee.createAnswer()); } + /** + * Sets the size of the video the call sends, 320x240 unless changed. Call + * it before {@link #startMedia()}. Spatial layers need a larger picture: + * WebRTC encodes none below a certain size. + */ + void setVideoSize(int width, int height) { + this.width = width; + this.height = height; + } + /** * Waits for both ends to connect. */ @@ -309,7 +322,7 @@ private void feed() { } if (videoSource != null && videoFrames * 1000 / 30 <= elapsedMs) { - NativeI420Buffer buffer = NativeI420Buffer.allocate(WIDTH, HEIGHT); + NativeI420Buffer buffer = NativeI420Buffer.allocate(width, height); ByteBuffer y = buffer.getDataY(); // A moving gradient, so that frames differ from each other. From 3c5776b2603ce6f712f2435a46f65bba8c9c0d69 Mon Sep 17 00:00:00 2001 From: Alex Andres Date: Fri, 2 Oct 2026 08:42:07 +0200 Subject: [PATCH 3/3] feat: decode VP9 on the GPU with Media Foundation on Windows HardwareVideoDecoderFactory now decodes VP9 profile 0 on Windows, through the VP9 decoder Media Foundation has on Direct3D 11, the way it does H.264 and AV1. The factory offers VP9 where the GPU has the DXVA decoder profile (D3D11_DECODER_PROFILE_VP9_VLD_PROFILE0, with NV12 output) and Windows has a VP9 decoder that uses Direct3D 11: the VP9 Video Extensions of the Microsoft Store. Without either, VP9 stays with libvpx, and negotiation does not change. MFVideoDecoder takes VP9 as a third codec. Two things are particular to it: - A frame with spatial layers goes to libvpx. Its layers reach a decoder back to back without a superframe index; VideoToolbox does not decode that, and a Media Foundation decoder is not known to. - A VP9 stream states its size in the key frames only, so the decoder may have no output type to offer before it has seen one. That is no longer a failure to configure; the output type is set when the transform asks for it, as it is for a change of size. The VP9 tests of the decoder test class run on Windows too, not on macOS only. A VP9 decoder is required with -Dwebrtc.test.hardwareVp9Decoder=true on Windows, as an AV1 decoder is with its own property; on macOS the existing property applies. This has not been built or run on Windows. --- docs/guide/advanced/video-codecs.md | 4 +- .../video/codec/windows/MFVideoDecoder.h | 4 +- .../codec/windows/MFVideoDecoderFactory.h | 12 ++- .../video/codec/windows/MFVideoDecoder.cpp | 28 +++++- .../codec/windows/MFVideoDecoderFactory.cpp | 18 +++- .../codec/HardwareVideoDecoderFactory.java | 7 +- .../HardwareVideoDecoderIntegrationTest.java | 91 ++++++++++++------- 7 files changed, 112 insertions(+), 52 deletions(-) diff --git a/docs/guide/advanced/video-codecs.md b/docs/guide/advanced/video-codecs.md index c8cb8a55..2d759ea5 100644 --- a/docs/guide/advanced/video-codecs.md +++ b/docs/guide/advanced/video-codecs.md @@ -51,11 +51,11 @@ PeerConnectionFactory factory = PeerConnectionFactory.builder() | Platform | Hardware decoding | |---|---| -| Windows | H.264 and AV1, on GPUs that decode them, through the Media Foundation decoders of Windows on Direct3D 11 (DXVA) | +| Windows | H.264, AV1 and VP9 (profile 0), on GPUs that decode them, through the Media Foundation decoders of Windows on Direct3D 11 (DXVA) | | Linux | Not yet; decoding is in software | | macOS | H.264 through VideoToolbox, as with `DefaultVideoDecoderFactory`, and VP9 (profile 0) through VideoToolbox's VP9 decoder, on Macs that have one | -AV1 on Windows needs the *AV1 Video Extension*, which Windows 11 includes. Decoded frames are copied from GPU memory back to system memory, where WebRTC's frames are, so hardware decoding pays off mostly at high resolutions and with many streams; at low resolutions WebRTC's software decoders are about as cheap. A hardware decoder that fails, or turns out to decode in software, is replaced by the software decoder, which starts with the next key frame. +AV1 on Windows needs the *AV1 Video Extension*, which Windows 11 includes, and VP9 the *VP9 Video Extensions*, a free component of the Microsoft Store that a system may not have; without it, VP9 is decoded in software. Streams with spatial layers (SVC) are decoded in software too, since their layers reach a decoder without the index that says where each ends. Decoded frames are copied from GPU memory back to system memory, where WebRTC's frames are, so hardware decoding pays off mostly at high resolutions and with many streams; at low resolutions WebRTC's software decoders are about as cheap. A hardware decoder that fails, or turns out to decode in software, is replaced by the software decoder, which starts with the next key frame. On macOS, VP9 decoding in hardware saves CPU: in a measurement on an Apple M2 it took a fraction of the processor time libvpx needs, but multi-threaded libvpx was about as fast on the clock. VideoToolbox decodes VP9 in hardware only where the Mac has a decoder for it, which Apple silicon does; a Mac without one decodes VP9 with libvpx as before. Profile 2 (10 bit), sizes below 64x64 or above 4096x4096, and streams with spatial layers (SVC) are decoded with libvpx too. diff --git a/webrtc-jni/src/main/cpp/include/media/video/codec/windows/MFVideoDecoder.h b/webrtc-jni/src/main/cpp/include/media/video/codec/windows/MFVideoDecoder.h index 55ae783b..6321a901 100644 --- a/webrtc-jni/src/main/cpp/include/media/video/codec/windows/MFVideoDecoder.h +++ b/webrtc-jni/src/main/cpp/include/media/video/codec/windows/MFVideoDecoder.h @@ -38,7 +38,7 @@ namespace jni { - // Decodes H.264 or AV1 on the GPU, with the decoder transform of Windows + // Decodes H.264, AV1 or VP9 on the GPU, with the decoder transform of Windows // for the codec, which decodes through DXVA on the Direct3D 11 device it // is given. // @@ -52,7 +52,7 @@ namespace jni class MFVideoDecoder : public webrtc::VideoDecoder { public: - // The codec is H.264 or AV1. + // The codec is H.264, AV1 or VP9. explicit MFVideoDecoder(webrtc::VideoCodecType codec); ~MFVideoDecoder() override; diff --git a/webrtc-jni/src/main/cpp/include/media/video/codec/windows/MFVideoDecoderFactory.h b/webrtc-jni/src/main/cpp/include/media/video/codec/windows/MFVideoDecoderFactory.h index 5dfe2640..13e5248b 100644 --- a/webrtc-jni/src/main/cpp/include/media/video/codec/windows/MFVideoDecoderFactory.h +++ b/webrtc-jni/src/main/cpp/include/media/video/codec/windows/MFVideoDecoderFactory.h @@ -27,14 +27,15 @@ namespace jni { - // Creates the Media Foundation decoders that decode on the GPU, for H.264 - // and AV1, whichever the GPU decodes and Windows has a decoder for that + // Creates the Media Foundation decoders that decode on the GPU, for H.264, + // AV1 and VP9, whichever the GPU decodes and Windows has a decoder for that // uses Direct3D 11. It offers them in the formats WebRTC's software - // decoders offer too: H.264 in all its profiles, and AV1 in profile 0. + // decoders offer too: H.264 in all its profiles, and AV1 and VP9 in + // profile 0. class MFVideoDecoderFactory : public webrtc::VideoDecoderFactory { public: - // Returns a factory, or null if the GPU decodes neither codec. + // Returns a factory, or null if the GPU decodes none of the codecs. static std::unique_ptr Create(); ~MFVideoDecoderFactory() override = default; @@ -44,11 +45,12 @@ namespace jni const webrtc::SdpVideoFormat & format) override; private: - MFVideoDecoderFactory(bool h264, bool av1); + MFVideoDecoderFactory(bool h264, bool av1, bool vp9); private: const bool h264; const bool av1; + const bool vp9; }; } diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/windows/MFVideoDecoder.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/windows/MFVideoDecoder.cpp index 2bbd88fc..3d3c7657 100644 --- a/webrtc-jni/src/main/cpp/src/media/video/codec/windows/MFVideoDecoder.cpp +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/windows/MFVideoDecoder.cpp @@ -56,7 +56,14 @@ namespace jni const GUID & InputFormat(webrtc::VideoCodecType codec) { - return codec == webrtc::kVideoCodecAV1 ? MFVideoFormat_AV1 : MFVideoFormat_H264; + switch (codec) { + case webrtc::kVideoCodecAV1: + return MFVideoFormat_AV1; + case webrtc::kVideoCodecVP9: + return MFVideoFormat_VP90; + default: + return MFVideoFormat_H264; + } } } @@ -174,6 +181,16 @@ namespace jni if (SUCCEEDED(hr)) { hr = SetOutputType(); + + // A VP9 stream states its size in the key frames only, and its + // decoder may offer no output type before it has seen one. It then + // asks for the type again with MF_E_TRANSFORM_STREAM_CHANGE, which + // DrainOutput answers, and a decoder that cannot give one at all + // fails there, and the software decoder takes over. + if (FAILED(hr) && codec == webrtc::kVideoCodecVP9 && !resolution.Valid()) { + RTC_LOG(LS_INFO) << implementationName << " offers no output type yet, hr=" << hr; + hr = S_OK; + } } if (SUCCEEDED(hr)) { hr = transform->ProcessMessage(MFT_MESSAGE_NOTIFY_BEGIN_STREAMING, 0); @@ -241,6 +258,15 @@ namespace jni return WEBRTC_VIDEO_CODEC_ERR_PARAMETER; } + // The layers of a VP9 frame with spatial layers reach a decoder back to + // back, without the superframe index that tells where one ends. libvpx + // takes them that way; a hardware decoder is not known to. + if (codec == webrtc::kVideoCodecVP9 + && (image.SpatialIndex().value_or(0) > 0 || image.SpatialLayerFrameSize(1).has_value())) + { + return WEBRTC_VIDEO_CODEC_FALLBACK_SOFTWARE; + } + ComPtr buffer; HRESULT hr = MFCreateMemoryBuffer(static_cast(image.size()), &buffer); diff --git a/webrtc-jni/src/main/cpp/src/media/video/codec/windows/MFVideoDecoderFactory.cpp b/webrtc-jni/src/main/cpp/src/media/video/codec/windows/MFVideoDecoderFactory.cpp index 4f6defed..ad4db2d3 100644 --- a/webrtc-jni/src/main/cpp/src/media/video/codec/windows/MFVideoDecoderFactory.cpp +++ b/webrtc-jni/src/main/cpp/src/media/video/codec/windows/MFVideoDecoderFactory.cpp @@ -58,6 +58,7 @@ namespace jni { bool h264 = false; bool av1 = false; + bool vp9 = false; try { ComInitializer comInitializer; @@ -86,23 +87,27 @@ namespace jni && HasDirect3DDecoder(MFVideoFormat_H264, manager.Get()); av1 = SupportsDecoderProfile(device.Get(), D3D11_DECODER_PROFILE_AV1_VLD_PROFILE0) && HasDirect3DDecoder(MFVideoFormat_AV1, manager.Get()); + vp9 = SupportsDecoderProfile(device.Get(), D3D11_DECODER_PROFILE_VP9_VLD_PROFILE0) + && HasDirect3DDecoder(MFVideoFormat_VP90, manager.Get()); } catch (...) { return nullptr; } - RTC_LOG(LS_INFO) << "Media Foundation hardware decoders, H.264: " << h264 << ", AV1: " << av1; + RTC_LOG(LS_INFO) << "Media Foundation hardware decoders, H.264: " << h264 << ", AV1: " << av1 + << ", VP9: " << vp9; - if (!h264 && !av1) { + if (!h264 && !av1 && !vp9) { return nullptr; } - return std::unique_ptr(new MFVideoDecoderFactory(h264, av1)); + return std::unique_ptr(new MFVideoDecoderFactory(h264, av1, vp9)); } - MFVideoDecoderFactory::MFVideoDecoderFactory(bool h264, bool av1) : + MFVideoDecoderFactory::MFVideoDecoderFactory(bool h264, bool av1, bool vp9) : h264(h264), - av1(av1) + av1(av1), + vp9(vp9) { } @@ -116,6 +121,9 @@ namespace jni if (av1) { formats.push_back(webrtc::SdpVideoFormat::AV1Profile0()); } + if (vp9) { + formats.push_back(webrtc::SdpVideoFormat::VP9Profile0()); + } return formats; } diff --git a/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java b/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java index 63cabaea..458cb4cf 100644 --- a/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java +++ b/webrtc/src/main/java/dev/onvoid/webrtc/media/video/codec/HardwareVideoDecoderFactory.java @@ -34,8 +34,11 @@ * .build(); * } *

- * On Windows, H.264 and AV1 are decoded on the GPU through Direct3D 11 by the - * Media Foundation decoders of Windows, where the GPU decodes the codec. + * On Windows, H.264, AV1 and VP9 (profile 0) are decoded on the GPU through + * Direct3D 11 by the Media Foundation decoders of Windows, where the GPU + * decodes the codec and Windows has a decoder for it; VP9 needs the VP9 Video + * Extensions of the Microsoft Store. VP9 streams with spatial layers go to + * libvpx. * Decoded frames are copied back to system memory, which WebRTC's frames are * in, so hardware decoding saves CPU mostly at high resolutions. A hardware * decoder that fails to start, or fails while decoding, is replaced by the diff --git a/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java b/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java index 91523e76..1774ff03 100644 --- a/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java +++ b/webrtc/src/test/java/dev/onvoid/webrtc/HardwareVideoDecoderIntegrationTest.java @@ -46,7 +46,9 @@ * A machine without a hardware decoder skips the tests that need one, as CI * runners do. Set the system property {@code webrtc.test.hardwareDecoder} to * {@code true} on a machine that has one, to make those tests fail instead, - * and {@code webrtc.test.hardwareAv1Decoder} for a GPU that decodes AV1. + * and {@code webrtc.test.hardwareAv1Decoder} for a GPU that decodes AV1, and + * {@code webrtc.test.hardwareVp9Decoder} for one that decodes VP9 on Windows. + * On macOS the VP9 tests follow {@code webrtc.test.hardwareDecoder}. */ @Execution(ExecutionMode.SAME_THREAD) class HardwareVideoDecoderIntegrationTest extends TestBase { @@ -57,6 +59,8 @@ class HardwareVideoDecoderIntegrationTest extends TestBase { private static final boolean HARDWARE_AV1_REQUIRED = Boolean.getBoolean("webrtc.test.hardwareAv1Decoder"); + private static final boolean HARDWARE_VP9_REQUIRED = Boolean.getBoolean("webrtc.test.hardwareVp9Decoder"); + private static final String OS = System.getProperty("os.name").toLowerCase(Locale.ROOT); private static final Predicate H264 = codec -> @@ -138,14 +142,10 @@ void macDecodesH264WithVideoToolbox() throws Exception { } @Test - void macDecodesVp9WithVideoToolbox() throws Exception { - assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); - - PeerConnectionFactory hardware = PeerConnectionFactory.builder() - .setAudioDeviceModule(audioDevModule) - .setVideoDecoderFactory(new HardwareVideoDecoderFactory()) - .build(); + void hardwareDecodesVp9() throws Exception { + assumeVp9Platform(); + PeerConnectionFactory hardware = hardwareFactory(); String implementation; try { @@ -155,29 +155,22 @@ void macDecodesVp9WithVideoToolbox() throws Exception { hardware.dispose(); } - boolean hardwareUsed = implementation.contains("VideoToolbox"); - - if (HARDWARE_REQUIRED) { - assertTrue(hardwareUsed, implementation); - } - else { - assumeTrue(hardwareUsed, "no hardware decoder: " + implementation); - } + expectVp9Hardware(implementation); } @Test - void macDecodesVp9InSoftwareByDefault() throws Exception { - assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + void defaultDecodesVp9InSoftware() throws Exception { + assumeVp9Platform(); // Hardware decoding is opt-in; the shared factory decodes VP9 with libvpx. String implementation = decoderImplementation(factory, VP9); - assertFalse(implementation.contains("VideoToolbox"), implementation); + assertFalse(isHardware(implementation), implementation); } @Test - void macFollowsResolutionChange() throws Exception { - assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + void hardwareFollowsResolutionChange() throws Exception { + assumeVp9Platform(); PeerConnectionFactory hardware = PeerConnectionFactory.builder() .setAudioDeviceModule(audioDevModule) @@ -213,7 +206,7 @@ else if (width == 160 && frame.buffer.getHeight() == 120) { String implementation = decoderImplementationOf(call); - assumeTrue(implementation.contains("VideoToolbox"), "no hardware decoder: " + implementation); + expectVp9Hardware(implementation); // The sender restarts at half the size, with a key frame; the // decoder has to start a session for it. @@ -229,7 +222,7 @@ else if (width == 160 && frame.buffer.getHeight() == 120) { assertTrue(half.await(TIMEOUT_SECONDS, TimeUnit.SECONDS), "no frames at the new size"); // Still the hardware decoder, not a fallback. - assertTrue(decoderImplementationOf(call).contains("VideoToolbox")); + assertTrue(isHardware(decoderImplementationOf(call))); track.removeSink(sink); receiver.dispose(); @@ -240,12 +233,12 @@ else if (width == 160 && frame.buffer.getHeight() == 120) { } @Test - void macVp9NeedsNoKeyFrames() throws Exception { - assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + void vp9NeedsNoKeyFrames() throws Exception { + assumeVp9Platform(); CallResult result = receiveVideo(hardwareFactory(), VP9, 320, 240, null, 60); - assumeTrue(result.decoder.contains("VideoToolbox"), "no hardware decoder: " + result.decoder); + expectVp9Hardware(result.decoder); // An inter frame that fails to decode makes the receiver ask for a key // frame, and every key frame is then the only frame that decodes. A @@ -255,20 +248,20 @@ void macVp9NeedsNoKeyFrames() throws Exception { } @Test - void macVp9TemporalLayers() throws Exception { - assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + void vp9TemporalLayers() throws Exception { + assumeVp9Platform(); - // Temporal layers are one spatial layer, which VideoToolbox decodes. + // Temporal layers are one spatial layer, which the hardware decodes. CallResult result = receiveVideo(hardwareFactory(), VP9, 320, 240, "L1T3", 60); - assumeTrue(result.decoder.contains("VideoToolbox"), "no hardware decoder: " + result.decoder); + expectVp9Hardware(result.decoder); assertTrue(result.pliCount <= 1, "key frames requested: " + result.pliCount); } @Test - void macVp9SpatialLayers() throws Exception { - assumeTrue(OS.contains("mac"), "VideoToolbox is available on macOS only"); + void vp9SpatialLayers() throws Exception { + assumeVp9Platform(); // WebRTC encodes spatial layers only above a size, and drops them // below it. @@ -276,9 +269,10 @@ void macVp9SpatialLayers() throws Exception { assumeTrue(result.scalability.startsWith("L2"), "no spatial layers were encoded: " + result.scalability); - // The frames arrive, from libvpx: VideoToolbox does not decode a - // frame whose layers come without a superframe index. - assertFalse(result.decoder.contains("VideoToolbox"), result.decoder); + // The frames arrive, from libvpx: the layers of a frame come without a + // superframe index, which VideoToolbox does not decode, and which a + // Media Foundation decoder is not known to. + assertFalse(isHardware(result.decoder), result.decoder); } private PeerConnectionFactory hardwareFactory() { @@ -288,6 +282,33 @@ private PeerConnectionFactory hardwareFactory() { .build(); } + /** + * Skips the test where VP9 is not decoded in hardware: macOS and Windows. + */ + private static void assumeVp9Platform() { + assumeTrue(OS.contains("mac") || OS.contains("win"), + "VP9 is decoded in hardware on macOS and Windows only"); + } + + private static boolean isHardware(String implementation) { + return implementation.contains("VideoToolbox") || implementation.contains("MediaFoundation"); + } + + /** + * Where a hardware decoder is required, the test fails without one; + * elsewhere it is skipped. + */ + private static void expectVp9Hardware(String implementation) { + boolean required = OS.contains("mac") ? HARDWARE_REQUIRED : HARDWARE_VP9_REQUIRED; + + if (required) { + assertTrue(isHardware(implementation), implementation); + } + else { + assumeTrue(isHardware(implementation), "no hardware decoder: " + implementation); + } + } + /** * What a call that received video tells about its decoder. */