blob: ca4c2d1bf063de074a0c4b67901bb17bcd0c4f16 [file]
/*
* Copyright (c) 2026 The WebRTC project authors. All Rights Reserved.
*
* Use of this source code is governed by a BSD-style license
* that can be found in the LICENSE file in the root of the source
* tree. An additional intellectual property rights grant can be found
* in the file PATENTS. All contributing project authors may
* be found in the AUTHORS file in the root of the source tree.
*/
#include <algorithm>
#include <array>
#include <cstddef>
#include <cstdint>
#include <deque>
#include <memory>
#include <optional>
#include <set>
#include <string>
#include <tuple>
#include <utility>
#include <vector>
#include "api/environment/environment.h"
#include "api/media_stream_interface.h"
#include "api/scoped_refptr.h"
#include "api/test/create_frame_generator.h"
#include "api/test/frame_generator_interface.h"
#include "api/units/data_rate.h"
#include "api/units/data_size.h"
#include "api/units/frequency.h"
#include "api/units/time_delta.h"
#include "api/units/timestamp.h"
#include "api/video/i420_buffer.h"
#include "api/video/resolution.h"
#include "api/video/video_frame.h"
#include "api/video/video_frame_buffer.h"
#include "api/video_codecs/libaom_av1_encoder_factory.h"
#include "api/video_codecs/test/temporal_layer_pattern_for_test.h"
#include "api/video_codecs/test/video_codec_test_utils.h"
#include "api/video_codecs/video_decoder_factory.h"
#include "api/video_codecs/video_encoder_builders.h"
#include "api/video_codecs/video_encoder_builders_for_test.h"
#include "api/video_codecs/video_encoder_factory_interface.h"
#include "api/video_codecs/video_encoder_interface.h"
#include "api/video_codecs/video_encoding_general.h"
#include "rtc_base/checks.h"
#include "rtc_base/logging.h"
#include "test/create_test_environment.h"
#include "test/gmock.h"
#include "test/gtest.h"
#include "test/qp_parser_for_test.h"
#include "test/testsupport/file_utils.h"
#include "test/testsupport/frame_reader.h"
#include "test/testsupport/pendulum_frame_generator.h"
#include "test/testsupport/switching_frame_reader.h"
#include "test/time_controller/simulated_time_controller.h"
// This file contains tests evaluating the rate control requirements in
// `api/video_codecs/g3doc/video_encoder_api_v2.md`.
// These are not meant to be exhaustive and are not intended for continuous
// performance testing. See e.g. test/video_codec_tester.h for such purposes.
namespace webrtc {
namespace {
// The CBR settings used by the tests in this file, see
// `VideoEncoderFactoryInterface::StaticEncoderSettings::Cbr`. The transmission
// delay the buffer sizes allow for also bounds how uneven a temporal layer
// allocation can reasonably be, since a frame asking for a significant part of
// that budget cannot be delivered in time.
constexpr TimeDelta kCbrMaxBufferSize = TimeDelta::Millis(1000);
constexpr TimeDelta kCbrTargetBufferSize = TimeDelta::Millis(600);
constexpr double kMaxIntraBitrateFactor = 3.0;
// Tracks accumulated data sizes and duration for actual encoded bytes and ideal
// CBR bytes.
struct AccumulatedData {
DataSize actual = DataSize::Zero();
DataSize ideal = DataSize::Zero();
TimeDelta duration = TimeDelta::Zero();
std::optional<double> psnr;
int temporal_id = 0;
bool is_keyframe = false;
void Add(const AccumulatedData& other) {
actual += other.actual;
ideal += other.ideal;
duration += other.duration;
}
void Subtract(const AccumulatedData& other) {
actual -= other.actual;
ideal -= other.ideal;
duration -= other.duration;
}
double deviation_pct() const { return 100.0 * (actual / ideal - 1.0); }
};
class VideoEncoderRateControlTestBase : public ::testing::Test {
protected:
VideoEncoderRateControlTestBase() = default;
bool SupportsCbr() const {
return encoder_factory_->GetEncoderCapabilities()
.bitrate_control()
.rc_modes()
.contains(VideoEncoderFactoryInterface::RateControlMode::kCbr);
}
bool SupportsCbrSetting(
VideoEncoderFactoryInterface::CbrSetting setting) const {
return encoder_factory_->GetEncoderCapabilities()
.bitrate_control()
.supported_cbr_settings()
.contains(setting);
}
int MaxTemporalLayers() const {
return encoder_factory_->GetEncoderCapabilities()
.prediction_constraints()
.max_temporal_layers();
}
bool SupportsTemporalLayers(int num_temporal_layers) const {
return MaxTemporalLayers() >= num_temporal_layers;
}
int MaxSpatialLayers() const {
return encoder_factory_->GetEncoderCapabilities()
.prediction_constraints()
.max_spatial_layers();
}
int NumReferenceBuffers() const {
return encoder_factory_->GetEncoderCapabilities()
.prediction_constraints()
.num_buffers();
}
// TestDecoder is only needed in order to produce PSNR.
void EnableDecoder() {
decoder_factory_ = CreateTestDecoderFactory();
test_decoder_ = std::make_unique<TestDecoder>(
env_, decoder_factory_.get(), encoder_factory_->CodecName());
RTC_CHECK(test_decoder_->IsSupported());
}
void SetUpCbrEncoder(
Resolution resolution,
double max_intra_bitrate_factor = kMaxIntraBitrateFactor) {
ASSERT_TRUE(SupportsCbr());
VideoEncoderFactoryInterface::StaticEncoderSettings static_settings =
StaticEncoderSettingsBuilder()
.MaxEncodeDimensions(resolution)
.EncodingFormat({.sub_sampling = EncodingFormat::SubSampling::k420,
.bit_depth = 8})
.CbrRcMode(kCbrMaxBufferSize, kCbrTargetBufferSize,
max_intra_bitrate_factor)
.MaxNumberOfThreads(1)
.Build();
encoder_ = encoder_factory_->CreateEncoder(static_settings, {});
RTC_CHECK(encoder_ != nullptr);
frame_generator_ = CreateFrameGenerator();
RTC_CHECK(frame_generator_ != nullptr);
current_timestamp_ = Timestamp::Zero();
is_first_frame_ = true;
test_decoder_.reset();
decoder_factory_.reset();
encoded_frames_.clear();
temporal_layer_pattern_.reset();
}
void SetFrameGenerator(
std::unique_ptr<test::FrameGeneratorInterface> frame_generator) {
RTC_CHECK(frame_generator != nullptr);
frame_generator_ = std::move(frame_generator);
}
// Returns the next frame of the frame generator, scaled to `resolution`.
scoped_refptr<VideoFrameBuffer> NextFrame(Resolution resolution) {
test::FrameGeneratorInterface::VideoFrameData frame_data =
frame_generator_->NextFrame();
RTC_CHECK(frame_data.buffer != nullptr);
if (frame_data.buffer->width() == resolution.width &&
frame_data.buffer->height() == resolution.height) {
return frame_data.buffer;
}
scoped_refptr<I420Buffer> scaled_buffer =
I420Buffer::Create(resolution.width, resolution.height);
scaled_buffer->ScaleFrom(*frame_data.buffer->ToI420());
return scaled_buffer;
}
// Parameters for encoding a sequence of frames in rate control tests.
struct EncodeSettings {
int num_frames = 0;
DataRate target_bitrate = DataRate::Zero();
// Time interval between the start of successive frames.
TimeDelta frame_interval = TimeDelta::Zero();
// Nominal frame duration reported to the CBR rate controller. If omitted,
// defaults to `frame_interval`. Must not be set together with
// `temporal_layer_pattern`, which derives the duration of each frame from
// `frame_interval`.
std::optional<TimeDelta> frame_duration;
Resolution resolution;
VideoTrackInterface::ContentHint content_hint =
VideoTrackInterface::ContentHint::kNone;
// When true, repeats the same image buffer instead of generating new
// frames.
bool repeat_frame = false;
// When set, frames are encoded as the temporal layer structure of this
// pattern prescribes. Ownership is transferred to the fixture, which keeps
// the pattern alive so that it can span several `Encode` calls; a
// subsequent call leaving this unset continues the same pattern. If no
// pattern has been set, all frames are encoded in a single temporal layer,
// referencing and updating buffer 0.
std::unique_ptr<TemporalLayerPatternForTest> temporal_layer_pattern;
};
void Encode(EncodeSettings settings) {
ASSERT_TRUE(settings.temporal_layer_pattern == nullptr ||
!settings.frame_duration);
if (settings.temporal_layer_pattern != nullptr) {
temporal_layer_pattern_ = std::move(settings.temporal_layer_pattern);
}
scoped_refptr<VideoFrameBuffer> frame;
for (int i = 0; i < settings.num_frames; ++i) {
if (frame == nullptr || !settings.repeat_frame) {
frame = NextFrame(settings.resolution);
}
std::optional<TemporalLayerPatternForTest::FrameConfig> frame_config;
if (temporal_layer_pattern_ != nullptr) {
frame_config = temporal_layer_pattern_->NextFrameConfig();
}
const int temporal_id = frame_config ? frame_config->temporal_id : 0;
// A temporal layer is given a share of the stream bitrate, and the
// frames of the layer split that share between them. A layer that only
// holds every fourth frame therefore gives each of its frames four times
// the bit budget its share of the bitrate would suggest, which is what
// `frame_budget_factor` accounts for. The duration always stays the
// interval to the next frame of the stream.
const TimeDelta frame_duration =
settings.frame_duration.value_or(settings.frame_interval);
const DataRate target_bitrate =
frame_config ? settings.target_bitrate * frame_config->rate_factor
: settings.target_bitrate;
const DataSize ideal_frame_size =
settings.target_bitrate * frame_duration *
(frame_config
? temporal_layer_pattern_->frame_budget_factor(temporal_id)
: 1.0);
EncOut out;
Fb builder;
builder.Res(settings.resolution)
.T(temporal_id)
.Cbr({.duration = frame_duration, .target_bitrate = target_bitrate})
.Out(out);
const bool is_keyframe = is_first_frame_;
if (is_first_frame_) {
builder.Key().Upd(0);
is_first_frame_ = false;
} else if (frame_config) {
builder.Delta().Upd(frame_config->update_buffer);
if (frame_config->reference_buffer) {
builder.Ref({*frame_config->reference_buffer});
}
} else {
builder.Delta().Ref({0}).Upd(0);
}
encoder_->Encode(
frame,
TemporalUnitSettings(settings.content_hint, current_timestamp_),
ToVec({builder.Build()}));
ASSERT_THAT(out, HasBitstreamAndMetaData());
std::optional<double> psnr;
if (test_decoder_ != nullptr) {
VideoFrame decoded = test_decoder_->Decode(out.bitstream);
psnr = Psnr(frame->ToI420(), decoded);
}
encoded_frames_.push_back(
{.actual = DataSize::Bytes(out.bitstream.size()),
.ideal = ideal_frame_size,
.duration = frame_duration,
.psnr = psnr,
.temporal_id = temporal_id,
.is_keyframe = is_keyframe});
current_timestamp_ += settings.frame_interval;
time_controller_.AdvanceTime(settings.frame_interval);
}
}
void VerifyTotalDeviation(double max_deviation_pct) {
AccumulatedData total;
for (const auto& frame : encoded_frames_) {
total.Add(frame);
}
double total_deviation_pct = total.deviation_pct();
RTC_LOG(LS_VERBOSE) << "total_bytes=" << total.actual.bytes()
<< " optimal=" << total.ideal.bytes()
<< " deviation=" << total_deviation_pct << "%";
EXPECT_NEAR(total_deviation_pct, 0.0, max_deviation_pct)
<< "Bitrate deviation " << total_deviation_pct
<< "% exceeded tolerance " << max_deviation_pct
<< "% (actual: " << total.actual.bytes()
<< " bytes, target: " << total.ideal.bytes() << " bytes)";
}
void VerifyFrameBasedSlidingWindowBitrateDeviation(
int window_frames,
double min_allowed_dev_pct,
double max_allowed_dev_pct) {
AccumulatedData window_data;
std::deque<AccumulatedData> window;
std::optional<double> max_window_dev_pct;
std::optional<double> min_window_dev_pct;
for (const auto& frame : encoded_frames_) {
window.push_back(frame);
window_data.Add(frame);
if (window.size() > static_cast<size_t>(window_frames)) {
window_data.Subtract(window.front());
window.pop_front();
}
if (window.size() == static_cast<size_t>(window_frames)) {
double window_dev_pct = window_data.deviation_pct();
if (!max_window_dev_pct || window_dev_pct > *max_window_dev_pct) {
max_window_dev_pct = window_dev_pct;
}
if (!min_window_dev_pct || window_dev_pct < *min_window_dev_pct) {
min_window_dev_pct = window_dev_pct;
}
}
}
ASSERT_TRUE(min_window_dev_pct.has_value());
ASSERT_TRUE(max_window_dev_pct.has_value());
RTC_LOG(LS_VERBOSE) << "sliding " << window_frames
<< "-frame window deviation range: ["
<< *min_window_dev_pct << "%, " << *max_window_dev_pct
<< "%]";
EXPECT_GE(*min_window_dev_pct, min_allowed_dev_pct);
EXPECT_LE(*max_window_dev_pct, max_allowed_dev_pct);
}
void VerifyTimeBasedSlidingWindowBitrateDeviation(
TimeDelta window_duration,
std::optional<double> min_allowed_dev_pct,
double max_allowed_dev_pct) {
AccumulatedData window_data;
std::deque<AccumulatedData> window;
std::optional<double> max_window_dev_pct;
std::optional<double> min_window_dev_pct;
for (const auto& frame : encoded_frames_) {
window.push_back(frame);
window_data.Add(frame);
while (window_data.duration >= window_duration) {
double window_dev_pct = window_data.deviation_pct();
if (!max_window_dev_pct || window_dev_pct > *max_window_dev_pct) {
max_window_dev_pct = window_dev_pct;
}
if (!min_window_dev_pct || window_dev_pct < *min_window_dev_pct) {
min_window_dev_pct = window_dev_pct;
}
window_data.Subtract(window.front());
window.pop_front();
}
}
ASSERT_TRUE(min_window_dev_pct.has_value());
ASSERT_TRUE(max_window_dev_pct.has_value());
RTC_LOG(LS_VERBOSE) << "sliding " << window_duration.seconds()
<< "s window deviation range: [" << *min_window_dev_pct
<< "%, " << *max_window_dev_pct << "%]";
if (min_allowed_dev_pct.has_value()) {
EXPECT_GE(*min_window_dev_pct, *min_allowed_dev_pct);
}
EXPECT_LE(*max_window_dev_pct, max_allowed_dev_pct);
}
// Verifies that the encoder acted on the requested temporal layer
// allocation. Each layer is checked against the bit budget that was
// requested for it, in the same way `VerifyTotalDeviation` checks the stream
// as a whole. Additionally, since every sensible allocation gives the lower
// temporal layers a larger per frame bit budget than the higher ones, the
// encoded frames must follow that order too.
void VerifyTemporalLayerAllocation(double max_deviation_pct) {
ASSERT_TRUE(temporal_layer_pattern_ != nullptr);
const int num_temporal_layers =
temporal_layer_pattern_->num_temporal_layers();
std::vector<AccumulatedData> per_layer(num_temporal_layers);
std::vector<int> frames_per_layer(num_temporal_layers, 0);
std::vector<double> psnr_sum_per_layer(num_temporal_layers, 0.0);
for (const AccumulatedData& frame : encoded_frames_) {
ASSERT_LT(frame.temporal_id, num_temporal_layers);
per_layer[frame.temporal_id].Add(frame);
++frames_per_layer[frame.temporal_id];
psnr_sum_per_layer[frame.temporal_id] += frame.psnr.value_or(0.0);
}
for (int tid = 0; tid < num_temporal_layers; ++tid) {
ASSERT_GT(frames_per_layer[tid], 0);
RTC_LOG(LS_VERBOSE) << "T" << tid << " frames=" << frames_per_layer[tid]
<< " bytes/frame="
<< per_layer[tid].actual.bytes() /
frames_per_layer[tid]
<< " deviation=" << per_layer[tid].deviation_pct()
<< "% psnr="
<< psnr_sum_per_layer[tid] / frames_per_layer[tid];
}
for (int tid = 0; tid < num_temporal_layers; ++tid) {
const double deviation_pct = per_layer[tid].deviation_pct();
EXPECT_NEAR(deviation_pct, 0.0, max_deviation_pct)
<< "T" << tid << " bitrate deviation " << deviation_pct
<< "% exceeded tolerance " << max_deviation_pct
<< "% (actual: " << per_layer[tid].actual.bytes()
<< " bytes, target: " << per_layer[tid].ideal.bytes() << " bytes)";
}
// Lower temporal layers are given a larger per frame bit budget, so the
// encoded frames have to follow the same order. A distribution that asks
// for the same budget on both sides of a layer boundary says nothing about
// the order the frames should come out in, so those pairs are skipped.
for (int tid = 1; tid < num_temporal_layers; ++tid) {
const double requested_below =
static_cast<double>(per_layer[tid - 1].ideal.bytes()) /
frames_per_layer[tid - 1];
const double requested_above =
static_cast<double>(per_layer[tid].ideal.bytes()) /
frames_per_layer[tid];
if (requested_below <= requested_above) {
continue;
}
EXPECT_GT(per_layer[tid - 1].actual.bytes() / frames_per_layer[tid - 1],
per_layer[tid].actual.bytes() / frames_per_layer[tid])
<< "T" << (tid - 1) << " frames are not larger than T" << tid
<< " frames";
}
}
// The mean PSNR of the frames belonging to temporal layer `temporal_id`.
double MeanPsnrOfTemporalLayer(int temporal_id) const {
double sum = 0.0;
int count = 0;
for (const AccumulatedData& frame : encoded_frames_) {
if (frame.temporal_id == temporal_id) {
RTC_CHECK(frame.psnr.has_value());
sum += *frame.psnr;
++count;
}
}
RTC_CHECK_GT(count, 0);
return sum / count;
}
// The mean PSNR of all encoded frames.
double MeanPsnr() const {
double sum = 0.0;
for (const AccumulatedData& frame : encoded_frames_) {
RTC_CHECK(frame.psnr.has_value());
sum += *frame.psnr;
}
RTC_CHECK(!encoded_frames_.empty());
return sum / encoded_frames_.size();
}
GlobalSimulatedTimeController time_controller_{Timestamp::Zero()};
Environment env_{CreateTestEnvironment({.time = &time_controller_})};
std::unique_ptr<VideoEncoderFactoryInterface> encoder_factory_;
std::unique_ptr<VideoEncoderInterface> encoder_;
std::unique_ptr<VideoDecoderFactory> decoder_factory_;
std::unique_ptr<TestDecoder> test_decoder_;
std::unique_ptr<test::FrameGeneratorInterface> frame_generator_;
std::unique_ptr<TemporalLayerPatternForTest> temporal_layer_pattern_;
std::vector<AccumulatedData> encoded_frames_;
Timestamp current_timestamp_ = Timestamp::Zero();
bool is_first_frame_ = true;
};
class VideoEncoderRateControlTest
: public VideoEncoderRateControlTestBase,
public ::testing::WithParamInterface<FactoryCreator> {
protected:
void SetUp() override { encoder_factory_ = GetParam()(); }
};
TEST_P(VideoEncoderRateControlTest, ConstantQpMatchesBitstreamAndEncoderQp) {
VideoEncoderFactoryInterface::Capabilities capabilities =
encoder_factory_->GetEncoderCapabilities();
if (!capabilities.bitrate_control().rc_modes().contains(
VideoEncoderFactoryInterface::RateControlMode::kCqp)) {
GTEST_SKIP() << "Encoder does not support CQP mode.";
}
int min_qp = capabilities.bitrate_control().min_qp();
int max_qp = capabilities.bitrate_control().max_qp();
VideoEncoderFactoryInterface::StaticEncoderSettings static_settings =
StaticEncoderSettingsBuilder()
.MaxEncodeDimensions(kDefaultResolution)
.EncodingFormat({.sub_sampling = EncodingFormat::SubSampling::k420,
.bit_depth = 8})
.CqpRcMode()
.MaxNumberOfThreads(1)
.Build();
QpParserForTest qp_parser;
std::unique_ptr<test::FrameGeneratorInterface> frame_generator =
CreateFrameGenerator();
std::unique_ptr<VideoEncoderInterface> enc =
encoder_factory_->CreateEncoder(static_settings, {});
ASSERT_NE(enc, nullptr);
int64_t timestamp_ms = 0;
bool is_first_frame = true;
for (int qp = min_qp; qp <= max_qp; ++qp) {
scoped_refptr<VideoFrameBuffer> frame = frame_generator->NextFrame().buffer;
EncOut out;
if (is_first_frame) {
enc->Encode(
frame, TemporalUnitSettings(Timestamp::Millis(timestamp_ms)),
ToVec({Fb().Cqp(qp).Res(kDefaultResolution).Upd(0).Key().Out(out)}));
is_first_frame = false;
} else {
enc->Encode(
frame, TemporalUnitSettings(Timestamp::Millis(timestamp_ms)),
ToVec(
{Fb().Cqp(qp).Res(kDefaultResolution).Ref({0}).Upd(0).Out(out)}));
}
timestamp_ms += 100;
ASSERT_THAT(out, HasBitstreamAndMetaData());
const EncodedData& ed = std::get<EncodedData>(out.res);
// libaom quantizer resolution has step 4 across the 0-255 qindex range.
EXPECT_NEAR(ed.encoded_qp, qp, 4);
std::optional<uint32_t> parsed_qp = qp_parser.Parse(
encoder_factory_->CodecName(), /*spatial_idx=*/0, out.bitstream);
ASSERT_TRUE(parsed_qp.has_value())
<< "Failed to parse QP from bitstream for codec "
<< encoder_factory_->CodecName() << " at target QP " << qp;
EXPECT_EQ(*parsed_qp, static_cast<uint32_t>(ed.encoded_qp));
}
}
constexpr Resolution kQvgaResolution = {.width = 320, .height = 180};
constexpr Resolution kVgaResolution = {.width = 640, .height = 360};
constexpr Resolution kHdResolution = {.width = 1280, .height = 720};
struct FixedBitrateTestParams {
std::string name;
Resolution resolution;
DataRate target_bitrate;
TimeDelta duration;
double max_deviation_pct;
};
class FixedBitrateRateControlTest
: public VideoEncoderRateControlTestBase,
public ::testing::WithParamInterface<
std::tuple<FactoryCreator, FixedBitrateTestParams>> {
protected:
void SetUp() override { encoder_factory_ = std::get<0>(GetParam())(); }
};
TEST_P(FixedBitrateRateControlTest, AdheresToTargetBitrate) {
if (!SupportsCbr()) {
GTEST_SKIP() << "Encoder does not support CBR mode.";
}
const FixedBitrateTestParams& params = std::get<1>(GetParam());
SetUpCbrEncoder(params.resolution);
constexpr TimeDelta kFrameInterval = 1 / Frequency::Hertz(30);
const int num_frames =
(params.duration.us() + kFrameInterval.us() / 2) / kFrameInterval.us();
Encode({.num_frames = num_frames,
.target_bitrate = params.target_bitrate,
.frame_interval = kFrameInterval,
.resolution = params.resolution});
VerifyTotalDeviation(params.max_deviation_pct);
}
// Verifies that the intra frame allowance has an effect, by comparing the
// keyframe produced with the smallest possible allowance against the one
// produced with the default allowance.
TEST_P(VideoEncoderRateControlTest, MaxIntraBitrateFactorLimitsKeyframeSize) {
if (!SupportsCbr()) {
GTEST_SKIP() << "Encoder does not support CBR mode.";
}
if (!SupportsCbrSetting(
VideoEncoderFactoryInterface::CbrSetting::kMaxIntraBitrateFactor)) {
GTEST_SKIP() << "Encoder does not limit the size of intra frames.";
}
constexpr TimeDelta kFrameInterval = 1 / Frequency::Hertz(30);
constexpr DataRate kTargetBitrate = DataRate::KilobitsPerSec(300);
auto encode_keyframe = [&](double max_intra_bitrate_factor) {
SetUpCbrEncoder(kQvgaResolution, max_intra_bitrate_factor);
Encode({.num_frames = 1,
.target_bitrate = kTargetBitrate,
.frame_interval = kFrameInterval,
.resolution = kQvgaResolution});
return encoded_frames_.front().actual;
};
DataSize keyframe_with_min_allowance = encode_keyframe(1.0);
DataSize keyframe_with_default_allowance =
encode_keyframe(kMaxIntraBitrateFactor);
EXPECT_LT(keyframe_with_min_allowance, keyframe_with_default_allowance);
}
// Verifies that dynamically changing the bitrate target follows the target
// within reasonable bounds, evaluated over a 2-second sliding window and
// overall accumulated bytes.
TEST_P(VideoEncoderRateControlTest, ChangingBitrateTargetVga) {
if (!SupportsCbr()) {
GTEST_SKIP() << "Encoder does not support CBR mode.";
}
SetUpCbrEncoder(kVgaResolution);
constexpr TimeDelta kFrameInterval = 1 / Frequency::Hertz(30);
constexpr int kWindowFrames = 60; // 2-second sliding window (30 fps).
// 1. 500 kbps for 2s (60 frames).
Encode({.num_frames = 60,
.target_bitrate = DataRate::KilobitsPerSec(500),
.frame_interval = kFrameInterval,
.resolution = kVgaResolution});
// 2. Drop to 100 kbps for 1s (30 frames).
Encode({.num_frames = 30,
.target_bitrate = DataRate::KilobitsPerSec(100),
.frame_interval = kFrameInterval,
.resolution = kVgaResolution});
// 3. Increase by 50 kbps every 200 ms (6 frames) until reaching 500 kbps.
for (int rate_kbps = 150; rate_kbps < 500; rate_kbps += 50) {
Encode({.num_frames = 6,
.target_bitrate = DataRate::KilobitsPerSec(rate_kbps),
.frame_interval = kFrameInterval,
.resolution = kVgaResolution});
}
// 4. Stay at 500 kbps for 1s (30 frames).
Encode({.num_frames = 30,
.target_bitrate = DataRate::KilobitsPerSec(500),
.frame_interval = kFrameInterval,
.resolution = kVgaResolution});
// (1) Check that deviation from optimal behavior over any 2-second sliding
// window does not exceed 20%.
VerifyFrameBasedSlidingWindowBitrateDeviation(kWindowFrames,
/*min_allowed_dev_pct=*/-20.0,
/*max_allowed_dev_pct=*/20.0);
// (2) Check that the sum of bytes sent is within 5% of the optimal behavior.
VerifyTotalDeviation(/*max_deviation_pct=*/5.0);
}
// Verifies that the CBR targets are adhered to even when input frame rate
// changes. Evaluated over a 2-second sliding window and overall accumulated
// bytes.
TEST_P(VideoEncoderRateControlTest, ChangingFramerateVga) {
if (!SupportsCbr()) {
GTEST_SKIP() << "Encoder does not support CBR mode.";
}
SetUpCbrEncoder(kVgaResolution);
constexpr DataRate kTargetBitrate = DataRate::KilobitsPerSec(500);
auto encode_segment = [&](Frequency framerate, TimeDelta duration) {
TimeDelta frame_interval = 1 / framerate;
int num_frames =
(duration.us() + frame_interval.us() / 2) / frame_interval.us();
Encode({.num_frames = num_frames,
.target_bitrate = kTargetBitrate,
.frame_interval = frame_interval,
.resolution = kVgaResolution});
};
// 1. 30 fps for 2s.
encode_segment(Frequency::Hertz(30), TimeDelta::Seconds(2));
// 2. Drop to 10 fps for 1s.
encode_segment(Frequency::Hertz(10), TimeDelta::Seconds(1));
// 3. Step up gradually back to 30 fps (15 fps, 20 fps, 25 fps for 200ms
// each).
encode_segment(Frequency::Hertz(15), TimeDelta::Millis(200));
encode_segment(Frequency::Hertz(20), TimeDelta::Millis(200));
encode_segment(Frequency::Hertz(25), TimeDelta::Millis(200));
// 4. Stay at 30 fps for 2s.
encode_segment(Frequency::Hertz(30), TimeDelta::Seconds(2));
// (1) Check that deviation from optimal behavior over any 1-second sliding
// window does not exceed 20%.
VerifyTimeBasedSlidingWindowBitrateDeviation(TimeDelta::Seconds(1),
/*min_allowed_dev_pct=*/-20.0,
/*max_allowed_dev_pct=*/20.0);
// (2) Check that the sum of bytes sent is within 5% of the optimal behavior.
VerifyTotalDeviation(/*max_deviation_pct=*/5.0);
}
// Verifies that the encoder adheres to CBR target bitrate when the video input
// periodically switches between different camera views every 5 seconds.
TEST_P(VideoEncoderRateControlTest, CameraSwitchingHd) {
if (!SupportsCbr()) {
GTEST_SKIP() << "Encoder does not support CBR mode.";
}
constexpr Resolution kResolution = kHdResolution;
constexpr Frequency kFramerate = Frequency::Hertz(30);
constexpr TimeDelta kFrameInterval = 1 / kFramerate;
constexpr DataRate kTargetBitrate = DataRate::KilobitsPerSec(2000);
constexpr TimeDelta kDuration = TimeDelta::Seconds(30);
constexpr TimeDelta kSwitchInterval = TimeDelta::Seconds(5);
const int num_frames =
(kDuration.us() + kFrameInterval.us() / 2) / kFrameInterval.us();
const std::vector<std::string> clip_paths = {
test::ResourcePath("ConferenceMotion_1280_720_50", "yuv"),
test::ResourcePath("FourPeople_1280x720_30", "yuv"),
test::ResourcePath("reference_less_video_test_file", "y4m"),
};
std::unique_ptr<test::FrameGeneratorInterface> generator =
test::CreateSwitchingFrameGenerator(
clip_paths,
{.width = static_cast<size_t>(kResolution.width),
.height = static_cast<size_t>(kResolution.height)},
kFramerate.hertz(), kSwitchInterval,
test::YuvFrameReaderImpl::RepeatMode::kPingPong);
SetUpCbrEncoder(kResolution);
SetFrameGenerator(std::move(generator));
Encode({.num_frames = num_frames,
.target_bitrate = kTargetBitrate,
.frame_interval = kFrameInterval,
.resolution = kResolution});
VerifyTotalDeviation(/*max_deviation_pct=*/5.0);
}
// Verified that the encoder adheres to CBR target bitrate when the video input
// has very high complexity both in terms of motion and high frequency detail.
TEST_P(VideoEncoderRateControlTest, SyntheticChaoticMotionStressHd) {
constexpr Resolution kResolution = {.width = 1280, .height = 720};
constexpr Frequency kFramerate = Frequency::Hertz(30);
constexpr TimeDelta kFrameInterval = 1 / kFramerate;
constexpr DataRate kTargetBitrate = DataRate::KilobitsPerSec(2000);
constexpr TimeDelta kDuration = TimeDelta::Seconds(10);
const int num_frames =
(kDuration.us() + kFrameInterval.us() / 2) / kFrameInterval.us();
test::PendulumFrameGenerator::Config config;
config.target_resolution = {
.width = static_cast<size_t>(kResolution.width),
.height = static_cast<size_t>(kResolution.height)};
config.fps = kFramerate.hertz();
config.min_zoom = 1.2;
config.max_zoom = 3.0;
config.zoom_speed = 0.3;
config.noise_level = 20;
SetUpCbrEncoder(kResolution);
SetFrameGenerator(std::make_unique<test::PendulumFrameGenerator>(config));
Encode({.num_frames = num_frames,
.target_bitrate = kTargetBitrate,
.frame_interval = kFrameInterval,
.resolution = kResolution});
VerifyTotalDeviation(/*max_deviation_pct=*/5.0);
}
// The number of groups of pictures a sequence has to cover. The base layer
// contributes a single frame to each, so this is also the number of base layer
// frames the per layer measurements are based on. Eight is enough to be within
// a few percentage points of the value this converges to, well inside the
// tolerances below.
constexpr int kMinGroupsOfPictures = 8;
// How far a single temporal layer may be off the share of the bitrate it was
// allocated. This is wider than the tolerance on the stream as a whole because
// the encoder is free to move bits between the layers as long as the total
// holds, and because a layer holds only a fraction of the bits, so the cost of
// starting the sequence weighs more heavily on it. The worst layer measured
// over the sequences below is 15% off.
//
// TODO(bugs.webrtc.org/496266459): Most of what is left is that start-up cost,
// which the encoder works off over a window far longer than these sequences;
// running them for 30s instead brings the worst layer to 5.6%. That triples
// the runtime of these tests, so it is not worth it until the tolerance has to
// be this tight to catch something.
constexpr double kMaxLayerDeviationPct = 20.0;
// A frame cannot be given an arbitrarily large share of the bitrate: the rate
// controller has to keep its buffer from draining, so a single frame asking
// for a significant part of the buffer will simply not be delivered. Temporal
// layer distributions are only exercised while they stay below this.
constexpr TimeDelta kMaxFrameBudget = kCbrTargetBufferSize / 4;
struct TemporalLayerTestParams {
std::string name;
Resolution resolution;
DataRate target_bitrate;
// Returns the bitrate fractions for a given number of temporal layers, see
// `TemporalLayerPatternForTest`.
std::vector<double> (*distribution)(int num_temporal_layers);
};
class TemporalLayerRateControlTest
: public VideoEncoderRateControlTestBase,
public ::testing::WithParamInterface<
std::tuple<FactoryCreator, TemporalLayerTestParams>> {
protected:
void SetUp() override { encoder_factory_ = std::get<0>(GetParam())(); }
};
// Verifies that the encoder adheres to the target bitrate when the bit budget
// is distributed over a temporal layer structure, and that it acts on the
// requested per temporal layer distribution, for every temporal layer count
// the encoder supports.
TEST_P(TemporalLayerRateControlTest, AdheresToLayerAllocation) {
if (!SupportsCbr()) {
GTEST_SKIP() << "Encoder does not support CBR mode.";
}
if (!SupportsTemporalLayers(2)) {
GTEST_SKIP() << "Encoder does not support temporal layers.";
}
const TemporalLayerTestParams& params = std::get<1>(GetParam());
constexpr TimeDelta kFrameInterval = 1 / Frequency::Hertz(30);
constexpr TimeDelta kMinDuration = TimeDelta::Seconds(10);
for (int num_temporal_layers = 2; num_temporal_layers <= MaxTemporalLayers();
++num_temporal_layers) {
SCOPED_TRACE(num_temporal_layers);
auto pattern = std::make_unique<TemporalLayerPatternForTest>(
num_temporal_layers, NumReferenceBuffers(),
params.distribution(num_temporal_layers));
// The base layer budget only grows with the number of layers, so no
// higher layer count is realizable either.
if (kFrameInterval * pattern->frame_budget_factor(0) > kMaxFrameBudget) {
break;
}
const int num_frames =
std::max<int>(kMinDuration / kFrameInterval,
kMinGroupsOfPictures *
TemporalLayerPatternForTest::FramesPerGroupOfPictures(
num_temporal_layers));
SetUpCbrEncoder(params.resolution);
EnableDecoder();
Encode({.num_frames = num_frames,
.target_bitrate = params.target_bitrate,
.frame_interval = kFrameInterval,
.resolution = params.resolution,
.temporal_layer_pattern = std::move(pattern)});
VerifyTotalDeviation(/*max_deviation_pct=*/5.0);
VerifyTemporalLayerAllocation(kMaxLayerDeviationPct);
// A receiver that only decodes the base layer sees a lower frame rate at a
// higher quality per frame, so base layer frames must not be worse than
// the average frame.
EXPECT_GT(MeanPsnrOfTemporalLayer(0), MeanPsnr());
}
}
// Verifies that the encoder adheres to the target bitrate when the bit budget
// is distributed over spatial layers, and that each spatial layer gets the
// share it was allocated, for every spatial layer count the encoder supports.
//
// This uses the simplest spatial structure, SxT1: every temporal unit holds a
// frame for each spatial layer, predicted from the same layer in the previous
// temporal unit and from the layer below in the same temporal unit. Unlike the
// scalability modes of that name, all layers have the same resolution, since
// only the bitrate allocation is of interest here.
TEST_P(VideoEncoderRateControlTest, SpatialLayerAllocation) {
if (!SupportsCbr()) {
GTEST_SKIP() << "Encoder does not support CBR mode.";
}
if (MaxSpatialLayers() < 2) {
GTEST_SKIP() << "Encoder does not support spatial layers.";
}
constexpr Resolution kResolution = kQvgaResolution;
constexpr TimeDelta kFrameInterval = 1 / Frequency::Hertz(30);
constexpr DataRate kTargetBitrate = DataRate::KilobitsPerSec(300);
constexpr int kNumTemporalUnits = TimeDelta::Seconds(10) / kFrameInterval;
for (int num_spatial_layers = 2; num_spatial_layers <= MaxSpatialLayers();
++num_spatial_layers) {
SCOPED_TRACE(num_spatial_layers);
SetUpCbrEncoder(kResolution);
// Each layer is given a larger share of the bitrate than the one below
// it, as it would be if it had a higher resolution. The exact split does
// not matter, only that the layers are asked for different amounts.
const int weight_sum = num_spatial_layers * (num_spatial_layers + 1) / 2;
std::vector<DataRate> layer_bitrates;
for (int sid = 0; sid < num_spatial_layers; ++sid) {
layer_bitrates.push_back(kTargetBitrate * (sid + 1) / weight_sum);
}
std::vector<AccumulatedData> per_layer(num_spatial_layers);
for (int tu = 0; tu < kNumTemporalUnits; ++tu) {
std::vector<EncOut> outs(num_spatial_layers);
std::vector<VideoEncoderInterface::FrameEncodeSettings> frame_settings;
for (int sid = 0; sid < num_spatial_layers; ++sid) {
Fb builder;
builder.Res(kResolution)
.S(sid)
.Cbr({.duration = kFrameInterval,
.target_bitrate = layer_bitrates[sid]})
.Upd(sid)
.Out(outs[sid]);
if (tu == 0 && sid == 0) {
builder.Key();
} else {
std::vector<int> references;
if (tu > 0) {
references.push_back(sid);
}
if (sid > 0) {
references.push_back(sid - 1);
}
builder.Delta().Ref(references);
}
frame_settings.push_back(builder.Build());
}
encoder_->Encode(NextFrame(kResolution),
TemporalUnitSettings(current_timestamp_),
std::move(frame_settings));
for (int sid = 0; sid < num_spatial_layers; ++sid) {
ASSERT_THAT(outs[sid], HasBitstreamAndMetaData());
const AccumulatedData frame = {
.actual = DataSize::Bytes(outs[sid].bitstream.size()),
.ideal = layer_bitrates[sid] * kFrameInterval,
.duration = kFrameInterval,
.is_keyframe = tu == 0 && sid == 0};
encoded_frames_.push_back(frame);
per_layer[sid].Add(frame);
}
current_timestamp_ += kFrameInterval;
time_controller_.AdvanceTime(kFrameInterval);
}
VerifyTotalDeviation(/*max_deviation_pct=*/5.0);
// The keyframe is part of the base layer, so that layer carries most of
// the cost of starting the sequence, and the more layers the bitrate is
// split over, the smaller its share and the more that cost weighs.
for (int sid = 0; sid < num_spatial_layers; ++sid) {
EXPECT_NEAR(per_layer[sid].deviation_pct(), 0.0, 8.0)
<< "S" << sid << " (actual: " << per_layer[sid].actual.bytes()
<< " bytes, target: " << per_layer[sid].ideal.bytes() << " bytes)";
}
}
}
// TODO(bugs.webrtc.org/496266459): Add tempo-spatial layer allocation tests,
// e.g. structures where not all temporal units have all spatial layers.
// Verifies that the encoder behaves well in screenshare scenarios with mostly
// static content combined with intermittent slide changes at high resolution.
TEST_P(VideoEncoderRateControlTest, ScreenshareSlideChangesFullHd) {
if (!SupportsCbr()) {
GTEST_SKIP() << "Encoder does not support CBR mode.";
}
constexpr Resolution kFullHdResolution = {.width = 1920, .height = 1080};
constexpr Resolution kResolution = kFullHdResolution;
constexpr Frequency kFramerate = Frequency::Hertz(30);
constexpr TimeDelta kFrameInterval = 1 / kFramerate;
constexpr DataRate kTargetBitrate = DataRate::KilobitsPerSec(2500);
constexpr TimeDelta kSlideDuration = TimeDelta::Seconds(2);
constexpr TimeDelta kTotalDuration = TimeDelta::Seconds(6);
const int num_frames =
(kTotalDuration.us() + kFrameInterval.us() / 2) / kFrameInterval.us();
const int frames_per_slide =
(kSlideDuration.us() + kFrameInterval.us() / 2) / kFrameInterval.us();
const std::vector<std::string> slides = {
test::ResourcePath("web_screenshot_1850_1110", "yuv"),
test::ResourcePath("presentation_1850_1110", "yuv"),
test::ResourcePath("difficult_photo_1850_1110", "yuv"),
};
std::unique_ptr<test::FrameGeneratorInterface> slide_generator =
test::CreateFromYuvFileFrameGenerator(slides, /*width=*/1850,
/*height=*/1110, frames_per_slide);
SetUpCbrEncoder(kResolution);
EnableDecoder();
SetFrameGenerator(std::move(slide_generator));
AccumulatedData total;
double sum_delta_psnr = 0.0;
int delta_count = 0;
for (int i = 0; i < num_frames; ++i) {
Encode({.num_frames = 1,
.target_bitrate = kTargetBitrate,
.frame_interval = kFrameInterval,
.resolution = kResolution,
.content_hint = VideoTrackInterface::ContentHint::kDetailed});
size_t frame_idx = encoded_frames_.size() - 1;
DataSize size = encoded_frames_[frame_idx].actual;
int offset_in_slide = i % frames_per_slide;
if (offset_in_slide == 0) {
// Transition frame to a new slide.
EXPECT_GT(size.bytes(), 0);
EXPECT_LE(size, kTargetBitrate * TimeDelta::Millis(575));
} else {
// Delta frames during static hold or refinement.
EXPECT_LE(size, kTargetBitrate * TimeDelta::Millis(330));
if (encoded_frames_[frame_idx].psnr.has_value()) {
sum_delta_psnr += *encoded_frames_[frame_idx].psnr;
++delta_count;
}
}
total.Add(encoded_frames_[frame_idx]);
}
// Total bytes should stay safely within the allocated CBR channel budget.
EXPECT_GT(total.actual.bytes(), 0);
EXPECT_LE(total.actual, total.ideal * 1.05);
// Quality must remain high during static slide periods despite low bitrate.
ASSERT_GT(delta_count, 0);
EXPECT_GT(sum_delta_psnr / delta_count, 40.0);
}
// Verifies that the encoder adheres to CBR targets during screenshare scrolling
// with alternating scrolling and paused intervals.
TEST_P(VideoEncoderRateControlTest, ScreenshareScrollingHd) {
if (!SupportsCbr()) {
GTEST_SKIP() << "Encoder does not support CBR mode.";
}
constexpr Resolution kResolution = kHdResolution;
constexpr Frequency kFramerate = Frequency::Hertz(30);
constexpr TimeDelta kFrameInterval = 1 / kFramerate;
constexpr DataRate kTargetBitrate = DataRate::KilobitsPerSec(2000);
constexpr TimeDelta kScrollDuration = TimeDelta::Seconds(2);
constexpr TimeDelta kPauseDuration = TimeDelta::Seconds(1);
constexpr TimeDelta kTotalDuration = TimeDelta::Seconds(6);
const int num_frames =
(kTotalDuration.us() + kFrameInterval.us() / 2) / kFrameInterval.us();
const std::vector<std::string> slides = {
test::ResourcePath("difficult_photo_1850_1110", "yuv"),
test::ResourcePath("web_screenshot_1850_1110", "yuv"),
};
std::unique_ptr<test::FrameGeneratorInterface> scroll_generator =
test::CreateScrollingInputFromYuvFilesFrameGenerator(
&env_.clock(), slides, /*source_width=*/1850,
/*source_height=*/1110, kResolution.width, kResolution.height,
kScrollDuration.ms(), kPauseDuration.ms());
SetUpCbrEncoder(kResolution);
EnableDecoder();
SetFrameGenerator(std::move(scroll_generator));
Encode({.num_frames = num_frames,
.target_bitrate = kTargetBitrate,
.frame_interval = kFrameInterval,
.resolution = kResolution,
.content_hint = VideoTrackInterface::ContentHint::kDetailed});
// During screenshare scrolling, bitrate undershoot during pause intervals is
// expected and allowed; only verify that overshoot is bounded.
VerifyTimeBasedSlidingWindowBitrateDeviation(
TimeDelta::Seconds(3),
/*min_allowed_dev_pct=*/std::nullopt,
/*max_allowed_dev_pct=*/20.0);
AccumulatedData total;
std::optional<double> min_pause_psnr;
double sum_pause_psnr = 0.0;
int pause_count = 0;
// 60 frames scroll (2s), 30 frames pause (1s), repeated.
for (size_t i = 0; i < encoded_frames_.size(); ++i) {
total.Add(encoded_frames_[i]);
int mod = i % 90;
if (mod >= 60 && encoded_frames_[i].psnr.has_value()) {
double p = *encoded_frames_[i].psnr;
min_pause_psnr = min_pause_psnr ? std::min(*min_pause_psnr, p) : p;
sum_pause_psnr += p;
++pause_count;
}
}
// Total bytes should stay safely within the allocated CBR channel budget.
EXPECT_GT(total.actual.bytes(), 0);
EXPECT_LE(total.actual, total.ideal * 1.05);
// Quality must not dip when bitrate drops during pauses.
ASSERT_TRUE(min_pause_psnr.has_value());
EXPECT_GT(*min_pause_psnr, 33.0);
EXPECT_GT(sum_pause_psnr / pause_count, 40.0);
}
// Verifies that the rate controller behaves well when entering "Zero-Hz" mode
// (where static frames are encoded at 1 Hz repeat rate while each frame's
// encoded CBR duration remains 1/target_fps) and subsequent resumption of
// regular motion.
TEST_P(VideoEncoderRateControlTest, ScreenshareZeroHzHd) {
if (!SupportsCbr()) {
GTEST_SKIP() << "Encoder does not support CBR mode.";
}
constexpr Resolution kResolution = kHdResolution;
constexpr Frequency kFramerate = Frequency::Hertz(30);
constexpr TimeDelta kFrameInterval = 1 / kFramerate;
constexpr DataRate kTargetBitrate = DataRate::KilobitsPerSec(1500);
SetUpCbrEncoder(kResolution);
EnableDecoder();
SetFrameGenerator(test::CreateFromYuvFileFrameGenerator(
{test::ResourcePath("FourPeople_1280x720_30", "yuv")}, kResolution.width,
kResolution.height, /*frame_repeat_count=*/1));
// Phase 1: 2 seconds of normal 30 fps motion.
constexpr int kPhase1Frames = 60;
Encode({.num_frames = kPhase1Frames,
.target_bitrate = kTargetBitrate,
.frame_interval = kFrameInterval,
.resolution = kResolution,
.content_hint = VideoTrackInterface::ContentHint::kDetailed});
// Phase 2: Enter Zero-Hz mode. We hold a static frame and encode at 1 Hz
// (timestamp advances by 1s), but CBR duration is still 1/target_fps
// (33.3ms).
constexpr int kZeroHzFrames = 3;
constexpr TimeDelta kZeroHzRepeatPeriod = TimeDelta::Seconds(1);
size_t phase2_start_idx = encoded_frames_.size();
Encode({
.num_frames = kZeroHzFrames,
.target_bitrate = kTargetBitrate,
.frame_interval = kZeroHzRepeatPeriod,
.frame_duration = kFrameInterval,
.resolution = kResolution,
.content_hint = VideoTrackInterface::ContentHint::kDetailed,
.repeat_frame = true,
});
for (size_t i = phase2_start_idx; i < phase2_start_idx + kZeroHzFrames; ++i) {
// Each 1 Hz frame must not burst to fill the entire 1-second gap; its ideal
// CBR allocation is 1 nominal frame duration (target_bitrate *
// kFrameInterval).
EXPECT_LE(encoded_frames_[i].actual,
kTargetBitrate * kFrameInterval * 125 / 100);
// Quality must remain high for static repeat frames.
ASSERT_TRUE(encoded_frames_[i].psnr.has_value());
EXPECT_GT(*encoded_frames_[i].psnr, 40.0);
}
// Phase 3: Resume motion at 30 fps for 3 seconds.
constexpr int kPhase3Frames = 90;
Encode({.num_frames = kPhase3Frames,
.target_bitrate = kTargetBitrate,
.frame_interval = kFrameInterval,
.resolution = kResolution,
.content_hint = VideoTrackInterface::ContentHint::kDetailed});
// Verify that the encoder survived Zero-Hz and after resuming motion,
// the Phase 3 frames converge to the target bitrate.
AccumulatedData phase3_total;
for (size_t i = phase2_start_idx + kZeroHzFrames; i < encoded_frames_.size();
++i) {
phase3_total.Add(encoded_frames_[i]);
}
RTC_LOG(LS_VERBOSE) << "Phase 3 total deviation: "
<< phase3_total.deviation_pct()
<< "% (actual=" << phase3_total.actual.bytes()
<< " bytes, ideal=" << phase3_total.ideal.bytes()
<< " bytes)";
EXPECT_NEAR(phase3_total.deviation_pct(), 0.0, 5.0);
}
std::unique_ptr<VideoEncoderFactoryInterface> CreateLibaomAv1EncoderFactory() {
return std::make_unique<LibaomAv1EncoderFactory>();
}
INSTANTIATE_TEST_SUITE_P(LibaomAv1,
VideoEncoderRateControlTest,
::testing::Values(CreateLibaomAv1EncoderFactory));
const FixedBitrateTestParams kFixedBitrateConfigs[] = {
{"VgaNormalBitrate", kVgaResolution, DataRate::KilobitsPerSec(500),
TimeDelta::Seconds(5), 5.0},
{"VgaLowBitrate", kVgaResolution, DataRate::KilobitsPerSec(100),
TimeDelta::Seconds(10), 5.0},
{"VgaHighBitrate", kVgaResolution, DataRate::KilobitsPerSec(1500),
TimeDelta::Seconds(5), 5.0},
{"QvgaNormalBitrate", kQvgaResolution, DataRate::KilobitsPerSec(125),
TimeDelta::Seconds(5), 6.0},
{"QvgaLowBitrate", kQvgaResolution, DataRate::KilobitsPerSec(25),
TimeDelta::Seconds(10), 10.0},
// TODO(bugs.webrtc.org/496266459): Bring this back to 5%. Most of what it
// covers is the cost of starting a sequence, which the encoder works off
// over a window far longer than the five seconds measured here, so a
// longer measurement rather than a wider tolerance is the way down.
{"QvgaHighBitrate", kQvgaResolution, DataRate::KilobitsPerSec(375),
TimeDelta::Seconds(5), 6.0},
{"HdNormalBitrate", kHdResolution, DataRate::KilobitsPerSec(2000),
TimeDelta::Seconds(5), 5.0},
{"HdLowBitrate", kHdResolution, DataRate::KilobitsPerSec(400),
TimeDelta::Seconds(10), 5.0},
{"HdHighBitrate", kHdResolution, DataRate::KilobitsPerSec(6000),
TimeDelta::Seconds(5), 5.0},
};
std::string FixedBitrateTestName(
const ::testing::TestParamInfo<
std::tuple<FactoryCreator, FixedBitrateTestParams>>& info) {
return std::get<1>(info.param).name;
}
INSTANTIATE_TEST_SUITE_P(
LibaomAv1,
FixedBitrateRateControlTest,
::testing::Combine(::testing::Values(CreateLibaomAv1EncoderFactory),
::testing::ValuesIn(kFixedBitrateConfigs)),
FixedBitrateTestName);
const TemporalLayerTestParams kTemporalLayerConfigs[] = {
// Halving the per frame bit budget for every temporal layer makes it
// proportional to the prediction distance, see `GeometricDistribution`.
// The base layer frame budget then grows as `2^N/(N+1)` frame intervals -
// 67 ms at three layers, but 533 ms at seven - so this stops short of
// `max_temporal_layers()` once it exceeds `kMaxFrameBudget`.
{"GeometricVga", kVgaResolution, DataRate::KilobitsPerSec(600),
[](int num_temporal_layers) {
return TemporalLayerPatternForTest::GeometricDistribution(
num_temporal_layers, /*ratio=*/0.5);
}},
// Only spans `N:1` between the base and top layer instead of the geometric
// `2^(N-1):1`, so it stays realizable all the way up. The pattern period
// doubles for every added layer, which makes covering enough groups of
// pictures expensive at high layer counts, hence the lower resolution.
{"LinearQvga", kQvgaResolution, DataRate::KilobitsPerSec(300),
&TemporalLayerPatternForTest::LinearDistribution},
};
std::string TemporalLayerTestName(
const ::testing::TestParamInfo<
std::tuple<FactoryCreator, TemporalLayerTestParams>>& info) {
return std::get<1>(info.param).name;
}
INSTANTIATE_TEST_SUITE_P(
LibaomAv1,
TemporalLayerRateControlTest,
::testing::Combine(::testing::Values(CreateLibaomAv1EncoderFactory),
::testing::ValuesIn(kTemporalLayerConfigs)),
TemporalLayerTestName);
} // namespace
} // namespace webrtc