Generalize TemporalLayerRateTracker into CbrLayerRateTracker. The tracker now covers spatial as well as temporal layers and produces an encoder agnostic CumulativeCbrAllocation: per spatial layer bitrates that are cumulative over the temporal layers, together with the temporal layer frame rate factors. Encoder wrappers only translate it to their own types and decide themselves when to apply it. - The tracker is fed once per temporal unit with the FrameEncodeSettings passed to Encode, instead of once per frame. Temporal unit boundaries are therefore known rather than guessed from the spatial ids. - Spatial layers without a frame in a temporal unit keep their last bitrate. Some rate controllers otherwise set the bitrate to zero, disabling the layer with unintended side effects. - The priming after a keyframe now applies to all spatial layers of the temporal unit that follows it, not only to single spatial layer streams. - The libaom wrapper only hands the encoder config and the SVC parameters to libaom when they differ from what was last applied. Both paths recompute the per layer rate control state from the frame rate of the most recently encoded frame, and reset the buffer of any layer whose per frame budget then appears to have changed by more than a factor of two. With spatial layers running at different frame rates this reset the lower layers before each of their frames, which made them overshoot their targets severalfold. Bug: webrtc:496266459 Change-Id: I260b348328ce9dfc2595a080b719e20b37b6890d Reviewed-on: https://webrtc-review.googlesource.com/c/src/+/506562 Reviewed-by: Sergey Silkin <ssilkin@webrtc.org> Commit-Queue: Erik Språng <sprang@webrtc.org> Cr-Commit-Position: refs/heads/main@{#48812}
diff --git a/modules/video_coding/BUILD.gn b/modules/video_coding/BUILD.gn index d8790cf..6327529 100644 --- a/modules/video_coding/BUILD.gn +++ b/modules/video_coding/BUILD.gn
@@ -324,6 +324,8 @@ sources = [ "utility/bandwidth_quality_scaler.cc", "utility/bandwidth_quality_scaler.h", + "utility/cbr_layer_rate_tracker.cc", + "utility/cbr_layer_rate_tracker.h", "utility/decoded_frames_history.cc", "utility/decoded_frames_history.h", "utility/frame_dropper.cc", @@ -345,8 +347,6 @@ "utility/simulcast_rate_allocator.h", "utility/simulcast_utility.cc", "utility/simulcast_utility.h", - "utility/temporal_layer_rate_tracker.cc", - "utility/temporal_layer_rate_tracker.h", "utility/vp8_constants.h", "utility/vp8_header_parser.cc", "utility/vp8_header_parser.h", @@ -371,6 +371,7 @@ "../../api/video:video_frame", "../../api/video:video_frame_type", "../../api/video_codecs:video_codecs_api", + "../../api/video_codecs:video_encoder_interface", "../../common_video", "../../rtc_base:bitstream_reader", "../../rtc_base:checks", @@ -1040,7 +1041,7 @@ ":sframe_packet_buffer_unittest", ":simulcast_rate_allocator_unittest", ":svc_config_unittest", - ":temporal_layer_rate_tracker_unittest", + ":cbr_layer_rate_tracker_unittest", ":video_codec_initializer_unittest", ":video_decoder_database_unittest", ":video_encoder_decoder_instantiation_tests", @@ -1595,11 +1596,14 @@ ] } - rtc_cc_test("temporal_layer_rate_tracker_unittest") { - sources = [ "utility/temporal_layer_rate_tracker_unittest.cc" ] + rtc_cc_test("cbr_layer_rate_tracker_unittest") { + sources = [ "utility/cbr_layer_rate_tracker_unittest.cc" ] deps = [ ":video_coding_utility", "../../api/units:data_rate", + "../../api/units:time_delta", + "../../api/video_codecs:video_encoder_factory_interface", + "../../api/video_codecs:video_encoder_interface", "../../test:test_support", ] }
diff --git a/modules/video_coding/codecs/av1/libaom_av1_encoder_v2.cc b/modules/video_coding/codecs/av1/libaom_av1_encoder_v2.cc index b28e729..e9931da 100644 --- a/modules/video_coding/codecs/av1/libaom_av1_encoder_v2.cc +++ b/modules/video_coding/codecs/av1/libaom_av1_encoder_v2.cc
@@ -39,8 +39,8 @@ #include "api/video_codecs/video_encoder_factory_interface.h" #include "api/video_codecs/video_encoder_interface.h" #include "api/video_codecs/video_encoding_general.h" +#include "modules/video_coding/utility/cbr_layer_rate_tracker.h" #include "modules/video_coding/utility/reference_buffer_tracker.h" -#include "modules/video_coding/utility/temporal_layer_rate_tracker.h" #include "rtc_base/checks.h" #include "rtc_base/logging.h" #include "rtc_base/numerics/rational.h" @@ -531,6 +531,15 @@ return std::clamp(target_qp / 4, 0, kMaxQuantizer); } +// Returns true if `value` has to be handed to libaom, i.e. if nothing has been +// applied yet or if it differs from what was last applied. The libaom config +// structs are plain C structs, so a bytewise comparison is sufficient. +template <typename T> +bool NeedsApply(const std::optional<T>& applied, const T& value) { + static_assert(std::is_trivially_copyable_v<T>); + return !applied.has_value() || std::memcmp(&*applied, &value, sizeof(T)) != 0; +} + } // namespace aom_svc_params_t LibaomAv1EncoderV2::GetSvcParams( @@ -556,9 +565,26 @@ // compared to the full stream, and `layer_target_bitrate` is its bitrate. // libaom recovers the per frame budget of an individual layer by // differentiating adjacent layers, see `av1_update_temporal_layer_framerate`. + // Both come from `rate_tracker_`, which accumulates them from the per frame + // reports seen so far. + const CumulativeCbrAllocation& allocation = rate_tracker_->allocation(); for (int tid = 0; tid < svc_params.number_temporal_layers; ++tid) { - framerate_factor_view[tid] = rate_tracker_->FramerateFactor(tid); + framerate_factor_view[tid] = allocation.framerate_factor[tid]; } + auto set_cbr_layer_rates = [&](int spatial_id) { + for (int tid = 0; tid < svc_params.number_temporal_layers; ++tid) { + const int id = spatial_id * svc_params.number_temporal_layers + tid; + // A layer with zero bitrate is considered disabled by libaom, so always + // leave at least 1 kbps. + layer_target_bitrate_view[id] = + std::max<int>(1, allocation.bitrate[spatial_id][tid].kbps()); + // When libaom is configured with `AOM_CBR` it will still limit QP to + // stay between `min_quantizers` and `max_quantizers'. Set + // `max_quantizers` to max QP to avoid the encoder overshooting. + max_quantizers_view[id] = kMaxQuantizer; + min_quantizers_view[id] = 0; + } + }; // If the scaling factor is left at zero for unused layers a division by zero // will happen inside libaom, default all layers to one. @@ -589,25 +615,7 @@ [&](auto&& arg) { using T = std::decay_t<decltype(arg)>; if constexpr (std::is_same_v<T, Cbr>) { - // Rate control in this API is expressed per frame: a frame states - // the bitrate of the temporal layer it belongs to, and that layer - // alone. libaom instead wants the bitrate of the stream formed by - // all the layers up to and including a given one, which - // `rate_tracker_` accumulates from the per frame reports seen so - // far. - for (int id = first_layer_id; id < last_layer_id; ++id) { - const DataRate layer_bitrate = rate_tracker_->CumulativeBitrate( - settings.spatial_id(), /*temporal_id=*/id - first_layer_id); - // A layer with zero bitrate is considered disabled by libaom, so - // always leave at least 1 kbps. - layer_target_bitrate_view[id] = - std::max<int>(1, layer_bitrate.kbps()); - // When libaom is configured with `AOM_CBR` it will still limit QP - // to stay between `min_quantizers` and `max_quantizers'. Set - // `max_quantizers` to max QP to avoid the encoder overshooting. - max_quantizers_view[id] = kMaxQuantizer; - min_quantizers_view[id] = 0; - } + set_cbr_layer_rates(settings.spatial_id()); } else if constexpr (std::is_same_v<T, Cqp>) { int quantizer = ToLibaomQuantizer(arg.target_qp); for (int id = first_layer_id; id < last_layer_id; ++id) { @@ -627,6 +635,20 @@ settings.rate_options()); } + // Spatial layers below the lowest one in this temporal unit get no + // `aom_codec_encode` call, so their bitrate is not used for this temporal + // unit. A zero bitrate would however make libaom empty their rate control + // buffers, and changing it back and forth would reconfigure libaom on every + // temporal unit, so they keep the bitrate they have in `allocation`. + // TODO(bugs.webrtc.org/496266459): Unused layers between the ones in this + // temporal unit still get a zero bitrate, since that is what makes libaom + // skip them. + if (std::holds_alternative<Cbr>(frame_settings[0].rate_options())) { + for (int sid = 0; sid < frame_settings[0].spatial_id(); ++sid) { + set_cbr_layer_rates(sid); + } + } + if (RTC_LOG_CHECK_LEVEL(LS_VERBOSE)) { StringBuilder sb; sb << "GetSvcParams layer bitrates kbps"; @@ -707,9 +729,11 @@ last_resolution_in_buffer_ = {}; reference_buffer_tracker_.Reset(); - rate_tracker_ = std::make_unique<TemporalLayerRateTracker>(); + rate_tracker_ = std::make_unique<CbrLayerRateTracker>(); content_type_.reset(); effort_level_by_spatial_id_.fill(std::nullopt); + applied_cfg_.reset(); + applied_svc_params_.reset(); if (aom_codec_err_t ret = aom_codec_enc_config_default( aom_codec_av1_cx(), &cfg_, AOM_USAGE_REALTIME); @@ -830,13 +854,7 @@ return; } - for (const FrameEncodeSettings& settings : frame_settings) { - if (const Cbr* cbr = std::get_if<Cbr>(&settings.rate_options())) { - rate_tracker_->Update( - settings.spatial_id(), settings.temporal_id(), cbr->target_bitrate, - settings.frame_type() == VideoEncoderInterface::FrameType::kKeyframe); - } - } + rate_tracker_->OnTemporalUnit(frame_settings); if (content_type_ != tu_settings.content_hint()) { if (tu_settings.content_hint() == ContentHint::kText || @@ -851,12 +869,14 @@ } if (cfg_.rc_end_usage == AOM_CBR) { - // The target bitrate of the current frame only describes the temporal - // layer it belongs to, so the bitrate of the stream as a whole is taken - // from `rate_tracker_`. + // The target bitrate of the current frame only describes the layer it + // belongs to, so the bitrate of the stream as a whole is taken from + // `rate_tracker_`. Spatial layers without a frame in this temporal unit + // are included, as they keep their bitrate in `GetSvcParams`. + const CumulativeCbrAllocation& allocation = rate_tracker_->allocation(); DataRate accum_rate = DataRate::Zero(); - for (const FrameEncodeSettings& settings : frame_settings) { - accum_rate += rate_tracker_->StreamBitrate(settings.spatial_id()); + for (int sid = 0; sid <= frame_settings.back().spatial_id(); ++sid) { + accum_rate += allocation.SpatialLayerBitrate(sid); } cfg_.rc_target_bitrate = accum_rate.kbps(); // Let the rate controller use the full quantizer range, matching the per @@ -894,13 +914,21 @@ // The bitrates calculated internally in libaom when `AV1E_SET_SVC_PARAMS` is // called depends on the currently configured `cfg_.rc_target_bitrate`. If the // total target bitrate is not updated first a division by zero could happen. - if (aom_codec_err_t ret = aom_codec_enc_config_set(&ctx_, &cfg_); - ret != AOM_CODEC_OK) { - RTC_LOG(LS_ERROR) << "aom_codec_enc_config_set returned " << ret; - return; + if (NeedsApply(applied_cfg_, cfg_)) { + applied_cfg_.reset(); + if (aom_codec_err_t ret = aom_codec_enc_config_set(&ctx_, &cfg_); + ret != AOM_CODEC_OK) { + RTC_LOG(LS_ERROR) << "aom_codec_enc_config_set returned " << ret; + return; + } + applied_cfg_ = cfg_; } aom_svc_params_t svc_params = GetSvcParams(*frame_buffer, frame_settings); - SET_OR_RETURN(AV1E_SET_SVC_PARAMS, &svc_params); + if (NeedsApply(applied_svc_params_, svc_params)) { + applied_svc_params_.reset(); + SET_OR_RETURN(AV1E_SET_SVC_PARAMS, &svc_params); + applied_svc_params_ = svc_params; + } // The libaom AV1 encoder requires that `aom_codec_encode` is called for // every spatial layer, even if no frame should be encoded for that layer.
diff --git a/modules/video_coding/codecs/av1/libaom_av1_encoder_v2.h b/modules/video_coding/codecs/av1/libaom_av1_encoder_v2.h index b94f08a..e44d878 100644 --- a/modules/video_coding/codecs/av1/libaom_av1_encoder_v2.h +++ b/modules/video_coding/codecs/av1/libaom_av1_encoder_v2.h
@@ -24,8 +24,8 @@ #include "api/video/video_frame_buffer.h" #include "api/video_codecs/video_encoder_factory_interface.h" #include "api/video_codecs/video_encoder_interface.h" +#include "modules/video_coding/utility/cbr_layer_rate_tracker.h" #include "modules/video_coding/utility/reference_buffer_tracker.h" -#include "modules/video_coding/utility/temporal_layer_rate_tracker.h" #include "third_party/libaom/source/libaom/aom/aom_codec.h" #include "third_party/libaom/source/libaom/aom/aom_encoder.h" #include "third_party/libaom/source/libaom/aom/aom_image.h" @@ -59,6 +59,12 @@ aom_img_ptr image_to_encode_ = aom_img_ptr(nullptr, aom_img_free); aom_codec_ctx_t ctx_{}; aom_codec_enc_cfg_t cfg_{}; + // The configuration and SVC parameters last handed to libaom. Reapplying + // them has side effects even when nothing changed, e.g. the rate control of + // layers running at a lower frame rate than the last encoded one is reset, + // so they are only applied when they differ from these. + std::optional<aom_codec_enc_cfg_t> applied_cfg_; + std::optional<aom_svc_params_t> applied_svc_params_; std::optional<ContentHint> content_type_; std::array<std::optional<int>, kMaxSpatialLayers> effort_level_by_spatial_id_; @@ -67,7 +73,7 @@ ReferenceBufferTracker reference_buffer_tracker_{kNumBuffers}; // Recreated on every `InitEncode`, since what it has learned only describes // the configuration it was fed. - std::unique_ptr<TemporalLayerRateTracker> rate_tracker_; + std::unique_ptr<CbrLayerRateTracker> rate_tracker_; }; } // namespace webrtc
diff --git a/modules/video_coding/utility/cbr_layer_rate_tracker.cc b/modules/video_coding/utility/cbr_layer_rate_tracker.cc new file mode 100644 index 0000000..1a3f8f8 --- /dev/null +++ b/modules/video_coding/utility/cbr_layer_rate_tracker.cc
@@ -0,0 +1,243 @@ +/* + * Copyright (c) 2026 The WebRTC project authors. All Rights Reserved. + * + * Use of this source code is governed by a BSD-style license + * that can be found in the LICENSE file in the root of the source + * tree. An additional intellectual property rights grant can be found + * in the file PATENTS. All contributing project authors may + * be found in the AUTHORS file in the root of the source tree. + */ + +#include "modules/video_coding/utility/cbr_layer_rate_tracker.h" + +#include <algorithm> +#include <cmath> +#include <optional> +#include <span> +#include <variant> + +#include "api/units/data_rate.h" +#include "api/video_codecs/video_encoder_interface.h" +#include "rtc_base/checks.h" +#include "rtc_base/numerics/exp_filter.h" + +namespace webrtc { +namespace { + +using FrameEncodeSettings = VideoEncoderInterface::FrameEncodeSettings; + +// How often the frames of a layer occur is a property of the temporal +// structure, which either stays the same forever or changes wholesale. The +// samples are constant as long as the structure is, so there is no ripple to +// suppress and the filter only has to average out the jitter of structures +// whose layers do not occur at a fixed interval. A slow filter is what lets a +// single out of phase interval, as seen after a keyframe or a dropped frame, +// pass without disturbing the estimate. +constexpr float kFrameIntervalAlpha = 0.9f; + +// The number of frames between consecutive frames of temporal layer `tid` in a +// dyadic L1Tx pattern with `num_layers` temporal layers. The topmost layer +// holds every other frame, the one below it every fourth, and so on, with the +// base layer matching the layer just above it. +double DyadicFrameInterval(int tid, int num_layers) { + if (num_layers <= 1) { + return 1.0; + } + return 1 << (num_layers - std::max(tid, 1)); +} + +float Filtered(const ExpFilter& filter) { + const float value = filter.filtered(); + return value == ExpFilter::kValueUndefined ? 0.0f : value; +} + +} // namespace + +DataRate CumulativeCbrAllocation::SpatialLayerBitrate(int spatial_id) const { + RTC_DCHECK_GE(spatial_id, 0); + RTC_DCHECK_LT(spatial_id, kMaxSpatialLayers); + return bitrate[spatial_id][kMaxTemporalLayers - 1]; +} + +DataRate CumulativeCbrAllocation::TotalBitrate() const { + DataRate total = DataRate::Zero(); + for (int sid = 0; sid < kMaxSpatialLayers; ++sid) { + total += SpatialLayerBitrate(sid); + } + return total; +} + +CbrLayerRateTracker::CbrLayerRateTracker() + : frame_interval_filters_(kMaxTemporalLayers, + ExpFilter(kFrameIntervalAlpha)) { + // Until a frame of a higher layer shows up the stream is assumed to consist + // of a single temporal layer holding all of the frames. + frame_interval_filters_[0].Apply(1.0f, 1.0f); +} + +void CbrLayerRateTracker::PrimeStandardPattern(int num_layers) { + RTC_DCHECK_GT(num_layers, 1); + RTC_DCHECK_LE(num_layers, kMaxTemporalLayers); + num_temporal_layers_ = std::max(num_temporal_layers_, num_layers); + + for (int tid = 0; tid < kMaxTemporalLayers; ++tid) { + frame_interval_filters_[tid].Reset(kFrameIntervalAlpha); + if (tid < num_layers) { + frame_interval_filters_[tid].Apply( + 1.0f, static_cast<float>(DyadicFrameInterval(tid, num_layers))); + } + } +} + +void CbrLayerRateTracker::PrimeLayerRates(int num_layers, + int spatial_id, + DataRate delta_bitrate) { + // Only the bitrates of the base layer, taken from the keyframe, and of the + // topmost layer, taken from the frame at hand, are known. In the recommended + // distribution, where the per frame bit budget halves for every step up the + // temporal layer stack, every layer above the base one ends up with the same + // share of the bitrate: the frames of a layer are twice as many but half as + // large as those of the layer below it. Assume that is the case here, which + // leaves the base layer as observed on the keyframe. The guesses are not + // recorded as stated bitrates, so the first frame of a layer that turns out + // to hold something else is not mistaken for a change of the allocation. + for (int tid = 1; tid < num_layers; ++tid) { + layer_rates_[spatial_id][tid].estimate = delta_bitrate; + } +} + +void CbrLayerRateTracker::UpdateLayerRates(int spatial_id, + int temporal_id, + DataRate layer_bitrate) { + SpatialLayerRates& layers = layer_rates_[spatial_id]; + LayerRate& layer = layers[temporal_id]; + + // A caller that changes the allocation normally scales the whole stream, so + // a layer that moves is taken to speak for the layers that have not reported + // since. What is carried over is only the part of the change the tracker did + // not already assume, and a layer that restates the bitrate it already had + // says nothing at all, which together keep a change of the distribution from + // being handed back and forth between the layers. + if (layer.stated.has_value() && *layer.stated != layer_bitrate && + layer.estimate > DataRate::Zero()) { + const double change = layer_bitrate / layer.estimate; + for (int tid = 0; tid < kMaxTemporalLayers; ++tid) { + if (tid != temporal_id) { + layers[tid].estimate = layers[tid].estimate * change; + } + } + } + + layer.estimate = layer_bitrate; + layer.stated = layer_bitrate; +} + +void CbrLayerRateTracker::OnTemporalUnit( + std::span<const FrameEncodeSettings> frames) { + // The temporal id of the first frame of the temporal unit is what the + // cadence is measured from. + std::optional<int> unit_temporal_id; + bool has_keyframe = false; + for (const FrameEncodeSettings& frame : frames) { + if (!std::holds_alternative<FrameEncodeSettings::Cbr>( + frame.rate_options())) { + continue; + } + RTC_DCHECK_GE(frame.spatial_id(), 0); + RTC_DCHECK_LT(frame.spatial_id(), kMaxSpatialLayers); + RTC_DCHECK_GE(frame.temporal_id(), 0); + RTC_DCHECK_LT(frame.temporal_id(), kMaxTemporalLayers); + if (!unit_temporal_id.has_value()) { + unit_temporal_id = frame.temporal_id(); + } + has_keyframe |= + frame.frame_type() == VideoEncoderInterface::FrameType::kKeyframe; + } + if (!unit_temporal_id.has_value()) { + return; + } + const int tid = *unit_temporal_id; + + // A keyframe restarts the temporal structure, and since it belongs to the + // base layer the temporal unit that follows it reveals the layer count. + const bool prime = last_unit_had_keyframe_ && !has_keyframe && tid > 0; + last_unit_had_keyframe_ = has_keyframe; + if (prime) { + PrimeStandardPattern(tid + 1); + } + + for (const FrameEncodeSettings& frame : frames) { + const auto* cbr = + std::get_if<FrameEncodeSettings::Cbr>(&frame.rate_options()); + if (cbr == nullptr) { + continue; + } + if (prime) { + PrimeLayerRates(tid + 1, frame.spatial_id(), cbr->target_bitrate); + } + UpdateLayerRates(frame.spatial_id(), frame.temporal_id(), + cbr->target_bitrate); + num_temporal_layers_ = + std::max(num_temporal_layers_, frame.temporal_id() + 1); + } + + ++temporal_unit_count_; + if (last_unit_of_layer_[tid].has_value()) { + frame_interval_filters_[tid].Apply( + 1.0f, temporal_unit_count_ - *last_unit_of_layer_[tid]); + } + last_unit_of_layer_[tid] = temporal_unit_count_; + + UpdateAllocation(); +} + +double CbrLayerRateTracker::FrameFraction(int temporal_id) const { + const double interval = Filtered(frame_interval_filters_[temporal_id]); + return interval > 0.0 ? 1.0 / interval : 0.0; +} + +int CbrLayerRateTracker::FramerateFactor(int temporal_id) const { + if (temporal_id >= num_temporal_layers_ - 1) { + return 1; + } + + double total_fraction = 0.0; + double cumulative_fraction = 0.0; + for (int tid = 0; tid < num_temporal_layers_; ++tid) { + const double fraction = FrameFraction(tid); + total_fraction += fraction; + if (tid <= temporal_id) { + cumulative_fraction += fraction; + } + } + + if (cumulative_fraction <= 0.0) { + return 1; + } + return std::max( + 1, static_cast<int>(std::round(total_fraction / cumulative_fraction))); +} + +DataRate CbrLayerRateTracker::CumulativeBitrate(int spatial_id, + int temporal_id) const { + DataRate bitrate = DataRate::Zero(); + for (int tid = 0; tid <= std::min(temporal_id, num_temporal_layers_ - 1); + ++tid) { + bitrate += layer_rates_[spatial_id][tid].estimate; + } + return bitrate; +} + +void CbrLayerRateTracker::UpdateAllocation() { + allocation_.num_temporal_layers = num_temporal_layers_; + for (int tid = 0; tid < kMaxTemporalLayers; ++tid) { + allocation_.framerate_factor[tid] = FramerateFactor(tid); + } + for (int sid = 0; sid < kMaxSpatialLayers; ++sid) { + for (int tid = 0; tid < kMaxTemporalLayers; ++tid) { + allocation_.bitrate[sid][tid] = CumulativeBitrate(sid, tid); + } + } +} + +} // namespace webrtc
diff --git a/modules/video_coding/utility/cbr_layer_rate_tracker.h b/modules/video_coding/utility/cbr_layer_rate_tracker.h new file mode 100644 index 0000000..51cf0dc --- /dev/null +++ b/modules/video_coding/utility/cbr_layer_rate_tracker.h
@@ -0,0 +1,183 @@ +/* + * Copyright (c) 2026 The WebRTC project authors. All Rights Reserved. + * + * Use of this source code is governed by a BSD-style license + * that can be found in the LICENSE file in the root of the source + * tree. An additional intellectual property rights grant can be found + * in the file PATENTS. All contributing project authors may + * be found in the AUTHORS file in the root of the source tree. + */ + +#ifndef MODULES_VIDEO_CODING_UTILITY_CBR_LAYER_RATE_TRACKER_H_ +#define MODULES_VIDEO_CODING_UTILITY_CBR_LAYER_RATE_TRACKER_H_ + +#include <array> +#include <optional> +#include <span> + +#include "absl/container/inlined_vector.h" +#include "api/units/data_rate.h" +#include "api/video_codecs/video_encoder_interface.h" +#include "rtc_base/numerics/exp_filter.h" + +namespace webrtc { + +// The rates of all the layers of a stream, in the form encoder libraries tend +// to want them. Unlike `VideoBitrateAllocation` the bitrates are cumulative +// over the temporal layers: the entry for a temporal layer describes the stream +// made up of that layer and all the temporal layers below it, within the same +// spatial layer. They are not cumulative over the spatial layers. +struct CumulativeCbrAllocation { + static constexpr int kMaxSpatialLayers = 4; + static constexpr int kMaxTemporalLayers = 4; + + // The bitrate of all temporal layers of spatial layer `spatial_id` combined. + DataRate SpatialLayerBitrate(int spatial_id) const; + + // The bitrate of all layers combined. + DataRate TotalBitrate() const; + + bool operator==(const CumulativeCbrAllocation& other) const = default; + + // The number of temporal layers seen so far, or the number guessed from the + // start of the stream. At least one. + int num_temporal_layers = 1; + + // How many times lower the frame rate of the stream made up of the temporal + // layers up to and including a given one is compared to the frame rate of + // the full stream. Shared by all spatial layers. Rounded to an integer, which + // is exact for the dyadic temporal structures used in practice. Entries from + // the topmost temporal layer and up are one. + std::array<int, kMaxTemporalLayers> framerate_factor = {1, 1, 1, 1}; + + // `bitrate[spatial_id][temporal_id]` is the bitrate of the stream made up of + // the temporal layers up to and including `temporal_id` within spatial layer + // `spatial_id`. Entries above the topmost temporal layer repeat its value, + // and spatial layers never heard from are zero. + std::array<std::array<DataRate, kMaxTemporalLayers>, kMaxSpatialLayers> + bitrate = {}; +}; + +// Aggregates rate control decisions that are made per frame into a +// `CumulativeCbrAllocation`. +// +// Callers of `VideoEncoderInterface` state the bitrate of the layer a frame +// belongs to, one frame at a time. Encoder libraries on the other hand tend to +// want the bitrate and the frame rate of the stream formed by all the temporal +// layers up to and including a given one, which this class derives by tracking +// the bitrate stated for each layer together with how often frames of each +// temporal layer occur. +// +// Only the layers present in a temporal unit are heard from, so the bitrate of +// the others has to be assumed until they report. A caller that changes the +// allocation normally scales all of the temporal layers of a spatial layer by +// the same factor, so a change seen on one of them is taken to apply to those +// that have not reported since. The assumption is dropped as soon as a layer +// states a bitrate of its own, and a layer repeating the bitrate it already had +// is not taken as evidence of anything, which is what keeps a change of the +// distribution between the layers from ping ponging around rather than +// settling. Spatial layers are independent of each other in this respect. A +// layer without a frame in a temporal unit, be it a temporal or a spatial one, +// keeps the bitrate it last had. +// +// How often the frames of a temporal layer occur cannot be stated by the +// caller and is measured instead, and smoothed. All spatial layers are assumed +// to share one temporal structure, though not necessarily in phase: the cadence +// is measured from the first frame of each temporal unit, so a structure that +// shifts the spatial layers relative to each other, such as L2T2_KEY_SHIFT, is +// described correctly as well. To avoid a transient at the start of a stream, +// the state is primed on the first temporal unit after a keyframe under the +// assumption that a standard dyadic temporal pattern is used: in such a pattern +// that temporal unit belongs to the topmost temporal layer, which reveals the +// layer count and thereby how often the frames of every layer occur. Structures +// the priming does not predict are learned instead, which takes a few +// repetitions of the pattern. +// +// An instance only ever describes one configuration of one encoder. Replace it +// rather than trying to reuse it when the encoder is reconfigured. +class CbrLayerRateTracker { + public: + static constexpr int kMaxSpatialLayers = + CumulativeCbrAllocation::kMaxSpatialLayers; + static constexpr int kMaxTemporalLayers = + CumulativeCbrAllocation::kMaxTemporalLayers; + + CbrLayerRateTracker(); + + // Records the frames of one temporal unit, as passed to + // `VideoEncoderInterface::Encode`. Temporal units must be passed in encode + // order. Frames that do not use `Cbr` rate options are ignored. + void OnTemporalUnit( + std::span<const VideoEncoderInterface::FrameEncodeSettings> frames); + + // The allocation as of the most recent temporal unit. + const CumulativeCbrAllocation& allocation() const { return allocation_; } + + private: + // What is known about the bitrate of one temporal layer of one spatial + // layer. + struct LayerRate { + // The bitrate the tracker believes the layer has: what the caller stated + // for it scaled by the changes seen on the other layers since, or what the + // priming guessed. + DataRate estimate = DataRate::Zero(); + // The bitrate the caller most recently stated for the layer, if any. Only + // a departure from this counts as a change of the allocation. + std::optional<DataRate> stated; + }; + using SpatialLayerRates = std::array<LayerRate, kMaxTemporalLayers>; + + // `ExpFilter` is not assignable, so the filters are held in a container that + // can be filled with copies of a prototype on construction. + using TemporalLayerFilters = + absl::InlinedVector<ExpFilter, kMaxTemporalLayers>; + + // Records that `layer_bitrate` was allocated to temporal layer `temporal_id` + // of spatial layer `spatial_id`, and carries the change over to the layers + // that have not reported since. + void UpdateLayerRates(int spatial_id, + int temporal_id, + DataRate layer_bitrate); + + // Primes the frame intervals as if a standard dyadic pattern with + // `num_layers` temporal layers was in use. + void PrimeStandardPattern(int num_layers); + + // Guesses the bitrates of the temporal layers of spatial layer `spatial_id` + // in a standard dyadic pattern with `num_layers` temporal layers, where the + // just observed topmost layer frame reported `delta_bitrate`. + void PrimeLayerRates(int num_layers, int spatial_id, DataRate delta_bitrate); + + // The share of the frames of the stream that belong to `temporal_id`, or + // zero if no two frames of that layer have been seen yet. + double FrameFraction(int temporal_id) const; + + int FramerateFactor(int temporal_id) const; + DataRate CumulativeBitrate(int spatial_id, int temporal_id) const; + + // Recomputes `allocation_` from the state. + void UpdateAllocation(); + + int num_temporal_layers_ = 1; + bool last_unit_had_keyframe_ = false; + // The number of temporal units seen so far, which doubles as the index of + // the current one, and the index of the temporal unit each temporal layer + // was last seen in. + int temporal_unit_count_ = 0; + std::array<std::optional<int>, kMaxTemporalLayers> last_unit_of_layer_; + + // What each temporal layer of each spatial layer is believed to hold. + std::array<SpatialLayerRates, kMaxSpatialLayers> layer_rates_; + + // The number of temporal units between consecutive frames of each temporal + // layer. Filtering the interval rather than the share of the frames each + // layer holds directly means the samples are constant for a fixed temporal + // structure, so the filter can be quick without the estimate rippling. + TemporalLayerFilters frame_interval_filters_; + + CumulativeCbrAllocation allocation_; +}; + +} // namespace webrtc + +#endif // MODULES_VIDEO_CODING_UTILITY_CBR_LAYER_RATE_TRACKER_H_
diff --git a/modules/video_coding/utility/cbr_layer_rate_tracker_unittest.cc b/modules/video_coding/utility/cbr_layer_rate_tracker_unittest.cc new file mode 100644 index 0000000..e4582b3 --- /dev/null +++ b/modules/video_coding/utility/cbr_layer_rate_tracker_unittest.cc
@@ -0,0 +1,708 @@ +/* + * Copyright (c) 2026 The WebRTC project authors. All Rights Reserved. + * + * Use of this source code is governed by a BSD-style license + * that can be found in the LICENSE file in the root of the source + * tree. An additional intellectual property rights grant can be found + * in the file PATENTS. All contributing project authors may + * be found in the AUTHORS file in the root of the source tree. + */ + +#include "modules/video_coding/utility/cbr_layer_rate_tracker.h" + +#include <array> +#include <optional> +#include <vector> + +#include "api/units/data_rate.h" +#include "api/units/time_delta.h" +#include "api/video_codecs/video_encoder_builders.h" +#include "api/video_codecs/video_encoder_interface.h" +#include "test/gtest.h" + +namespace webrtc { +namespace { + +using FrameType = VideoEncoderInterface::FrameType; + +constexpr DataRate kStreamBitrate = DataRate::KilobitsPerSec(600); +// The tracker does not look at the duration. +constexpr TimeDelta kFrameDuration = TimeDelta::Millis(33); + +struct Frame { + int spatial_id; + int temporal_id; + DataRate bitrate; + bool is_keyframe; +}; + +// Feeds the tracker one temporal unit made up of `frames`. +void EncodeTemporalUnit(CbrLayerRateTracker& tracker, + const std::vector<Frame>& frames) { + std::vector<VideoEncoderInterface::FrameEncodeSettings> settings; + for (const Frame& frame : frames) { + settings.push_back(FrameEncodeSettingsBuilder() + .SpatialId(frame.spatial_id) + .TemporalId(frame.temporal_id) + .CbrRateOptions(kFrameDuration, frame.bitrate) + .FrameType(frame.is_keyframe + ? FrameType::kKeyframe + : FrameType::kDeltaFrame) + .Build()); + } + tracker.OnTemporalUnit(settings); +} + +// Feeds the tracker a temporal unit holding a single frame. +void EncodeFrame(CbrLayerRateTracker& tracker, + int spatial_id, + int temporal_id, + DataRate bitrate, + bool is_keyframe) { + EncodeTemporalUnit(tracker, + {{spatial_id, temporal_id, bitrate, is_keyframe}}); +} + +// The bitrate of temporal layer `tid` in a dyadic L1Tx pattern with +// `num_layers` temporal layers, where the per frame bit budget is halved for +// every step up the temporal layer stack and the stream as a whole targets +// `kStreamBitrate`. Matches +// `TemporalLayerPatternForTest::GeometricDistribution` with a ratio of 0.5. +// +// Every layer above the base one ends up with the same share of the bitrate, +// since its frames are twice as many but half as large as those of the layer +// below it. The base layer holds as many frames as the layer just above it, +// each twice the size, so it gets twice the share. +DataRate GeometricLayerBitrate(int tid, int num_layers) { + return kStreamBitrate * ((tid == 0 ? 2.0 : 1.0) / (num_layers + 1)); +} + +// Feeds the tracker a keyframe followed by the first delta frame of a dyadic +// L1Tx pattern, which is all the priming needs. +void EncodeKeyframeAndFirstDeltaFrame(CbrLayerRateTracker& tracker, + int num_layers) { + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/0, + GeometricLayerBitrate(0, num_layers), /*is_keyframe=*/true); + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/num_layers - 1, + GeometricLayerBitrate(num_layers - 1, num_layers), + /*is_keyframe=*/false); +} + +// A dyadic L1T3 pattern. Its first two frames are the ones +// `EncodeKeyframeAndFirstDeltaFrame` feeds, so a run of the pattern picks up +// at frame two. +constexpr std::array<int, 4> kL1T3Pattern = {0, 2, 1, 2}; + +int L1T3TemporalId(int frame) { + return kL1T3Pattern[frame % kL1T3Pattern.size()]; +} + +// Feeds `num_frames` frames of the L1T3 pattern, starting at `first_frame`, +// with every layer allocated `scale` times its share of `kStreamBitrate`. +void EncodeL1T3Frames(CbrLayerRateTracker& tracker, + int first_frame, + int num_frames, + double scale = 1.0) { + for (int frame = first_frame; frame < first_frame + num_frames; ++frame) { + const int temporal_id = L1T3TemporalId(frame); + EncodeFrame(tracker, /*spatial_id=*/0, temporal_id, + GeometricLayerBitrate(temporal_id, 3) * scale, + /*is_keyframe=*/false); + } +} + +// As `kL1T3Pattern`, for four temporal layers. +constexpr std::array<int, 8> kL1T4Pattern = {0, 3, 2, 3, 1, 3, 2, 3}; + +int L1T4TemporalId(int frame) { + return kL1T4Pattern[frame % kL1T4Pattern.size()]; +} + +// The share of `stream_bitrate` that a geometric distribution over +// `num_layers` temporal layers gives to `temporal_id`, and the share it gives +// to all the layers up to and including it. +DataRate LayerShare(DataRate stream_bitrate, int temporal_id, int num_layers) { + return stream_bitrate * ((temporal_id == 0 ? 2.0 : 1.0) / (num_layers + 1)); +} + +DataRate CumulativeShare(DataRate stream_bitrate, + int temporal_id, + int num_layers) { + return stream_bitrate * + (static_cast<double>(temporal_id + 2) / (num_layers + 1)); +} + +TEST(CbrLayerRateTrackerTest, SingleLayerReportsTheFullBitrate) { + CbrLayerRateTracker tracker; + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/0, kStreamBitrate, + /*is_keyframe=*/true); + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/0, kStreamBitrate, + /*is_keyframe=*/false); + + EXPECT_EQ(tracker.allocation().num_temporal_layers, 1); + EXPECT_EQ(tracker.allocation().framerate_factor[0], 1); + EXPECT_EQ(tracker.allocation().bitrate[0][0], kStreamBitrate); + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), kStreamBitrate); +} + +TEST(CbrLayerRateTrackerTest, NothingIsKnownBeforeTheFirstFrame) { + CbrLayerRateTracker tracker; + + EXPECT_EQ(tracker.allocation().num_temporal_layers, 1); + EXPECT_EQ(tracker.allocation().framerate_factor[0], 1); + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), DataRate::Zero()); +} + +TEST(CbrLayerRateTrackerTest, PrimesTwoLayersOnFirstFrameAfterKeyframe) { + CbrLayerRateTracker tracker; + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/2); + + EXPECT_EQ(tracker.allocation().num_temporal_layers, 2); + EXPECT_EQ(tracker.allocation().framerate_factor[0], 2); + EXPECT_EQ(tracker.allocation().framerate_factor[1], 1); + // The layers hold half of the frames each, and a base layer frame is twice + // the size of a T1 frame, so the base layer accounts for two thirds of the + // bitrate. + EXPECT_NEAR(tracker.allocation().bitrate[0][0].kbps(), 400, 1); + EXPECT_NEAR(tracker.allocation().SpatialLayerBitrate(0).kbps(), + kStreamBitrate.kbps(), 1); +} + +TEST(CbrLayerRateTrackerTest, PrimesThreeLayersOnFirstFrameAfterKeyframe) { + CbrLayerRateTracker tracker; + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); + + EXPECT_EQ(tracker.allocation().num_temporal_layers, 3); + EXPECT_EQ(tracker.allocation().framerate_factor[0], 4); + EXPECT_EQ(tracker.allocation().framerate_factor[1], 2); + EXPECT_EQ(tracker.allocation().framerate_factor[2], 1); + EXPECT_NEAR(tracker.allocation().bitrate[0][0].kbps(), 300, 1); + EXPECT_NEAR(tracker.allocation().bitrate[0][1].kbps(), 450, 1); + EXPECT_NEAR(tracker.allocation().SpatialLayerBitrate(0).kbps(), + kStreamBitrate.kbps(), 1); +} + +TEST(CbrLayerRateTrackerTest, PrimesFourLayersOnFirstFrameAfterKeyframe) { + CbrLayerRateTracker tracker; + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/4); + + EXPECT_EQ(tracker.allocation().num_temporal_layers, 4); + EXPECT_EQ(tracker.allocation().framerate_factor[0], 8); + EXPECT_EQ(tracker.allocation().framerate_factor[1], 4); + EXPECT_EQ(tracker.allocation().framerate_factor[2], 2); + EXPECT_EQ(tracker.allocation().framerate_factor[3], 1); + EXPECT_NEAR(tracker.allocation().bitrate[0][0].kbps(), 240, 1); + EXPECT_NEAR(tracker.allocation().bitrate[0][1].kbps(), 360, 1); + EXPECT_NEAR(tracker.allocation().bitrate[0][2].kbps(), 480, 1); + EXPECT_NEAR(tracker.allocation().SpatialLayerBitrate(0).kbps(), + kStreamBitrate.kbps(), 1); +} + +TEST(CbrLayerRateTrackerTest, SpatialLayersShareTheTemporalStructure) { + CbrLayerRateTracker tracker; + // Two spatial layers, the upper one with twice the bitrate of the lower one, + // in a dyadic L1T2 pattern. + for (int frame = 0; frame < 16; ++frame) { + const int temporal_id = frame % 2; + const DataRate bitrate = GeometricLayerBitrate(temporal_id, 2); + EncodeTemporalUnit( + tracker, {{0, temporal_id, bitrate / 3, /*is_keyframe=*/frame == 0}, + {1, temporal_id, bitrate * 2 / 3, /*is_keyframe=*/false}}); + } + + EXPECT_EQ(tracker.allocation().num_temporal_layers, 2); + EXPECT_EQ(tracker.allocation().framerate_factor[0], 2); + EXPECT_EQ(tracker.allocation().framerate_factor[1], 1); + EXPECT_NEAR(tracker.allocation().SpatialLayerBitrate(0).kbps(), + kStreamBitrate.kbps() / 3, 5); + EXPECT_NEAR(tracker.allocation().SpatialLayerBitrate(1).kbps(), + 2 * kStreamBitrate.kbps() / 3, 5); +} + +TEST(CbrLayerRateTrackerTest, ConvergesOnANonGeometricDistribution) { + CbrLayerRateTracker tracker; + // A distribution where the per frame bit budget decreases linearly with the + // temporal id instead of geometrically, so the layers above the base one do + // not end up with equal shares and the priming does not predict it. With the + // frames of an L1T3 pattern distributed 1:1:2 over the layers and per frame + // budgets of 3:2:1, the shares come out as 3:2:2. + const std::array<DataRate, 3> kLayerBitrates = { + kStreamBitrate * 3 / 7, + kStreamBitrate * 2 / 7, + kStreamBitrate * 2 / 7, + }; + constexpr std::array<int, 4> kPattern = {0, 2, 1, 2}; + + EncodeFrame(tracker, 0, 0, kLayerBitrates[0], /*is_keyframe=*/true); + for (int frame = 1; frame < 40; ++frame) { + const int temporal_id = kPattern[frame % 4]; + EncodeFrame(tracker, 0, temporal_id, kLayerBitrates[temporal_id], + /*is_keyframe=*/false); + } + + EXPECT_EQ(tracker.allocation().framerate_factor[0], 4); + EXPECT_EQ(tracker.allocation().framerate_factor[1], 2); + EXPECT_EQ(tracker.allocation().framerate_factor[2], 1); + EXPECT_NEAR(tracker.allocation().bitrate[0][0].kbps(), + kStreamBitrate.kbps() * 3 / 7, 10); + EXPECT_NEAR(tracker.allocation().bitrate[0][1].kbps(), + kStreamBitrate.kbps() * 5 / 7, 10); + EXPECT_NEAR(tracker.allocation().SpatialLayerBitrate(0).kbps(), + kStreamBitrate.kbps(), 10); +} + +TEST(CbrLayerRateTrackerTest, FollowsAChangeOfTheStreamBitrate) { + CbrLayerRateTracker tracker; + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/2); + + // The stream bitrate is halved, keeping the same distribution over the + // temporal layers. Redistributing the same shape over a new bitrate is + // picked up as soon as every layer has been seen once. + for (int frame = 0; frame < 4; ++frame) { + const int temporal_id = frame % 2; + EncodeFrame(tracker, 0, temporal_id, + GeometricLayerBitrate(temporal_id, 2) / 2, + /*is_keyframe=*/false); + } + + EXPECT_NEAR(tracker.allocation().SpatialLayerBitrate(0).kbps(), + kStreamBitrate.kbps() / 2, 5); + EXPECT_NEAR(tracker.allocation().bitrate[0][0].kbps(), 200, 5); +} + +TEST(CbrLayerRateTrackerTest, SingleLayerFollowsTheBitrateImmediately) { + CbrLayerRateTracker tracker; + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/0, kStreamBitrate, + /*is_keyframe=*/true); + + // With a single temporal layer every frame states the bitrate of the whole + // stream, so there is nothing to average over and no reason to lag behind. + const DataRate kLowerBitrate = kStreamBitrate / 5; + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/0, kLowerBitrate, + /*is_keyframe=*/false); + + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), kLowerBitrate); +} + +TEST(CbrLayerRateTrackerTest, ConvergesOnAShiftedKeyPattern) { + // In L2T2_KEY_SHIFT the spatial layers run antiphase: within a temporal unit + // they are on different temporal layers, and the base layer of the lower + // spatial layer holds two frames in a row right after the keyframe. The + // frame after the keyframe is therefore not on the topmost layer and the + // priming does not kick in, leaving the cadence to be learned. + // + // t=0: S0T0, S1T0 t=1: S0T0, S1T1 t=2: S0T1, S1T0 ... + CbrLayerRateTracker tracker; + const DataRate kLowerBitrate = kStreamBitrate / 3; + const DataRate kUpperBitrate = kStreamBitrate * 2 / 3; + EncodeTemporalUnit(tracker, + {{0, 0, kLowerBitrate * 2 / 3, /*is_keyframe=*/true}, + {1, 0, kUpperBitrate * 2 / 3, /*is_keyframe=*/false}}); + + for (int temporal_unit = 1; temporal_unit <= 16; ++temporal_unit) { + const int lower_temporal_id = temporal_unit % 2 == 1 ? 0 : 1; + const int upper_temporal_id = 1 - lower_temporal_id; + EncodeTemporalUnit( + tracker, {{0, lower_temporal_id, + kLowerBitrate * (lower_temporal_id == 0 ? 2.0 / 3 : 1.0 / 3), + /*is_keyframe=*/false}, + {1, upper_temporal_id, + kUpperBitrate * (upper_temporal_id == 0 ? 2.0 / 3 : 1.0 / 3), + /*is_keyframe=*/false}}); + + // Both layers have been seen twice after four temporal units, which is all + // it takes to measure how often their frames occur. + if (temporal_unit >= 4) { + EXPECT_EQ(tracker.allocation().framerate_factor[0], 2) + << " at temporal unit " << temporal_unit; + } + } + + EXPECT_EQ(tracker.allocation().num_temporal_layers, 2); + EXPECT_EQ(tracker.allocation().framerate_factor[1], 1); + EXPECT_NEAR(tracker.allocation().SpatialLayerBitrate(0).kbps(), + kLowerBitrate.kbps(), 5); + EXPECT_NEAR(tracker.allocation().SpatialLayerBitrate(1).kbps(), + kUpperBitrate.kbps(), 5); +} + +TEST(CbrLayerRateTrackerTest, ConvergesOnANonDyadicCadence) { + // A pattern where only every third frame belongs to the base layer. The + // priming assumes a dyadic pattern, in which the base layer holds every + // other frame, so the cadence has to be corrected from there. + CbrLayerRateTracker tracker; + // The base layer frames are twice the size of the ones above them but only + // half as many, so the two layers end up with the same bitrate. + const DataRate kLayerBitrate = kStreamBitrate / 2; + + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/0, kLayerBitrate, + /*is_keyframe=*/true); + for (int frame = 1; frame <= 40; ++frame) { + EncodeFrame(tracker, /*spatial_id=*/0, + /*temporal_id=*/frame % 3 == 0 ? 0 : 1, kLayerBitrate, + /*is_keyframe=*/false); + + // Measured: the cadence estimate settles on the correct value by the + // twentieth frame, less than a second of video, and stays there. The + // frames of the upper layer come in pairs, so the interval between them + // alternates between one and two and the estimate has to average the two + // out before it can be trusted to the nearest integer. + if (frame >= 20) { + EXPECT_EQ(tracker.allocation().framerate_factor[0], 3) + << " at frame " << frame; + } + } + + EXPECT_EQ(tracker.allocation().num_temporal_layers, 2); + EXPECT_EQ(tracker.allocation().framerate_factor[1], 1); + EXPECT_NEAR(tracker.allocation().bitrate[0][0].kbps(), kLayerBitrate.kbps(), + 5); + EXPECT_NEAR(tracker.allocation().SpatialLayerBitrate(0).kbps(), + kStreamBitrate.kbps(), 5); +} + +// The layer the caller happens to state a new allocation on first must not +// matter: the change is carried over to the layers that have not reported yet, +// so the cumulative bitrate is right for the very next frame either way. The +// three tests below place the change on each of the layers in turn. +TEST(CbrLayerRateTrackerTest, CarriesAChangeSeenOnTheBaseLayerOver) { + CbrLayerRateTracker tracker; + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); + // Run the pattern until every layer has stated a bitrate of its own. The + // frame that follows belongs to the base layer. + EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/6); + ASSERT_EQ(L1T3TemporalId(8), 0); + + EncodeL1T3Frames(tracker, /*first_frame=*/8, /*num_frames=*/1, /*scale=*/0.5); + + EXPECT_EQ(tracker.allocation().bitrate[0][0], + GeometricLayerBitrate(0, 3) / 2); + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), kStreamBitrate / 2); +} + +TEST(CbrLayerRateTrackerTest, CarriesAChangeSeenOnAMiddleLayerOver) { + CbrLayerRateTracker tracker; + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); + EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/4); + ASSERT_EQ(L1T3TemporalId(6), 1); + + EncodeL1T3Frames(tracker, /*first_frame=*/6, /*num_frames=*/1, /*scale=*/0.5); + + EXPECT_EQ(tracker.allocation().bitrate[0][0], + GeometricLayerBitrate(0, 3) / 2); + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), kStreamBitrate / 2); +} + +TEST(CbrLayerRateTrackerTest, CarriesAChangeSeenOnTheTopLayerOver) { + CbrLayerRateTracker tracker; + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); + EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/5); + ASSERT_EQ(L1T3TemporalId(7), 2); + + EncodeL1T3Frames(tracker, /*first_frame=*/7, /*num_frames=*/1, /*scale=*/0.5); + + EXPECT_EQ(tracker.allocation().bitrate[0][0], + GeometricLayerBitrate(0, 3) / 2); + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), kStreamBitrate / 2); +} + +TEST(CbrLayerRateTrackerTest, DoesNotCountACarriedOverChangeTwice) { + CbrLayerRateTracker tracker; + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); + EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/4); + + // The allocation is halved, and halved again before the layers that were + // only assumed to follow along have stated anything themselves. Each of them + // then restates what was already assumed, which must leave the estimates + // where they are. + EncodeL1T3Frames(tracker, /*first_frame=*/6, /*num_frames=*/1, /*scale=*/0.5); + EncodeL1T3Frames(tracker, /*first_frame=*/7, /*num_frames=*/1, + /*scale=*/0.25); + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), kStreamBitrate / 4); + + EncodeL1T3Frames(tracker, /*first_frame=*/8, /*num_frames=*/4, + /*scale=*/0.25); + EXPECT_EQ(tracker.allocation().bitrate[0][0], + GeometricLayerBitrate(0, 3) / 4); + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), kStreamBitrate / 4); +} + +TEST(CbrLayerRateTrackerTest, SettlesOnANewDistribution) { + CbrLayerRateTracker tracker; + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); + EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/6); + + // The stream bitrate is kept, but distributed 3:2:2 rather than 2:1:1. The + // first layer to state its new share is taken to speak for the others, which + // is wrong here, so a layer that was carried along is only put right the + // next time it states a bitrate of its own. + const std::array<DataRate, 3> kNewBitrates = { + kStreamBitrate * 3 / 7, + kStreamBitrate * 2 / 7, + kStreamBitrate * 2 / 7, + }; + // Splitting the stream bitrate in sevenths does not come out even, so the + // layers are what the total is held against. + const DataRate kNewStreamBitrate = + kNewBitrates[0] + kNewBitrates[1] + kNewBitrates[2]; + for (int frame = 8; frame < 16; ++frame) { + const int temporal_id = L1T3TemporalId(frame); + EncodeFrame(tracker, 0, temporal_id, kNewBitrates[temporal_id], + /*is_keyframe=*/false); + + // The base layer, the last to be heard from a second time, settles on the + // frame that follows a full pattern. + if (frame >= 12) { + EXPECT_EQ(tracker.allocation().bitrate[0][0], kNewBitrates[0]) + << " at frame " << frame; + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), kNewStreamBitrate) + << " at frame " << frame; + } + } + + EXPECT_EQ(tracker.allocation().bitrate[0][1], + kNewBitrates[0] + kNewBitrates[1]); +} + +TEST(CbrLayerRateTrackerTest, DoesNotReadARepeatedBitrateAsAChange) { + CbrLayerRateTracker tracker; + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); + EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/6); + + // Only the base layer is given more, which the tracker cannot tell from the + // start of a change of the whole allocation. The layers above it then keep + // restating what they had, and a restatement says nothing, so the estimates + // must settle rather than swing back and forth between the two readings. + const DataRate kNewBaseBitrate = GeometricLayerBitrate(0, 3) * 1.5; + const DataRate kNewStreamBitrate = kNewBaseBitrate + + GeometricLayerBitrate(1, 3) + + GeometricLayerBitrate(2, 3); + for (int frame = 8; frame < 40; ++frame) { + const int temporal_id = L1T3TemporalId(frame); + EncodeFrame(tracker, 0, temporal_id, + temporal_id == 0 ? kNewBaseBitrate + : GeometricLayerBitrate(temporal_id, 3), + /*is_keyframe=*/false); + + // One pattern is enough for every layer to have been heard from. + if (frame >= 12) { + EXPECT_EQ(tracker.allocation().bitrate[0][0], kNewBaseBitrate) + << " at frame " << frame; + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), kNewStreamBitrate) + << " at frame " << frame; + } + } +} + +TEST(CbrLayerRateTrackerTest, DoesNotReadADepartureFromAGuessAsAChange) { + CbrLayerRateTracker tracker; + // The priming guesses what the layers above the base one hold. The first + // frame of such a layer states the real value, which is not a change of the + // allocation however far off the guess was, so the other layers must be left + // alone. + EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); + const DataRate kMiddleLayerBitrate = GeometricLayerBitrate(1, 3) * 4; + + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/1, kMiddleLayerBitrate, + /*is_keyframe=*/false); + + EXPECT_EQ(tracker.allocation().bitrate[0][0], GeometricLayerBitrate(0, 3)); + EXPECT_EQ(tracker.allocation().bitrate[0][1], + GeometricLayerBitrate(0, 3) + kMiddleLayerBitrate); +} + +TEST(CbrLayerRateTrackerTest, KeepsTheSpatialLayersApart) { + // Two spatial layers in an L1T2 pattern, the lower one at 300 kbps and the + // upper one at 600 kbps, both split 2:1 between their temporal layers. + constexpr DataRate kLowerStreamBitrate = DataRate::KilobitsPerSec(300); + constexpr DataRate kUpperStreamBitrate = DataRate::KilobitsPerSec(600); + CbrLayerRateTracker tracker; + for (int frame = 0; frame < 4; ++frame) { + const double share = frame % 2 == 0 ? 2.0 / 3 : 1.0 / 3; + EncodeTemporalUnit(tracker, {{0, frame % 2, kLowerStreamBitrate * share, + /*is_keyframe=*/frame == 0}, + {1, frame % 2, kUpperStreamBitrate * share, + /*is_keyframe=*/false}}); + } + + // Halving the lower spatial layer says nothing about the upper one, which + // is left out of this temporal unit. + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/0, + kLowerStreamBitrate * (2.0 / 3) / 2, /*is_keyframe=*/false); + + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), + kLowerStreamBitrate / 2); + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(1), kUpperStreamBitrate); +} + +// The two simulations below drive the tracker the way a congestion controller +// would drive a stream with temporal layers, which the encoder level tests do +// not cover: those run a single temporal layer, where the bitrate of the layer +// and of the stream are the same thing. +TEST(CbrLayerRateTrackerTest, FollowsAStagedBitrateSweep) { + // The bitrate profile of the encoder level ChangingBitrateTargetVga test, + // over an L1T3 pattern at 30 fps. + constexpr int kNumLayers = 3; + CbrLayerRateTracker tracker; + int frame = 0; + + auto encode = [&](DataRate stream_bitrate, int num_frames) { + for (int i = 0; i < num_frames; ++i, ++frame) { + const int temporal_id = L1T3TemporalId(frame); + EncodeFrame(tracker, /*spatial_id=*/0, temporal_id, + LayerShare(stream_bitrate, temporal_id, kNumLayers), + /*is_keyframe=*/frame == 0); + + // Every step of the sweep scales the whole allocation, so the frame at + // hand states everything there is to know about the change and the + // bitrate it is encoded against is right away. + EXPECT_EQ(tracker.allocation().bitrate[0][temporal_id], + CumulativeShare(stream_bitrate, temporal_id, kNumLayers)) + << " at frame " << frame << " of T" << temporal_id; + } + }; + + encode(DataRate::KilobitsPerSec(500), 60); + encode(DataRate::KilobitsPerSec(100), 30); + for (int kbps = 150; kbps < 500; kbps += 50) { + encode(DataRate::KilobitsPerSec(kbps), 6); + } + encode(DataRate::KilobitsPerSec(500), 30); + + EXPECT_EQ(tracker.allocation().SpatialLayerBitrate(0), + DataRate::KilobitsPerSec(500)); +} + +TEST(CbrLayerRateTrackerTest, RecoversOnceTheTargetHoldsStill) { + // A target that moves on every single frame, which is more than a congestion + // controller would ask for, denies the tracker the one thing that tells a + // change of the whole allocation from a change of the distribution: a layer + // restating the bitrate it already had. Whatever the stream does while a + // layer waits for its first turn is then never accounted for, and the skew + // that leaves behind, measured at up to 10% of the cumulative bitrate, + // stays until the target settles. + constexpr int kNumLayers = 4; + CbrLayerRateTracker tracker; + auto stream_bitrate_at = [](int frame) { + // A sawtooth between 300 and 900 kbps in steps of 30, which divides evenly + // into fifths and keeps the layer bitrates whole. + return DataRate::KilobitsPerSec(300 + 30 * (frame % 21)); + }; + + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/0, + LayerShare(stream_bitrate_at(0), 0, kNumLayers), + /*is_keyframe=*/true); + EncodeFrame(tracker, /*spatial_id=*/0, /*temporal_id=*/kNumLayers - 1, + LayerShare(stream_bitrate_at(1), kNumLayers - 1, kNumLayers), + /*is_keyframe=*/false); + for (int frame = 2; frame < 40; ++frame) { + const int temporal_id = L1T4TemporalId(frame); + EncodeFrame(tracker, + /*spatial_id=*/0, temporal_id, + LayerShare(stream_bitrate_at(frame), temporal_id, kNumLayers), + /*is_keyframe=*/false); + } + + // A layer restating what it holds is what puts the estimates right, so they + // are all in order again once the pattern has come around. + constexpr DataRate kHeldBitrate = DataRate::KilobitsPerSec(600); + for (int frame = 40; frame < 80; ++frame) { + const int temporal_id = L1T4TemporalId(frame); + EncodeFrame(tracker, /*spatial_id=*/0, temporal_id, + LayerShare(kHeldBitrate, temporal_id, kNumLayers), + /*is_keyframe=*/false); + + if (frame >= 40 + 1 * static_cast<int>(kL1T4Pattern.size())) { + EXPECT_EQ(tracker.allocation().bitrate[0][temporal_id], + CumulativeShare(kHeldBitrate, temporal_id, kNumLayers)) + << " at frame " << frame << " of T" << temporal_id; + } + } +} + +TEST(CbrLayerRateTrackerTest, PrimesAllSpatialLayersOnFirstUnitAfterKeyframe) { + // L2T3: the temporal unit after the keyframe is on the topmost temporal + // layer in both spatial layers, so both are primed. + CbrLayerRateTracker tracker; + EncodeTemporalUnit(tracker, {{0, 0, GeometricLayerBitrate(0, 3) / 3, + /*is_keyframe=*/true}, + {1, 0, GeometricLayerBitrate(0, 3) * 2 / 3, + /*is_keyframe=*/false}}); + EncodeTemporalUnit(tracker, {{0, 2, GeometricLayerBitrate(2, 3) / 3, + /*is_keyframe=*/false}, + {1, 2, GeometricLayerBitrate(2, 3) * 2 / 3, + /*is_keyframe=*/false}}); + + const CumulativeCbrAllocation& allocation = tracker.allocation(); + EXPECT_EQ(allocation.num_temporal_layers, 3); + EXPECT_EQ(allocation.framerate_factor[0], 4); + EXPECT_EQ(allocation.framerate_factor[1], 2); + EXPECT_EQ(allocation.framerate_factor[2], 1); + EXPECT_NEAR(allocation.bitrate[0][0].kbps(), 100, 1); + EXPECT_NEAR(allocation.bitrate[0][1].kbps(), 150, 1); + EXPECT_NEAR(allocation.SpatialLayerBitrate(0).kbps(), 200, 1); + EXPECT_NEAR(allocation.bitrate[1][0].kbps(), 200, 1); + EXPECT_NEAR(allocation.bitrate[1][1].kbps(), 300, 1); + EXPECT_NEAR(allocation.SpatialLayerBitrate(1).kbps(), 400, 1); + EXPECT_NEAR(allocation.TotalBitrate().kbps(), kStreamBitrate.kbps(), 2); +} + +TEST(CbrLayerRateTrackerTest, AbsentSpatialLayersKeepTheirBitrate) { + // Three spatial layers where a spatial layer being present means that all + // layers above it are present too, and the lowest present one follows the + // dyadic cadence {0, 2, 1, 2}. All frames are on the base temporal layer, so + // the spatial layers simply run at different frame rates. + constexpr std::array<int, 4> kLowestLayer = {0, 2, 1, 2}; + const std::array<DataRate, 3> kLayerBitrates = { + DataRate::KilobitsPerSec(50), DataRate::KilobitsPerSec(100), + DataRate::KilobitsPerSec(150)}; + CbrLayerRateTracker tracker; + + std::optional<CumulativeCbrAllocation> previous; + for (int temporal_unit = 0; temporal_unit < 16; ++temporal_unit) { + std::vector<Frame> frames; + for (int sid = kLowestLayer[temporal_unit % 4]; sid < 3; ++sid) { + frames.push_back({sid, 0, kLayerBitrates[sid], + /*is_keyframe=*/temporal_unit == 0 && sid == 0}); + } + EncodeTemporalUnit(tracker, frames); + + // Once every layer has been heard from, the allocation must not move + // whichever layers the temporal unit holds, or encoders that reconfigure + // on every change would do so for nothing. + if (temporal_unit >= 1) { + ASSERT_TRUE(previous.has_value()); + EXPECT_EQ(tracker.allocation(), *previous) + << " at temporal unit " << temporal_unit; + } + previous = tracker.allocation(); + } + + const CumulativeCbrAllocation& allocation = tracker.allocation(); + EXPECT_EQ(allocation.num_temporal_layers, 1); + for (int sid = 0; sid < 3; ++sid) { + EXPECT_EQ(allocation.SpatialLayerBitrate(sid), kLayerBitrates[sid]); + for (int tid = 0; tid < CumulativeCbrAllocation::kMaxTemporalLayers; + ++tid) { + EXPECT_EQ(allocation.bitrate[sid][tid], kLayerBitrates[sid]); + } + } + EXPECT_EQ(allocation.TotalBitrate(), DataRate::KilobitsPerSec(300)); +} + +TEST(CbrLayerRateTrackerTest, IgnoresFramesWithoutCbr) { + CbrLayerRateTracker tracker; + std::vector<VideoEncoderInterface::FrameEncodeSettings> settings; + settings.push_back(FrameEncodeSettingsBuilder() + .SpatialId(0) + .TemporalId(0) + .CqpRateOptions(/*target_qp=*/30) + .FrameType(FrameType::kKeyframe) + .Build()); + tracker.OnTemporalUnit(settings); + + EXPECT_EQ(tracker.allocation(), CumulativeCbrAllocation()); +} + +} // namespace +} // namespace webrtc
diff --git a/modules/video_coding/utility/temporal_layer_rate_tracker.cc b/modules/video_coding/utility/temporal_layer_rate_tracker.cc deleted file mode 100644 index 4316f11..0000000 --- a/modules/video_coding/utility/temporal_layer_rate_tracker.cc +++ /dev/null
@@ -1,197 +0,0 @@ -/* - * Copyright (c) 2026 The WebRTC project authors. All Rights Reserved. - * - * Use of this source code is governed by a BSD-style license - * that can be found in the LICENSE file in the root of the source - * tree. An additional intellectual property rights grant can be found - * in the file PATENTS. All contributing project authors may - * be found in the AUTHORS file in the root of the source tree. - */ - -#include "modules/video_coding/utility/temporal_layer_rate_tracker.h" - -#include <algorithm> -#include <cmath> - -#include "api/units/data_rate.h" -#include "rtc_base/checks.h" -#include "rtc_base/numerics/exp_filter.h" - -namespace webrtc { -namespace { - -// How often the frames of a layer occur is a property of the temporal -// structure, which either stays the same forever or changes wholesale. The -// samples are constant as long as the structure is, so there is no ripple to -// suppress and the filter only has to average out the jitter of structures -// whose layers do not occur at a fixed interval. A slow filter is what lets a -// single out of phase interval, as seen after a keyframe or a dropped frame, -// pass without disturbing the estimate. -constexpr float kFrameIntervalAlpha = 0.9f; - -// The number of frames between consecutive frames of temporal layer `tid` in a -// dyadic L1Tx pattern with `num_layers` temporal layers. The topmost layer -// holds every other frame, the one below it every fourth, and so on, with the -// base layer matching the layer just above it. -double DyadicFrameInterval(int tid, int num_layers) { - if (num_layers <= 1) { - return 1.0; - } - return 1 << (num_layers - std::max(tid, 1)); -} - -float Filtered(const ExpFilter& filter) { - const float value = filter.filtered(); - return value == ExpFilter::kValueUndefined ? 0.0f : value; -} - -} // namespace - -TemporalLayerRateTracker::TemporalLayerRateTracker() - : frame_interval_filters_(kMaxTemporalLayers, - ExpFilter(kFrameIntervalAlpha)) { - // Until a frame of a higher layer shows up the stream is assumed to consist - // of a single temporal layer holding all of the frames. - frame_interval_filters_[0].Apply(1.0f, 1.0f); -} - -void TemporalLayerRateTracker::PrimeStandardPattern(int num_layers, - int spatial_id, - DataRate delta_bitrate) { - RTC_DCHECK_GT(num_layers, 1); - RTC_DCHECK_LE(num_layers, kMaxTemporalLayers); - num_temporal_layers_ = std::max(num_temporal_layers_, num_layers); - - for (int tid = 0; tid < kMaxTemporalLayers; ++tid) { - frame_interval_filters_[tid].Reset(kFrameIntervalAlpha); - if (tid < num_layers) { - frame_interval_filters_[tid].Apply( - 1.0f, static_cast<float>(DyadicFrameInterval(tid, num_layers))); - } - } - - // Only the bitrates of the base layer, taken from the keyframe, and of the - // topmost layer, taken from the frame at hand, are known. In the recommended - // distribution, where the per frame bit budget halves for every step up the - // temporal layer stack, every layer above the base one ends up with the same - // share of the bitrate: the frames of a layer are twice as many but half as - // large as those of the layer below it. Assume that is the case here, which - // leaves the base layer as observed on the keyframe. The guesses are not - // recorded as stated bitrates, so the first frame of a layer that turns out - // to hold something else is not mistaken for a change of the allocation. - for (int tid = 1; tid < num_layers; ++tid) { - layer_rates_[spatial_id][tid].estimate = delta_bitrate; - } -} - -void TemporalLayerRateTracker::UpdateLayerRates(int spatial_id, - int temporal_id, - DataRate layer_bitrate) { - SpatialLayerRates& layers = layer_rates_[spatial_id]; - LayerRate& layer = layers[temporal_id]; - - // A caller that changes the allocation normally scales the whole stream, so - // a layer that moves is taken to speak for the layers that have not reported - // since. What is carried over is only the part of the change the tracker did - // not already assume, and a layer that restates the bitrate it already had - // says nothing at all, which together keep a change of the distribution from - // being handed back and forth between the layers. - if (layer.stated.has_value() && *layer.stated != layer_bitrate && - layer.estimate > DataRate::Zero()) { - const double change = layer_bitrate / layer.estimate; - for (int tid = 0; tid < kMaxTemporalLayers; ++tid) { - if (tid != temporal_id) { - layers[tid].estimate = layers[tid].estimate * change; - } - } - } - - layer.estimate = layer_bitrate; - layer.stated = layer_bitrate; -} - -void TemporalLayerRateTracker::Update(int spatial_id, - int temporal_id, - DataRate layer_bitrate, - bool is_keyframe) { - RTC_DCHECK_GE(spatial_id, 0); - RTC_DCHECK_LT(spatial_id, kMaxSpatialLayers); - RTC_DCHECK_GE(temporal_id, 0); - RTC_DCHECK_LT(temporal_id, kMaxTemporalLayers); - - if (is_keyframe) { - // A keyframe restarts the temporal structure, and since it belongs to the - // base layer the frame that follows it reveals the layer count. - last_frame_was_keyframe_ = true; - } else if (last_frame_was_keyframe_) { - last_frame_was_keyframe_ = false; - if (temporal_id > 0) { - PrimeStandardPattern(temporal_id + 1, spatial_id, layer_bitrate); - } - } - - UpdateLayerRates(spatial_id, temporal_id, layer_bitrate); - num_temporal_layers_ = std::max(num_temporal_layers_, temporal_id + 1); - - // All spatial layers of a temporal unit are assumed to run the same temporal - // pattern, so only the first of them advances the cadence. A spatial id that - // does not exceed the previous one means a new temporal unit started. - if (!last_updated_spatial_id_.has_value() || - spatial_id <= *last_updated_spatial_id_) { - ++temporal_unit_count_; - if (last_unit_of_layer_[temporal_id].has_value()) { - frame_interval_filters_[temporal_id].Apply( - 1.0f, temporal_unit_count_ - *last_unit_of_layer_[temporal_id]); - } - last_unit_of_layer_[temporal_id] = temporal_unit_count_; - } - last_updated_spatial_id_ = spatial_id; -} - -double TemporalLayerRateTracker::FrameFraction(int temporal_id) const { - const double interval = Filtered(frame_interval_filters_[temporal_id]); - return interval > 0.0 ? 1.0 / interval : 0.0; -} - -int TemporalLayerRateTracker::FramerateFactor(int temporal_id) const { - RTC_DCHECK_GE(temporal_id, 0); - if (temporal_id >= num_temporal_layers_ - 1) { - return 1; - } - - double total_fraction = 0.0; - double cumulative_fraction = 0.0; - for (int tid = 0; tid < num_temporal_layers_; ++tid) { - const double fraction = FrameFraction(tid); - total_fraction += fraction; - if (tid <= temporal_id) { - cumulative_fraction += fraction; - } - } - - if (cumulative_fraction <= 0.0) { - return 1; - } - return std::max( - 1, static_cast<int>(std::round(total_fraction / cumulative_fraction))); -} - -DataRate TemporalLayerRateTracker::CumulativeBitrate(int spatial_id, - int temporal_id) const { - RTC_DCHECK_GE(spatial_id, 0); - RTC_DCHECK_LT(spatial_id, kMaxSpatialLayers); - RTC_DCHECK_GE(temporal_id, 0); - - DataRate bitrate = DataRate::Zero(); - for (int tid = 0; tid <= std::min(temporal_id, num_temporal_layers_ - 1); - ++tid) { - bitrate += layer_rates_[spatial_id][tid].estimate; - } - return bitrate; -} - -DataRate TemporalLayerRateTracker::StreamBitrate(int spatial_id) const { - return CumulativeBitrate(spatial_id, num_temporal_layers_ - 1); -} - -} // namespace webrtc
diff --git a/modules/video_coding/utility/temporal_layer_rate_tracker.h b/modules/video_coding/utility/temporal_layer_rate_tracker.h deleted file mode 100644 index 9d029a3..0000000 --- a/modules/video_coding/utility/temporal_layer_rate_tracker.h +++ /dev/null
@@ -1,153 +0,0 @@ -/* - * Copyright (c) 2026 The WebRTC project authors. All Rights Reserved. - * - * Use of this source code is governed by a BSD-style license - * that can be found in the LICENSE file in the root of the source - * tree. An additional intellectual property rights grant can be found - * in the file PATENTS. All contributing project authors may - * be found in the AUTHORS file in the root of the source tree. - */ - -#ifndef MODULES_VIDEO_CODING_UTILITY_TEMPORAL_LAYER_RATE_TRACKER_H_ -#define MODULES_VIDEO_CODING_UTILITY_TEMPORAL_LAYER_RATE_TRACKER_H_ - -#include <array> -#include <optional> - -#include "absl/container/inlined_vector.h" -#include "api/units/data_rate.h" -#include "rtc_base/numerics/exp_filter.h" - -namespace webrtc { - -// Aggregates rate control decisions that are made per frame into the per -// temporal layer quantities encoder libraries want. -// -// Callers of `VideoEncoderInterface` state the bitrate of the temporal layer a -// frame belongs to, one frame at a time. Encoder libraries on the other hand -// tend to want the bitrate and the frame rate of the stream formed by all the -// layers up to and including a given one, which this class derives by tracking -// the bitrate stated for each layer together with how often frames of that -// layer occur. -// -// Only one layer is heard from per frame, so the bitrate of the others has to -// be assumed until they report. A caller that changes the allocation normally -// scales all of the layers by the same factor, so a change seen on one layer is -// taken to apply to the layers that have not reported since. The assumption is -// dropped as soon as a layer states a bitrate of its own, and a layer repeating -// the bitrate it already had is not taken as evidence of anything, which is -// what keeps a change of the distribution between the layers from ping ponging -// around rather than settling. A layer that stops being used keeps the bitrate -// it last had, which is only felt if it is taken back into use while a layer -// below it has gone stale as well. -// -// How often the frames of a layer occur cannot be stated by the caller and is -// measured instead, and smoothed. To avoid a transient at the start of a -// stream, the state is primed on the first frame after a keyframe under the -// assumption that a standard dyadic L1Tx pattern is used: in such a pattern -// that frame belongs to the topmost temporal layer, which reveals the layer -// count and thereby how often the frames of every layer occur. Structures the -// priming does not predict are learned instead, which takes a few repetitions -// of the pattern. -// -// An instance only ever describes one configuration of one encoder. Replace it -// rather than trying to reuse it when the encoder is reconfigured. -class TemporalLayerRateTracker { - public: - // Rate control is tracked separately per spatial layer, since the layers - // have separate bitrates. The spatial layers are assumed to run the same - // temporal pattern, though not necessarily in phase: the cadence is measured - // from the first spatial layer of each temporal unit, so a pattern that - // shifts the layers relative to each other, such as L2T2_KEY_SHIFT, is - // described correctly as well. - static constexpr int kMaxSpatialLayers = 4; - static constexpr int kMaxTemporalLayers = 4; - - TemporalLayerRateTracker(); - - // Records that a frame of temporal layer `temporal_id` in spatial layer - // `spatial_id` was encoded, and that the layer it belongs to was allocated - // `layer_bitrate`. Frames must be passed in encode order. - void Update(int spatial_id, - int temporal_id, - DataRate layer_bitrate, - bool is_keyframe); - - // The number of temporal layers seen so far, or the number the priming - // guessed. At least one. - int num_temporal_layers() const { return num_temporal_layers_; } - - // How many times lower the frame rate of the stream made up of the temporal - // layers up to and including `temporal_id` is compared to the frame rate of - // the full stream. Rounded to an integer, which is exact for the dyadic - // temporal structures used in practice. At least one. - int FramerateFactor(int temporal_id) const; - - // The bitrate of the stream made up of the temporal layers up to and - // including `temporal_id` within spatial layer `spatial_id`. - DataRate CumulativeBitrate(int spatial_id, int temporal_id) const; - - // The bitrate of all temporal layers of spatial layer `spatial_id` combined. - DataRate StreamBitrate(int spatial_id) const; - - private: - // What is known about the bitrate of one temporal layer of one spatial - // layer. - struct LayerRate { - // The bitrate the tracker believes the layer has: what the caller stated - // for it scaled by the changes seen on the other layers since, or what the - // priming guessed. - DataRate estimate = DataRate::Zero(); - // The bitrate the caller most recently stated for the layer, if any. Only - // a departure from this counts as a change of the allocation. - std::optional<DataRate> stated; - }; - using SpatialLayerRates = std::array<LayerRate, kMaxTemporalLayers>; - - // `ExpFilter` is not assignable, so the filters are held in a container that - // can be filled with copies of a prototype on construction. - using TemporalLayerFilters = - absl::InlinedVector<ExpFilter, kMaxTemporalLayers>; - - // Records that `layer_bitrate` was allocated to temporal layer `temporal_id` - // of spatial layer `spatial_id`, and carries the change over to the layers - // that have not reported since. - void UpdateLayerRates(int spatial_id, - int temporal_id, - DataRate layer_bitrate); - - // Primes the state as if a standard dyadic L1Tx pattern with `num_layers` - // temporal layers was in use, where the just observed topmost layer frame of - // spatial layer `spatial_id` reported `delta_bitrate`. - void PrimeStandardPattern(int num_layers, - int spatial_id, - DataRate delta_bitrate); - - // The share of the frames of the stream that belong to `temporal_id`, or - // zero if no two frames of that layer have been seen yet. - double FrameFraction(int temporal_id) const; - - int num_temporal_layers_ = 1; - bool last_frame_was_keyframe_ = false; - // The spatial id of the previous frame, used to tell the frames of a new - // temporal unit from the remaining spatial layers of the current one. - std::optional<int> last_updated_spatial_id_; - // The number of temporal units seen so far, which doubles as the index of - // the current one, and the index of the temporal unit each temporal layer - // was last seen in. - int temporal_unit_count_ = 0; - std::array<std::optional<int>, kMaxTemporalLayers> last_unit_of_layer_; - - // What each temporal layer of each spatial layer is believed to hold. - std::array<SpatialLayerRates, kMaxSpatialLayers> layer_rates_; - - // The number of temporal units between consecutive frames of each temporal - // layer. Filtering the interval rather than the share of the frames each - // layer holds directly means the samples are constant for a fixed temporal - // structure, so the filter can be quick without the estimate rippling. - TemporalLayerFilters frame_interval_filters_; -}; - -} // namespace webrtc - -#endif // MODULES_VIDEO_CODING_UTILITY_TEMPORAL_LAYER_RATE_TRACKER_H_
diff --git a/modules/video_coding/utility/temporal_layer_rate_tracker_unittest.cc b/modules/video_coding/utility/temporal_layer_rate_tracker_unittest.cc deleted file mode 100644 index f97d597..0000000 --- a/modules/video_coding/utility/temporal_layer_rate_tracker_unittest.cc +++ /dev/null
@@ -1,563 +0,0 @@ -/* - * Copyright (c) 2026 The WebRTC project authors. All Rights Reserved. - * - * Use of this source code is governed by a BSD-style license - * that can be found in the LICENSE file in the root of the source - * tree. An additional intellectual property rights grant can be found - * in the file PATENTS. All contributing project authors may - * be found in the AUTHORS file in the root of the source tree. - */ - -#include "modules/video_coding/utility/temporal_layer_rate_tracker.h" - -#include <array> - -#include "api/units/data_rate.h" -#include "test/gtest.h" - -namespace webrtc { -namespace { - -constexpr DataRate kStreamBitrate = DataRate::KilobitsPerSec(600); - -// The bitrate of temporal layer `tid` in a dyadic L1Tx pattern with -// `num_layers` temporal layers, where the per frame bit budget is halved for -// every step up the temporal layer stack and the stream as a whole targets -// `kStreamBitrate`. Matches -// `TemporalLayerPatternForTest::GeometricDistribution` with a ratio of 0.5. -// -// Every layer above the base one ends up with the same share of the bitrate, -// since its frames are twice as many but half as large as those of the layer -// below it. The base layer holds as many frames as the layer just above it, -// each twice the size, so it gets twice the share. -DataRate GeometricLayerBitrate(int tid, int num_layers) { - return kStreamBitrate * ((tid == 0 ? 2.0 : 1.0) / (num_layers + 1)); -} - -// Feeds the tracker a keyframe followed by the first delta frame of a dyadic -// L1Tx pattern, which is all the priming needs. -void EncodeKeyframeAndFirstDeltaFrame(TemporalLayerRateTracker& tracker, - int num_layers) { - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/0, - GeometricLayerBitrate(0, num_layers), /*is_keyframe=*/true); - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/num_layers - 1, - GeometricLayerBitrate(num_layers - 1, num_layers), - /*is_keyframe=*/false); -} - -// A dyadic L1T3 pattern. Its first two frames are the ones -// `EncodeKeyframeAndFirstDeltaFrame` feeds, so a run of the pattern picks up -// at frame two. -constexpr std::array<int, 4> kL1T3Pattern = {0, 2, 1, 2}; - -int L1T3TemporalId(int frame) { - return kL1T3Pattern[frame % kL1T3Pattern.size()]; -} - -// Feeds `num_frames` frames of the L1T3 pattern, starting at `first_frame`, -// with every layer allocated `scale` times its share of `kStreamBitrate`. -void EncodeL1T3Frames(TemporalLayerRateTracker& tracker, - int first_frame, - int num_frames, - double scale = 1.0) { - for (int frame = first_frame; frame < first_frame + num_frames; ++frame) { - const int temporal_id = L1T3TemporalId(frame); - tracker.Update(/*spatial_id=*/0, temporal_id, - GeometricLayerBitrate(temporal_id, 3) * scale, - /*is_keyframe=*/false); - } -} - -// As `kL1T3Pattern`, for four temporal layers. -constexpr std::array<int, 8> kL1T4Pattern = {0, 3, 2, 3, 1, 3, 2, 3}; - -int L1T4TemporalId(int frame) { - return kL1T4Pattern[frame % kL1T4Pattern.size()]; -} - -// The share of `stream_bitrate` that a geometric distribution over -// `num_layers` temporal layers gives to `temporal_id`, and the share it gives -// to all the layers up to and including it. -DataRate LayerShare(DataRate stream_bitrate, int temporal_id, int num_layers) { - return stream_bitrate * ((temporal_id == 0 ? 2.0 : 1.0) / (num_layers + 1)); -} - -DataRate CumulativeShare(DataRate stream_bitrate, - int temporal_id, - int num_layers) { - return stream_bitrate * - (static_cast<double>(temporal_id + 2) / (num_layers + 1)); -} - -TEST(TemporalLayerRateTrackerTest, SingleLayerReportsTheFullBitrate) { - TemporalLayerRateTracker tracker; - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/0, kStreamBitrate, - /*is_keyframe=*/true); - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/0, kStreamBitrate, - /*is_keyframe=*/false); - - EXPECT_EQ(tracker.num_temporal_layers(), 1); - EXPECT_EQ(tracker.FramerateFactor(0), 1); - EXPECT_EQ(tracker.CumulativeBitrate(/*spatial_id=*/0, /*temporal_id=*/0), - kStreamBitrate); - EXPECT_EQ(tracker.StreamBitrate(/*spatial_id=*/0), kStreamBitrate); -} - -TEST(TemporalLayerRateTrackerTest, NothingIsKnownBeforeTheFirstFrame) { - TemporalLayerRateTracker tracker; - - EXPECT_EQ(tracker.num_temporal_layers(), 1); - EXPECT_EQ(tracker.FramerateFactor(0), 1); - EXPECT_EQ(tracker.StreamBitrate(/*spatial_id=*/0), DataRate::Zero()); -} - -TEST(TemporalLayerRateTrackerTest, PrimesTwoLayersOnFirstFrameAfterKeyframe) { - TemporalLayerRateTracker tracker; - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/2); - - EXPECT_EQ(tracker.num_temporal_layers(), 2); - EXPECT_EQ(tracker.FramerateFactor(0), 2); - EXPECT_EQ(tracker.FramerateFactor(1), 1); - // The layers hold half of the frames each, and a base layer frame is twice - // the size of a T1 frame, so the base layer accounts for two thirds of the - // bitrate. - EXPECT_NEAR(tracker.CumulativeBitrate(0, 0).kbps(), 400, 1); - EXPECT_NEAR(tracker.StreamBitrate(0).kbps(), kStreamBitrate.kbps(), 1); -} - -TEST(TemporalLayerRateTrackerTest, PrimesThreeLayersOnFirstFrameAfterKeyframe) { - TemporalLayerRateTracker tracker; - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); - - EXPECT_EQ(tracker.num_temporal_layers(), 3); - EXPECT_EQ(tracker.FramerateFactor(0), 4); - EXPECT_EQ(tracker.FramerateFactor(1), 2); - EXPECT_EQ(tracker.FramerateFactor(2), 1); - EXPECT_NEAR(tracker.CumulativeBitrate(0, 0).kbps(), 300, 1); - EXPECT_NEAR(tracker.CumulativeBitrate(0, 1).kbps(), 450, 1); - EXPECT_NEAR(tracker.StreamBitrate(0).kbps(), kStreamBitrate.kbps(), 1); -} - -TEST(TemporalLayerRateTrackerTest, PrimesFourLayersOnFirstFrameAfterKeyframe) { - TemporalLayerRateTracker tracker; - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/4); - - EXPECT_EQ(tracker.num_temporal_layers(), 4); - EXPECT_EQ(tracker.FramerateFactor(0), 8); - EXPECT_EQ(tracker.FramerateFactor(1), 4); - EXPECT_EQ(tracker.FramerateFactor(2), 2); - EXPECT_EQ(tracker.FramerateFactor(3), 1); - EXPECT_NEAR(tracker.CumulativeBitrate(0, 0).kbps(), 240, 1); - EXPECT_NEAR(tracker.CumulativeBitrate(0, 1).kbps(), 360, 1); - EXPECT_NEAR(tracker.CumulativeBitrate(0, 2).kbps(), 480, 1); - EXPECT_NEAR(tracker.StreamBitrate(0).kbps(), kStreamBitrate.kbps(), 1); -} - -TEST(TemporalLayerRateTrackerTest, SpatialLayersShareTheTemporalStructure) { - TemporalLayerRateTracker tracker; - // Two spatial layers, the upper one with twice the bitrate of the lower one, - // in a dyadic L1T2 pattern. - for (int frame = 0; frame < 16; ++frame) { - const int temporal_id = frame % 2; - const DataRate bitrate = GeometricLayerBitrate(temporal_id, 2); - tracker.Update(/*spatial_id=*/0, temporal_id, bitrate / 3, - /*is_keyframe=*/frame == 0); - tracker.Update(/*spatial_id=*/1, temporal_id, bitrate * 2 / 3, - /*is_keyframe=*/frame == 0); - } - - EXPECT_EQ(tracker.num_temporal_layers(), 2); - EXPECT_EQ(tracker.FramerateFactor(0), 2); - EXPECT_EQ(tracker.FramerateFactor(1), 1); - EXPECT_NEAR(tracker.StreamBitrate(0).kbps(), kStreamBitrate.kbps() / 3, 5); - EXPECT_NEAR(tracker.StreamBitrate(1).kbps(), 2 * kStreamBitrate.kbps() / 3, - 5); -} - -TEST(TemporalLayerRateTrackerTest, ConvergesOnANonGeometricDistribution) { - TemporalLayerRateTracker tracker; - // A distribution where the per frame bit budget decreases linearly with the - // temporal id instead of geometrically, so the layers above the base one do - // not end up with equal shares and the priming does not predict it. With the - // frames of an L1T3 pattern distributed 1:1:2 over the layers and per frame - // budgets of 3:2:1, the shares come out as 3:2:2. - const std::array<DataRate, 3> kLayerBitrates = { - kStreamBitrate * 3 / 7, - kStreamBitrate * 2 / 7, - kStreamBitrate * 2 / 7, - }; - constexpr std::array<int, 4> kPattern = {0, 2, 1, 2}; - - tracker.Update(0, 0, kLayerBitrates[0], /*is_keyframe=*/true); - for (int frame = 1; frame < 40; ++frame) { - const int temporal_id = kPattern[frame % 4]; - tracker.Update(0, temporal_id, kLayerBitrates[temporal_id], - /*is_keyframe=*/false); - } - - EXPECT_EQ(tracker.FramerateFactor(0), 4); - EXPECT_EQ(tracker.FramerateFactor(1), 2); - EXPECT_EQ(tracker.FramerateFactor(2), 1); - EXPECT_NEAR(tracker.CumulativeBitrate(0, 0).kbps(), - kStreamBitrate.kbps() * 3 / 7, 10); - EXPECT_NEAR(tracker.CumulativeBitrate(0, 1).kbps(), - kStreamBitrate.kbps() * 5 / 7, 10); - EXPECT_NEAR(tracker.StreamBitrate(0).kbps(), kStreamBitrate.kbps(), 10); -} - -TEST(TemporalLayerRateTrackerTest, FollowsAChangeOfTheStreamBitrate) { - TemporalLayerRateTracker tracker; - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/2); - - // The stream bitrate is halved, keeping the same distribution over the - // temporal layers. Redistributing the same shape over a new bitrate is - // picked up as soon as every layer has been seen once. - for (int frame = 0; frame < 4; ++frame) { - const int temporal_id = frame % 2; - tracker.Update(0, temporal_id, GeometricLayerBitrate(temporal_id, 2) / 2, - /*is_keyframe=*/false); - } - - EXPECT_NEAR(tracker.StreamBitrate(0).kbps(), kStreamBitrate.kbps() / 2, 5); - EXPECT_NEAR(tracker.CumulativeBitrate(0, 0).kbps(), 200, 5); -} - -TEST(TemporalLayerRateTrackerTest, SingleLayerFollowsTheBitrateImmediately) { - TemporalLayerRateTracker tracker; - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/0, kStreamBitrate, - /*is_keyframe=*/true); - - // With a single temporal layer every frame states the bitrate of the whole - // stream, so there is nothing to average over and no reason to lag behind. - const DataRate kLowerBitrate = kStreamBitrate / 5; - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/0, kLowerBitrate, - /*is_keyframe=*/false); - - EXPECT_EQ(tracker.StreamBitrate(/*spatial_id=*/0), kLowerBitrate); -} - -TEST(TemporalLayerRateTrackerTest, ConvergesOnAShiftedKeyPattern) { - // In L2T2_KEY_SHIFT the spatial layers run antiphase: within a temporal unit - // they are on different temporal layers, and the base layer of the lower - // spatial layer holds two frames in a row right after the keyframe. The - // frame after the keyframe is therefore not on the topmost layer and the - // priming does not kick in, leaving the cadence to be learned. - // - // t=0: S0T0, S1T0 t=1: S0T0, S1T1 t=2: S0T1, S1T0 ... - TemporalLayerRateTracker tracker; - const DataRate kLowerBitrate = kStreamBitrate / 3; - const DataRate kUpperBitrate = kStreamBitrate * 2 / 3; - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/0, kLowerBitrate * 2 / 3, - /*is_keyframe=*/true); - tracker.Update(/*spatial_id=*/1, /*temporal_id=*/0, kUpperBitrate * 2 / 3, - /*is_keyframe=*/true); - - for (int temporal_unit = 1; temporal_unit <= 16; ++temporal_unit) { - const int lower_temporal_id = temporal_unit % 2 == 1 ? 0 : 1; - const int upper_temporal_id = 1 - lower_temporal_id; - tracker.Update(/*spatial_id=*/0, lower_temporal_id, - kLowerBitrate * (lower_temporal_id == 0 ? 2.0 / 3 : 1.0 / 3), - /*is_keyframe=*/false); - tracker.Update(/*spatial_id=*/1, upper_temporal_id, - kUpperBitrate * (upper_temporal_id == 0 ? 2.0 / 3 : 1.0 / 3), - /*is_keyframe=*/false); - - // Both layers have been seen twice after four temporal units, which is all - // it takes to measure how often their frames occur. - if (temporal_unit >= 4) { - EXPECT_EQ(tracker.FramerateFactor(0), 2) - << " at temporal unit " << temporal_unit; - } - } - - EXPECT_EQ(tracker.num_temporal_layers(), 2); - EXPECT_EQ(tracker.FramerateFactor(1), 1); - EXPECT_NEAR(tracker.StreamBitrate(0).kbps(), kLowerBitrate.kbps(), 5); - EXPECT_NEAR(tracker.StreamBitrate(1).kbps(), kUpperBitrate.kbps(), 5); -} - -TEST(TemporalLayerRateTrackerTest, ConvergesOnANonDyadicCadence) { - // A pattern where only every third frame belongs to the base layer. The - // priming assumes a dyadic pattern, in which the base layer holds every - // other frame, so the cadence has to be corrected from there. - TemporalLayerRateTracker tracker; - // The base layer frames are twice the size of the ones above them but only - // half as many, so the two layers end up with the same bitrate. - const DataRate kLayerBitrate = kStreamBitrate / 2; - - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/0, kLayerBitrate, - /*is_keyframe=*/true); - for (int frame = 1; frame <= 40; ++frame) { - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/frame % 3 == 0 ? 0 : 1, - kLayerBitrate, /*is_keyframe=*/false); - - // Measured: the cadence estimate settles on the correct value by the - // twentieth frame, less than a second of video, and stays there. The - // frames of the upper layer come in pairs, so the interval between them - // alternates between one and two and the estimate has to average the two - // out before it can be trusted to the nearest integer. - if (frame >= 20) { - EXPECT_EQ(tracker.FramerateFactor(0), 3) << " at frame " << frame; - } - } - - EXPECT_EQ(tracker.num_temporal_layers(), 2); - EXPECT_EQ(tracker.FramerateFactor(1), 1); - EXPECT_NEAR(tracker.CumulativeBitrate(0, 0).kbps(), kLayerBitrate.kbps(), 5); - EXPECT_NEAR(tracker.StreamBitrate(0).kbps(), kStreamBitrate.kbps(), 5); -} - -// The layer the caller happens to state a new allocation on first must not -// matter: the change is carried over to the layers that have not reported yet, -// so the cumulative bitrate is right for the very next frame either way. The -// three tests below place the change on each of the layers in turn. -TEST(TemporalLayerRateTrackerTest, CarriesAChangeSeenOnTheBaseLayerOver) { - TemporalLayerRateTracker tracker; - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); - // Run the pattern until every layer has stated a bitrate of its own. The - // frame that follows belongs to the base layer. - EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/6); - ASSERT_EQ(L1T3TemporalId(8), 0); - - EncodeL1T3Frames(tracker, /*first_frame=*/8, /*num_frames=*/1, /*scale=*/0.5); - - EXPECT_EQ(tracker.CumulativeBitrate(0, 0), GeometricLayerBitrate(0, 3) / 2); - EXPECT_EQ(tracker.StreamBitrate(0), kStreamBitrate / 2); -} - -TEST(TemporalLayerRateTrackerTest, CarriesAChangeSeenOnAMiddleLayerOver) { - TemporalLayerRateTracker tracker; - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); - EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/4); - ASSERT_EQ(L1T3TemporalId(6), 1); - - EncodeL1T3Frames(tracker, /*first_frame=*/6, /*num_frames=*/1, /*scale=*/0.5); - - EXPECT_EQ(tracker.CumulativeBitrate(0, 0), GeometricLayerBitrate(0, 3) / 2); - EXPECT_EQ(tracker.StreamBitrate(0), kStreamBitrate / 2); -} - -TEST(TemporalLayerRateTrackerTest, CarriesAChangeSeenOnTheTopLayerOver) { - TemporalLayerRateTracker tracker; - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); - EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/5); - ASSERT_EQ(L1T3TemporalId(7), 2); - - EncodeL1T3Frames(tracker, /*first_frame=*/7, /*num_frames=*/1, /*scale=*/0.5); - - EXPECT_EQ(tracker.CumulativeBitrate(0, 0), GeometricLayerBitrate(0, 3) / 2); - EXPECT_EQ(tracker.StreamBitrate(0), kStreamBitrate / 2); -} - -TEST(TemporalLayerRateTrackerTest, DoesNotCountACarriedOverChangeTwice) { - TemporalLayerRateTracker tracker; - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); - EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/4); - - // The allocation is halved, and halved again before the layers that were - // only assumed to follow along have stated anything themselves. Each of them - // then restates what was already assumed, which must leave the estimates - // where they are. - EncodeL1T3Frames(tracker, /*first_frame=*/6, /*num_frames=*/1, /*scale=*/0.5); - EncodeL1T3Frames(tracker, /*first_frame=*/7, /*num_frames=*/1, - /*scale=*/0.25); - EXPECT_EQ(tracker.StreamBitrate(0), kStreamBitrate / 4); - - EncodeL1T3Frames(tracker, /*first_frame=*/8, /*num_frames=*/4, - /*scale=*/0.25); - EXPECT_EQ(tracker.CumulativeBitrate(0, 0), GeometricLayerBitrate(0, 3) / 4); - EXPECT_EQ(tracker.StreamBitrate(0), kStreamBitrate / 4); -} - -TEST(TemporalLayerRateTrackerTest, SettlesOnANewDistribution) { - TemporalLayerRateTracker tracker; - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); - EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/6); - - // The stream bitrate is kept, but distributed 3:2:2 rather than 2:1:1. The - // first layer to state its new share is taken to speak for the others, which - // is wrong here, so a layer that was carried along is only put right the - // next time it states a bitrate of its own. - const std::array<DataRate, 3> kNewBitrates = { - kStreamBitrate * 3 / 7, - kStreamBitrate * 2 / 7, - kStreamBitrate * 2 / 7, - }; - // Splitting the stream bitrate in sevenths does not come out even, so the - // layers are what the total is held against. - const DataRate kNewStreamBitrate = - kNewBitrates[0] + kNewBitrates[1] + kNewBitrates[2]; - for (int frame = 8; frame < 16; ++frame) { - const int temporal_id = L1T3TemporalId(frame); - tracker.Update(0, temporal_id, kNewBitrates[temporal_id], - /*is_keyframe=*/false); - - // The base layer, the last to be heard from a second time, settles on the - // frame that follows a full pattern. - if (frame >= 12) { - EXPECT_EQ(tracker.CumulativeBitrate(0, 0), kNewBitrates[0]) - << " at frame " << frame; - EXPECT_EQ(tracker.StreamBitrate(0), kNewStreamBitrate) - << " at frame " << frame; - } - } - - EXPECT_EQ(tracker.CumulativeBitrate(0, 1), kNewBitrates[0] + kNewBitrates[1]); -} - -TEST(TemporalLayerRateTrackerTest, DoesNotReadARepeatedBitrateAsAChange) { - TemporalLayerRateTracker tracker; - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); - EncodeL1T3Frames(tracker, /*first_frame=*/2, /*num_frames=*/6); - - // Only the base layer is given more, which the tracker cannot tell from the - // start of a change of the whole allocation. The layers above it then keep - // restating what they had, and a restatement says nothing, so the estimates - // must settle rather than swing back and forth between the two readings. - const DataRate kNewBaseBitrate = GeometricLayerBitrate(0, 3) * 1.5; - const DataRate kNewStreamBitrate = kNewBaseBitrate + - GeometricLayerBitrate(1, 3) + - GeometricLayerBitrate(2, 3); - for (int frame = 8; frame < 40; ++frame) { - const int temporal_id = L1T3TemporalId(frame); - tracker.Update(0, temporal_id, - temporal_id == 0 ? kNewBaseBitrate - : GeometricLayerBitrate(temporal_id, 3), - /*is_keyframe=*/false); - - // One pattern is enough for every layer to have been heard from. - if (frame >= 12) { - EXPECT_EQ(tracker.CumulativeBitrate(0, 0), kNewBaseBitrate) - << " at frame " << frame; - EXPECT_EQ(tracker.StreamBitrate(0), kNewStreamBitrate) - << " at frame " << frame; - } - } -} - -TEST(TemporalLayerRateTrackerTest, DoesNotReadADepartureFromAGuessAsAChange) { - TemporalLayerRateTracker tracker; - // The priming guesses what the layers above the base one hold. The first - // frame of such a layer states the real value, which is not a change of the - // allocation however far off the guess was, so the other layers must be left - // alone. - EncodeKeyframeAndFirstDeltaFrame(tracker, /*num_layers=*/3); - const DataRate kMiddleLayerBitrate = GeometricLayerBitrate(1, 3) * 4; - - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/1, kMiddleLayerBitrate, - /*is_keyframe=*/false); - - EXPECT_EQ(tracker.CumulativeBitrate(0, 0), GeometricLayerBitrate(0, 3)); - EXPECT_EQ(tracker.CumulativeBitrate(0, 1), - GeometricLayerBitrate(0, 3) + kMiddleLayerBitrate); -} - -TEST(TemporalLayerRateTrackerTest, KeepsTheSpatialLayersApart) { - // Two spatial layers in an L1T2 pattern, the lower one at 300 kbps and the - // upper one at 600 kbps, both split 2:1 between their temporal layers. - constexpr DataRate kLowerStreamBitrate = DataRate::KilobitsPerSec(300); - constexpr DataRate kUpperStreamBitrate = DataRate::KilobitsPerSec(600); - TemporalLayerRateTracker tracker; - for (int frame = 0; frame < 4; ++frame) { - const double share = frame % 2 == 0 ? 2.0 / 3 : 1.0 / 3; - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/frame % 2, - kLowerStreamBitrate * share, /*is_keyframe=*/frame == 0); - tracker.Update(/*spatial_id=*/1, /*temporal_id=*/frame % 2, - kUpperStreamBitrate * share, /*is_keyframe=*/frame == 0); - } - - // Halving the lower spatial layer says nothing about the upper one. - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/0, - kLowerStreamBitrate * (2.0 / 3) / 2, /*is_keyframe=*/false); - - EXPECT_EQ(tracker.StreamBitrate(0), kLowerStreamBitrate / 2); - EXPECT_EQ(tracker.StreamBitrate(1), kUpperStreamBitrate); -} - -// The two simulations below drive the tracker the way a congestion controller -// would drive a stream with temporal layers, which the encoder level tests do -// not cover: those run a single temporal layer, where the bitrate of the layer -// and of the stream are the same thing. -TEST(TemporalLayerRateTrackerTest, FollowsAStagedBitrateSweep) { - // The bitrate profile of the encoder level ChangingBitrateTargetVga test, - // over an L1T3 pattern at 30 fps. - constexpr int kNumLayers = 3; - TemporalLayerRateTracker tracker; - int frame = 0; - - auto encode = [&](DataRate stream_bitrate, int num_frames) { - for (int i = 0; i < num_frames; ++i, ++frame) { - const int temporal_id = L1T3TemporalId(frame); - tracker.Update(/*spatial_id=*/0, temporal_id, - LayerShare(stream_bitrate, temporal_id, kNumLayers), - /*is_keyframe=*/frame == 0); - - // Every step of the sweep scales the whole allocation, so the frame at - // hand states everything there is to know about the change and the - // bitrate it is encoded against is right away. - EXPECT_EQ(tracker.CumulativeBitrate(/*spatial_id=*/0, temporal_id), - CumulativeShare(stream_bitrate, temporal_id, kNumLayers)) - << " at frame " << frame << " of T" << temporal_id; - } - }; - - encode(DataRate::KilobitsPerSec(500), 60); - encode(DataRate::KilobitsPerSec(100), 30); - for (int kbps = 150; kbps < 500; kbps += 50) { - encode(DataRate::KilobitsPerSec(kbps), 6); - } - encode(DataRate::KilobitsPerSec(500), 30); - - EXPECT_EQ(tracker.StreamBitrate(/*spatial_id=*/0), - DataRate::KilobitsPerSec(500)); -} - -TEST(TemporalLayerRateTrackerTest, RecoversOnceTheTargetHoldsStill) { - // A target that moves on every single frame, which is more than a congestion - // controller would ask for, denies the tracker the one thing that tells a - // change of the whole allocation from a change of the distribution: a layer - // restating the bitrate it already had. Whatever the stream does while a - // layer waits for its first turn is then never accounted for, and the skew - // that leaves behind, measured at up to 10% of the cumulative bitrate, - // stays until the target settles. - constexpr int kNumLayers = 4; - TemporalLayerRateTracker tracker; - auto stream_bitrate_at = [](int frame) { - // A sawtooth between 300 and 900 kbps in steps of 30, which divides evenly - // into fifths and keeps the layer bitrates whole. - return DataRate::KilobitsPerSec(300 + 30 * (frame % 21)); - }; - - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/0, - LayerShare(stream_bitrate_at(0), 0, kNumLayers), - /*is_keyframe=*/true); - tracker.Update(/*spatial_id=*/0, /*temporal_id=*/kNumLayers - 1, - LayerShare(stream_bitrate_at(1), kNumLayers - 1, kNumLayers), - /*is_keyframe=*/false); - for (int frame = 2; frame < 40; ++frame) { - const int temporal_id = L1T4TemporalId(frame); - tracker.Update( - /*spatial_id=*/0, temporal_id, - LayerShare(stream_bitrate_at(frame), temporal_id, kNumLayers), - /*is_keyframe=*/false); - } - - // A layer restating what it holds is what puts the estimates right, so they - // are all in order again once the pattern has come around. - constexpr DataRate kHeldBitrate = DataRate::KilobitsPerSec(600); - for (int frame = 40; frame < 80; ++frame) { - const int temporal_id = L1T4TemporalId(frame); - tracker.Update(/*spatial_id=*/0, temporal_id, - LayerShare(kHeldBitrate, temporal_id, kNumLayers), - /*is_keyframe=*/false); - - if (frame >= 40 + 1 * static_cast<int>(kL1T4Pattern.size())) { - EXPECT_EQ(tracker.CumulativeBitrate(/*spatial_id=*/0, temporal_id), - CumulativeShare(kHeldBitrate, temporal_id, kNumLayers)) - << " at frame " << frame << " of T" << temporal_id; - } - } -} - -} // namespace -} // namespace webrtc