Keep muted audio frames muted when resampling ResamplerHelper resampled muted frames as well and wrote the output through mutable_data(), which clears the muted flag. When resampling, ChannelReceive therefore returned muted NetEq output as kNormal, so the mixer mixed silence instead of skipping the source, and the frames weren't counted in decoded_muted_output. - ResamplerHelper keeps a muted frame that follows a muted frame muted and doesn't resample it. The first muted frame after audio is still resampled and comes out unmuted, since it holds the resampler's tail of that audio. After muted frames, the resampler is primed with silence. - ChannelReceive counts decoded_muted_output from NetEq's muted state before resampling. Bug: none Change-Id: I8bb8c5572911c172239005038554feacbb8f87cb Reviewed-on: https://webrtc-review.googlesource.com/c/src/+/506980 Reviewed-by: Jakob Ivarsson‎ <jakobi@webrtc.org> Commit-Queue: Tomas Gunnarsson <tommi@webrtc.org> Cr-Commit-Position: refs/heads/main@{#48827}
diff --git a/audio/channel_receive.cc b/audio/channel_receive.cc index e9599b9..10fe97b 100644 --- a/audio/channel_receive.cc +++ b/audio/channel_receive.cc
@@ -449,13 +449,16 @@ last_playout_time_ = now; } - resampler_helper_.MaybeResample(sample_rate_hz, audio_frame); - + // Update the stats with NetEq's muted state before MaybeResample(), which + // unmutes a frame whose resampled output can still hold the tail of the + // audio before it. { MutexLock lock(&call_stats_mutex_); call_stats_.DecodedByNetEq(audio_frame->speech_type_, audio_frame->muted()); } + resampler_helper_.MaybeResample(sample_rate_hz, audio_frame); + bool has_capture_time = false; for (const auto& packet_info : audio_frame->packet_infos_) { if (packet_info.absolute_capture_time().has_value()) {
diff --git a/audio/channel_receive_unittest.cc b/audio/channel_receive_unittest.cc index b2022cd..ac3d0be 100644 --- a/audio/channel_receive_unittest.cc +++ b/audio/channel_receive_unittest.cc
@@ -421,6 +421,40 @@ ::testing::ElementsAre(1111, 2222)); } +// Parameterized on the sample rate requested by the mixer. With a rate other +// than kSampleRateHz, the NetEq output is resampled. +class ChannelReceiveOutputRateTest : public ChannelReceiveTest, + public ::testing::WithParamInterface<int> { +}; + +TEST_P(ChannelReceiveOutputRateTest, ReportsMutedNetEqOutputAsMuted) { + const int mixer_sample_rate_hz = GetParam(); + auto channel = CreateTestChannelReceive(); + channel->StartPlayout(); + channel->OnRtpPacket(CreateRtpPacket()); + + // Once the packet has been played out, NetEq fades out and then enters the + // muted state. Pull up to 5 seconds of 10 ms frames. + AudioFrame audio_frame; + bool muted = false; + for (int i = 0; i < 500 && !muted; ++i) { + muted = + channel->GetAudioFrameWithInfo(mixer_sample_rate_hz, &audio_frame) == + AudioMixer::Source::AudioFrameInfo::kMuted; + } + EXPECT_TRUE(muted); + // When resampling, the first muted NetEq frame can still hold the + // resampler's tail of the audio before it, so it is reported as normal and + // only the next one as muted. + const bool resampling = mixer_sample_rate_hz != kSampleRateHz; + EXPECT_EQ(channel->GetDecodingCallStatistics().decoded_muted_output, + resampling ? 2 : 1); +} + +INSTANTIATE_TEST_SUITE_P(All, + ChannelReceiveOutputRateTest, + ::testing::Values(kSampleRateHz, 48000)); + } // namespace } // namespace voe } // namespace webrtc
diff --git a/modules/audio_coding/acm2/acm_resampler.cc b/modules/audio_coding/acm2/acm_resampler.cc index ddefdbe..48115d6 100644 --- a/modules/audio_coding/acm2/acm_resampler.cc +++ b/modules/audio_coding/acm2/acm_resampler.cc
@@ -57,8 +57,28 @@ } } + // The resampler's delay is shorter than a frame, so a muted frame that + // follows a muted frame resamples to silence. Frames rejected above don't + // reach the resampler and don't count. + const bool previous_frame_muted = last_frame_muted_; + last_frame_muted_ = audio_frame->muted(); + const bool silent = audio_frame->muted() && previous_frame_muted; + + if (need_resampling && silent) { + // Keep the frame muted instead of resampling silence. Audio after it is + // still resampled as if after silence: the resampler's last input was a + // muted frame, or the resampler gets primed with silence below. + audio_frame->SetSampleRateAndChannelSize(desired_sample_rate_hz); + return true; + } + if (need_resampling && !resampled_last_output_frame_) { // Prime the resampler with the last frame. + if (previous_frame_muted) { + // `last_audio_buffer_` holds zeros only up to the size of the muted + // frame it stored last, and this frame can be larger. + absl::c_fill(last_audio_buffer_, 0); + } InterleavedView<const int16_t> src(last_audio_buffer_.data(), audio_frame->samples_per_channel(), audio_frame->num_channels()); @@ -78,7 +98,6 @@ audio_frame->SetSampleRateAndChannelSize(desired_sample_rate_hz); InterleavedView<int16_t> dst = audio_frame->mutable_data( audio_frame->samples_per_channel(), audio_frame->num_channels()); - // TODO(tommi): Don't resample muted audio frames. resampler_.Resample(src, dst); resampled_last_output_frame_ = true; } else {
diff --git a/modules/audio_coding/acm2/acm_resampler.h b/modules/audio_coding/acm2/acm_resampler.h index 8b6cee0..e9dd839 100644 --- a/modules/audio_coding/acm2/acm_resampler.h +++ b/modules/audio_coding/acm2/acm_resampler.h
@@ -30,11 +30,18 @@ ResamplerHelper(); // Resamples audio_frame if it is not already in desired_sample_rate_hz. + // A muted frame that follows a muted frame stays muted: its resampled + // output would be silent. A muted frame that follows audio is resampled + // and comes out unmuted, since the output still holds the tail of that + // audio. bool MaybeResample(int desired_sample_rate_hz, AudioFrame* audio_frame); private: PushResampler<int16_t> resampler_; bool resampled_last_output_frame_ = true; + // Whether the last frame that MaybeResample() accepted was muted. The + // resampler starts out silent. + bool last_frame_muted_ = true; std::array<int16_t, AudioFrame::kMaxDataSizeSamples> last_audio_buffer_; };
diff --git a/modules/audio_coding/acm2/acm_resampler_unittest.cc b/modules/audio_coding/acm2/acm_resampler_unittest.cc index 49713a9..1a9e28c 100644 --- a/modules/audio_coding/acm2/acm_resampler_unittest.cc +++ b/modules/audio_coding/acm2/acm_resampler_unittest.cc
@@ -12,9 +12,11 @@ #include <cstddef> #include <cstdint> +#include <optional> #include <vector> #include "api/audio/audio_frame.h" +#include "api/audio/audio_view.h" #include "test/gtest.h" namespace webrtc { @@ -78,5 +80,127 @@ EXPECT_EQ(audio_frame.num_channels_, kChannels); } +namespace { + +constexpr int kInputSampleRateHz = 16000; +constexpr size_t kInputSamplesPerChannel = kInputSampleRateHz / 100; +// Twice the input rate: the resampling ratio is exact in floating point, so +// the outputs of two resamplers can be compared sample by sample. +constexpr int kOutputSampleRateHz = 32000; +constexpr size_t kOutputSamplesPerChannel = kOutputSampleRateHz / 100; + +// Sets `frame` to 10 ms of mono audio at `sample_rate_hz` with all samples +// equal to `value`, or to a muted frame if `value` is std::nullopt. +void SetFrame(std::optional<int16_t> value, + int sample_rate_hz, + AudioFrame* frame) { + const size_t samples_per_channel = + SampleRateToDefaultChannelSize(sample_rate_hz); + const std::vector<int16_t> samples(samples_per_channel, value.value_or(0)); + frame->UpdateFrame(/*timestamp=*/0, value ? samples.data() : nullptr, + samples_per_channel, sample_rate_hz, + AudioFrame::kNormalSpeech, AudioFrame::kVadActive); +} + +std::vector<int16_t> Samples(const AudioFrame& frame) { + InterleavedView<const int16_t> samples = frame.data_view(); + return std::vector<int16_t>(samples.begin(), samples.end()); +} + +} // namespace + +TEST(ResamplerHelperTest, KeepsMutedFrameAfterMutedFrameMuted) { + ResamplerHelper resampler; + AudioFrame audio_frame; + SetFrame(1000, kInputSampleRateHz, &audio_frame); + ASSERT_TRUE(resampler.MaybeResample(kOutputSampleRateHz, &audio_frame)); + + // The first muted frame after audio comes out unmuted: the resampled output + // starts with the tail of that audio. + SetFrame(std::nullopt, kInputSampleRateHz, &audio_frame); + ASSERT_TRUE(resampler.MaybeResample(kOutputSampleRateHz, &audio_frame)); + EXPECT_FALSE(audio_frame.muted()); + EXPECT_NE(audio_frame.data_view()[0], 0); + + // The next muted frame would resample to silence, so it stays muted. + SetFrame(std::nullopt, kInputSampleRateHz, &audio_frame); + ASSERT_TRUE(resampler.MaybeResample(kOutputSampleRateHz, &audio_frame)); + EXPECT_TRUE(audio_frame.muted()); + EXPECT_EQ(audio_frame.sample_rate_hz_, kOutputSampleRateHz); + EXPECT_EQ(audio_frame.samples_per_channel(), kOutputSamplesPerChannel); +} + +// Keeping muted frames muted doesn't change the audio: the output is the same +// as that of a resampler that gets zeros instead of muted frames. +TEST(ResamplerHelperTest, KeepingMutedFramesMutedDoesNotChangeTheOutput) { + // Larger frames than at kInputSampleRateHz. Resampling from this rate to + // kOutputSampleRateHz is exact in floating point too. + constexpr int kLargeFrameSampleRateHz = 48000; + struct Step { + bool muted; + int desired_sample_rate_hz; + int input_sample_rate_hz = kInputSampleRateHz; + }; + constexpr Step kSteps[] = { + {false, kOutputSampleRateHz}, + {true, kOutputSampleRateHz}, + {true, kOutputSampleRateHz}, + {false, kOutputSampleRateHz}, + // Muted without resampling, then resampling resumes while muted. + {true, kInputSampleRateHz}, + {true, kOutputSampleRateHz}, + {false, kOutputSampleRateHz}, + // The same, but the audio before and after the muted frames comes in + // larger frames than the muted frames. + {false, kLargeFrameSampleRateHz, kLargeFrameSampleRateHz}, + {true, kInputSampleRateHz}, + {true, kOutputSampleRateHz}, + {false, kOutputSampleRateHz, kLargeFrameSampleRateHz}, + }; + ResamplerHelper resampler; + ResamplerHelper reference; + AudioFrame audio_frame; + AudioFrame reference_frame; + int step_index = 0; + for (const Step& step : kSteps) { + SCOPED_TRACE(step_index++); + const int16_t value = step.muted ? 0 : 1000; + SetFrame(step.muted ? std::nullopt : std::make_optional(value), + step.input_sample_rate_hz, &audio_frame); + SetFrame(value, step.input_sample_rate_hz, &reference_frame); + ASSERT_TRUE( + resampler.MaybeResample(step.desired_sample_rate_hz, &audio_frame)); + ASSERT_TRUE( + reference.MaybeResample(step.desired_sample_rate_hz, &reference_frame)); + EXPECT_EQ(Samples(audio_frame), Samples(reference_frame)); + } +} + +// A frame that MaybeResample() rejects doesn't reach the resampler, so a muted +// frame after it still gets the tail of the audio before it. +TEST(ResamplerHelperTest, MutedFrameAfterRejectedFrameGetsTheTail) { + // 10 ms of kChannels at kTooHighSampleRateHz don't fit in an AudioFrame. + constexpr size_t kChannels = 8; + constexpr int kTooHighSampleRateHz = 192000; + const std::vector<int16_t> audio(kInputSamplesPerChannel * kChannels, 1000); + ResamplerHelper resampler; + AudioFrame audio_frame; + auto set_frame = [&](const int16_t* data) { + audio_frame.UpdateFrame(/*timestamp=*/0, data, kInputSamplesPerChannel, + kInputSampleRateHz, AudioFrame::kNormalSpeech, + AudioFrame::kVadActive, kChannels); + }; + set_frame(audio.data()); + ASSERT_TRUE(resampler.MaybeResample(kOutputSampleRateHz, &audio_frame)); + + set_frame(/*data=*/nullptr); + EXPECT_FALSE(resampler.MaybeResample(kTooHighSampleRateHz, &audio_frame)); + + set_frame(/*data=*/nullptr); + ASSERT_TRUE(resampler.MaybeResample(kOutputSampleRateHz, &audio_frame)); + EXPECT_FALSE(audio_frame.muted()); + EXPECT_NE(audio_frame.data_view()[0], 0); +} + } // namespace acm2 } // namespace webrtc