You need to enable JavaScript to run this app.
优惠活动
大模型
产品
解决方案
定价
更多

Windows C++桌面录音应用回声消除功能开发需求

问题描述

我们有一款基于Windows C++开发的音频录制应用,通过系统级麦克风活动检测启动录制,生成麦克风输入和扬声器输出两个独立原始文件,录制停止后合并为单一声立体声文件(本地用户音频与远程扬声器输入各占一个声道)。当前应用功能正常,但缺少音频预处理与回声消除功能:当用户未使用耳机时,麦克风会拾取扬声器播放的声音并混入本地声道,导致远程参与者听到回声。需要在麦克风录制代码的TODO位置添加回声消除逻辑,实现麦克风仅录制本地用户音频。

相关核心代码片段:

bool AudioCapture::CaptureMicrophoneAudio(const wchar_t* outputFilePath)
{
    // Activate audio client
    IAudioClient* pAudioClient = nullptr;
    HRESULT hr = pDevice->Activate(__uuidof(IAudioClient), CLSCTX_ALL, NULL, (void**)&pAudioClient);
    if (FAILED(hr)) {
        logger->Log("Error activating audio client: ");
        pDevice->Release();
        //CoUninitialize();
        return false;
    }

    // Get mix format
    WAVEFORMATEX* pWaveFormat = nullptr;
    hr = pAudioClient->GetMixFormat(&pWaveFormat);
    if (FAILED(hr)) {
        logger->Log("Error getting mix format: ");
        pAudioClient->Release();
        pDevice->Release();
        //CoUninitialize();
        return false;
    }

    logger->Log("WAVEFORMATEX: nChannels - " + std::to_string(pWaveFormat->nChannels));
    logger->Log("WAVEFORMATEX: nSamplesPerSec - " + std::to_string(pWaveFormat->nSamplesPerSec));
    logger->Log("WAVEFORMATEX: wBitsPerSample - " + std::to_string(pWaveFormat->wBitsPerSample));
    logger->Log("WAVEFORMATEX: wFormatTag - " + std::to_string(pWaveFormat->wFormatTag));
    logger->Log("WAVEFORMATEX: cbSize - " + std::to_string(pWaveFormat->cbSize));
    logger->Log("WAVEFORMATEX: nAvgBytesPerSec - " + std::to_string(pWaveFormat->nAvgBytesPerSec));
    logger->Log("WAVEFORMATEX: nBlockAlign - " + std::to_string(pWaveFormat->nBlockAlign));

    nMicChannels = pWaveFormat->nChannels;
    nMicSamplesPerSec = pWaveFormat->nSamplesPerSec;
    wMicBitsPerSample = pWaveFormat->wBitsPerSample;
    nMicAvgBytesPerSec = pWaveFormat->nAvgBytesPerSec;

    // Initialize audio client with the mix format
    hr = pAudioClient->Initialize(AUDCLNT_SHAREMODE_SHARED, 0, 0, 0, pWaveFormat, NULL);
    if (FAILED(hr)) {
        logger->Log("Error initializing audio client: ");
        CoTaskMemFree(pWaveFormat);
        pAudioClient->Release();
        pDevice->Release();
        //CoUninitialize();
        return false;
    }

    // Get capture client
    IAudioCaptureClient* pCaptureClient = nullptr;
    hr = pAudioClient->GetService(__uuidof(IAudioCaptureClient), (void**)&pCaptureClient);
    if (FAILED(hr)) {
        logger->Log("Error getting capture client: ");
        CoTaskMemFree(pWaveFormat);
        pAudioClient->Release();
        pDevice->Release();
       // CoUninitialize();
        return false;
    }

    // Open binary file for writing
    std::string tempFilePath = Utils::GetTempFilename("Audio_mic_capture", "raw");
    logger->Log(tempFilePath);
    std::ofstream outFile(tempFilePath, std::ios::binary);
    if (!outFile.is_open()) {
        logger->Log("Error opening binary file for writing");
        CoTaskMemFree(pWaveFormat);
        pCaptureClient->Release();
        pAudioClient->Release();
        pDevice->Release();
      //  CoUninitialize();
        return false;
    }

    // Start capturing
    hr = pAudioClient->Start();
    if (FAILED(hr))
    {
        logger->Log("Error starting audio client: ");
        CoTaskMemFree(pWaveFormat);
        pCaptureClient->Release();
        pAudioClient->Release();
        pDevice->Release();
     //   CoUninitialize();
        return false;
    }

    // Main loop for capturing and writing to file
    while (!exitFlag) 
    {        
        // Capture audio data
        // Read audio data from pCaptureClient
        BYTE* pData;
        UINT32 numFramesAvailable;
        DWORD flags;
        hr = pCaptureClient->GetBuffer(&pData, &numFramesAvailable, &flags, NULL, NULL);
        if (FAILED(hr)) 
        {
            logger->Log("Error getting audio buffer: ");
            break;
        }

        //
        // TODO: Apply echo cancellation algorithm to pData
        //

        // Write audio data to file
        int count = numFramesAvailable * pWaveFormat->nBlockAlign;
        outFile.write(reinterpret_cast<const char*>(pData), count);

        // Release the buffer
        hr = pCaptureClient->ReleaseBuffer(numFramesAvailable);
        if (FAILED(hr)) 
        {
            logger->Log("Error releasing audio buffer: ");
            break;
        }
    }

    logger->Log("Capturing Completed");

    // Close binary file
    outFile.close();

    // Stop capturing
    pAudioClient->Stop();

    // Clean up resources
    CoTaskMemFree(pWaveFormat);
    pCaptureClient->Release();
    pAudioClient->Release();
    pDevice->Release();

   // CoUninitialize();

    // convert to mp3 and delete temp file
    std::string tempFilePathMp3 = Utils::GetTempFilename("Audio_capture_mic", "mp3");
    tempMicFilePathWMp3 = std::wstring(tempFilePathMp3.begin(), tempFilePathMp3.end());
    std::wstring tempFilePathW(tempFilePath.begin(), tempFilePath.end());
    if (!convertToMp3(tempFilePathW, tempMicFilePathWMp3, nMicChannels, nMicSamplesPerSec, wMicBitsPerSample, nMicAvgBytesPerSec))
    {
        logger->Log("Failed to convert to MP3");
    }
    else
    {
        logger->Log("Saved to MP3 data");
    }

    // Remove temp file
    std::remove(tempFilePath.c_str());

    return true;
}
实现方案与技术参考

针对Windows平台的回声消除需求,推荐以下三种可行方案,按易用性、效果优先级排序:

方案一:使用Windows系统内置回声消除API(推荐)

Windows自带音频效果框架,支持硬件/软件级回声消除,无需引入第三方库,兼容性强,适合大多数场景。

实现步骤:

  1. 启用系统回声消除效果:在初始化IAudioClient时,指定启用回声消除标志,或通过IAudioEffectsManager主动添加AEC效果。
  2. 修改音频客户端初始化参数:

代码修改示例:

在IAudioClient::Initialize调用时添加回声消除标志:

// 替换原初始化代码,添加AEC标志
hr = pAudioClient->Initialize(
    AUDCLNT_SHAREMODE_SHARED,
    AUDCLNT_STREAMFLAGS_ECHO_CANCELLATION | AUDCLNT_STREAMFLAGS_AUTOCONVERTPCM,
    0,
    0,
    pWaveFormat,
    NULL
);

如果系统不支持直接标志启用,可通过效果管理器手动添加:

// 激活AudioClient后,获取效果管理器
IAudioEffectsManager* pEffectsManager = nullptr;
hr = pAudioClient->GetService(__uuidof(IAudioEffectsManager), (void**)&pEffectsManager);
if (SUCCEEDED(hr)) {
    IAudioEffect* pAecEffect = nullptr;
    // 查找系统回声消除效果
    hr = pEffectsManager->GetEffect(CLSID_AudioEchoCancellation, &pAecEffect);
    if (SUCCEEDED(hr)) {
        pAecEffect->SetEnabled(TRUE);
        logger->Log("Enabled system echo cancellation");
        pAecEffect->Release();
    }
    pEffectsManager->Release();
}

方案二:使用WebRTC开源回声消除库

WebRTC的AEC模块是业界成熟的回声消除实现,效果优于系统内置方案,适合对音质要求较高的场景。

实现步骤:

  1. 集成WebRTC音频处理模块(编译modules/audio_processing或使用预编译库)。
  2. 初始化AEC实例,匹配麦克风与扬声器的音频参数。
  3. 在捕获循环中,用扬声器输出音频作为参考,处理麦克风数据。

代码修改示例:

  1. 类中添加WebRTC相关成员:
#include <webrtc/modules/audio_processing/include/audio_processing.h>
#include <webrtc/modules/audio_processing/echo_cancellation.h>

class AudioCapture {
private:
    webrtc::AudioProcessing* apm_ = nullptr;
    std::vector<int16_t> speaker_ref_buffer_; // 存储扬声器参考音频
    // ... 其他成员
};
  1. 在CaptureMicrophoneAudio初始化阶段配置AEC:
// 初始化WebRTC音频处理器
apm_ = webrtc::AudioProcessing::Create();
auto* aec_ = apm_->echo_cancellation();
aec_->enable_stream_delay_estimator(true);
aec_->enable(true);

// 设置音频参数(需与麦克风、扬声器格式一致)
webrtc::ProcessingConfig config;
config.input_stream().set_sample_rate_hz(nMicSamplesPerSec);
config.input_stream().set_num_channels(nMicChannels);
config.output_stream().set_sample_rate_hz(nMicSamplesPerSec);
config.output_stream().set_num_channels(nMicChannels);
apm_->Initialize(config);
  1. 在TODO位置添加回声消除处理:
// TODO: Apply echo cancellation algorithm to pData
// 1. 转换麦克风数据为int16_t格式(假设是16位PCM)
int sample_bytes = wMicBitsPerSample / 8;
int total_samples = numFramesAvailable * nMicChannels;
std::vector<int16_t> mic_data(total_samples);
memcpy(mic_data.data(), pData, total_samples * sample_bytes);

// 2. 获取对应帧的扬声器参考音频(需从扬声器录制模块同步获取)
std::vector<int16_t> ref_data(total_samples);
if (!speaker_ref_buffer_.empty()) {
    memcpy(ref_data.data(), speaker_ref_buffer_.data(), total_samples * sample_bytes);
    speaker_ref_buffer_.clear();
}

// 3. 调用WebRTC AEC处理
webrtc::AudioFrame mic_frame;
mic_frame.UpdateFrame(0, mic_data.data(), total_samples, nMicSamplesPerSec, 
                      webrtc::AudioFrame::kNormalSpeech, webrtc::AudioFrame::kVadUnknown, nMicChannels);

webrtc::AudioFrame ref_frame;
ref_frame.UpdateFrame(0, ref_data.data(), total_samples, nMicSamplesPerSec, 
                      webrtc::AudioFrame::kNormalSpeech, webrtc::AudioFrame::kVadUnknown, nMicChannels);

apm_->ProcessStream(&mic_frame, &ref_frame);

// 4. 将处理后的数据写回原缓冲区
memcpy(pData, mic_frame.data(), total_samples * sample_bytes);

方案三:基础自适应滤波实现(临时测试用)

如果不想依赖外部资源,可实现简单的NLMS自适应滤波算法,但效果远不如专业方案,仅适合临时验证。

代码示例(简化版):

// 预先初始化滤波器权重(长度根据回声延迟估算,比如400ms对应的采样数)
static std::vector<float> filter_weights(nMicSamplesPerSec * 0.4f, 0.0f);
const float mu = 0.01f; // 步长系数

// 转换麦克风和参考音频为float格式
int total_samples = numFramesAvailable * nMicChannels;
std::vector<float> mic_float(total_samples);
std::vector<float> ref_float(total_samples);
// 此处省略格式转换代码(从BYTE转float)

// NLMS算法处理
for (int i = 0; i < total_samples; ++i) {
    // 估算回声
    float echo_est = 0.0f;
    for (size_t j = 0; j < filter_weights.size(); ++j) {
        if (i >= (int)j) echo_est += filter_weights[j] * ref_float[i - j];
    }
    // 计算误差(消除回声后的信号)
    float error = mic_float[i] - echo_est;
    // 更新滤波器权重
    for (size_t j = 0; j < filter_weights.size(); ++j) {
        if (i >= (int)j) filter_weights[j] += mu * error * ref_float[i - j];
    }
    mic_float[i] = error;
}

// 将处理后的数据转换回原格式写回pData
// 此处省略格式转换代码(从float转BYTE)
关键注意事项
  • 时间同步:回声消除的核心是扬声器参考音频与麦克风捕获音频的时间对齐,需确保两者帧延迟一致,可通过调整缓冲区大小或添加延迟补偿解决。
  • 格式匹配:麦克风与扬声器的音频格式(采样率、位深、通道数)必须统一,否则需先做格式转换。
  • 错误处理:系统API或开源库调用需做好异常捕获,避免因设备不支持导致应用崩溃。

内容的提问来源于stack exchange,提问作者kran

相关产品推荐
方舟 Agent Plan

超全模态模型 × Harness 升级,最新支持 Deepseek-V4.1-Flash、GLM-5.3 系列、Doubao-Seedream-5.0-pro、Kimi-K3 (部分), 限时 9.9 元起

最近更新时间:2026.06.27 11:47:02