Python向Unity传输WAV音频文件出现高音调噪音问题求助
问题分析
你的核心问题是音频格式解析不匹配:
- 测试用的WAV文件是16位PCM格式(2字节/采样),但你在Unity中错误地将接收的字节按32位浮点(4字节/采样)解析,导致采样数计算错误(实际采样数是接收字节数/2,但你按/4计算),播放速度直接翻倍,同时格式不匹配产生大量噪音。
- 硬编码跳过44字节WAV头不可靠,部分WAV文件包含额外元数据块,头长度可能超过44字节。
修正方案
步骤1:修正Python发送端
先解析WAV头获取正确的音频参数,再发送参数+音频数据,避免硬编码假设格式:
import socket import struct HOST = '127.0.0.1' PORT = 12345 def read_wav_info(file_path): with open(file_path, 'rb') as f: # 读取WAV头 f.read(4) # RIFF标识 f.read(4) # 文件总大小 f.read(4) # WAVE标识 f.read(4) # fmt块标识 fmt_size = struct.unpack('<I', f.read(4))[0] audio_format = struct.unpack('<H', f.read(2))[0] channels = struct.unpack('<H', f.read(2))[0] sample_rate = struct.unpack('<I', f.read(4))[0] f.read(4) # 字节率 f.read(2) # 块对齐值 bits_per_sample = struct.unpack('<H', f.read(2))[0] # 跳过额外的fmt块数据(如果有的话) if fmt_size > 16: f.read(fmt_size - 16) # 定位到data块 while True: chunk_id = f.read(4) if chunk_id == b'data': break chunk_size = struct.unpack('<I', f.read(4))[0] f.read(chunk_size) data_size = struct.unpack('<I', f.read(4))[0] audio_data = f.read(data_size) return sample_rate, channels, bits_per_sample, audio_data # 读取音频信息和原始数据 sample_rate, channels, bits_per_sample, audio_data = read_wav_info('BabyElephantWalk60.wav') # 创建Socket并发送数据 server_socket = socket.socket(socket.AF_INET, socket.SOCK_STREAM) server_socket.bind((HOST, PORT)) server_socket.listen() print(f"Server listening on {HOST}:{PORT}") client_socket, addr = server_socket.accept() print(f"Connected to {addr}") # 先发送音频参数(采样率、声道数、位深度、数据长度) client_socket.send(struct.pack('<III', sample_rate, channels, bits_per_sample)) client_socket.send(struct.pack('<I', len(audio_data))) # 发送音频原始数据 client_socket.sendall(audio_data) print(f"Sent: {sample_rate}Hz, {channels} channels, {bits_per_sample}bit, {len(audio_data)} bytes") client_socket.close() server_socket.close()
步骤2:修正Unity接收端
根据接收的参数,正确解析16位PCM数据为AudioClip所需的float采样:
using System; using System.Net.Sockets; using System.IO; using UnityEngine; public class RealTimeAudioReceiver : MonoBehaviour { public AudioSource audioSource; private TcpClient client; private NetworkStream stream; private BinaryReader reader; private byte[] receivedAudioData; private int sampleRate; private int channels; private int bitsPerSample; private bool play = false; private void Start() { Debug.Log("Time.timeScale: " + Time.timeScale); ConnectToServer(); } private void Update() { if (play) { audioSource.Play(); play = false; } } private void ConnectToServer() { try { client = new TcpClient("127.0.0.1", 12345); stream = client.GetStream(); reader = new BinaryReader(stream); StartCoroutine(ReceiveAudioData()); } catch (Exception e) { Debug.LogError("Error connecting to server: " + e.Message); } } private System.Collections.IEnumerator ReceiveAudioData() { // 先接收音频参数 sampleRate = reader.ReadInt32(); channels = reader.ReadInt32(); bitsPerSample = reader.ReadInt32(); int dataLength = reader.ReadInt32(); Debug.Log($"Received params: {sampleRate}Hz, {channels} channels, {bitsPerSample}bit, {dataLength} bytes"); // 按指定长度接收音频数据 receivedAudioData = new byte[dataLength]; int bytesRead = 0; while (bytesRead < dataLength) { int read = reader.Read(receivedAudioData, bytesRead, dataLength - bytesRead); if (read == 0) break; bytesRead += read; } CreateAndPlayAudioClip(); // 清理资源 reader.Close(); stream.Close(); client.Close(); yield return null; } private void CreateAndPlayAudioClip() { if (receivedAudioData == null || receivedAudioData.Length == 0) return; int sampleCount = receivedAudioData.Length / (bitsPerSample / 8); AudioClip clip = AudioClip.Create("ReceivedAudio", sampleCount, channels, sampleRate, false); float[] samples = new float[sampleCount]; if (bitsPerSample == 16) { // 16位PCM整数转float(映射到[-1,1]范围) for (int i = 0; i < sampleCount; i++) { short rawSample = BitConverter.ToInt16(receivedAudioData, i * 2); samples[i] = rawSample / 32768f; } } else if (bitsPerSample == 32) { // 兼容32位浮点格式 for (int i = 0; i < sampleCount; i++) { samples[i] = BitConverter.ToSingle(receivedAudioData, i * 4); } } clip.SetData(samples, 0); audioSource.clip = clip; play = true; } }
关键修正点
- 格式匹配:将16位PCM整数转换为float时,除以32768(16位有符号整数的最大值),将数值映射到Unity AudioClip要求的[-1,1]范围。
- 动态参数同步:从WAV头解析真实参数并发送给Unity,避免硬编码导致的参数不匹配。
- 可靠数据接收:先发送数据长度,再按长度接收,解决TCP粘包或数据截断问题。
- 正确采样数计算:采样数=总字节数/(位深度/8),16位格式下为字节数/2,修正了之前采样数减半导致的播放加速问题。
内容的提问来源于stack exchange,提问作者Fantuan
相关产品推荐
相关产品推荐

