Google Cloud Speech-to-Text API 400音频超时错误排查求助
问题
使用Google Cloud Speech-to-Text API转录访谈音频,已设置3次重试机制,但反复收到400 Audio Timeout Error: Long duration elapsed without audio. Audio should be sent close to real time错误,尝试常规方案无效。
控制台输出
Converted retries value: 3 -> Retry(total=3, connect=None, read=None, redirect=None, status=None) Making request: POST https://oauth2.googleapis.com/token Starting new HTTPS connection (1): oauth2.googleapis.com:443 https://oauth2.googleapis.com:443 "POST /token HTTP/1.1" 200 None Starting new HTTPS connection (1): storage.googleapis.com:443 https://storage.googleapis.com:443 "GET /storage/v1/b/nicarg?projection=noAcl&prettyPrint=false HTTP/1.1" 200 727 https://storage.googleapis.com:443 "GET /download/storage/v1/b/*****/o/******.m4a?alt=media HTTP/1.1" 200 56660932 subprocess.call(['ffmpeg', '-y', '-f', 'mp4', '-i', 'audio\\******.m4a', '-acodec', 'pcm_s16le', '-vn', '-f', 'wav', '-']) Authenticating credentials... Authentication successful! Creating a streaming recognizer... Attempting to transcribe. Maximum retries: 3. Loading stream... Transcribing... Transcribing Segments: 0%| | 0/12 [00:00<?, ?it/s] Error during Google Cloud Speech-to-Text API request: 400 Audio Timeout Error: Long duration elapsed without audio. Audio should be sent close to real time. Loading stream... Transcribing... Transcribing Segments: 0%| | 0/12 [00:00<?, ?it/s] Error during Google Cloud Speech-to-Text API request: 400 Audio Timeout Error: Long duration elapsed without audio. Audio should be sent close to real time. Loading stream... Transcribing... Transcribing Segments: 0%| | 0/12 [00:00<?, ?it/s] Error during Google Cloud Speech-to-Text API request: 400 Audio Timeout Error: Long duration elapsed without audio. Audio should be sent close to real time. Maximum retry attempts reached. Unable to transcribe the segments.
完整代码
import os import sys from pydub import AudioSegment from pydub.silence import split_on_silence from transcriber import Transcriber import logging from google.cloud import storage from google.cloud import speech from tqdm import tqdm import time from google.api_core.exceptions import DeadlineExceeded, ServiceUnavailable import random # Constants GOOGLE_CREDENTIALS_PATH = "********.json" BUCKET_NAME = "*****" AUDIO_FILE_NAME = "*****.m4a" AUDIO_PATH = os.path.join("audio", AUDIO_FILE_NAME) REMOTE_AUDIO_PATH = AUDIO_FILE_NAME SEGMENT_DURATION = 5 * 60000 LANGUAGE = "es-419" class Transcriptor: """ Functions to process audio into transcribable text. """ @staticmethod def download_audio_from_cloud(audio_path, remote_path): client = storage.Client.from_service_account_json(GOOGLE_CREDENTIALS_PATH) bucket = client.get_bucket(BUCKET_NAME) blob = bucket.blob(remote_path) blob.download_to_filename(audio_path) @staticmethod def split_and_transcribe(audio_path, segment_duration=SEGMENT_DURATION, language=LANGUAGE): """ Splits the audio into segments and then transcribes each segment. Parameters: - audio_path (path): Audio file to process. - segment_duration (int): Segment duration. - language: Language of processing. Returns: - Audio transcripted. """ Transcriptor.download_audio_from_cloud(audio_path, REMOTE_AUDIO_PATH) # Load audio file audio = AudioSegment.from_file(audio_path, format="m4a") # Calculate the number of segments based on the specified duration num_segments = int(len(audio) / segment_duration) + 1 # Split the audio into segments audio_segments = [audio[i * segment_duration:(i + 1) * segment_duration] for i in range(num_segments)] # Transcribe the segments using Google Cloud Speech-to-Text API result_transcription = Transcriptor.transcribe_cloud( f"gs://{BUCKET_NAME}/{AUDIO_FILE_NAME}", language, audio_segments, total=num_segments ) return result_transcription @staticmethod def transcribe_cloud(gcs_uri, language=LANGUAGE, audio_segments=None, total=100, max_retries=3): """ Transcribes audio segments using Google Cloud Speech-to-Text API. Parameters: - gcs_uri (str): Google Cloud storage URI for the audio file. - language: Language of processing (always set as es-419) - audio_segments: List of audio segments to transcribe. Returns: - Transcript of the audio segments. """ # Authenticate credentials try: print("Authenticating credentials...") client = speech.SpeechClient.from_service_account_json(GOOGLE_CREDENTIALS_PATH) print("Authentication successful!") except Exception as e: logging.error(f"Authentication error: {e}. Verify Google credentials and try again.", exc_info=True) return None # Create a streaming recognizer with the given config print("Creating a streaming recognizer...") config = speech.RecognitionConfig( encoding=speech.RecognitionConfig.AudioEncoding.LINEAR16, sample_rate_hertz=16000, language_code=language, enable_automatic_punctuation=True, ) streaming_config = speech.StreamingRecognitionConfig( config=config, interim_results=True ) num_tries = 0 # Attempting to transcribe print(f"Attempting to transcribe. Maximum retries: {max_retries}.") while num_tries < max_retries: try: # Loading stream print("Loading stream...") requests = ( speech.StreamingRecognizeRequest(audio_content=segment.raw_data if segment else b'') for segment in audio_segments ) # Call the streaming_recognize method with the generator of requests responses = client.streaming_recognize(config=streaming_config, requests=requests) # Transcribing print("Transcribing...") transcript_builder = [] # Process interim and final results with tqdm(total=total, desc="Transcribing Segments") as pbar: for response in responses: for result in response.results: for alternative in result.alternatives: transcript_builder.append(f"\nTranscript: {alternative.transcript}") transcript_builder.append(f"\nConfidence: {alternative.confidence}\n") # Update the progress bar pbar.update(1) transcript = "".join(transcript_builder) if transcript: logging.info("Transcription successful!") print(transcript) return transcript else: logging.error("Transcribing result is empty. Retrying the segments.") num_tries += 1 except DeadlineExceeded as timeout_error: logging.warning(f"Timeout error: {timeout_error}. Retrying the segments. Retry attempt {num_tries}/{max_retries}.") num_tries += 1 except ServiceUnavailable as service_unavailable_error: logging.warning(f"Service Unavailable: {service_unavailable_error}. Retrying the segments.") num_tries += 1 except Exception as e: logging.error(f"Error during Google Cloud Speech-to-Text API request: {e}") num_tries += 1 # Exponential backoff: wait for a random time between 2^nums_tries seconds wait_time = random.uniform(0, 2**num_tries) time.sleep(wait_time) logging.error("Maximum retry attempts reached. Unable to transcribe the segments.") sys.exit(1) if __name__ == "__main__": # Set up logging logging.basicConfig(filename="logs/transcriptor.log", level=logging.DEBUG) # Set up console logging console_handler = logging.StreamHandler() console_handler.setLevel(logging.DEBUG) logging.getLogger().addHandler(console_handler) # Set Google Cloud credentials os.environ["GOOGLE_APPLICATION_CREDENTIALS"] = GOOGLE_CREDENTIALS_PATH # Transcribe audio and pass the text to the Transcriber try: result_transcription = Transcriptor.split_and_transcribe(AUDIO_PATH, SEGMENT_DURATION, LANGUAGE) app = Transcriber(result_transcription) app.run() except Exception as e: logging.exception(f"An error occurred: {e}")
解决方案
问题根源
- API选型错误:Streaming Recognition API专为实时音频流设计,要求音频按接近实时的节奏发送。一次性发送5分钟长的片段,不符合流式传输预期,触发超时。
- 音频格式不匹配:代码中直接使用
AudioSegment.raw_data作为LINEAR16编码内容,但未确保采样率、声道数与RecognitionConfig一致(配置设为16000Hz单声道,原始m4a可能不匹配)。 - 批量处理场景不适用流式API:预录制长音频应使用异步批量识别API(
long_running_recognize),而非流式API。
修复步骤
1. 替换为异步批量识别API
修改transcribe_cloud方法,使用适合长音频的异步接口,无需手动拆分(API自动处理):
@staticmethod def transcribe_cloud(gcs_uri, language=LANGUAGE, max_retries=3): """ 使用Google Cloud Speech-to-Text异步API转录长音频 """ try: print("Authenticating credentials...") client = speech.SpeechClient.from_service_account_json(GOOGLE_CREDENTIALS_PATH) print("Authentication successful!") except Exception as e: logging.error(f"Authentication error: {e}. Verify Google credentials and try again.", exc_info=True) return None # 匹配原始音频的实际格式,用ffmpeg -i 音频文件查看参数 config = speech.RecognitionConfig( encoding=speech.RecognitionConfig.AudioEncoding.MP4, sample_rate_hertz=44100, language_code=language, enable_automatic_punctuation=True, ) audio = speech.RecognitionAudio(uri=gcs_uri) num_tries = 0 print(f"Attempting to transcribe. Maximum retries: {max_retries}.") while num_tries < max_retries: try: print("Starting asynchronous transcription...") operation = client.long_running_recognize(config=config, audio=audio) print("Waiting for operation to complete...") # 根据音频长度调整超时时间,示例为1小时 response = operation.result(timeout=3600) transcript_builder = [] for result in response.results: transcript_builder.append(f"\nTranscript: {result.alternatives[0].transcript}") transcript_builder.append(f"\nConfidence: {result.alternatives[0].confidence}\n") transcript = "".join(transcript_builder) if transcript: logging.info("Transcription successful!") print(transcript) return transcript else: logging.error("Transcription result is empty. Retrying...") num_tries += 1 except DeadlineExceeded as timeout_error: logging.warning(f"Timeout error: {timeout_error}. Retrying attempt {num_tries+1}/{max_retries}.") num_tries += 1 except ServiceUnavailable as service_unavailable_error: logging.warning(f"Service Unavailable: {service_unavailable_error}. Retrying...") num_tries += 1 except Exception as e: logging.error(f"Error during transcription: {e}") num_tries += 1 wait_time = random.uniform(0, 2**num_tries) time.sleep(wait_time) logging.error("Maximum retry attempts reached.") sys.exit(1)
2. 调整音频格式配置
- 用
ffmpeg -i audio/your_file.m4a查看原始音频参数,确认采样率、编码格式。 - 对应修改
RecognitionConfig中的encoding和sample_rate_hertz:m4a文件使用AudioEncoding.MP4,采样率匹配实际值。
3. 简化调用逻辑
更新split_and_transcribe方法,直接传入GCS URI即可,无需手动拆分音频:
@staticmethod def split_and_transcribe(audio_path, segment_duration=SEGMENT_DURATION, language=LANGUAGE): Transcriptor.download_audio_from_cloud(audio_path, REMOTE_AUDIO_PATH) # 直接调用异步API处理完整音频 result_transcription = Transcriptor.transcribe_cloud( f"gs://{BUCKET_NAME}/{AUDIO_FILE_NAME}", language ) return result_transcription
关键说明
- 异步API支持最长480分钟的音频,自动处理拆分和转录,适合预录制内容。
- 流式API仅用于实时场景(如语音通话、直播),要求持续、及时发送音频数据。
- 必须确保音频格式与配置完全匹配,否则会导致识别失败或错误。
内容的提问来源于stack exchange,提问作者9scorp4
相关产品推荐
相关产品推荐

