diff --git a/main/xiaozhi-server/config.yaml b/main/xiaozhi-server/config.yaml index b2cc9fe6..1fca578c 100644 --- a/main/xiaozhi-server/config.yaml +++ b/main/xiaozhi-server/config.yaml @@ -70,6 +70,12 @@ enable_stop_tts_notify: false # 说完话是否开启提示音,音效地址 stop_tts_notify_voice: "config/assets/tts_notify.mp3" +# TTS音频发送延迟配置 +# tts_audio_send_delay: 控制音频包发送间隔 +# 0: 使用精确时间控制,严格匹配音频帧率(默认,运行时按音频帧率计算) +# > 0: 使用固定延迟(毫秒)发送,例如: 60 +tts_audio_send_delay: 0 + exit_commands: - "退出" - "关闭" @@ -987,4 +993,4 @@ TTS: # sample_rate: 24000 # 采样率:16000, 8000, 24000 # volume: 50 # 音量:0-100 # speed: 50 # 语速:0-100 - # pitch: 50 # 语调:0-100 \ No newline at end of file + # pitch: 50 # 语调:0-100 diff --git a/main/xiaozhi-server/core/handle/sendAudioHandle.py b/main/xiaozhi-server/core/handle/sendAudioHandle.py index 8beda577..a820ee46 100644 --- a/main/xiaozhi-server/core/handle/sendAudioHandle.py +++ b/main/xiaozhi-server/core/handle/sendAudioHandle.py @@ -90,6 +90,9 @@ async def sendAudio(conn, audios, frame_duration=60): if audios is None or len(audios) == 0: return + # 获取发送延迟配置 + send_delay = conn.config.get("tts_audio_send_delay", -1) / 1000.0 + if isinstance(audios, bytes): if conn.client_abort: return @@ -107,16 +110,21 @@ async def sendAudio(conn, audios, frame_duration=60): flow_control = conn.audio_flow_control current_time = time.perf_counter() - # 计算预期发送时间 - expected_time = flow_control["start_time"] + ( - flow_control["packet_count"] * frame_duration / 1000 - ) - delay = expected_time - current_time - if delay > 0: - await asyncio.sleep(delay) + + if send_delay > 0: + # 使用固定延迟 + await asyncio.sleep(send_delay) else: - # 纠正误差 - flow_control["start_time"] += abs(delay) + # 计算预期发送时间 + expected_time = flow_control["start_time"] + ( + flow_control["packet_count"] * frame_duration / 1000 + ) + delay = expected_time - current_time + if delay > 0: + await asyncio.sleep(delay) + else: + # 纠正误差 + flow_control["start_time"] += abs(delay) if conn.conn_from_mqtt_gateway: # 计算时间戳和序列号 @@ -164,12 +172,16 @@ async def sendAudio(conn, audios, frame_duration=60): # 重置没有声音的状态 conn.last_activity_time = time.time() * 1000 - # 计算预期发送时间 - expected_time = start_time + (play_position / 1000) - current_time = time.perf_counter() - delay = expected_time - current_time - if delay > 0: - await asyncio.sleep(delay) + if send_delay > 0: + # 固定延迟模式 + await asyncio.sleep(send_delay) + else: + # 计算预期发送时间 + expected_time = start_time + (play_position / 1000) + current_time = time.perf_counter() + delay = expected_time - current_time + if delay > 0: + await asyncio.sleep(delay) if conn.conn_from_mqtt_gateway: # 计算时间戳和序列号(使用当前的数据包索引确保连续性)