mirror of
https://github.com/xinnan-tech/xiaozhi-esp32-server.git
synced 2026-07-26 00:53:54 +08:00
update:长按说话不走VAD直接触发ASR识别
This commit is contained in:
@@ -30,7 +30,13 @@ class ListenTextMessageHandler(TextMessageHandler):
|
|||||||
conn.client_have_voice = True
|
conn.client_have_voice = True
|
||||||
conn.client_voice_stop = True
|
conn.client_voice_stop = True
|
||||||
if len(conn.asr_audio) > 0:
|
if len(conn.asr_audio) > 0:
|
||||||
await handleAudioMessage(conn, b"")
|
# 手动模式下直接触发ASR识别,不需要再调用handleAudioMessage
|
||||||
|
asr_audio_task = conn.asr_audio.copy()
|
||||||
|
conn.asr_audio.clear()
|
||||||
|
conn.reset_vad_states()
|
||||||
|
|
||||||
|
if len(asr_audio_task) > 0:
|
||||||
|
await conn.asr.handle_voice_stop(conn, asr_audio_task)
|
||||||
elif msg_json["state"] == "detect":
|
elif msg_json["state"] == "detect":
|
||||||
conn.client_have_voice = False
|
conn.client_have_voice = False
|
||||||
conn.asr_audio.clear()
|
conn.asr_audio.clear()
|
||||||
|
|||||||
@@ -56,21 +56,28 @@ class ASRProviderBase(ABC):
|
|||||||
async def receive_audio(self, conn, audio, audio_have_voice):
|
async def receive_audio(self, conn, audio, audio_have_voice):
|
||||||
if conn.client_listen_mode == "auto" or conn.client_listen_mode == "realtime":
|
if conn.client_listen_mode == "auto" or conn.client_listen_mode == "realtime":
|
||||||
have_voice = audio_have_voice
|
have_voice = audio_have_voice
|
||||||
else:
|
|
||||||
have_voice = conn.client_have_voice
|
|
||||||
|
|
||||||
conn.asr_audio.append(audio)
|
conn.asr_audio.append(audio)
|
||||||
if not have_voice and not conn.client_have_voice:
|
if not have_voice and not conn.client_have_voice:
|
||||||
conn.asr_audio = conn.asr_audio[-10:]
|
conn.asr_audio = conn.asr_audio[-10:]
|
||||||
return
|
return
|
||||||
|
else:
|
||||||
|
# 手动模式:总是缓存音频,忽略VAD检测结果
|
||||||
|
conn.asr_audio.append(audio)
|
||||||
|
|
||||||
if conn.client_voice_stop:
|
if conn.client_voice_stop:
|
||||||
asr_audio_task = conn.asr_audio.copy()
|
asr_audio_task = conn.asr_audio.copy()
|
||||||
conn.asr_audio.clear()
|
conn.asr_audio.clear()
|
||||||
conn.reset_vad_states()
|
conn.reset_vad_states()
|
||||||
|
|
||||||
if len(asr_audio_task) > 15:
|
# 手动模式下允许短语音识别,自动模式保持原有限制
|
||||||
await self.handle_voice_stop(conn, asr_audio_task)
|
if conn.client_listen_mode == "auto" or conn.client_listen_mode == "realtime":
|
||||||
|
if len(asr_audio_task) > 15:
|
||||||
|
await self.handle_voice_stop(conn, asr_audio_task)
|
||||||
|
else:
|
||||||
|
# 手动模式:只要有音频就进行识别
|
||||||
|
if len(asr_audio_task) > 0:
|
||||||
|
await self.handle_voice_stop(conn, asr_audio_task)
|
||||||
|
|
||||||
# 处理语音停止
|
# 处理语音停止
|
||||||
async def handle_voice_stop(self, conn, asr_audio_task: List[bytes]):
|
async def handle_voice_stop(self, conn, asr_audio_task: List[bytes]):
|
||||||
|
|||||||
@@ -45,6 +45,10 @@ class VADProvider(VADProviderBase):
|
|||||||
pass
|
pass
|
||||||
|
|
||||||
def is_vad(self, conn, opus_packet):
|
def is_vad(self, conn, opus_packet):
|
||||||
|
# 手动模式:直接返回True,不进行实时VAD检测,所有音频都缓存
|
||||||
|
if conn.client_listen_mode not in ["auto", "realtime"]:
|
||||||
|
return True
|
||||||
|
|
||||||
try:
|
try:
|
||||||
pcm_frame = self.decoder.decode(opus_packet, 960)
|
pcm_frame = self.decoder.decode(opus_packet, 960)
|
||||||
conn.client_audio_buffer.extend(pcm_frame) # 将新数据加入缓冲区
|
conn.client_audio_buffer.extend(pcm_frame) # 将新数据加入缓冲区
|
||||||
|
|||||||
Reference in New Issue
Block a user