feat: 增加asr,tts文件上报功能

This commit is contained in:
goodyhao
2025-04-30 17:56:52 +08:00
parent 4f829017ed
commit 48f8c6c5b7
21 changed files with 706 additions and 6 deletions
+104 -2
View File
@@ -30,6 +30,7 @@ from core.mcp.manager import MCPManager
from config.config_loader import get_private_config_from_api
from config.manage_api_client import DeviceNotFoundException, DeviceBindException
from core.utils.output_counter import add_device_output
from core.handle.ttsReportHandle import enqueue_tts_report
TAG = __name__
@@ -54,6 +55,7 @@ class ConnectionHandler:
self.websocket = None
self.headers = None
self.device_id = None
self.client_ip = None
self.client_ip_info = {}
self.session_id = None
@@ -72,6 +74,13 @@ class ConnectionHandler:
self.audio_play_queue = queue.Queue()
self.executor = ThreadPoolExecutor(max_workers=10)
# 上报线程标志
self.session_open_time = time.time()
self.tts_report_queue = queue.Queue()
self.asr_report_queue = queue.Queue()
self.asr_report_thread = None
self.tts_report_thread = None
# 依赖的组件
self.vad = _vad
self.asr = _asr
@@ -153,6 +162,7 @@ class ConnectionHandler:
# 认证通过,继续处理
self.websocket = ws
self.device_id = self.headers.get("device-id", None)
self.session_id = str(uuid.uuid4())
# 启动超时检查任务
@@ -301,6 +311,26 @@ class ConnectionHandler:
self._initialize_memory()
"""加载意图识别"""
self._initialize_intent()
"""初始化上报线程"""
self._init_report_threads()
def _init_report_threads(self):
"""初始化ASR和TTS上报线程"""
if self.asr_report_thread is None or not self.asr_report_thread.is_alive():
self.asr_report_thread = threading.Thread(
target=self._asr_report_worker,
daemon=True
)
self.asr_report_thread.start()
self.logger.bind(tag=TAG).info("ASR上报线程已启动")
if self.tts_report_thread is None or not self.tts_report_thread.is_alive():
self.tts_report_thread = threading.Thread(
target=self._tts_report_worker,
daemon=True
)
self.tts_report_thread.start()
self.logger.bind(tag=TAG).info("TTS上报线程已启动")
def _initialize_private_config(self):
read_config_from_api = self.config.get("read_config_from_api", False)
@@ -427,8 +457,7 @@ class ConnectionHandler:
def _initialize_memory(self):
"""初始化记忆模块"""
device_id = self.headers.get("device-id", None)
self.memory.init_memory(device_id, self.llm)
self.memory.init_memory(self.device_id, self.llm)
def _initialize_intent(self):
if (
@@ -862,6 +891,9 @@ class ConnectionHandler:
f"TTS生成:文件路径: {tts_file}"
)
if os.path.exists(tts_file):
# 在这里上报TTS数据(使用文件路径)
enqueue_tts_report(self, text, tts_file)
opus_datas, duration = self.tts.audio_to_opus_data(tts_file)
else:
self.logger.bind(tag=TAG).error(
@@ -918,6 +950,72 @@ class ConnectionHandler:
f"audio_play_priority priority_thread: {text} {e}"
)
def _asr_report_worker(self):
"""ASR上报工作线程"""
# 提前导入避免循环引用问题
from core.handle.asrReportHandle import report_asr
while not self.stop_event.is_set():
try:
# 从队列获取数据,设置超时以便定期检查停止事件
item = self.asr_report_queue.get(timeout=1)
if item is None: # 检测毒丸对象
break
text, file_path = item
try:
# 执行上报(传入文件路径)
await_result = report_asr(self, text, file_path)
# 使用asyncio.run_coroutine_threadsafe执行异步操作
future = asyncio.run_coroutine_threadsafe(await_result, self.loop)
future.result()
except Exception as e:
self.logger.bind(tag=TAG).error(f"ASR上报线程异常: {e}")
finally:
# 标记任务完成
self.asr_report_queue.task_done()
except queue.Empty:
continue
except Exception as e:
self.logger.bind(tag=TAG).error(f"ASR上报工作线程异常: {e}")
self.logger.bind(tag=TAG).info("ASR上报线程已退出")
def _tts_report_worker(self):
"""TTS上报工作线程"""
# 提前导入避免循环引用问题
from core.handle.ttsReportHandle import report_tts
while not self.stop_event.is_set():
try:
# 从队列获取数据,设置超时以便定期检查停止事件
item = self.tts_report_queue.get(timeout=1)
if item is None: # 检测毒丸对象
break
text, audio_data = item
try:
# 执行上报(传入二进制数据)
await_result = report_tts(self, text, audio_data)
# 使用asyncio.run_coroutine_threadsafe执行异步操作
future = asyncio.run_coroutine_threadsafe(await_result, self.loop)
future.result()
except Exception as e:
self.logger.bind(tag=TAG).error(f"TTS上报线程异常: {e}")
finally:
# 标记任务完成
self.tts_report_queue.task_done()
except queue.Empty:
continue
except Exception as e:
self.logger.bind(tag=TAG).error(f"TTS上报工作线程异常: {e}")
self.logger.bind(tag=TAG).info("TTS上报线程已退出")
def speak_and_play(self, text, text_index=0):
if text is None or len(text) <= 0:
self.logger.bind(tag=TAG).info(f"无需tts转换,query为空,{text}")
@@ -963,6 +1061,10 @@ class ConnectionHandler:
self.executor.shutdown(wait=False, cancel_futures=True)
self.executor = None
# 添加毒丸对象到上报队列确保线程退出
self.asr_report_queue.put(None)
self.tts_report_queue.put(None)
# 清空任务队列
self.clear_queues()
@@ -0,0 +1,100 @@
"""
ASR上报功能已集成到ConnectionHandler类中。
上报功能包括:
1. 每个连接对象拥有自己的上报队列和处理线程
2. 上报线程的生命周期与连接对象绑定
3. 使用ConnectionHandler.enqueue_asr_report方法进行上报
具体实现请参考core/connection.py中的相关代码。
"""
import os
from config.logger import setup_logging
from config.manage_api_client import report
TAG = __name__
logger = setup_logging()
async def report_asr(conn, text, file_path):
"""执行ASR上报操作
Args:
conn: 连接对象
text: 识别文本
file_path: 音频文件路径(可以为None或空字符串,表示纯文本上报)
"""
audio_data = None
try:
# 处理无音频的纯文本上报
if not file_path or not os.path.exists(file_path):
# 纯文本上报时使用空音频数据
result = await report(
mac_address=conn.device_id,
session_id=conn.session_id,
sort=int(conn.session_open_time),
chat_type=1, # ASR类型为1
content=text,
audio=b'', # 空音频数据
file_extension="wav"
)
logger.bind(tag=TAG).info(f"纯文本上报成功: {conn.device_id}, {conn.session_id}")
else:
# 读取文件为二进制数据
with open(file_path, 'rb') as f:
audio_data = f.read()
# 正常ASR上报(带音频)
result = await report(
mac_address=conn.device_id,
session_id=conn.session_id,
sort=int(conn.session_open_time),
chat_type=1, # ASR类型为1
content=text,
audio=audio_data,
file_extension="wav"
)
logger.bind(tag=TAG).info(f"ASR上报成功: {conn.device_id}, {conn.session_id},文件: {file_path}")
return result
except Exception as e:
logger.bind(tag=TAG).error(f"ASR上报失败: {e}")
return None
finally:
# 清理资源
if file_path and os.path.exists(file_path):
try:
os.remove(file_path)
logger.bind(tag=TAG).debug(f"ASR上报后删除文件: {file_path}")
except Exception as e:
logger.bind(tag=TAG).error(f"ASR上报后删除文件失败: {e}")
# 手动清理audio_data
if audio_data:
del audio_data
def enqueue_asr_report(conn, text, audio):
"""将ASR数据加入上报队列
Args:
conn: 连接对象
text: 识别文本
audio: 音频数据(可以为空列表,表示纯文本上报)
"""
try:
if not audio or len(audio) == 0:
# 纯文本上报,不需要保存文件
file_path = None
else:
# 保存音频数据到文件
file_path = conn.asr.save_audio_to_file(audio, conn.session_id)
# 使用连接对象的队列,传入文件路径
conn.asr_report_queue.put((text, file_path))
if not audio or len(audio) == 0:
logger.bind(tag=TAG).info(f"纯文本数据已加入上报队列: {conn.device_id}, {text[:20] if text else ''}...")
else:
logger.bind(tag=TAG).info(f"ASR数据已加入上报队列: {conn.device_id}, 文件: {file_path}")
except Exception as e:
logger.bind(tag=TAG).error(f"加入ASR上报队列失败: {e}")
@@ -4,6 +4,7 @@ from core.utils.util import remove_punctuation_and_length
from core.handle.sendAudioHandle import send_stt_message
from core.handle.intentHandler import handle_user_intent
from core.utils.output_counter import check_device_output_limit
from core.handle.asrReportHandle import enqueue_asr_report
TAG = __name__
logger = setup_logging()
@@ -40,6 +41,9 @@ async def handleAudioMessage(conn, audio):
logger.bind(tag=TAG).info(f"识别文本: {text}")
text_len, _ = remove_punctuation_and_length(text)
if text_len > 0:
# 使用自定义模块进行上报
enqueue_asr_report(conn, text, conn.asr_audio)
await startToChat(conn, text)
else:
conn.asr_server_receive = True
@@ -6,6 +6,7 @@ from core.utils.util import remove_punctuation_and_length
from core.handle.receiveAudioHandle import startToChat, handleAudioMessage
from core.handle.sendAudioHandle import send_stt_message, send_tts_message
from core.handle.iotHandle import handleIotDescriptors, handleIotStatus
from core.handle.asrReportHandle import enqueue_asr_report
import asyncio
TAG = __name__
@@ -54,8 +55,12 @@ async def handleTextMessage(conn, message):
await send_stt_message(conn, text)
await send_tts_message(conn, "stop", None)
elif is_wakeup_words:
# 上报纯文字数据(复用ASR上报功能,但不提供音频数据)
enqueue_asr_report(conn, "嘿,你好呀", [])
await startToChat(conn, "嘿,你好呀")
else:
# 上报纯文字数据(复用ASR上报功能,但不提供音频数据)
enqueue_asr_report(conn, text, [])
# 否则需要LLM对文字内容进行答复
await startToChat(conn, text)
elif msg_json["type"] == "iot":
@@ -0,0 +1,70 @@
"""
TTS上报功能已集成到ConnectionHandler类中。
上报功能包括:
1. 每个连接对象拥有自己的上报队列和处理线程
2. 上报线程的生命周期与连接对象绑定
3. 使用ConnectionHandler.enqueue_tts_report方法进行上报
具体实现请参考core/connection.py中的相关代码。
"""
import os
from config.logger import setup_logging
from config.manage_api_client import report
TAG = __name__
logger = setup_logging()
async def report_tts(conn, text, audio_data):
"""执行TTS上报操作
Args:
conn: 连接对象
text: 合成文本
audio_data: 音频二进制数据
"""
try:
# 执行上报
result = await report(
mac_address=conn.device_id,
session_id=conn.session_id,
sort=int(conn.session_open_time),
chat_type=2, # TTS类型为2
content=text,
audio=audio_data,
file_extension="wav"
)
logger.bind(tag=TAG).info(f"TTS上报成功: {conn.device_id}, {conn.session_id}, 数据大小: {len(audio_data)} 字节")
return result
except Exception as e:
logger.bind(tag=TAG).error(f"TTS上报失败: {e}")
return None
finally:
# 手动清理audio_data引用,帮助垃圾回收
del audio_data
def enqueue_tts_report(conn, text, file_path):
"""将TTS数据加入上报队列
Args:
conn: 连接对象
text: 合成文本
file_path: TTS音频文件路径
"""
try:
# 检查文件是否存在
if not file_path or not os.path.exists(file_path):
logger.bind(tag=TAG).error(f"加入TTS上报队列失败: 文件不存在 {file_path}")
return
# 立即读取文件为二进制数据,因为外部会删除文件
with open(file_path, 'rb') as f:
audio_data = f.read()
# 使用连接对象的队列,传入文本和二进制数据而非文件路径
conn.tts_report_queue.put((text, audio_data))
logger.bind(tag=TAG).info(f"TTS数据已加入上报队列: {conn.device_id}, 文件大小: {len(audio_data)} 字节")
except Exception as e:
logger.bind(tag=TAG).error(f"加入TTS上报队列失败: {e}, 文件: {file_path}")
+1 -1
View File
@@ -484,4 +484,4 @@ def analyze_emotion(text):
if emotion in top_emotions:
return emotion
return top_emotions[0] # 如果都不在优先级列表里,返回第一个
return top_emotions[0] # 如果都不在优先级列表里,返回第一个