mirror of
https://github.com/xinnan-tech/xiaozhi-esp32-server.git
synced 2026-07-24 16:13:54 +08:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
af6c8ab935 | ||
|
|
4acb5f924a | ||
|
|
b1b99f1066 | ||
|
|
004ee6f143 | ||
|
|
19455f89d3 | ||
|
|
18d98e3bd0 | ||
|
|
4a9297dc65 | ||
|
|
49fefb41a6 | ||
|
|
2fa64dfdd0 | ||
|
|
ed91e5d607 | ||
|
|
0bf0b0d381 |
@@ -145,6 +145,7 @@ tmp
|
|||||||
.history
|
.history
|
||||||
.DS_Store
|
.DS_Store
|
||||||
main/xiaozhi-server/data
|
main/xiaozhi-server/data
|
||||||
|
main/xiaozhi-server/config/assets/wakeup_words.*
|
||||||
main/manager-web/node_modules
|
main/manager-web/node_modules
|
||||||
.config.yaml
|
.config.yaml
|
||||||
.secrets.yaml
|
.secrets.yaml
|
||||||
|
|||||||
@@ -57,6 +57,14 @@ delete_audio: true
|
|||||||
close_connection_no_voice_time: 120
|
close_connection_no_voice_time: 120
|
||||||
# TTS请求超时时间(秒)
|
# TTS请求超时时间(秒)
|
||||||
tts_timeout: 10
|
tts_timeout: 10
|
||||||
|
# 开启唤醒词加速
|
||||||
|
enable_wakeup_words_response_cache: true
|
||||||
|
# 开场是否回复唤醒词
|
||||||
|
enable_greeting: true
|
||||||
|
# 说完话是否开启提示音
|
||||||
|
enable_stop_tts_notify: false
|
||||||
|
# 说完话是否开启提示音,音效地址
|
||||||
|
stop_tts_notify_voice: "config/assets/tts_notify.mp3"
|
||||||
|
|
||||||
CMD_exit:
|
CMD_exit:
|
||||||
- "退出"
|
- "退出"
|
||||||
@@ -514,3 +522,16 @@ module_test:
|
|||||||
- "你好,请介绍一下你自己"
|
- "你好,请介绍一下你自己"
|
||||||
- "What's the weather like today?"
|
- "What's the weather like today?"
|
||||||
- "请用100字概括量子计算的基本原理和应用前景"
|
- "请用100字概括量子计算的基本原理和应用前景"
|
||||||
|
|
||||||
|
# 唤醒词,用于识别唤醒词还是讲话内容
|
||||||
|
wakeup_words:
|
||||||
|
- "你好小智"
|
||||||
|
- "你好小志"
|
||||||
|
- "小爱同学"
|
||||||
|
- "你好小鑫"
|
||||||
|
- "你好小新"
|
||||||
|
- "小美同学"
|
||||||
|
- "小龙小龙"
|
||||||
|
- "喵喵同学"
|
||||||
|
- "小滨小滨"
|
||||||
|
- "小冰小冰"
|
||||||
Binary file not shown.
Binary file not shown.
@@ -9,7 +9,7 @@ import traceback
|
|||||||
import threading
|
import threading
|
||||||
import websockets
|
import websockets
|
||||||
from typing import Dict, Any
|
from typing import Dict, Any
|
||||||
import plugins_func.loadplugins
|
from plugins_func.loadplugins import auto_import_modules
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
from core.utils.dialogue import Message, Dialogue
|
from core.utils.dialogue import Message, Dialogue
|
||||||
from core.handle.textHandle import handleTextMessage
|
from core.handle.textHandle import handleTextMessage
|
||||||
@@ -25,6 +25,8 @@ from core.utils.auth_code_gen import AuthCodeGenerator
|
|||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
|
|
||||||
|
auto_import_modules('plugins_func.functions')
|
||||||
|
|
||||||
|
|
||||||
class TTSException(RuntimeError):
|
class TTSException(RuntimeError):
|
||||||
pass
|
pass
|
||||||
@@ -109,11 +111,15 @@ class ConnectionHandler:
|
|||||||
|
|
||||||
# 进行认证
|
# 进行认证
|
||||||
await self.auth.authenticate(self.headers)
|
await self.auth.authenticate(self.headers)
|
||||||
|
|
||||||
device_id = self.headers.get("device-id", None)
|
device_id = self.headers.get("device-id", None)
|
||||||
self.memory.init_memory(device_id, self.llm)
|
|
||||||
self.intent.set_llm(self.llm)
|
|
||||||
|
|
||||||
|
# 认证通过,继续处理
|
||||||
|
self.websocket = ws
|
||||||
|
self.session_id = str(uuid.uuid4())
|
||||||
|
|
||||||
|
self.welcome_msg = self.config["xiaozhi"]
|
||||||
|
self.welcome_msg["session_id"] = self.session_id
|
||||||
|
await self.websocket.send(json.dumps(self.welcome_msg))
|
||||||
# Load private configuration if device_id is provided
|
# Load private configuration if device_id is provided
|
||||||
bUsePrivateConfig = self.config.get("use_private_config", False)
|
bUsePrivateConfig = self.config.get("use_private_config", False)
|
||||||
self.logger.bind(tag=TAG).info(f"bUsePrivateConfig: {bUsePrivateConfig}, device_id: {device_id}")
|
self.logger.bind(tag=TAG).info(f"bUsePrivateConfig: {bUsePrivateConfig}, device_id: {device_id}")
|
||||||
@@ -141,17 +147,8 @@ class ConnectionHandler:
|
|||||||
self.private_config = None
|
self.private_config = None
|
||||||
raise
|
raise
|
||||||
|
|
||||||
# 认证通过,继续处理
|
|
||||||
self.websocket = ws
|
|
||||||
self.session_id = str(uuid.uuid4())
|
|
||||||
|
|
||||||
self.welcome_msg = self.config["xiaozhi"]
|
|
||||||
self.welcome_msg["session_id"] = self.session_id
|
|
||||||
await self.websocket.send(json.dumps(self.welcome_msg))
|
|
||||||
|
|
||||||
# 异步初始化
|
# 异步初始化
|
||||||
await self.loop.run_in_executor(None, self._initialize_components)
|
self.executor.submit(self._initialize_components)
|
||||||
|
|
||||||
# tts 消化线程
|
# tts 消化线程
|
||||||
tts_priority = threading.Thread(target=self._tts_priority_thread, daemon=True)
|
tts_priority = threading.Thread(target=self._tts_priority_thread, daemon=True)
|
||||||
tts_priority.start()
|
tts_priority.start()
|
||||||
@@ -187,16 +184,26 @@ class ConnectionHandler:
|
|||||||
await handleAudioMessage(self, message)
|
await handleAudioMessage(self, message)
|
||||||
|
|
||||||
def _initialize_components(self):
|
def _initialize_components(self):
|
||||||
|
"""加载插件"""
|
||||||
|
self.func_handler = FunctionHandler(self)
|
||||||
|
|
||||||
|
"""加载提示词"""
|
||||||
self.prompt = self.config["prompt"]
|
self.prompt = self.config["prompt"]
|
||||||
if self.private_config:
|
if self.private_config:
|
||||||
self.prompt = self.private_config.private_config.get("prompt", self.prompt)
|
self.prompt = self.private_config.private_config.get("prompt", self.prompt)
|
||||||
|
|
||||||
self.client_ip_info = get_ip_info(self.client_ip)
|
|
||||||
self.logger.bind(tag=TAG).info(f"Client ip info: {self.client_ip_info}")
|
|
||||||
self.prompt = self.prompt + f"\n我在:{self.client_ip_info}"
|
|
||||||
self.dialogue.put(Message(role="system", content=self.prompt))
|
self.dialogue.put(Message(role="system", content=self.prompt))
|
||||||
|
|
||||||
self.func_handler = FunctionHandler(self)
|
"""加载记忆"""
|
||||||
|
device_id = self.headers.get("device-id", None)
|
||||||
|
self.memory.init_memory(device_id, self.llm)
|
||||||
|
self.intent.set_llm(self.llm)
|
||||||
|
|
||||||
|
"""加载位置信息"""
|
||||||
|
self.client_ip_info = get_ip_info(self.client_ip)
|
||||||
|
if self.client_ip_info is not None and "city" in self.client_ip_info:
|
||||||
|
self.logger.bind(tag=TAG).info(f"Client ip info: {self.client_ip_info}")
|
||||||
|
self.prompt = self.prompt + f"\nuser location:{self.client_ip_info}"
|
||||||
|
self.dialogue.update_system_message(self.prompt)
|
||||||
|
|
||||||
def change_system_prompt(self, prompt):
|
def change_system_prompt(self, prompt):
|
||||||
self.prompt = prompt
|
self.prompt = prompt
|
||||||
|
|||||||
@@ -1,8 +1,72 @@
|
|||||||
import json
|
import json
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
|
from core.handle.sendAudioHandle import send_stt_message
|
||||||
|
from core.utils.util import remove_punctuation_and_length
|
||||||
|
import shutil
|
||||||
|
import asyncio
|
||||||
|
import os
|
||||||
|
import random
|
||||||
|
import time
|
||||||
|
|
||||||
logger = setup_logging()
|
logger = setup_logging()
|
||||||
|
|
||||||
|
WAKEUP_CONFIG = {"dir": "config/assets/", "file_name": "wakeup_words", "create_time": time.time(), "refresh_time": 10,
|
||||||
|
"words": ["很高兴见到你", "你好啊", "我们又见面了", "最近可好?", "很高兴再次和你谈话", "在干嘛"]}
|
||||||
|
|
||||||
|
|
||||||
async def handleHelloMessage(conn):
|
async def handleHelloMessage(conn):
|
||||||
await conn.websocket.send(json.dumps(conn.welcome_msg))
|
await conn.websocket.send(json.dumps(conn.welcome_msg))
|
||||||
|
|
||||||
|
|
||||||
|
async def checkWakeupWords(conn, text):
|
||||||
|
enable_wakeup_words_response_cache = conn.config["enable_wakeup_words_response_cache"]
|
||||||
|
"""是否开启唤醒词加速"""
|
||||||
|
if not enable_wakeup_words_response_cache:
|
||||||
|
return False
|
||||||
|
"""检查是否是唤醒词"""
|
||||||
|
_, text = remove_punctuation_and_length(text)
|
||||||
|
if text in conn.config.get("wakeup_words"):
|
||||||
|
await send_stt_message(conn, text)
|
||||||
|
conn.tts_first_text_index = 0
|
||||||
|
conn.tts_last_text_index = 0
|
||||||
|
conn.llm_finish_task = True
|
||||||
|
|
||||||
|
file = getWakeupWordFile(WAKEUP_CONFIG["file_name"])
|
||||||
|
if file is None:
|
||||||
|
asyncio.create_task(wakeupWordsResponse(conn))
|
||||||
|
return False
|
||||||
|
opus_packets, duration = conn.tts.audio_to_opus_data(file)
|
||||||
|
conn.audio_play_queue.put((opus_packets, text, 0))
|
||||||
|
if time.time() - WAKEUP_CONFIG["create_time"] > WAKEUP_CONFIG["refresh_time"]:
|
||||||
|
asyncio.create_task(wakeupWordsResponse(conn))
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
def getWakeupWordFile(file_name):
|
||||||
|
for file in os.listdir(WAKEUP_CONFIG["dir"]):
|
||||||
|
if "my_"+file_name in file:
|
||||||
|
return f"config/assets/{file}"
|
||||||
|
|
||||||
|
"""查找config/assets/目录下名称为wakeup_words的文件"""
|
||||||
|
for file in os.listdir(WAKEUP_CONFIG["dir"]):
|
||||||
|
if file_name in file:
|
||||||
|
return f"config/assets/{file}"
|
||||||
|
return None
|
||||||
|
|
||||||
|
|
||||||
|
async def wakeupWordsResponse(conn):
|
||||||
|
"""唤醒词响应"""
|
||||||
|
wakeup_word = random.choice(WAKEUP_CONFIG["words"])
|
||||||
|
result = conn.llm.response_no_stream(conn.config["prompt"], wakeup_word)
|
||||||
|
tts_file = await asyncio.to_thread(conn.tts.to_tts, result)
|
||||||
|
if tts_file is not None and os.path.exists(tts_file):
|
||||||
|
file_type = os.path.splitext(tts_file)[1]
|
||||||
|
if file_type:
|
||||||
|
file_type = file_type.lstrip('.')
|
||||||
|
old_file = getWakeupWordFile("my_" +WAKEUP_CONFIG["file_name"])
|
||||||
|
if old_file is not None:
|
||||||
|
os.remove(old_file)
|
||||||
|
"""将文件挪到"wakeup_words.mp3"""
|
||||||
|
shutil.move(tts_file, WAKEUP_CONFIG["dir"] + "my_" +WAKEUP_CONFIG["file_name"] + "." + file_type)
|
||||||
|
WAKEUP_CONFIG["create_time"] = time.time()
|
||||||
|
|||||||
@@ -2,6 +2,7 @@ from config.logger import setup_logging
|
|||||||
import json
|
import json
|
||||||
import uuid
|
import uuid
|
||||||
from core.handle.sendAudioHandle import send_stt_message
|
from core.handle.sendAudioHandle import send_stt_message
|
||||||
|
from core.handle.helloHandle import checkWakeupWords
|
||||||
from core.utils.util import remove_punctuation_and_length
|
from core.utils.util import remove_punctuation_and_length
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
@@ -12,6 +13,10 @@ async def handle_user_intent(conn, text):
|
|||||||
# 检查是否有明确的退出命令
|
# 检查是否有明确的退出命令
|
||||||
if await check_direct_exit(conn, text):
|
if await check_direct_exit(conn, text):
|
||||||
return True
|
return True
|
||||||
|
# 检查是否是唤醒词
|
||||||
|
if await checkWakeupWords(conn, text):
|
||||||
|
return True
|
||||||
|
|
||||||
if conn.use_function_call_mode:
|
if conn.use_function_call_mode:
|
||||||
# 使用支持function calling的聊天方法,不再进行意图分析
|
# 使用支持function calling的聊天方法,不再进行意图分析
|
||||||
return False
|
return False
|
||||||
@@ -35,6 +40,7 @@ async def check_direct_exit(conn, text):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
|
|
||||||
async def analyze_intent_with_llm(conn, text):
|
async def analyze_intent_with_llm(conn, text):
|
||||||
"""使用LLM分析用户意图"""
|
"""使用LLM分析用户意图"""
|
||||||
if not hasattr(conn, 'intent') or not conn.intent:
|
if not hasattr(conn, 'intent') or not conn.intent:
|
||||||
|
|||||||
@@ -7,12 +7,26 @@ from core.utils.util import remove_punctuation_and_length, get_string_no_punctua
|
|||||||
TAG = __name__
|
TAG = __name__
|
||||||
logger = setup_logging()
|
logger = setup_logging()
|
||||||
|
|
||||||
|
|
||||||
async def sendAudioMessage(conn, audios, text, text_index=0):
|
async def sendAudioMessage(conn, audios, text, text_index=0):
|
||||||
# 发送句子开始消息
|
# 发送句子开始消息
|
||||||
if text_index == conn.tts_first_text_index:
|
if text_index == conn.tts_first_text_index:
|
||||||
logger.bind(tag=TAG).info(f"发送第一段语音: {text}")
|
logger.bind(tag=TAG).info(f"发送第一段语音: {text}")
|
||||||
await send_tts_message(conn, "sentence_start", text)
|
await send_tts_message(conn, "sentence_start", text)
|
||||||
|
|
||||||
|
# 播放音频
|
||||||
|
await sendAudio(conn, audios)
|
||||||
|
|
||||||
|
await send_tts_message(conn, "sentence_end", text)
|
||||||
|
|
||||||
|
# 发送结束消息(如果是最后一个文本)
|
||||||
|
if conn.llm_finish_task and text_index == conn.tts_last_text_index:
|
||||||
|
await send_tts_message(conn, 'stop', None)
|
||||||
|
if conn.close_after_chat:
|
||||||
|
await conn.close()
|
||||||
|
|
||||||
|
# 播放音频
|
||||||
|
async def sendAudio(conn, audios):
|
||||||
# 流控参数优化
|
# 流控参数优化
|
||||||
original_frame_duration = 60 # 原始帧时长(毫秒)
|
original_frame_duration = 60 # 原始帧时长(毫秒)
|
||||||
adjusted_frame_duration = int(original_frame_duration * 0.8) # 缩短20%
|
adjusted_frame_duration = int(original_frame_duration * 0.8) # 缩短20%
|
||||||
@@ -42,13 +56,6 @@ async def sendAudioMessage(conn, audios, text, text_index=0):
|
|||||||
if compensation > 0:
|
if compensation > 0:
|
||||||
await asyncio.sleep(compensation)
|
await asyncio.sleep(compensation)
|
||||||
|
|
||||||
await send_tts_message(conn, "sentence_end", text)
|
|
||||||
|
|
||||||
# 发送结束消息(如果是最后一个文本)
|
|
||||||
if conn.llm_finish_task and text_index == conn.tts_last_text_index:
|
|
||||||
await send_tts_message(conn, 'stop', None)
|
|
||||||
if conn.close_after_chat:
|
|
||||||
await conn.close()
|
|
||||||
|
|
||||||
async def send_tts_message(conn, state, text=None):
|
async def send_tts_message(conn, state, text=None):
|
||||||
"""发送 TTS 状态消息"""
|
"""发送 TTS 状态消息"""
|
||||||
@@ -60,10 +67,20 @@ async def send_tts_message(conn, state, text=None):
|
|||||||
if text is not None:
|
if text is not None:
|
||||||
message["text"] = text
|
message["text"] = text
|
||||||
|
|
||||||
await conn.websocket.send(json.dumps(message))
|
# TTS播放结束
|
||||||
if state == "stop":
|
if state == "stop":
|
||||||
|
# 播放提示音
|
||||||
|
tts_notify = conn.config.get("enable_stop_tts_notify", False)
|
||||||
|
if tts_notify:
|
||||||
|
stop_tts_notify_voice = conn.config.get("stop_tts_notify_voice", "config/assets/tts_notify.mp3")
|
||||||
|
audios, duration = conn.tts.audio_to_opus_data(stop_tts_notify_voice)
|
||||||
|
await sendAudio(conn, audios)
|
||||||
|
# 清除服务端讲话状态
|
||||||
conn.clearSpeakStatus()
|
conn.clearSpeakStatus()
|
||||||
|
|
||||||
|
# 发送消息到客户端
|
||||||
|
await conn.websocket.send(json.dumps(message))
|
||||||
|
|
||||||
|
|
||||||
async def send_stt_message(conn, text):
|
async def send_stt_message(conn, text):
|
||||||
"""发送 STT 状态消息"""
|
"""发送 STT 状态消息"""
|
||||||
|
|||||||
@@ -2,7 +2,9 @@ from config.logger import setup_logging
|
|||||||
import json
|
import json
|
||||||
from core.handle.abortHandle import handleAbortMessage
|
from core.handle.abortHandle import handleAbortMessage
|
||||||
from core.handle.helloHandle import handleHelloMessage
|
from core.handle.helloHandle import handleHelloMessage
|
||||||
|
from core.utils.util import remove_punctuation_and_length
|
||||||
from core.handle.receiveAudioHandle import startToChat, handleAudioMessage
|
from core.handle.receiveAudioHandle import startToChat, handleAudioMessage
|
||||||
|
from core.handle.sendAudioHandle import send_stt_message, send_tts_message
|
||||||
from core.handle.iotHandle import handleIotDescriptors, handleIotStatus
|
from core.handle.iotHandle import handleIotDescriptors, handleIotStatus
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
@@ -38,7 +40,21 @@ async def handleTextMessage(conn, message):
|
|||||||
conn.client_have_voice = False
|
conn.client_have_voice = False
|
||||||
conn.asr_audio.clear()
|
conn.asr_audio.clear()
|
||||||
if "text" in msg_json:
|
if "text" in msg_json:
|
||||||
await startToChat(conn, msg_json["text"])
|
text = msg_json["text"]
|
||||||
|
_, text = remove_punctuation_and_length(text)
|
||||||
|
|
||||||
|
# 识别是否是唤醒词
|
||||||
|
is_wakeup_words = text in conn.config.get("wakeup_words")
|
||||||
|
# 是否开启唤醒词回复
|
||||||
|
enable_greeting = conn.config.get("enable_greeting", True)
|
||||||
|
|
||||||
|
if is_wakeup_words and not enable_greeting:
|
||||||
|
# 如果是唤醒词,且关闭了唤醒词回复,就不用回答
|
||||||
|
await send_stt_message(conn, text)
|
||||||
|
await send_tts_message(conn, "stop", None)
|
||||||
|
else:
|
||||||
|
# 否则需要LLM对文字内容进行答复
|
||||||
|
await startToChat(conn, text)
|
||||||
elif msg_json["type"] == "iot":
|
elif msg_json["type"] == "iot":
|
||||||
if "descriptors" in msg_json:
|
if "descriptors" in msg_json:
|
||||||
await handleIotDescriptors(conn, msg_json["descriptors"])
|
await handleIotDescriptors(conn, msg_json["descriptors"])
|
||||||
|
|||||||
@@ -36,7 +36,7 @@ class TTSProviderBase(ABC):
|
|||||||
|
|
||||||
return tmp_file
|
return tmp_file
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.bind(tag=TAG).info(f"Failed to generate TTS file: {e}")
|
logger.bind(tag=TAG).error(f"Failed to generate TTS file: {e}")
|
||||||
return None
|
return None
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
|
|||||||
@@ -73,9 +73,7 @@ def get_ip_info(ip_addr):
|
|||||||
resp = requests.get(url).json()
|
resp = requests.get(url).json()
|
||||||
|
|
||||||
ip_info = {
|
ip_info = {
|
||||||
"city": resp.get("cityName"),
|
"city": resp.get("cityName")
|
||||||
"region": resp.get("regionName"),
|
|
||||||
"country": resp.get("countryName")
|
|
||||||
}
|
}
|
||||||
return ip_info
|
return ip_info
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
|
|||||||
@@ -23,5 +23,3 @@ def auto_import_modules(package_name):
|
|||||||
full_module_name = f"{package_name}.{module_name}"
|
full_module_name = f"{package_name}.{module_name}"
|
||||||
importlib.import_module(full_module_name)
|
importlib.import_module(full_module_name)
|
||||||
#logger.bind(tag=TAG).info(f"模块 '{full_module_name}' 已加载")
|
#logger.bind(tag=TAG).info(f"模块 '{full_module_name}' 已加载")
|
||||||
|
|
||||||
auto_import_modules('plugins_func.functions')
|
|
||||||
Reference in New Issue
Block a user