mirror of
https://github.com/xinnan-tech/xiaozhi-esp32-server.git
synced 2026-07-29 18:23:59 +08:00
add:支持手动按住说话
This commit is contained in:
@@ -27,16 +27,16 @@
|
|||||||
## 已实现
|
## 已实现
|
||||||
|
|
||||||
- `xiaozhi-esp32` 通信 WebSocket 协议
|
- `xiaozhi-esp32` 通信 WebSocket 协议
|
||||||
- 支持实时打断对话,试一下在聊天的时候说一句`你好小智`
|
- 支持唤醒对话、手动对话、实时打断对话
|
||||||
- 支持国语、粤语、英语、日语、韩语 5 种语言识别(FunASR(默认))
|
- 支持国语、粤语、英语、日语、韩语 5 种语言识别(FunASR(默认))
|
||||||
- 自由更换 LLM(支持ChatGLM(默认)、Dify、DeepSeek)
|
- 自由更换 LLM(支持ChatGLM(默认)、Dify、DeepSeek)
|
||||||
- 自由更换 TTS(支持EdgeTTS(默认)、火山引擎豆包TTS)
|
- 自由更换 TTS(支持EdgeTTS(默认)、火山引擎豆包TTS)
|
||||||
|
|
||||||
## 正在实现
|
## 正在实现
|
||||||
|
|
||||||
- 按键手动对话
|
|
||||||
- 长时间不聊天进入休眠状态
|
- 长时间不聊天进入休眠状态
|
||||||
- 对话记忆
|
- 对话记忆
|
||||||
|
- 更换心情模式
|
||||||
|
|
||||||
## 本项目依赖服务
|
## 本项目依赖服务
|
||||||
|
|
||||||
|
|||||||
+2
-2
@@ -32,16 +32,16 @@ To fully experience this project, follow these steps:
|
|||||||
## Implemented
|
## Implemented
|
||||||
|
|
||||||
- `xiaozhi-esp32` WebSocket communication protocol
|
- `xiaozhi-esp32` WebSocket communication protocol
|
||||||
- Support interrupt dialogue in real time, try to say a sentence when chatting `你好小智`
|
- Supports wake-word initiated dialogue, manual dialogue, and real-time interruption of dialogue.
|
||||||
- Support for 5 languages: Mandarin, Cantonese, English, Japanese, Korean (FunASR - default)
|
- Support for 5 languages: Mandarin, Cantonese, English, Japanese, Korean (FunASR - default)
|
||||||
- Flexible LLM switching (ChatGLM - default, Dify, DeepSeek)
|
- Flexible LLM switching (ChatGLM - default, Dify, DeepSeek)
|
||||||
- Flexible TTS switching (EdgeTTS - default, ByteDance Doubao TTS)
|
- Flexible TTS switching (EdgeTTS - default, ByteDance Doubao TTS)
|
||||||
|
|
||||||
## In Progress
|
## In Progress
|
||||||
|
|
||||||
- Manual button-triggered dialogue
|
|
||||||
- Sleep mode after inactivity
|
- Sleep mode after inactivity
|
||||||
- Dialogue memory
|
- Dialogue memory
|
||||||
|
- Change the mood mode
|
||||||
|
|
||||||
## Dependencies
|
## Dependencies
|
||||||
|
|
||||||
|
|||||||
+4
-15
@@ -12,8 +12,6 @@ from collections import deque
|
|||||||
from core.utils.util import is_segment
|
from core.utils.util import is_segment
|
||||||
from core.utils.dialogue import Message, Dialogue
|
from core.utils.dialogue import Message, Dialogue
|
||||||
from core.handle.textHandle import handleTextMessage
|
from core.handle.textHandle import handleTextMessage
|
||||||
from core.handle.abortHandle import handleAbortMessage
|
|
||||||
from core.handle.helloHandle import handleHelloMessage
|
|
||||||
from core.utils.util import get_string_no_punctuation_or_emoji
|
from core.utils.util import get_string_no_punctuation_or_emoji
|
||||||
from concurrent.futures import ThreadPoolExecutor, TimeoutError
|
from concurrent.futures import ThreadPoolExecutor, TimeoutError
|
||||||
from core.handle.audioHandle import handleAudioMessage, sendAudioMessage
|
from core.handle.audioHandle import handleAudioMessage, sendAudioMessage
|
||||||
@@ -29,7 +27,10 @@ class ConnectionHandler:
|
|||||||
self.session_id = None
|
self.session_id = None
|
||||||
self.prompt = None
|
self.prompt = None
|
||||||
self.welcome_msg = None
|
self.welcome_msg = None
|
||||||
|
|
||||||
|
# 客户端状态相关
|
||||||
self.client_abort = False
|
self.client_abort = False
|
||||||
|
self.client_listen_mode = "auto"
|
||||||
|
|
||||||
# 线程任务相关
|
# 线程任务相关
|
||||||
self.loop = asyncio.get_event_loop()
|
self.loop = asyncio.get_event_loop()
|
||||||
@@ -91,22 +92,10 @@ class ConnectionHandler:
|
|||||||
async def _route_message(self, message):
|
async def _route_message(self, message):
|
||||||
"""消息路由"""
|
"""消息路由"""
|
||||||
if isinstance(message, str):
|
if isinstance(message, str):
|
||||||
await self._handle_text(message)
|
await handleTextMessage(self, message)
|
||||||
elif isinstance(message, bytes):
|
elif isinstance(message, bytes):
|
||||||
await handleAudioMessage(self, message)
|
await handleAudioMessage(self, message)
|
||||||
|
|
||||||
async def _handle_text(self, message):
|
|
||||||
"""处理文本消息"""
|
|
||||||
self.logger.info(f"收到文本消息:{message}")
|
|
||||||
try:
|
|
||||||
msg_json = json.loads(message)
|
|
||||||
if msg_json["type"] == "hello":
|
|
||||||
await handleHelloMessage(self, "你好")
|
|
||||||
if msg_json["type"] == "abort":
|
|
||||||
await handleAbortMessage(self)
|
|
||||||
except json.JSONDecodeError:
|
|
||||||
await handleTextMessage(self, message)
|
|
||||||
|
|
||||||
def _initialize_components(self):
|
def _initialize_components(self):
|
||||||
self.prompt = self.config["prompt"]
|
self.prompt = self.config["prompt"]
|
||||||
# 赋予LLM时间观念
|
# 赋予LLM时间观念
|
||||||
|
|||||||
+21
-15
@@ -11,7 +11,10 @@ async def handleAudioMessage(conn, audio):
|
|||||||
if not conn.asr_server_receive:
|
if not conn.asr_server_receive:
|
||||||
logger.debug(f"前期数据处理中,暂停接收")
|
logger.debug(f"前期数据处理中,暂停接收")
|
||||||
return
|
return
|
||||||
have_voice = conn.vad.is_vad(conn, audio)
|
if conn.client_listen_mode == "auto":
|
||||||
|
have_voice = conn.vad.is_vad(conn, audio)
|
||||||
|
else:
|
||||||
|
have_voice = conn.client_have_voice
|
||||||
|
|
||||||
# 如果本次没有声音,本段也没声音,就把声音丢弃了
|
# 如果本次没有声音,本段也没声音,就把声音丢弃了
|
||||||
if have_voice == False and conn.client_have_voice == False:
|
if have_voice == False and conn.client_have_voice == False:
|
||||||
@@ -26,25 +29,28 @@ async def handleAudioMessage(conn, audio):
|
|||||||
logger.info(f"识别文本: {text}")
|
logger.info(f"识别文本: {text}")
|
||||||
text_len = remove_punctuation_and_length(text)
|
text_len = remove_punctuation_and_length(text)
|
||||||
if text_len > 0:
|
if text_len > 0:
|
||||||
stt_text = get_string_no_punctuation_or_emoji(text)
|
await startToChat(conn, text)
|
||||||
await conn.websocket.send(json.dumps({
|
|
||||||
"type": "stt",
|
|
||||||
"text": stt_text,
|
|
||||||
"session_id": conn.session_id}
|
|
||||||
))
|
|
||||||
await conn.websocket.send(
|
|
||||||
json.dumps({
|
|
||||||
"type": "llm",
|
|
||||||
"text": "😊",
|
|
||||||
"emotion": "happy",
|
|
||||||
"session_id": conn.session_id}
|
|
||||||
))
|
|
||||||
conn.executor.submit(conn.chat, text)
|
|
||||||
else:
|
else:
|
||||||
conn.asr_server_receive = True
|
conn.asr_server_receive = True
|
||||||
conn.asr_audio.clear()
|
conn.asr_audio.clear()
|
||||||
conn.reset_vad_states()
|
conn.reset_vad_states()
|
||||||
|
|
||||||
|
async def startToChat(conn, text):
|
||||||
|
stt_text = get_string_no_punctuation_or_emoji(text)
|
||||||
|
await conn.websocket.send(json.dumps({
|
||||||
|
"type": "stt",
|
||||||
|
"text": stt_text,
|
||||||
|
"session_id": conn.session_id}
|
||||||
|
))
|
||||||
|
await conn.websocket.send(
|
||||||
|
json.dumps({
|
||||||
|
"type": "llm",
|
||||||
|
"text": "😊",
|
||||||
|
"emotion": "happy",
|
||||||
|
"session_id": conn.session_id}
|
||||||
|
))
|
||||||
|
conn.executor.submit(conn.chat, text)
|
||||||
|
|
||||||
|
|
||||||
async def sendAudioMessage(conn, audios, duration, text):
|
async def sendAudioMessage(conn, audios, duration, text):
|
||||||
base_delay = conn.tts_duration
|
base_delay = conn.tts_duration
|
||||||
|
|||||||
@@ -1,7 +1,37 @@
|
|||||||
import logging
|
import logging
|
||||||
|
import json
|
||||||
|
from core.handle.abortHandle import handleAbortMessage
|
||||||
|
from core.handle.helloHandle import handleHelloMessage
|
||||||
|
from core.handle.audioHandle import startToChat
|
||||||
|
|
||||||
logger = logging.getLogger(__name__)
|
logger = logging.getLogger(__name__)
|
||||||
|
|
||||||
|
|
||||||
async def handleTextMessage(conn, message):
|
async def handleTextMessage(conn, message):
|
||||||
|
"""处理文本消息"""
|
||||||
|
logger.info(f"收到文本消息:{message}")
|
||||||
|
try:
|
||||||
|
msg_json = json.loads(message)
|
||||||
|
if msg_json["type"] == "hello":
|
||||||
|
await handleHelloMessage(conn, "你好")
|
||||||
|
elif msg_json["type"] == "abort":
|
||||||
|
await handleAbortMessage(conn)
|
||||||
|
elif msg_json["type"] == "listen":
|
||||||
|
if "mode" in msg_json:
|
||||||
|
conn.client_listen_mode = msg_json["mode"]
|
||||||
|
logger.info(f"客户端拾音模式:{conn.client_listen_mode}")
|
||||||
|
if msg_json["state"] == "start":
|
||||||
|
conn.client_have_voice = True
|
||||||
|
conn.client_voice_stop = False
|
||||||
|
elif msg_json["state"] == "stop":
|
||||||
|
conn.client_have_voice = True
|
||||||
|
conn.client_voice_stop = True
|
||||||
|
elif msg_json["state"] == "detect":
|
||||||
|
conn.asr_server_receive = False
|
||||||
|
conn.client_have_voice = False
|
||||||
|
conn.asr_audio.clear()
|
||||||
|
if "text" in msg_json:
|
||||||
|
await startToChat(conn, msg_json["text"])
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
await handleTextMessage(conn, message)
|
||||||
await conn.websocket.send(message)
|
await conn.websocket.send(message)
|
||||||
|
|||||||
Reference in New Issue
Block a user