diff --git a/.github/workflows/docker-image.yml b/.github/workflows/docker-image.yml index e54cdd4e..aefa77d0 100644 --- a/.github/workflows/docker-image.yml +++ b/.github/workflows/docker-image.yml @@ -31,6 +31,9 @@ jobs: - name: Set up Docker Buildx uses: docker/setup-buildx-action@v3 + with: + driver-opts: | + network=host - name: Login to GitHub Container Registry uses: docker/login-action@v3 @@ -60,6 +63,10 @@ jobs: tags: | ${{ env.IS_VERSION == 'true' && format('ghcr.io/{0}:server_{1},ghcr.io/{0}:server_latest', github.repository, env.VERSION) || format('ghcr.io/{0}:server_latest', github.repository) }} platforms: linux/amd64,linux/arm64 + cache-from: type=gha + cache-to: type=gha,mode=max + build-args: | + BUILDKIT_PROGRESS=plain # 构建 manager-api 镜像 - name: Build and push manager-web @@ -70,4 +77,8 @@ jobs: push: true tags: | ${{ env.IS_VERSION == 'true' && format('ghcr.io/{0}:web_{1},ghcr.io/{0}:web_latest', github.repository, env.VERSION) || format('ghcr.io/{0}:web_latest', github.repository) }} - platforms: linux/amd64,linux/arm64 \ No newline at end of file + platforms: linux/amd64,linux/arm64 + cache-from: type=gha + cache-to: type=gha,mode=max + build-args: | + BUILDKIT_PROGRESS=plain \ No newline at end of file diff --git a/Dockerfile-server b/Dockerfile-server index 6e87f7c0..a12fbb11 100644 --- a/Dockerfile-server +++ b/Dockerfile-server @@ -3,10 +3,17 @@ FROM python:3.10-slim AS builder WORKDIR /app +# 配置pip使用国内镜像源(阿里云)并设置超时和重试 +RUN pip config set global.index-url https://mirrors.aliyun.com/pypi/simple/ && \ + pip config set global.trusted-host mirrors.aliyun.com && \ + pip config set global.timeout 120 && \ + pip config set install.retries 5 + COPY main/xiaozhi-server/requirements.txt . -# 安装Python依赖 -RUN pip install --no-cache-dir -r requirements.txt +# 安装Python依赖,使用并行下载 +RUN pip install --no-cache-dir --upgrade pip setuptools wheel && \ + pip install --no-cache-dir -r requirements.txt --default-timeout=120 --retries 5 # 第二阶段:生产镜像 FROM python:3.10-slim diff --git a/Dockerfile-web b/Dockerfile-web index 35c25ed2..a347a88e 100644 --- a/Dockerfile-web +++ b/Dockerfile-web @@ -18,12 +18,12 @@ FROM bellsoft/liberica-runtime-container:jre-21-glibc # 安装Nginx和字体库 RUN apk update && \ - apk add --no-cache nginx bash && \ - apk add --no-cache fontconfig ttf-dejavu msttcorefonts-installer && \ + apk add --no-cache nginx bash fontconfig ttf-dejavu && \ + apk add --no-cache --repository=http://dl-cdn.alpinelinux.org/alpine/edge/testing/ msttcorefonts-installer || true && \ rm -rf /var/cache/apk/* # 更新字体缓存 -RUN printf 'YES\n' | update-ms-fonts && fc-cache -f -v +RUN (printf 'YES\n' | update-ms-fonts || true) && fc-cache -f -v # 配置Nginx COPY docs/docker/nginx.conf /etc/nginx/nginx.conf diff --git a/README.md b/README.md index aa5ce96e..862e989a 100644 --- a/README.md +++ b/README.md @@ -283,6 +283,8 @@ Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/ | dify 接口调用 | Dify | - | | fastgpt 接口调用 | Fastgpt | - | | coze 接口调用 | Coze | - | +| xinference 接口调用 | Xinference | - | +| homeassistant 接口调用 | HomeAssistant | - | 实际上,任何支持 openai 接口调用的 LLM 均可接入使用。 @@ -302,8 +304,8 @@ Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/ | 使用方式 | 支持平台 | 免费平台 | |:---:|:---:|:---:| -| 接口调用 | EdgeTTS、火山引擎豆包TTS、腾讯云、阿里云TTS、CosyVoiceSiliconflow、TTS302AI、CozeCnTTS、GizwitsTTS、ACGNTTS、OpenAITTS、灵犀流式TTS | 灵犀流式TTS、EdgeTTS、CosyVoiceSiliconflow(部分) | -| 本地服务 | FishSpeech、GPT_SOVITS_V2、GPT_SOVITS_V3、MinimaxTTS | FishSpeech、GPT_SOVITS_V2、GPT_SOVITS_V3、MinimaxTTS | +| 接口调用 | EdgeTTS、火山引擎豆包TTS、腾讯云、阿里云TTS、阿里云流式TTS、CosyVoiceSiliconflow、TTS302AI、CozeCnTTS、GizwitsTTS、ACGNTTS、OpenAITTS、灵犀流式TTS、MinimaxTTS、火山双流式TTS | 灵犀流式TTS、EdgeTTS、CosyVoiceSiliconflow(部分) | +| 本地服务 | FishSpeech、GPT_SOVITS_V2、GPT_SOVITS_V3、Index-TTS、PaddleSpeech | Index-TTS、PaddleSpeech、FishSpeech、GPT_SOVITS_V2、GPT_SOVITS_V3 | --- @@ -320,7 +322,7 @@ Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/ | 使用方式 | 支持平台 | 免费平台 | |:---:|:---:|:---:| | 本地使用 | FunASR、SherpaASR | FunASR、SherpaASR | -| 接口调用 | DoubaoASR、FunASRServer、TencentASR、AliyunASR | FunASRServer | +| 接口调用 | DoubaoASR、Doubao流式ASR、FunASRServer、TencentASR、AliyunASR、Aliyun流式ASR、百度ASR、OpenAI ASR | FunASRServer | --- @@ -338,6 +340,7 @@ Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/ |:------:|:---------------:|:----:|:---------:|:--:| | Memory | mem0ai | 接口调用 | 1000次/月额度 | | | Memory | mem_local_short | 本地总结 | 免费 | | +| Memory | nomem | 无记忆模式 | 免费 | | --- @@ -347,6 +350,7 @@ Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/ |:------:|:-------------:|:----:|:-------:|:---------------------:| | Intent | intent_llm | 接口调用 | 根据LLM收费 | 通过大模型识别意图,通用性强 | | Intent | function_call | 接口调用 | 根据LLM收费 | 通过大模型函数调用完成意图,速度快,效果好 | +| Intent | nointent | 无意图模式 | 免费 | 不进行意图识别,直接返回对话结果 | --- diff --git a/docs/Deployment_all.md b/docs/Deployment_all.md index 135b7cf5..96b3d75c 100644 --- a/docs/Deployment_all.md +++ b/docs/Deployment_all.md @@ -476,7 +476,8 @@ ws://你电脑局域网的ip:8000/xiaozhi/v1/ 4、[如何部署MCP接入点](./mcp-endpoint-enable.md)
5、[如何接入MCP接入点](./mcp-endpoint-integration.md)
6、[如何开启声纹识别](./voiceprint-integration.md)
-10、[新闻插件源配置指南](./newsnow_plugin_config.md)
+7、[新闻插件源配置指南](./newsnow_plugin_config.md)
+8、[天气插件使用指南](./weather-integration.md)
## 语音克隆、本地语音部署相关教程 1、[如何部署集成index-tts本地语音](./index-stream-integration.md)
2、[如何部署集成fish-speech本地语音](./fish-speech-integration.md)
diff --git a/docs/weather-integration.md b/docs/weather-integration.md new file mode 100644 index 00000000..3b6ca2b6 --- /dev/null +++ b/docs/weather-integration.md @@ -0,0 +1,64 @@ +# 天气插件使用指南 + +## 概述 + +天气插件 `get_weather` 是小智ESP32语音助手的核心功能之一,支持通过语音查询全国各地的天气信息。插件基于和风天气API,提供实时天气和7天天气预报功能。 + +## API Key 申请指南 + +### 1. 注册和风天气账号 + +1. 访问 [和风天气控制台](https://console.qweather.com/) +2. 注册账号并完成邮箱验证 +3. 登录控制台 + +### 2. 创建应用获取API Key + +1. 进入控制台后,点击右侧["项目管理"](https://console.qweather.com/project?lang=zh) → "创建项目" +2. 填写项目信息: + - **项目名称**:如"小智语音助手" +3. 点击保存 +4. 项目创建完成后,在该项目中点击"创建凭据" +5. 填写凭据信息: + - **凭据名称**:如"小智语音助手" + - **身份认证方式**:选择"API Key" +6. 点击保存 +7. 在凭据中复制`API Key`,这是第一个关键的配置信息 + +### 3. 获取API Host + +1. 在控制台中点击["设置"](https://console.qweather.com/setting?lang=zh) → "API Host" +2. 查看分配给你的专属`API Host`地址,这个是第二个关键的配置信息 + +以上操作,会得到两个重要的配置信息:`API Key`和`API Host` + +## 配置方式(任选一种) + +### 方式1. 如果你使用了智控台部署(推荐) + +1. 登录智控台 +2. 进入"角色配置"页面 +3. 选择要配置的智能体 +4. 点击"编辑功能"按钮 +5. 在右侧参数配置区域找到"天气查询"插件 +6. 勾选"天气查询" +7. 将复制过来的第一个关键配置`API Key`,填入到`天气插件 API 密钥`里 +8. 将复制过来的第二个关键配置`API Host`,填入到`开发者 API Host`里 +9. 保存配置,再保存智能体配置 + +### 方式2. 如果你只是单模块xiaozhi-server部署 + +在 `data/.config.yaml` 中配置: + +1. 将复制过来的第一个关键配置`API Key`,填入到`api_key`里 +2. 将复制过来的第二个关键配置`API Host`,填入到`api_host`里 +3. 将你所在的城市填入到`default_location`里,例如`广州` + +```yaml +plugins: + get_weather: + api_key: "你的和风天气API密钥" + api_host: "你的和风天气API主机地址" + default_location: "你的默认查询城市" +``` + diff --git a/main/manager-api/src/main/java/xiaozhi/common/constant/Constant.java b/main/manager-api/src/main/java/xiaozhi/common/constant/Constant.java index bf747600..ba6567a4 100644 --- a/main/manager-api/src/main/java/xiaozhi/common/constant/Constant.java +++ b/main/manager-api/src/main/java/xiaozhi/common/constant/Constant.java @@ -237,7 +237,7 @@ public interface Constant { /** * 版本号 */ - public static final String VERSION = "0.7.5"; + public static final String VERSION = "0.7.6"; /** * 无效固件URL diff --git a/main/manager-mobile/src/pages/settings/index.vue b/main/manager-mobile/src/pages/settings/index.vue index 41d16c00..0cecd6bf 100644 --- a/main/manager-mobile/src/pages/settings/index.vue +++ b/main/manager-mobile/src/pages/settings/index.vue @@ -222,7 +222,7 @@ function showAbout() { title: `关于${import.meta.env.VITE_APP_TITLE}`, content: `${import.meta.env.VITE_APP_TITLE}\n\n基于 Vue.js 3 + uni-app 构建的跨平台移动端管理应用,为小智ESP32智能硬件提供设备管理、智能体配置等功能。\n\n© 2025 xiaozhi-esp32-server`, title: `关于小智智控台`, - content: `小智智控台\n\n基于 Vue.js 3 + uni-app 构建的跨平台移动端管理应用,为小智ESP32智能硬件提供设备管理、智能体配置等功能。\n\n© 2025 xiaozhi-esp32-server 0.7.5`, + content: `小智智控台\n\n基于 Vue.js 3 + uni-app 构建的跨平台移动端管理应用,为小智ESP32智能硬件提供设备管理、智能体配置等功能。\n\n© 2025 xiaozhi-esp32-server 0.7.6`, showCancel: false, confirmText: '确定', }) diff --git a/main/manager-web/src/components/FunctionDialog.vue b/main/manager-web/src/components/FunctionDialog.vue index 76c69c8e..f8adfef7 100644 --- a/main/manager-web/src/components/FunctionDialog.vue +++ b/main/manager-web/src/components/FunctionDialog.vue @@ -698,15 +698,10 @@ export default { font-size: 14px; height: 36px; box-sizing: border-box; - background-color: #f5f5f5; -} -::v-deep .el-input__inner { - background-color: #f5f5f5; - padding-right: 80px; -} - -.url-input { + ::v-deep .el-input__inner { + background-color: #f5f5f5 !important; + } ::v-deep .el-input__suffix { right: 0; diff --git a/main/xiaozhi-server/config/logger.py b/main/xiaozhi-server/config/logger.py index cd4d9965..5a815f06 100644 --- a/main/xiaozhi-server/config/logger.py +++ b/main/xiaozhi-server/config/logger.py @@ -5,7 +5,7 @@ from config.config_loader import load_config from config.settings import check_config_file from datetime import datetime -SERVER_VERSION = "0.7.5" +SERVER_VERSION = "0.7.6" _logger_initialized = False diff --git a/main/xiaozhi-server/core/handle/textHandle.py b/main/xiaozhi-server/core/handle/textHandle.py index f12b974a..b5e87783 100644 --- a/main/xiaozhi-server/core/handle/textHandle.py +++ b/main/xiaozhi-server/core/handle/textHandle.py @@ -1,163 +1,14 @@ -import json -import asyncio -from core.utils.util import filter_sensitive_info -from core.handle.abortHandle import handleAbortMessage -from core.handle.helloHandle import handleHelloMessage -from core.handle.reportHandle import enqueue_asr_report -from core.providers.tools.device_mcp import handle_mcp_message -from core.handle.receiveAudioHandle import startToChat, handleAudioMessage -from core.handle.sendAudioHandle import send_stt_message, send_tts_message -from core.providers.tools.device_iot import handleIotDescriptors, handleIotStatus +from core.handle.textMessageHandlerRegistry import TextMessageHandlerRegistry +from core.handle.textMessageProcessor import TextMessageProcessor TAG = __name__ +# 全局处理器注册表 +message_registry = TextMessageHandlerRegistry() + +# 创建全局消息处理器实例 +message_processor = TextMessageProcessor(message_registry) async def handleTextMessage(conn, message): """处理文本消息""" - try: - msg_json = json.loads(message) - if isinstance(msg_json, int): - conn.logger.bind(tag=TAG).info(f"收到文本消息:{message}") - await conn.websocket.send(message) - return - if msg_json["type"] == "hello": - conn.logger.bind(tag=TAG).info(f"收到hello消息:{message}") - await handleHelloMessage(conn, msg_json) - elif msg_json["type"] == "abort": - conn.logger.bind(tag=TAG).info(f"收到abort消息:{message}") - await handleAbortMessage(conn) - elif msg_json["type"] == "listen": - conn.logger.bind(tag=TAG).info(f"收到listen消息:{message}") - if "mode" in msg_json: - conn.client_listen_mode = msg_json["mode"] - conn.logger.bind(tag=TAG).debug( - f"客户端拾音模式:{conn.client_listen_mode}" - ) - if msg_json["state"] == "start": - conn.client_have_voice = True - conn.client_voice_stop = False - elif msg_json["state"] == "stop": - conn.client_have_voice = True - conn.client_voice_stop = True - if len(conn.asr_audio) > 0: - await handleAudioMessage(conn, b"") - elif msg_json["state"] == "detect": - conn.client_have_voice = False - conn.asr_audio.clear() - if "text" in msg_json: - original_text = msg_json["text"] # 保留设备上传的文本 - # 识别是否是唤醒词 - is_wakeup_words = original_text in conn.config.get("wakeup_words") - # 是否开启唤醒词回复 - enable_greeting = conn.config.get("enable_greeting", True) - - if is_wakeup_words and not enable_greeting: - # 如果是唤醒词,且关闭了唤醒词回复,就不用回答 - await send_stt_message(conn, original_text) - await send_tts_message(conn, "stop", None) - conn.client_is_speaking = False - elif is_wakeup_words and enable_greeting: - conn.just_woken_up = True - # 上报纯文字数据(复用ASR上报功能,但不提供音频数据) - enqueue_asr_report(conn, "嘿,你好呀", []) - await startToChat(conn, "嘿,你好呀") - else: - # 上报纯文字数据(复用ASR上报功能,但不提供音频数据) - enqueue_asr_report(conn, original_text, []) - # 否则需要LLM对文字内容进行答复 - await startToChat(conn, original_text) - elif msg_json["type"] == "iot": - conn.logger.bind(tag=TAG).info(f"收到iot消息:{message}") - if "descriptors" in msg_json: - asyncio.create_task(handleIotDescriptors(conn, msg_json["descriptors"])) - if "states" in msg_json: - asyncio.create_task(handleIotStatus(conn, msg_json["states"])) - elif msg_json["type"] == "mcp": - conn.logger.bind(tag=TAG).info(f"收到mcp消息:{message[:100]}") - if "payload" in msg_json: - asyncio.create_task( - handle_mcp_message(conn, conn.mcp_client, msg_json["payload"]) - ) - elif msg_json["type"] == "server": - # 记录日志时过滤敏感信息 - conn.logger.bind(tag=TAG).info( - f"收到服务器消息:{filter_sensitive_info(msg_json)}" - ) - # 如果配置是从API读取的,则需要验证secret - if not conn.read_config_from_api: - return - # 获取post请求的secret - post_secret = msg_json.get("content", {}).get("secret", "") - secret = conn.config["manager-api"].get("secret", "") - # 如果secret不匹配,则返回 - if post_secret != secret: - await conn.websocket.send( - json.dumps( - { - "type": "server", - "status": "error", - "message": "服务器密钥验证失败", - } - ) - ) - return - # 动态更新配置 - if msg_json["action"] == "update_config": - try: - # 更新WebSocketServer的配置 - if not conn.server: - await conn.websocket.send( - json.dumps( - { - "type": "server", - "status": "error", - "message": "无法获取服务器实例", - "content": {"action": "update_config"}, - } - ) - ) - return - - if not await conn.server.update_config(): - await conn.websocket.send( - json.dumps( - { - "type": "server", - "status": "error", - "message": "更新服务器配置失败", - "content": {"action": "update_config"}, - } - ) - ) - return - - # 发送成功响应 - await conn.websocket.send( - json.dumps( - { - "type": "server", - "status": "success", - "message": "配置更新成功", - "content": {"action": "update_config"}, - } - ) - ) - except Exception as e: - conn.logger.bind(tag=TAG).error(f"更新配置失败: {str(e)}") - await conn.websocket.send( - json.dumps( - { - "type": "server", - "status": "error", - "message": f"更新配置失败: {str(e)}", - "content": {"action": "update_config"}, - } - ) - ) - # 重启服务器 - elif msg_json["action"] == "restart": - await conn.handle_restart(msg_json) - else: - conn.logger.bind(tag=TAG).error(f"收到未知类型消息:{message}") - except json.JSONDecodeError: - await conn.websocket.send(message) + await message_processor.process_message(conn, message) diff --git a/main/xiaozhi-server/core/handle/textHandler/abortMessageHandler.py b/main/xiaozhi-server/core/handle/textHandler/abortMessageHandler.py new file mode 100644 index 00000000..dc540d24 --- /dev/null +++ b/main/xiaozhi-server/core/handle/textHandler/abortMessageHandler.py @@ -0,0 +1,16 @@ +from typing import Dict, Any + +from core.handle.abortHandle import handleAbortMessage +from core.handle.textMessageHandler import TextMessageHandler +from core.handle.textMessageType import TextMessageType + + +class AbortTextMessageHandler(TextMessageHandler): + """Abort消息处理器""" + + @property + def message_type(self) -> TextMessageType: + return TextMessageType.ABORT + + async def handle(self, conn, msg_json: Dict[str, Any]) -> None: + await handleAbortMessage(conn) diff --git a/main/xiaozhi-server/core/handle/textHandler/helloMessageHandler.py b/main/xiaozhi-server/core/handle/textHandler/helloMessageHandler.py new file mode 100644 index 00000000..1839814e --- /dev/null +++ b/main/xiaozhi-server/core/handle/textHandler/helloMessageHandler.py @@ -0,0 +1,16 @@ +from typing import Dict, Any + +from core.handle.helloHandle import handleHelloMessage +from core.handle.textMessageHandler import TextMessageHandler +from core.handle.textMessageType import TextMessageType + + +class HelloTextMessageHandler(TextMessageHandler): + """Hello消息处理器""" + + @property + def message_type(self) -> TextMessageType: + return TextMessageType.HELLO + + async def handle(self, conn, msg_json: Dict[str, Any]) -> None: + await handleHelloMessage(conn, msg_json) \ No newline at end of file diff --git a/main/xiaozhi-server/core/handle/textHandler/iotMessageHandler.py b/main/xiaozhi-server/core/handle/textHandler/iotMessageHandler.py new file mode 100644 index 00000000..335d08b0 --- /dev/null +++ b/main/xiaozhi-server/core/handle/textHandler/iotMessageHandler.py @@ -0,0 +1,20 @@ +import asyncio +from typing import Dict, Any + +from core.handle.textMessageHandler import TextMessageHandler +from core.handle.textMessageType import TextMessageType +from core.providers.tools.device_iot import handleIotStatus, handleIotDescriptors + + +class IotTextMessageHandler(TextMessageHandler): + """IOT消息处理器""" + + @property + def message_type(self) -> TextMessageType: + return TextMessageType.IOT + + async def handle(self, conn, msg_json: Dict[str, Any]) -> None: + if "descriptors" in msg_json: + asyncio.create_task(handleIotDescriptors(conn, msg_json["descriptors"])) + if "states" in msg_json: + asyncio.create_task(handleIotStatus(conn, msg_json["states"])) \ No newline at end of file diff --git a/main/xiaozhi-server/core/handle/textHandler/listenMessageHandler.py b/main/xiaozhi-server/core/handle/textHandler/listenMessageHandler.py new file mode 100644 index 00000000..97286dfe --- /dev/null +++ b/main/xiaozhi-server/core/handle/textHandler/listenMessageHandler.py @@ -0,0 +1,63 @@ +import time +from typing import Dict, Any + +from core.handle.receiveAudioHandle import handleAudioMessage, startToChat +from core.handle.reportHandle import enqueue_asr_report +from core.handle.sendAudioHandle import send_stt_message, send_tts_message +from core.handle.textMessageHandler import TextMessageHandler +from core.handle.textMessageType import TextMessageType +from core.utils.util import remove_punctuation_and_length + +TAG = __name__ + +class ListenTextMessageHandler(TextMessageHandler): + """Listen消息处理器""" + + @property + def message_type(self) -> TextMessageType: + return TextMessageType.LISTEN + + async def handle(self, conn, msg_json: Dict[str, Any]) -> None: + if "mode" in msg_json: + conn.client_listen_mode = msg_json["mode"] + conn.logger.bind(tag=TAG).debug( + f"客户端拾音模式:{conn.client_listen_mode}" + ) + if msg_json["state"] == "start": + conn.client_have_voice = True + conn.client_voice_stop = False + elif msg_json["state"] == "stop": + conn.client_have_voice = True + conn.client_voice_stop = True + if len(conn.asr_audio) > 0: + await handleAudioMessage(conn, b"") + elif msg_json["state"] == "detect": + conn.client_have_voice = False + conn.asr_audio.clear() + if "text" in msg_json: + conn.last_activity_time = time.time() * 1000 + original_text = msg_json["text"] # 保留原始文本 + filtered_len, filtered_text = remove_punctuation_and_length( + original_text + ) + + # 识别是否是唤醒词 + is_wakeup_words = filtered_text in conn.config.get("wakeup_words") + # 是否开启唤醒词回复 + enable_greeting = conn.config.get("enable_greeting", True) + + if is_wakeup_words and not enable_greeting: + # 如果是唤醒词,且关闭了唤醒词回复,就不用回答 + await send_stt_message(conn, original_text) + await send_tts_message(conn, "stop", None) + conn.client_is_speaking = False + elif is_wakeup_words: + conn.just_woken_up = True + # 上报纯文字数据(复用ASR上报功能,但不提供音频数据) + enqueue_asr_report(conn, "嘿,你好呀", []) + await startToChat(conn, "嘿,你好呀") + else: + # 上报纯文字数据(复用ASR上报功能,但不提供音频数据) + enqueue_asr_report(conn, original_text, []) + # 否则需要LLM对文字内容进行答复 + await startToChat(conn, original_text) \ No newline at end of file diff --git a/main/xiaozhi-server/core/handle/textHandler/mcpMessageHandler.py b/main/xiaozhi-server/core/handle/textHandler/mcpMessageHandler.py new file mode 100644 index 00000000..65876f24 --- /dev/null +++ b/main/xiaozhi-server/core/handle/textHandler/mcpMessageHandler.py @@ -0,0 +1,20 @@ +import asyncio +from typing import Dict, Any + +from core.handle.textMessageHandler import TextMessageHandler +from core.handle.textMessageType import TextMessageType +from core.providers.tools.device_mcp import handle_mcp_message + + +class McpTextMessageHandler(TextMessageHandler): + """MCP消息处理器""" + + @property + def message_type(self) -> TextMessageType: + return TextMessageType.MCP + + async def handle(self, conn, msg_json: Dict[str, Any]) -> None: + if "payload" in msg_json: + asyncio.create_task( + handle_mcp_message(conn, conn.mcp_client, msg_json["payload"]) + ) \ No newline at end of file diff --git a/main/xiaozhi-server/core/handle/textHandler/serverMessageHandler.py b/main/xiaozhi-server/core/handle/textHandler/serverMessageHandler.py new file mode 100644 index 00000000..b9a23588 --- /dev/null +++ b/main/xiaozhi-server/core/handle/textHandler/serverMessageHandler.py @@ -0,0 +1,92 @@ +import asyncio +import json +from typing import Dict, Any + +from core.handle.textMessageHandler import TextMessageHandler +from core.handle.textMessageType import TextMessageType +from core.providers.tools.device_mcp import handle_mcp_message + +TAG = __name__ + +class ServerTextMessageHandler(TextMessageHandler): + """MCP消息处理器""" + + @property + def message_type(self) -> TextMessageType: + return TextMessageType.SERVER + + async def handle(self, conn, msg_json: Dict[str, Any]) -> None: + # 如果配置是从API读取的,则需要验证secret + if not conn.read_config_from_api: + return + # 获取post请求的secret + post_secret = msg_json.get("content", {}).get("secret", "") + secret = conn.config["manager-api"].get("secret", "") + # 如果secret不匹配,则返回 + if post_secret != secret: + await conn.websocket.send( + json.dumps( + { + "type": "server", + "status": "error", + "message": "服务器密钥验证失败", + } + ) + ) + return + # 动态更新配置 + if msg_json["action"] == "update_config": + try: + # 更新WebSocketServer的配置 + if not conn.server: + await conn.websocket.send( + json.dumps( + { + "type": "server", + "status": "error", + "message": "无法获取服务器实例", + "content": {"action": "update_config"}, + } + ) + ) + return + + if not await conn.server.update_config(): + await conn.websocket.send( + json.dumps( + { + "type": "server", + "status": "error", + "message": "更新服务器配置失败", + "content": {"action": "update_config"}, + } + ) + ) + return + + # 发送成功响应 + await conn.websocket.send( + json.dumps( + { + "type": "server", + "status": "success", + "message": "配置更新成功", + "content": {"action": "update_config"}, + } + ) + ) + except Exception as e: + conn.logger.bind(tag=TAG).error(f"更新配置失败: {str(e)}") + await conn.websocket.send( + json.dumps( + { + "type": "server", + "status": "error", + "message": f"更新配置失败: {str(e)}", + "content": {"action": "update_config"}, + } + ) + ) + # 重启服务器 + elif msg_json["action"] == "restart": + await conn.handle_restart(msg_json) \ No newline at end of file diff --git a/main/xiaozhi-server/core/handle/textMessageHandler.py b/main/xiaozhi-server/core/handle/textMessageHandler.py new file mode 100644 index 00000000..f94a0bac --- /dev/null +++ b/main/xiaozhi-server/core/handle/textMessageHandler.py @@ -0,0 +1,21 @@ +from abc import abstractmethod, ABC +from typing import Dict, Any + +from core.handle.textMessageType import TextMessageType + +TAG = __name__ + + +class TextMessageHandler(ABC): + """消息处理器抽象基类""" + + @abstractmethod + async def handle(self, conn, msg_json: Dict[str, Any]) -> None: + """处理消息的抽象方法""" + pass + + @property + @abstractmethod + def message_type(self) -> TextMessageType: + """返回处理的消息类型""" + pass diff --git a/main/xiaozhi-server/core/handle/textMessageHandlerRegistry.py b/main/xiaozhi-server/core/handle/textMessageHandlerRegistry.py new file mode 100644 index 00000000..e90d7231 --- /dev/null +++ b/main/xiaozhi-server/core/handle/textMessageHandlerRegistry.py @@ -0,0 +1,45 @@ +from typing import Dict, Optional + +from core.handle.textHandler.abortMessageHandler import AbortTextMessageHandler +from core.handle.textHandler.helloMessageHandler import HelloTextMessageHandler +from core.handle.textHandler.iotMessageHandler import IotTextMessageHandler +from core.handle.textHandler.listenMessageHandler import ListenTextMessageHandler +from core.handle.textHandler.mcpMessageHandler import McpTextMessageHandler +from core.handle.textMessageHandler import TextMessageHandler +from core.handle.textHandler.serverMessageHandler import ServerTextMessageHandler + +TAG = __name__ + + +class TextMessageHandlerRegistry: + """消息处理器注册表""" + + def __init__(self): + self._handlers: Dict[str, TextMessageHandler] = {} + self._register_default_handlers() + + def _register_default_handlers(self) -> None: + """注册默认的消息处理器""" + handlers = [ + HelloTextMessageHandler(), + AbortTextMessageHandler(), + ListenTextMessageHandler(), + IotTextMessageHandler(), + McpTextMessageHandler(), + ServerTextMessageHandler(), + ] + + for handler in handlers: + self.register_handler(handler) + + def register_handler(self, handler: TextMessageHandler) -> None: + """注册消息处理器""" + self._handlers[handler.message_type.value] = handler + + def get_handler(self, message_type: str) -> Optional[TextMessageHandler]: + """获取消息处理器""" + return self._handlers.get(message_type) + + def get_supported_types(self) -> list: + """获取支持的消息类型""" + return list(self._handlers.keys()) diff --git a/main/xiaozhi-server/core/handle/textMessageProcessor.py b/main/xiaozhi-server/core/handle/textMessageProcessor.py new file mode 100644 index 00000000..0cae5e09 --- /dev/null +++ b/main/xiaozhi-server/core/handle/textMessageProcessor.py @@ -0,0 +1,41 @@ +import json + +from core.handle.textMessageHandlerRegistry import TextMessageHandlerRegistry + +TAG = __name__ + + +class TextMessageProcessor: + """消息处理器主类""" + + def __init__(self, registry: TextMessageHandlerRegistry): + self.registry = registry + + async def process_message(self, conn, message: str) -> None: + """处理消息的主入口""" + try: + # 解析JSON消息 + msg_json = json.loads(message) + + # 处理JSON消息 + if isinstance(msg_json, dict): + message_type = msg_json.get("type") + + # 记录日志 + conn.logger.bind(tag=TAG).info(f"收到{message_type}消息:{message}") + + # 获取并执行处理器 + handler = self.registry.get_handler(message_type) + if handler: + await handler.handle(conn, msg_json) + else: + conn.logger.bind(tag=TAG).error(f"收到未知类型消息:{message}") + # 处理纯数字消息 + elif isinstance(msg_json, int): + conn.logger.bind(tag=TAG).info(f"收到数字消息:{message}") + await conn.websocket.send(message) + + except json.JSONDecodeError: + # 非JSON消息直接转发 + conn.logger.bind(tag=TAG).error(f"解析到错误的消息:{message}") + await conn.websocket.send(message) diff --git a/main/xiaozhi-server/core/handle/textMessageType.py b/main/xiaozhi-server/core/handle/textMessageType.py new file mode 100644 index 00000000..53e71b71 --- /dev/null +++ b/main/xiaozhi-server/core/handle/textMessageType.py @@ -0,0 +1,11 @@ +from enum import Enum + + +class TextMessageType(Enum): + """消息类型枚举""" + HELLO = "hello" + ABORT = "abort" + LISTEN = "listen" + IOT = "iot" + MCP = "mcp" + SERVER = "server" diff --git a/main/xiaozhi-server/core/providers/tts/fishspeech.py b/main/xiaozhi-server/core/providers/tts/fishspeech.py index bbf19164..83b3bc3f 100644 --- a/main/xiaozhi-server/core/providers/tts/fishspeech.py +++ b/main/xiaozhi-server/core/providers/tts/fishspeech.py @@ -143,8 +143,8 @@ class TTSProvider(TTSProviderBase): data = { "text": text, "references": [ - ServeReferenceAudio(audio=audio if audio else b"", text=text) - for text, audio in zip(ref_texts, byte_audios) + ServeReferenceAudio(audio=audio if audio else b"", text=ref_text) + for ref_text, audio in zip(ref_texts, byte_audios) ], "reference_id": self.reference_id, "normalize": self.normalize, diff --git a/main/xiaozhi-server/core/utils/opus_encoder_utils.py b/main/xiaozhi-server/core/utils/opus_encoder_utils.py index 8ca406d3..23d26c39 100644 --- a/main/xiaozhi-server/core/utils/opus_encoder_utils.py +++ b/main/xiaozhi-server/core/utils/opus_encoder_utils.py @@ -5,8 +5,9 @@ Opus编码工具类 import logging import traceback + import numpy as np -from typing import Optional, Callable, Any +from typing import List, Optional from opuslib_next import Encoder from opuslib_next import constants @@ -55,14 +56,13 @@ class OpusEncoderUtils: self.encoder.reset_state() self.buffer = np.array([], dtype=np.int16) - def encode_pcm_to_opus_stream(self, pcm_data: bytes, end_of_stream: bool, callback: Callable[[Any], Any]): + def encode_pcm_to_opus(self, pcm_data: bytes, end_of_stream: bool) -> List[bytes]: """ - 将PCM数据编码为Opus格式,以流式方式进行处理 + 将PCM数据编码为Opus格式 Args: pcm_data: PCM字节数据 - end_of_stream: 是否为流的结束, - callback: opus处理方法 + end_of_stream: 是否为流的结束 Returns: Opus数据包列表 @@ -76,6 +76,7 @@ class OpusEncoderUtils: # 将新数据追加到缓冲区 self.buffer = np.append(self.buffer, new_samples) + opus_packets = [] offset = 0 # 处理所有完整帧 @@ -83,7 +84,7 @@ class OpusEncoderUtils: frame = self.buffer[offset : offset + self.total_frame_size] output = self._encode(frame) if output: - callback(output) + opus_packets.append(output) offset += self.total_frame_size # 保留未处理的样本 @@ -97,9 +98,11 @@ class OpusEncoderUtils: output = self._encode(last_frame) if output: - callback(output) + opus_packets.append(output) self.buffer = np.array([], dtype=np.int16) + return opus_packets + def _encode(self, frame: np.ndarray) -> Optional[bytes]: """编码一帧音频数据""" try: diff --git a/main/xiaozhi-server/core/utils/p3.py b/main/xiaozhi-server/core/utils/p3.py index 415e5366..c75b968e 100644 --- a/main/xiaozhi-server/core/utils/p3.py +++ b/main/xiaozhi-server/core/utils/p3.py @@ -1,12 +1,15 @@ -import io import struct -from typing import Callable, Any +def decode_opus_from_file(input_file): + """ + 从p3文件中解码 Opus 数据,并返回一个 Opus 数据包的列表以及总时长。 + """ + opus_datas = [] + total_frames = 0 + sample_rate = 16000 # 文件采样率 + frame_duration_ms = 60 # 帧时长 + frame_size = int(sample_rate * frame_duration_ms / 1000) -def decode_opus_from_file_stream(input_file, callback: Callable[[Any], Any]): - """ - 从p3文件中解码 Opus 数据,由 callback 处理 Opus 数据包。 - """ with open(input_file, 'rb') as f: while True: # 读取头部(4字节):[1字节类型,1字节保留,2字节长度] @@ -22,13 +25,23 @@ def decode_opus_from_file_stream(input_file, callback: Callable[[Any], Any]): if len(opus_data) != data_len: raise ValueError(f"Data length({len(opus_data)}) mismatch({data_len}) in the file.") - callback(opus_data) + opus_datas.append(opus_data) + total_frames += 1 + # 计算总时长 + total_duration = (total_frames * frame_duration_ms) / 1000.0 + return opus_datas, total_duration -def decode_opus_from_bytes_stream(input_bytes, callback: Callable[[Any], Any]): +def decode_opus_from_bytes(input_bytes): """ - 从p3二进制数据中解码 Opus 数据,由 callback 处理 Opus 数据包。 + 从p3二进制数据中解码 Opus 数据,并返回一个 Opus 数据包的列表以及总时长。 """ + import io + opus_datas = [] + total_frames = 0 + sample_rate = 16000 # 文件采样率 + frame_duration_ms = 60 # 帧时长 + frame_size = int(sample_rate * frame_duration_ms / 1000) f = io.BytesIO(input_bytes) while True: @@ -39,4 +52,8 @@ def decode_opus_from_bytes_stream(input_bytes, callback: Callable[[Any], Any]): opus_data = f.read(data_len) if len(opus_data) != data_len: raise ValueError(f"Data length({len(opus_data)}) mismatch({data_len}) in the bytes.") - callback(opus_data) + opus_datas.append(opus_data) + total_frames += 1 + + total_duration = (total_frames * frame_duration_ms) / 1000.0 + return opus_datas, total_duration \ No newline at end of file diff --git a/main/xiaozhi-server/plugins_func/functions/get_news_from_newsnow.py b/main/xiaozhi-server/plugins_func/functions/get_news_from_newsnow.py index 54641106..1d60aefd 100644 --- a/main/xiaozhi-server/plugins_func/functions/get_news_from_newsnow.py +++ b/main/xiaozhi-server/plugins_func/functions/get_news_from_newsnow.py @@ -125,7 +125,8 @@ def fetch_news_from_api(conn, source="thepaper"): ]["get_news_from_newsnow"].get("url"): api_url = conn.config["plugins"]["get_news_from_newsnow"]["url"] + source - response = requests.get(api_url, timeout=10) + headers = {"User-Agent": "Mozilla/5.0"} + response = requests.get(api_url, headers=headers, timeout=10) response.raise_for_status() data = response.json() @@ -144,7 +145,8 @@ def fetch_news_from_api(conn, source="thepaper"): def fetch_news_detail(url): """获取新闻详情页内容并使用MarkItDown清理HTML""" try: - response = requests.get(url, timeout=10) + headers = {"User-Agent": "Mozilla/5.0"} + response = requests.get(url, headers=headers, timeout=10) response.raise_for_status() # 使用MarkItDown清理HTML内容 diff --git a/main/xiaozhi-server/plugins_func/functions/play_music.py b/main/xiaozhi-server/plugins_func/functions/play_music.py index dd967d0a..2cbc4018 100644 --- a/main/xiaozhi-server/plugins_func/functions/play_music.py +++ b/main/xiaozhi-server/plugins_func/functions/play_music.py @@ -212,6 +212,7 @@ async def play_local_music(conn, specific_file=None): conn.logger.bind(tag=TAG).error(f"选定的音乐文件不存在: {music_path}") return text = _get_random_play_prompt(selected_music) + await send_stt_message(conn, text) conn.dialogue.put(Message(role="assistant", content=text)) if conn.intent_type == "intent_llm":