mirror of
https://github.com/xinnan-tech/xiaozhi-esp32-server.git
synced 2026-07-23 23:53:55 +08:00
Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
2d7d75c290 | ||
|
|
237e88ba89 | ||
|
|
48223e4f39 | ||
|
|
ce776f210c | ||
|
|
fb5e25ec16 | ||
|
|
c4f2411fee | ||
|
|
3c3be950e9 | ||
|
|
b24ee0b9a6 | ||
|
|
e1d5caa0fd | ||
|
|
35ba9b0e6b | ||
|
|
66de07823d | ||
|
|
fd96e71d04 | ||
|
|
3c85242efc | ||
|
|
3c288e1e59 | ||
|
|
4bf2f927b0 | ||
|
|
ad30030a1f | ||
|
|
02d66e8093 | ||
|
|
49eab95be9 | ||
|
|
b1bfaf5a5c | ||
|
|
843e605352 | ||
|
|
6a7fa6060b | ||
|
|
43ce1df0ed | ||
|
|
a6deb3af8b | ||
|
|
9c01f5f0b0 | ||
|
|
80b89a85ee | ||
|
|
2fd0bb4912 | ||
|
|
c0329618fe | ||
|
|
c4a240cb04 | ||
|
|
33d5761194 | ||
|
|
a58ddd5880 | ||
|
|
3bbb5f9b83 | ||
|
|
3d6bf800f3 | ||
|
|
a0b0a0e1e8 | ||
|
|
8f48e9ac0c | ||
|
|
bb42095ca0 | ||
|
|
c487f69676 | ||
|
|
3db46a84b2 | ||
|
|
69520391f2 | ||
|
|
4b212753fb | ||
|
|
d86a2cf9de | ||
|
|
625d079168 | ||
|
|
2586843654 | ||
|
|
8ea48f9875 | ||
|
|
d7f3b3caf5 | ||
|
|
b37d8abf62 | ||
|
|
cd930edd06 | ||
|
|
4956aa6d70 | ||
|
|
fd17465d94 | ||
|
|
87a9425241 | ||
|
|
b225e8afd5 | ||
|
|
297b6e69d5 | ||
|
|
d7f0e88801 | ||
|
|
d62b27814c | ||
|
|
19526878b3 | ||
|
|
14ecad720f | ||
|
|
81689c04a5 | ||
|
|
de6b9de341 | ||
|
|
7f864eb84d | ||
|
|
4af0c1e2ce | ||
|
|
6a4ed78812 | ||
|
|
23b881f364 | ||
|
|
22054cf8bc | ||
|
|
fc236b1d96 | ||
|
|
4f81454f93 | ||
|
|
b3d6f173f1 | ||
|
|
8d2ba39ab8 | ||
|
|
4260e5a1a7 | ||
|
|
17585c8294 | ||
|
|
d242a9e5c0 | ||
|
|
9ca98f6391 | ||
|
|
be7ef08f40 | ||
|
|
626692df29 | ||
|
|
631787a4f1 | ||
|
|
09cf6cdc2d | ||
|
|
baf174b059 | ||
|
|
49eb3178a9 | ||
|
|
6a6aceff1d | ||
|
|
b8349af3da | ||
|
|
258d783f8e | ||
|
|
761fc05331 | ||
|
|
0c8e943d1b | ||
|
|
9787ca60da | ||
|
|
b8e57aeff4 | ||
|
|
9a0240ef3e | ||
|
|
cf5ccafe7e | ||
|
|
ae64233986 | ||
|
|
6dda79ee10 | ||
|
|
335a855968 | ||
|
|
94c92c5f38 | ||
|
|
35fa1493b3 | ||
|
|
f78f6fc529 | ||
|
|
5be65216e2 | ||
|
|
24526ad206 | ||
|
|
d7564a65f7 | ||
|
|
866d61cfaf | ||
|
|
3f3f3fdaa6 | ||
|
|
e0da59096a | ||
|
|
40632019ac | ||
|
|
3c5563b62f | ||
|
|
562424b74d | ||
|
|
7a598d5839 | ||
|
|
574d34bc2c | ||
|
|
16a4ccdb12 | ||
|
|
920cf4f897 | ||
|
|
76ee2c5365 | ||
|
|
472106390d | ||
|
|
d97f8b2e9a | ||
|
|
98e2526f8a | ||
|
|
c30c4649a4 | ||
|
|
ede8676979 | ||
|
|
17fb60b7ae | ||
|
|
8119897818 | ||
|
|
0a9e5d2cea | ||
|
|
a5f74d7767 | ||
|
|
02cb9c35b3 | ||
|
|
fdc6dcb26e | ||
|
|
9fd010e91e | ||
|
|
c900498ce8 | ||
|
|
c4c84e44e1 | ||
|
|
191ac47353 | ||
|
|
851365fb58 | ||
|
|
ede2bc6a4e | ||
|
|
38780b6daa | ||
|
|
885b7a0b05 | ||
|
|
92e117c10a | ||
|
|
71db92ef61 | ||
|
|
118bb728a9 | ||
|
|
d7aac5e10f | ||
|
|
4359f1d5b3 | ||
|
|
8bb888c58e | ||
|
|
4c81585e32 | ||
|
|
0e6d1cd677 | ||
|
|
98368cbe06 | ||
|
|
0b4a4df1af | ||
|
|
c8a3d378b7 | ||
|
|
aa34473560 | ||
|
|
f479a789db | ||
|
|
d4ce501c75 | ||
|
|
c2643d0b80 | ||
|
|
078605c188 | ||
|
|
cd3e9ed494 | ||
|
|
8a5a9cc0b3 | ||
|
|
347fc4ca44 | ||
|
|
61db1192e2 | ||
|
|
d9d973b6aa | ||
|
|
cc4aed5a8e | ||
|
|
b44d52fb3e | ||
|
|
93265be974 | ||
|
|
2028405af6 | ||
|
|
bb9df3f54c | ||
|
|
264487574b | ||
|
|
6a00895af8 | ||
|
|
1136ca4b24 | ||
|
|
b153bc11f2 | ||
|
|
a067a6ba87 | ||
|
|
c271b88c87 | ||
|
|
1dfa619243 | ||
|
|
91bdc05a38 | ||
|
|
0e8edfe002 | ||
|
|
c1084923f7 | ||
|
|
b2381fc689 | ||
|
|
fa75f56ffb |
+10
-2
@@ -107,6 +107,7 @@ celerybeat.pid
|
|||||||
*.sage.py
|
*.sage.py
|
||||||
|
|
||||||
# Environments
|
# Environments
|
||||||
|
.env
|
||||||
.venv
|
.venv
|
||||||
env/
|
env/
|
||||||
venv/
|
venv/
|
||||||
@@ -145,24 +146,31 @@ tmp
|
|||||||
.history
|
.history
|
||||||
.DS_Store
|
.DS_Store
|
||||||
main/xiaozhi-server/data
|
main/xiaozhi-server/data
|
||||||
main/xiaozhi-server/config/assets/wakeup_words.*
|
|
||||||
main/manager-web/node_modules
|
main/manager-web/node_modules
|
||||||
.config.yaml
|
.config.yaml
|
||||||
.secrets.yaml
|
.secrets.yaml
|
||||||
.private_config.yaml
|
.private_config.yaml
|
||||||
|
.env.development
|
||||||
|
|
||||||
# model files
|
# model files
|
||||||
main/xiaozhi-server/models/SenseVoiceSmall/model.pt
|
main/xiaozhi-server/models/SenseVoiceSmall/model.pt
|
||||||
main/xiaozhi-server/models/sherpa-onnx*
|
main/xiaozhi-server/models/sherpa-onnx*
|
||||||
|
/main/xiaozhi-server/audio_ref/
|
||||||
|
/audio_ref/
|
||||||
|
/asr-models/iic/SenseVoiceSmall/
|
||||||
|
/main/xiaozhi-server/asr-models/iic/SenseVoiceSmall/
|
||||||
|
/models/SenseVoiceSmall/model.pt
|
||||||
my_wakeup_words.mp3
|
my_wakeup_words.mp3
|
||||||
!main/xiaozhi-server/config/assets/bind_code.wav
|
!main/xiaozhi-server/config/assets/bind_code.wav
|
||||||
|
!main/xiaozhi-server/config/assets/wakeup_words.wav
|
||||||
!main/xiaozhi-server/config/assets/bind_not_found.wav
|
!main/xiaozhi-server/config/assets/bind_not_found.wav
|
||||||
!main/xiaozhi-server/config/assets/bind_code/*.wav
|
!main/xiaozhi-server/config/assets/bind_code/*.wav
|
||||||
!main/xiaozhi-server/config/assets/max_output_size.wav
|
!main/xiaozhi-server/config/assets/max_output_size.wav
|
||||||
main/manager-api/.vscode
|
main/manager-api/.vscode
|
||||||
|
|
||||||
# Ignore webpack cache directory
|
# Ignore webpack cache directory
|
||||||
main/manager-web/.webpack_cache/
|
main/manager-web/.webpack_cache/
|
||||||
main/xiaozhi-server/mysql
|
main/xiaozhi-server/mysql
|
||||||
uploadfile
|
uploadfile
|
||||||
|
*.json
|
||||||
.vscode
|
.vscode
|
||||||
|
|||||||
@@ -121,6 +121,33 @@
|
|||||||
</a>
|
</a>
|
||||||
</td>
|
</td>
|
||||||
</tr>
|
</tr>
|
||||||
|
<tr>
|
||||||
|
<td>
|
||||||
|
<a href="https://www.bilibili.com/video/BV12J7WzBEaH" target="_blank">
|
||||||
|
<picture>
|
||||||
|
<img alt="实时打断" src="docs/images/demo10.png" />
|
||||||
|
</picture>
|
||||||
|
</a>
|
||||||
|
</td>
|
||||||
|
<td>
|
||||||
|
<a href="https://www.bilibili.com/video/BV1Co76z7EvK" target="_blank">
|
||||||
|
<picture>
|
||||||
|
<img alt="拍照识物品" src="docs/images/demo12.png" />
|
||||||
|
</picture>
|
||||||
|
</a>
|
||||||
|
</td>
|
||||||
|
<td>
|
||||||
|
<a href="https://www.bilibili.com/video/BV1TJ7WzzEo6" target="_blank">
|
||||||
|
<picture>
|
||||||
|
<img alt="多指令任务" src="docs/images/demo11.png" />
|
||||||
|
</picture>
|
||||||
|
</a>
|
||||||
|
</td>
|
||||||
|
<td>
|
||||||
|
</td>
|
||||||
|
<td>
|
||||||
|
</td>
|
||||||
|
</tr>
|
||||||
</table>
|
</table>
|
||||||
|
|
||||||
---
|
---
|
||||||
@@ -141,11 +168,11 @@
|
|||||||
本项目提供两种部署方式,请根据您的具体需求选择:
|
本项目提供两种部署方式,请根据您的具体需求选择:
|
||||||
|
|
||||||
#### 🚀 部署方式选择
|
#### 🚀 部署方式选择
|
||||||
|
| 部署方式 | 特点 | 适用场景 | 部署文档 | 配置要求 | 视频教程 |
|
||||||
|
|---------|------|---------|---------|---------|---------|
|
||||||
|
| **最简化安装** | 智能对话、IOT功能,数据存储在配置文件 | 低配置环境,无需数据库 | [Docker版](./docs/Deployment.md#%E6%96%B9%E5%BC%8F%E4%B8%80docker%E5%8F%AA%E8%BF%90%E8%A1%8Cserver) / [源码部署](./docs/Deployment.md#%E6%96%B9%E5%BC%8F%E4%BA%8C%E6%9C%AC%E5%9C%B0%E6%BA%90%E7%A0%81%E5%8F%AA%E8%BF%90%E8%A1%8Cserver)| 如果使用`FunASR`要2核4G,如果全API,要2核2G | - |
|
||||||
|
| **全模块安装** | 智能对话、IOT、OTA、智控台,数据存储在数据库 | 完整功能体验 |[Docker版](./docs/Deployment_all.md#%E6%96%B9%E5%BC%8F%E4%B8%80docker%E8%BF%90%E8%A1%8C%E5%85%A8%E6%A8%A1%E5%9D%97) / [源码部署](./docs/Deployment_all.md#%E6%96%B9%E5%BC%8F%E4%BA%8C%E6%9C%AC%E5%9C%B0%E6%BA%90%E7%A0%81%E8%BF%90%E8%A1%8C%E5%85%A8%E6%A8%A1%E5%9D%97) | 如果使用`FunASR`要4核8G,如果全API,要2核4G| [本地源码视频教程](https://www.bilibili.com/video/BV1wBJhz4Ewe) |
|
||||||
|
|
||||||
| 部署方式 | 特点 | 适用场景 | Docker部署文档 | 源码部署文档 |
|
|
||||||
|---------|------|---------|---------|---------|
|
|
||||||
| **最简化安装** | 智能对话、IOT功能,数据存储在配置文件 | 低配置环境,无需数据库 | [Docker只运行Server](./docs/Deployment.md#%E6%96%B9%E5%BC%8F%E4%B8%80docker%E5%8F%AA%E8%BF%90%E8%A1%8Cserver) | [本地源码只运行Server](./docs/Deployment.md#%E6%96%B9%E5%BC%8F%E4%BA%8C%E6%9C%AC%E5%9C%B0%E6%BA%90%E7%A0%81%E5%8F%AA%E8%BF%90%E8%A1%8Cserver)|
|
|
||||||
| **全模块安装** | 智能对话、IOT、OTA、智控台,数据存储在数据库 | 完整功能体验 |[Docker运行全模块](./docs/Deployment_all.md#%E6%96%B9%E5%BC%8F%E4%B8%80docker%E8%BF%90%E8%A1%8C%E5%85%A8%E6%A8%A1%E5%9D%97) | [本地源码运行全模块](./docs/Deployment_all.md#%E6%96%B9%E5%BC%8F%E4%BA%8C%E6%9C%AC%E5%9C%B0%E6%BA%90%E7%A0%81%E8%BF%90%E8%A1%8C%E5%85%A8%E6%A8%A1%E5%9D%97) |
|
|
||||||
|
|
||||||
> 💡 提示:以下是按最新代码部署后的测试平台,有需要可烧录测试,并发为6个,每天会清空数据
|
> 💡 提示:以下是按最新代码部署后的测试平台,有需要可烧录测试,并发为6个,每天会清空数据
|
||||||
|
|
||||||
@@ -157,9 +184,23 @@ OTA接口地址: https://2662r3426b.vicp.fun/xiaozhi/ota/
|
|||||||
Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/
|
Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/
|
||||||
```
|
```
|
||||||
|
|
||||||
|
#### 🚩 配置说明和推荐
|
||||||
|
> [!Note]
|
||||||
|
> 本项目默认的配置是`入门全免费`设置,如果想效果更优,推荐使用`全流式配置`。
|
||||||
|
>
|
||||||
|
> 本项目自`0.5.2`版本,已支持整个生命周期全流式,相比`0.5`版本以前,响应速度提升约`2.5秒`
|
||||||
|
|
||||||
|
|
||||||
|
| 模块名称 | 入门全免费设置 | 全流式配置 |
|
||||||
|
|---------|---------|------|
|
||||||
|
| ASR(语音识别) | FunASR(本地) | ✅DoubaoASR(火山流式语音识别) |
|
||||||
|
| LLM(大模型) | ChatGLMLLM(智谱glm-4-flash) | ✅DoubaoLLM(火山doubao-1-5-pro-32k-250115) |
|
||||||
|
| TTS(语音合成) | EdgeTTS(微软语音) | ✅HuoshanDoubleStreamTTS(火山双流式语音合成) |
|
||||||
|
| Intent(意图识别) | function_call(函数调用) | ✅function_call(函数调用) |
|
||||||
|
| Memory(记忆功能) | mem_local_short(本地短期记忆) | ✅mem_local_short(本地短期记忆) |
|
||||||
|
|
||||||
---
|
---
|
||||||
## 功能清单 ✨
|
## 功能清单 ✨
|
||||||
|
|
||||||
### 已实现 ✅
|
### 已实现 ✅
|
||||||
|
|
||||||
| 功能模块 | 描述 |
|
| 功能模块 | 描述 |
|
||||||
|
|||||||
+5
-1
@@ -108,7 +108,11 @@ VAD:
|
|||||||
|
|
||||||
参考教程[阿里云短信集成指南](./ali-sms-integration.md)
|
参考教程[阿里云短信集成指南](./ali-sms-integration.md)
|
||||||
|
|
||||||
### 9、更多问题,可联系我们反馈 💬
|
### 9、如何开启视觉模型实现拍照识物 📷
|
||||||
|
|
||||||
|
参考教程[视觉模型使用指南](./mcp-vision-integration.md)
|
||||||
|
|
||||||
|
### 10、更多问题,可联系我们反馈 💬
|
||||||
|
|
||||||
可以在[issues](https://github.com/xinnan-tech/xiaozhi-esp32-server/issues)提交您的问题。
|
可以在[issues](https://github.com/xinnan-tech/xiaozhi-esp32-server/issues)提交您的问题。
|
||||||
|
|
||||||
|
|||||||
Binary file not shown.
|
After Width: | Height: | Size: 386 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 518 KiB |
Binary file not shown.
|
After Width: | Height: | Size: 361 KiB |
@@ -0,0 +1,171 @@
|
|||||||
|
# 视觉模型使用指南
|
||||||
|
本教程分为两部分:
|
||||||
|
- 第一部分:单模块运行xiaozhi-server开启视觉模型
|
||||||
|
- 第二部分:全模块运行时,如何开启视觉模型
|
||||||
|
|
||||||
|
开启视觉模型前,你需要准备三件事:
|
||||||
|
- 你需要准备一台带摄像头的设备,而且这台设备已经在虾哥仓库里,实现了调用摄像头功能。例如`立创·实战派ESP32-S3开发板`
|
||||||
|
- 你设备固件的版本升级到1.6.6及以上
|
||||||
|
- 你已经成功跑通基础对话模块
|
||||||
|
|
||||||
|
## 单模块运行xiaozhi-server开启视觉模型
|
||||||
|
|
||||||
|
### 第一步确认网络
|
||||||
|
由于视觉模型会默认启动8003端口。
|
||||||
|
|
||||||
|
如果你是docker运行,请确认一下你的`docker-compose.yml`是否放了`8003`端口,如果没有就更新最新的`docker-compose.yml`文件
|
||||||
|
|
||||||
|
如果你是源码运行,确认防火墙是否放行`8003`端口
|
||||||
|
|
||||||
|
### 第二步选择你的视觉模型
|
||||||
|
打开你的`data/.config.yaml`文件,设置你的`selected_module.VLLM`设置为某个视觉模型。目前我们已经支持`openai`类型接口的视觉模型。`ChatGLMVLLM`就是其中一款兼容`openai`的模型。
|
||||||
|
|
||||||
|
```
|
||||||
|
selected_module:
|
||||||
|
VAD: ..
|
||||||
|
ASR: ..
|
||||||
|
LLM: ..
|
||||||
|
VLLM: ChatGLMVLLM
|
||||||
|
TTS: ..
|
||||||
|
Memory: ..
|
||||||
|
Intent: ..
|
||||||
|
```
|
||||||
|
|
||||||
|
假设我们使用`ChatGLMVLLM`作为视觉模型,那我们需要先登录[智谱AI](https://bigmodel.cn/usercenter/proj-mgmt/apikeys)网站,申请密钥。如果你之前已经申请过了密钥,可以复用这个密钥。
|
||||||
|
|
||||||
|
在你的配置文件中,增加这个配置,如果已经有了这个配置,就设置好你的api_key。
|
||||||
|
|
||||||
|
```
|
||||||
|
VLLM:
|
||||||
|
ChatGLMVLLM:
|
||||||
|
api_key: 你的api_key
|
||||||
|
```
|
||||||
|
|
||||||
|
### 第三步启动xiaozhi-server服务
|
||||||
|
如果你是源码,就输入命令启动
|
||||||
|
```
|
||||||
|
python app.py
|
||||||
|
```
|
||||||
|
如果你是docker运行,就重启容器
|
||||||
|
```
|
||||||
|
docker restart xiaozhi-esp32-server
|
||||||
|
```
|
||||||
|
|
||||||
|
启动后会输出以下内容的日志。
|
||||||
|
|
||||||
|
```
|
||||||
|
2025-06-01 **** - OTA接口是 http://192.168.4.7:8003/xiaozhi/ota/
|
||||||
|
2025-06-01 **** - 视觉分析接口是 http://192.168.4.7:8003/mcp/vision/explain
|
||||||
|
2025-06-01 **** - Websocket地址是 ws://192.168.4.7:8000/xiaozhi/v1/
|
||||||
|
2025-06-01 **** - =======上面的地址是websocket协议地址,请勿用浏览器访问=======
|
||||||
|
2025-06-01 **** - 如想测试websocket请用谷歌浏览器打开test目录下的test_page.html
|
||||||
|
2025-06-01 **** - =============================================================
|
||||||
|
```
|
||||||
|
|
||||||
|
启动后,使用使用浏览器打开日志里`视觉分析接口`连接。看看输出了什么?如果你是linux,没有浏览器,你可以执行这个命令:
|
||||||
|
```
|
||||||
|
curl -i 你的视觉分析接口
|
||||||
|
```
|
||||||
|
|
||||||
|
正常来说会这样显示
|
||||||
|
```
|
||||||
|
MCP Vision 接口运行正常,视觉解释接口地址是:http://xxxx:8003/mcp/vision/explain
|
||||||
|
```
|
||||||
|
|
||||||
|
请注意,如果你是公网部署,或者docker部署,一定要改一下你的`data/.config.yaml`里这个配置
|
||||||
|
```
|
||||||
|
server:
|
||||||
|
vision_explain: http://你的ip或者域名:端口号/mcp/vision/explain
|
||||||
|
```
|
||||||
|
|
||||||
|
为什么呢?因为视觉解释接口需要下发到设备,如果你的地址是局域网地址,或者是docker内部地址,设备是无法访问的。
|
||||||
|
|
||||||
|
假设你的公网地址是`111.111.111.111`,那么`vision_explain`应该这么配
|
||||||
|
|
||||||
|
```
|
||||||
|
server:
|
||||||
|
vision_explain: http://111.111.111.111:8003/mcp/vision/explain
|
||||||
|
```
|
||||||
|
|
||||||
|
如果你的MCP Vision 接口运行正常,且你也试着用浏览器访问正常打开下发的`视觉解释接口地址`,请继续下一步
|
||||||
|
|
||||||
|
### 第四步 设备唤醒开启
|
||||||
|
|
||||||
|
对设备说“请打开摄像头,说你你看到了什么”
|
||||||
|
|
||||||
|
留意xiaozhi-server的日志输出,看看有没有报错。
|
||||||
|
|
||||||
|
|
||||||
|
## 全模块运行时,如何开启视觉模型
|
||||||
|
|
||||||
|
### 第一步 确认网络
|
||||||
|
由于视觉模型会默认启动8003端口。
|
||||||
|
|
||||||
|
如果你是docker运行,请确认一下你的`docker-compose_all.yml`是否映射了`8003`端口,如果没有就更新最新的`docker-compose_all.yml`文件
|
||||||
|
|
||||||
|
如果你是源码运行,确认防火墙是否放行`8003`端口
|
||||||
|
|
||||||
|
### 第二步 确认你配置文件
|
||||||
|
|
||||||
|
打开你的`data/.config.yaml`文件,确认一下你的配置文件的结构,是否和`data/config_from_api.yaml`一样。如果不一样,或缺少某项,请补齐。
|
||||||
|
|
||||||
|
### 第三步 配置视觉模型密钥
|
||||||
|
|
||||||
|
那我们需要先登录[智谱AI](https://bigmodel.cn/usercenter/proj-mgmt/apikeys)网站,申请密钥。如果你之前已经申请过了密钥,可以复用这个密钥。
|
||||||
|
|
||||||
|
登录`智控台`,顶部菜单点击`模型配置`,在左侧栏点击`视觉打语言模型`,找到`VLLM_ChatGLMVLLM`,点击修改按钮,在弹框中,在`API密钥`输入你密钥,点击保存。
|
||||||
|
|
||||||
|
保存成功后,去到你需要测试的智能体哪里,点击`配置角色`,在打开的内容里,查看`视觉大语言模型(VLLM)`是否选择了刚才的视觉模型。点击保存。
|
||||||
|
|
||||||
|
### 第三步 启动xiaozhi-server模块
|
||||||
|
如果你是源码,就输入命令启动
|
||||||
|
```
|
||||||
|
python app.py
|
||||||
|
```
|
||||||
|
如果你是docker运行,就重启容器
|
||||||
|
```
|
||||||
|
docker restart xiaozhi-esp32-server
|
||||||
|
```
|
||||||
|
|
||||||
|
启动后会输出以下内容的日志。
|
||||||
|
|
||||||
|
```
|
||||||
|
2025-06-01 **** - 视觉分析接口是 http://192.168.4.7:8003/mcp/vision/explain
|
||||||
|
2025-06-01 **** - Websocket地址是 ws://192.168.4.7:8000/xiaozhi/v1/
|
||||||
|
2025-06-01 **** - =======上面的地址是websocket协议地址,请勿用浏览器访问=======
|
||||||
|
2025-06-01 **** - 如想测试websocket请用谷歌浏览器打开test目录下的test_page.html
|
||||||
|
2025-06-01 **** - =============================================================
|
||||||
|
```
|
||||||
|
|
||||||
|
启动后,使用使用浏览器打开日志里`视觉分析接口`连接。看看输出了什么?如果你是linux,没有浏览器,你可以执行这个命令:
|
||||||
|
```
|
||||||
|
curl -i 你的视觉分析接口
|
||||||
|
```
|
||||||
|
|
||||||
|
正常来说会这样显示
|
||||||
|
```
|
||||||
|
MCP Vision 接口运行正常,视觉解释接口地址是:http://xxxx:8003/mcp/vision/explain
|
||||||
|
```
|
||||||
|
|
||||||
|
请注意,如果你是公网部署,或者docker部署,一定要改一下你的`data/.config.yaml`里这个配置
|
||||||
|
```
|
||||||
|
server:
|
||||||
|
vision_explain: http://你的ip或者域名:端口号/mcp/vision/explain
|
||||||
|
```
|
||||||
|
|
||||||
|
为什么呢?因为视觉解释接口需要下发到设备,如果你的地址是局域网地址,或者是docker内部地址,设备是无法访问的。
|
||||||
|
|
||||||
|
假设你的公网地址是`111.111.111.111`,那么`vision_explain`应该这么配
|
||||||
|
|
||||||
|
```
|
||||||
|
server:
|
||||||
|
vision_explain: http://111.111.111.111:8003/mcp/vision/explain
|
||||||
|
```
|
||||||
|
|
||||||
|
如果你的MCP Vision 接口运行正常,且你也试着用浏览器访问正常打开下发的`视觉解释接口地址`,请继续下一步
|
||||||
|
|
||||||
|
### 第四步 设备唤醒开启
|
||||||
|
|
||||||
|
对设备说“请打开摄像头,说你你看到了什么”
|
||||||
|
|
||||||
|
留意xiaozhi-server的日志输出,看看有没有报错。
|
||||||
@@ -227,7 +227,7 @@ public interface Constant {
|
|||||||
/**
|
/**
|
||||||
* 版本号
|
* 版本号
|
||||||
*/
|
*/
|
||||||
public static final String VERSION = "0.4.4";
|
public static final String VERSION = "0.5.2";
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 无效固件URL
|
* 无效固件URL
|
||||||
|
|||||||
@@ -4,7 +4,6 @@ import java.util.List;
|
|||||||
import java.util.Objects;
|
import java.util.Objects;
|
||||||
|
|
||||||
import org.apache.shiro.authz.UnauthorizedException;
|
import org.apache.shiro.authz.UnauthorizedException;
|
||||||
import org.springframework.context.support.DefaultMessageSourceResolvable;
|
|
||||||
import org.springframework.dao.DuplicateKeyException;
|
import org.springframework.dao.DuplicateKeyException;
|
||||||
import org.springframework.validation.ObjectError;
|
import org.springframework.validation.ObjectError;
|
||||||
import org.springframework.web.bind.MethodArgumentNotValidException;
|
import org.springframework.web.bind.MethodArgumentNotValidException;
|
||||||
|
|||||||
+2
@@ -28,4 +28,6 @@ public class AgentChatHistoryReportDTO {
|
|||||||
private String content;
|
private String content;
|
||||||
@Schema(description = "base64编码的opus音频数据", example = "")
|
@Schema(description = "base64编码的opus音频数据", example = "")
|
||||||
private String audioBase64;
|
private String audioBase64;
|
||||||
|
@Schema(description = "上报时间,十位时间戳,空时默认使用当前时间", example = "1745657732")
|
||||||
|
private Long reportTime;
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -27,6 +27,9 @@ public class AgentDTO {
|
|||||||
@Schema(description = "大语言模型名称", example = "llm_model_01")
|
@Schema(description = "大语言模型名称", example = "llm_model_01")
|
||||||
private String llmModelName;
|
private String llmModelName;
|
||||||
|
|
||||||
|
@Schema(description = "视觉模型名称", example = "vllm_model_01")
|
||||||
|
private String vllmModelName;
|
||||||
|
|
||||||
@Schema(description = "记忆模型ID", example = "mem_model_01")
|
@Schema(description = "记忆模型ID", example = "mem_model_01")
|
||||||
private String memModelId;
|
private String memModelId;
|
||||||
|
|
||||||
|
|||||||
@@ -36,6 +36,9 @@ public class AgentEntity {
|
|||||||
@Schema(description = "大语言模型标识")
|
@Schema(description = "大语言模型标识")
|
||||||
private String llmModelId;
|
private String llmModelId;
|
||||||
|
|
||||||
|
@Schema(description = "VLLM模型标识")
|
||||||
|
private String vllmModelId;
|
||||||
|
|
||||||
@Schema(description = "语音合成模型标识")
|
@Schema(description = "语音合成模型标识")
|
||||||
private String ttsModelId;
|
private String ttsModelId;
|
||||||
|
|
||||||
|
|||||||
@@ -49,6 +49,11 @@ public class AgentTemplateEntity implements Serializable {
|
|||||||
*/
|
*/
|
||||||
private String llmModelId;
|
private String llmModelId;
|
||||||
|
|
||||||
|
/**
|
||||||
|
* VLLM模型标识
|
||||||
|
*/
|
||||||
|
private String vllmModelId;
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* 语音合成模型标识
|
* 语音合成模型标识
|
||||||
*/
|
*/
|
||||||
|
|||||||
+7
-5
@@ -47,7 +47,8 @@ public class AgentChatHistoryBizServiceImpl implements AgentChatHistoryBizServic
|
|||||||
public Boolean report(AgentChatHistoryReportDTO report) {
|
public Boolean report(AgentChatHistoryReportDTO report) {
|
||||||
String macAddress = report.getMacAddress();
|
String macAddress = report.getMacAddress();
|
||||||
Byte chatType = report.getChatType();
|
Byte chatType = report.getChatType();
|
||||||
log.info("小智设备聊天上报请求: macAddress={}, type={}", macAddress, chatType);
|
Long reportTimeMillis = null != report.getReportTime() ? report.getReportTime() * 1000 : System.currentTimeMillis();
|
||||||
|
log.info("小智设备聊天上报请求: macAddress={}, type={} reportTime={}", macAddress, chatType, reportTimeMillis);
|
||||||
|
|
||||||
// 根据设备MAC地址查询对应的默认智能体,判断是否需要上报
|
// 根据设备MAC地址查询对应的默认智能体,判断是否需要上报
|
||||||
AgentEntity agentEntity = agentService.getDefaultAgentByMacAddress(macAddress);
|
AgentEntity agentEntity = agentService.getDefaultAgentByMacAddress(macAddress);
|
||||||
@@ -59,10 +60,10 @@ public class AgentChatHistoryBizServiceImpl implements AgentChatHistoryBizServic
|
|||||||
String agentId = agentEntity.getId();
|
String agentId = agentEntity.getId();
|
||||||
|
|
||||||
if (Objects.equals(chatHistoryConf, Constant.ChatHistoryConfEnum.RECORD_TEXT.getCode())) {
|
if (Objects.equals(chatHistoryConf, Constant.ChatHistoryConfEnum.RECORD_TEXT.getCode())) {
|
||||||
saveChatText(report, agentId, macAddress, null);
|
saveChatText(report, agentId, macAddress, null, reportTimeMillis);
|
||||||
} else if (Objects.equals(chatHistoryConf, Constant.ChatHistoryConfEnum.RECORD_TEXT_AUDIO.getCode())) {
|
} else if (Objects.equals(chatHistoryConf, Constant.ChatHistoryConfEnum.RECORD_TEXT_AUDIO.getCode())) {
|
||||||
String audioId = saveChatAudio(report);
|
String audioId = saveChatAudio(report);
|
||||||
saveChatText(report, agentId, macAddress, audioId);
|
saveChatText(report, agentId, macAddress, audioId, reportTimeMillis);
|
||||||
}
|
}
|
||||||
|
|
||||||
// 更新设备最后对话时间
|
// 更新设备最后对话时间
|
||||||
@@ -92,8 +93,7 @@ public class AgentChatHistoryBizServiceImpl implements AgentChatHistoryBizServic
|
|||||||
/**
|
/**
|
||||||
* 组装上报数据
|
* 组装上报数据
|
||||||
*/
|
*/
|
||||||
private void saveChatText(AgentChatHistoryReportDTO report, String agentId, String macAddress, String audioId) {
|
private void saveChatText(AgentChatHistoryReportDTO report, String agentId, String macAddress, String audioId, Long reportTime) {
|
||||||
|
|
||||||
// 构建聊天记录实体
|
// 构建聊天记录实体
|
||||||
AgentChatHistoryEntity entity = AgentChatHistoryEntity.builder()
|
AgentChatHistoryEntity entity = AgentChatHistoryEntity.builder()
|
||||||
.macAddress(macAddress)
|
.macAddress(macAddress)
|
||||||
@@ -102,6 +102,8 @@ public class AgentChatHistoryBizServiceImpl implements AgentChatHistoryBizServic
|
|||||||
.chatType(report.getChatType())
|
.chatType(report.getChatType())
|
||||||
.content(report.getContent())
|
.content(report.getContent())
|
||||||
.audioId(audioId)
|
.audioId(audioId)
|
||||||
|
.createdAt(new Date(reportTime))
|
||||||
|
// NOTE(haotian): 2025/5/26 updateAt可以不设置,重点是createAt,而且这样可以看到上报延迟
|
||||||
.build();
|
.build();
|
||||||
|
|
||||||
// 保存数据
|
// 保存数据
|
||||||
|
|||||||
+3
@@ -102,6 +102,9 @@ public class AgentServiceImpl extends BaseServiceImpl<AgentDao, AgentEntity> imp
|
|||||||
// 获取 LLM 模型名称
|
// 获取 LLM 模型名称
|
||||||
dto.setLlmModelName(modelConfigService.getModelNameById(agent.getLlmModelId()));
|
dto.setLlmModelName(modelConfigService.getModelNameById(agent.getLlmModelId()));
|
||||||
|
|
||||||
|
// 获取 VLLM 模型名称
|
||||||
|
dto.setVllmModelName(modelConfigService.getModelNameById(agent.getVllmModelId()));
|
||||||
|
|
||||||
// 获取记忆模型名称
|
// 获取记忆模型名称
|
||||||
dto.setMemModelId(agent.getMemModelId());
|
dto.setMemModelId(agent.getMemModelId());
|
||||||
|
|
||||||
|
|||||||
+31
-6
@@ -72,6 +72,7 @@ public class ConfigServiceImpl implements ConfigService {
|
|||||||
null,
|
null,
|
||||||
null,
|
null,
|
||||||
null,
|
null,
|
||||||
|
null,
|
||||||
result,
|
result,
|
||||||
isCache);
|
isCache);
|
||||||
|
|
||||||
@@ -140,6 +141,7 @@ public class ConfigServiceImpl implements ConfigService {
|
|||||||
agent.getVadModelId(),
|
agent.getVadModelId(),
|
||||||
agent.getAsrModelId(),
|
agent.getAsrModelId(),
|
||||||
agent.getLlmModelId(),
|
agent.getLlmModelId(),
|
||||||
|
agent.getVllmModelId(),
|
||||||
agent.getTtsModelId(),
|
agent.getTtsModelId(),
|
||||||
agent.getMemModelId(),
|
agent.getMemModelId(),
|
||||||
agent.getIntentModelId(),
|
agent.getIntentModelId(),
|
||||||
@@ -241,6 +243,7 @@ public class ConfigServiceImpl implements ConfigService {
|
|||||||
String vadModelId,
|
String vadModelId,
|
||||||
String asrModelId,
|
String asrModelId,
|
||||||
String llmModelId,
|
String llmModelId,
|
||||||
|
String vllmModelId,
|
||||||
String ttsModelId,
|
String ttsModelId,
|
||||||
String memModelId,
|
String memModelId,
|
||||||
String intentModelId,
|
String intentModelId,
|
||||||
@@ -248,9 +251,10 @@ public class ConfigServiceImpl implements ConfigService {
|
|||||||
boolean isCache) {
|
boolean isCache) {
|
||||||
Map<String, String> selectedModule = new HashMap<>();
|
Map<String, String> selectedModule = new HashMap<>();
|
||||||
|
|
||||||
String[] modelTypes = { "VAD", "ASR", "TTS", "Memory", "Intent", "LLM" };
|
String[] modelTypes = { "VAD", "ASR", "TTS", "Memory", "Intent", "LLM", "VLLM" };
|
||||||
String[] modelIds = { vadModelId, asrModelId, ttsModelId, memModelId, intentModelId, llmModelId };
|
String[] modelIds = { vadModelId, asrModelId, ttsModelId, memModelId, intentModelId, llmModelId, vllmModelId };
|
||||||
String intentLLMModelId = null;
|
String intentLLMModelId = null;
|
||||||
|
String memLocalShortLLMModelId = null;
|
||||||
|
|
||||||
for (int i = 0; i < modelIds.length; i++) {
|
for (int i = 0; i < modelIds.length; i++) {
|
||||||
if (modelIds[i] == null) {
|
if (modelIds[i] == null) {
|
||||||
@@ -269,7 +273,7 @@ public class ConfigServiceImpl implements ConfigService {
|
|||||||
Map<String, Object> map = (Map<String, Object>) model.getConfigJson();
|
Map<String, Object> map = (Map<String, Object>) model.getConfigJson();
|
||||||
if ("intent_llm".equals(map.get("type"))) {
|
if ("intent_llm".equals(map.get("type"))) {
|
||||||
intentLLMModelId = (String) map.get("llm");
|
intentLLMModelId = (String) map.get("llm");
|
||||||
if (intentLLMModelId != null && intentLLMModelId.equals(llmModelId)) {
|
if (StringUtils.isNotBlank(intentLLMModelId) && intentLLMModelId.equals(llmModelId)) {
|
||||||
intentLLMModelId = null;
|
intentLLMModelId = null;
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
@@ -281,10 +285,31 @@ public class ConfigServiceImpl implements ConfigService {
|
|||||||
}
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
if ("Memory".equals(modelTypes[i])) {
|
||||||
|
Map<String, Object> map = (Map<String, Object>) model.getConfigJson();
|
||||||
|
if ("mem_local_short".equals(map.get("type"))) {
|
||||||
|
memLocalShortLLMModelId = (String) map.get("llm");
|
||||||
|
if (StringUtils.isNotBlank(memLocalShortLLMModelId)
|
||||||
|
&& memLocalShortLLMModelId.equals(llmModelId)) {
|
||||||
|
memLocalShortLLMModelId = null;
|
||||||
|
}
|
||||||
|
}
|
||||||
|
}
|
||||||
// 如果是LLM类型,且intentLLMModelId不为空,则添加附加模型
|
// 如果是LLM类型,且intentLLMModelId不为空,则添加附加模型
|
||||||
if ("LLM".equals(modelTypes[i]) && intentLLMModelId != null) {
|
if ("LLM".equals(modelTypes[i])) {
|
||||||
ModelConfigEntity intentLLM = modelConfigService.getModelById(intentLLMModelId, isCache);
|
if (StringUtils.isNotBlank(intentLLMModelId)) {
|
||||||
typeConfig.put(intentLLM.getId(), intentLLM.getConfigJson());
|
if (!typeConfig.containsKey(intentLLMModelId)) {
|
||||||
|
ModelConfigEntity intentLLM = modelConfigService.getModelById(intentLLMModelId, isCache);
|
||||||
|
typeConfig.put(intentLLM.getId(), intentLLM.getConfigJson());
|
||||||
|
}
|
||||||
|
}
|
||||||
|
if (StringUtils.isNotBlank(memLocalShortLLMModelId)) {
|
||||||
|
if (!typeConfig.containsKey(memLocalShortLLMModelId)) {
|
||||||
|
ModelConfigEntity memLocalShortLLM = modelConfigService
|
||||||
|
.getModelById(memLocalShortLLMModelId, isCache);
|
||||||
|
typeConfig.put(memLocalShortLLM.getId(), memLocalShortLLM.getConfigJson());
|
||||||
|
}
|
||||||
|
}
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
result.put(modelTypes[i], typeConfig);
|
result.put(modelTypes[i], typeConfig);
|
||||||
|
|||||||
@@ -0,0 +1,51 @@
|
|||||||
|
-- 本地短期记忆配置可以设置独立的LLM
|
||||||
|
update `ai_model_provider` set fields = '[{"key":"llm","label":"LLM模型","type":"string"}]' where id = 'SYSTEM_Memory_mem_local_short';
|
||||||
|
update `ai_model_config` set config_json = '{\"type\": \"mem_local_short\", \"llm\": \"LLM_ChatGLMLLM\"}' where id = 'Memory_mem_local_short';
|
||||||
|
|
||||||
|
-- 增加火山双流式TTS供应器和模型配置
|
||||||
|
delete from `ai_model_provider` where id = 'SYSTEM_TTS_HSDSTTS';
|
||||||
|
INSERT INTO `ai_model_provider` (`id`, `model_type`, `provider_code`, `name`, `fields`, `sort`, `creator`, `create_date`, `updater`, `update_date`) VALUES
|
||||||
|
('SYSTEM_TTS_HSDSTTS', 'TTS', 'huoshan_double_stream', '火山双流式语音合成', '[{"key":"ws_url","label":"WebSocket地址","type":"string"},{"key":"appid","label":"应用ID","type":"string"},{"key":"access_token","label":"访问令牌","type":"string"},{"key":"resource_id","label":"资源ID","type":"string"},{"key":"speaker","label":"默认音色","type":"string"}]', 13, 1, NOW(), 1, NOW());
|
||||||
|
|
||||||
|
delete from `ai_model_config` where id = 'TTS_HuoshanDoubleStreamTTS';
|
||||||
|
INSERT INTO `ai_model_config` VALUES ('TTS_HuoshanDoubleStreamTTS', 'TTS', 'HuoshanDoubleStreamTTS', '火山双流式语音合成', 0, 1, '{\"type\": \"huoshan_double_stream\", \"ws_url\": \"wss://openspeech.bytedance.com/api/v3/tts/bidirection\", \"appid\": \"你的火山引擎语音合成服务appid\", \"access_token\": \"你的火山引擎语音合成服务access_token\", \"resource_id\": \"volc.service_type.10029\", \"speaker\": \"zh_female_wanwanxiaohe_moon_bigtts\"}', NULL, NULL, 16, NULL, NULL, NULL, NULL);
|
||||||
|
|
||||||
|
-- 火山双流式TT模型配置说明文档
|
||||||
|
UPDATE `ai_model_config` SET
|
||||||
|
`doc_link` = 'https://console.volcengine.com/speech/service/10007',
|
||||||
|
`remark` = '火山引擎语音合成服务配置说明:
|
||||||
|
1. 访问 https://www.volcengine.com/ 注册并开通火山引擎账号
|
||||||
|
2. 访问 https://console.volcengine.com/speech/service/10007 开通语音合成大模型,购买音色
|
||||||
|
3. 在页面底部获取appid和access_token
|
||||||
|
5. 资源ID固定为:volc.service_type.10029(大模型语音合成及混音)
|
||||||
|
6. 填入配置文件中' WHERE `id` = 'TTS_HuoshanDoubleStreamTTS';
|
||||||
|
|
||||||
|
|
||||||
|
-- 添加火山双流式TTS音色
|
||||||
|
delete from `ai_tts_voice` where tts_model_id = 'TTS_HuoshanDoubleStreamTTS';
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0001', 'TTS_HuoshanDoubleStreamTTS', '爽快思思/Skye', 'zh_female_shuangkuaisisi_moon_bigtts', '中文、英文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/Skye.mp3', NULL, 1, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0002', 'TTS_HuoshanDoubleStreamTTS', '温暖阿虎/Alvin', 'zh_male_wennuanahu_moon_bigtts', '中文、英文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/Alvin.mp3', NULL, 2, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0003', 'TTS_HuoshanDoubleStreamTTS', '少年梓辛/Brayan', 'zh_male_shaonianzixin_moon_bigtts', '中文、英文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/Brayan.mp3', NULL, 3, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0004', 'TTS_HuoshanDoubleStreamTTS', '邻家女孩', 'zh_female_linjianvhai_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E9%82%BB%E5%AE%B6%E5%A5%B3%E5%AD%A9.mp3', NULL, 4, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0005', 'TTS_HuoshanDoubleStreamTTS', '渊博小叔', 'zh_male_yuanboxiaoshu_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E6%B8%8A%E5%8D%9A%E5%B0%8F%E5%8F%94.mp3', NULL, 5, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0006', 'TTS_HuoshanDoubleStreamTTS', '阳光青年', 'zh_male_yangguangqingnian_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E9%98%B3%E5%85%89%E9%9D%92%E5%B9%B4.mp3', NULL, 6, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0007', 'TTS_HuoshanDoubleStreamTTS', '京腔侃爷/Harmony', 'zh_male_jingqiangkanye_moon_bigtts', '中文、英文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/Harmony.mp3', NULL, 7, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0008', 'TTS_HuoshanDoubleStreamTTS', '湾湾小何', 'zh_female_wanwanxiaohe_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E6%B9%BE%E6%B9%BE%E5%B0%8F%E4%BD%95.mp3', NULL, 8, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0009', 'TTS_HuoshanDoubleStreamTTS', '湾区大叔', 'zh_female_wanqudashu_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E6%B9%BE%E5%8C%BA%E5%A4%A7%E5%8F%94.mp3', NULL, 9, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0010', 'TTS_HuoshanDoubleStreamTTS', '呆萌川妹', 'zh_female_daimengchuanmei_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E5%91%86%E8%90%8C%E5%B7%9D%E5%A6%B9.mp3', NULL, 10, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0011', 'TTS_HuoshanDoubleStreamTTS', '广州德哥', 'zh_male_guozhoudege_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E5%B9%BF%E5%B7%9E%E5%BE%B7%E5%93%A5.mp3', NULL, 11, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0012', 'TTS_HuoshanDoubleStreamTTS', '北京小爷', 'zh_male_beijingxiaoye_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E5%8C%97%E4%BA%AC%E5%B0%8F%E7%88%B7.mp3', NULL, 12, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0013', 'TTS_HuoshanDoubleStreamTTS', '浩宇小哥', 'zh_male_haoyuxiaoge_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E6%B5%A9%E5%AE%87%E5%B0%8F%E5%93%A5.mp3', NULL, 13, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0014', 'TTS_HuoshanDoubleStreamTTS', '广西远舟', 'zh_male_guangxiyuanzhou_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E5%B9%BF%E8%A5%BF%E8%BF%9C%E8%88%9F.mp3', NULL, 14, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0015', 'TTS_HuoshanDoubleStreamTTS', '妹坨洁儿', 'zh_female_meituojieer_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E5%A6%B9%E5%9D%A8%E6%B4%81%E5%84%BF.mp3', NULL, 15, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0016', 'TTS_HuoshanDoubleStreamTTS', '豫州子轩', 'zh_male_yuzhouzixuan_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E8%B1%AB%E5%B7%9E%E5%AD%90%E8%BD%A9.mp3', NULL, 16, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0017', 'TTS_HuoshanDoubleStreamTTS', '高冷御姐', 'zh_female_gaolengyujie_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E9%AB%98%E5%86%B7%E5%BE%A1%E5%A7%90.mp3', NULL, 17, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0018', 'TTS_HuoshanDoubleStreamTTS', '傲娇霸总', 'zh_male_aojiaobazong_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E5%82%B2%E5%A8%87%E9%9C%B8%E6%80%BB.mp3', NULL, 18, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0019', 'TTS_HuoshanDoubleStreamTTS', '魅力女友', 'zh_female_meilinvyou_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E9%AD%85%E5%8A%9B%E5%A5%B3%E5%8F%8B.mp3', NULL, 19, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0020', 'TTS_HuoshanDoubleStreamTTS', '深夜播客', 'zh_male_shenyeboke_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E6%B7%B1%E5%A4%9C%E6%92%AD%E5%AE%A2.mp3', NULL, 20, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0021', 'TTS_HuoshanDoubleStreamTTS', '柔美女友', 'zh_female_sajiaonvyou_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E6%9F%94%E7%BE%8E%E5%A5%B3%E5%8F%8B.mp3', NULL, 21, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0022', 'TTS_HuoshanDoubleStreamTTS', '撒娇学妹', 'zh_female_yuanqinvyou_moon_bigtts', '中文', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E6%92%92%E5%A8%87%E5%AD%A6%E5%A6%B9.mp3', NULL, 22, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0023', 'TTS_HuoshanDoubleStreamTTS', 'かずね(和音)', 'multi_male_jingqiangkanye_moon_bigtts', '日语、西语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/Javier.wav', NULL, 23, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0024', 'TTS_HuoshanDoubleStreamTTS', 'はるこ(晴子)', 'multi_female_shuangkuaisisi_moon_bigtts', '日语、西语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/Esmeralda.mp3', NULL, 24, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0025', 'TTS_HuoshanDoubleStreamTTS', 'あけみ(朱美)', 'multi_female_gaolengyujie_moon_bigtts', '日语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/%E6%9C%B1%E7%BE%8E.mp3', NULL, 25, NULL, NULL, NULL, NULL);
|
||||||
|
INSERT INTO `ai_tts_voice` VALUES ('TTS_HuoshanDoubleStreamTTS_0026', 'TTS_HuoshanDoubleStreamTTS', 'ひろし(広志)', 'multi_male_wanqudashu_moon_bigtts', '日语、西语', 'https://lf3-static.bytednsdoc.com/obj/eden-cn/lm_hz_ihsph/ljhwZthlaukjlkulzlp/portal/bigtts/Roberto.wav', NULL, 26, NULL, NULL, NULL, NULL);
|
||||||
@@ -0,0 +1,29 @@
|
|||||||
|
-- VLLM模型供应器
|
||||||
|
delete from `ai_model_provider` where id = 'SYSTEM_VLLM_openai';
|
||||||
|
INSERT INTO `ai_model_provider` (`id`, `model_type`, `provider_code`, `name`, `fields`, `sort`, `creator`, `create_date`, `updater`, `update_date`) VALUES
|
||||||
|
('SYSTEM_VLLM_openai', 'VLLM', 'openai', 'OpenAI接口', '[{"key":"base_url","label":"基础URL","type":"string"},{"key":"model_name","label":"模型名称","type":"string"},{"key":"api_key","label":"API密钥","type":"string"}]', 9, 1, NOW(), 1, NOW());
|
||||||
|
|
||||||
|
-- VLLM模型配置
|
||||||
|
delete from `ai_model_config` where id = 'VLLM_ChatGLMVLLM';
|
||||||
|
INSERT INTO `ai_model_config` VALUES ('VLLM_ChatGLMVLLM', 'VLLM', 'ChatGLMVLLM', '智谱视觉AI', 1, 1, '{\"type\": \"openai\", \"model_name\": \"glm-4v-flash\", \"base_url\": \"https://open.bigmodel.cn/api/paas/v4/\", \"api_key\": \"你的api_key\"}', NULL, NULL, 1, NULL, NULL, NULL, NULL);
|
||||||
|
|
||||||
|
-- 更新文档
|
||||||
|
UPDATE `ai_model_config` SET
|
||||||
|
`doc_link` = 'https://bigmodel.cn/usercenter/proj-mgmt/apikeys',
|
||||||
|
`remark` = '智谱视觉AI配置说明:
|
||||||
|
1. 访问 https://bigmodel.cn/usercenter/proj-mgmt/apikeys
|
||||||
|
2. 注册并获取API密钥
|
||||||
|
3. 填入配置文件中' WHERE `id` = 'VLLM_ChatGLMVLLM';
|
||||||
|
|
||||||
|
|
||||||
|
-- 添加参数
|
||||||
|
INSERT INTO `sys_params` (id, param_code, param_value, value_type, param_type, remark) VALUES (113, 'server.http_port', '8003', 'number', 1, 'http服务的端口,用于启动视觉分析接口');
|
||||||
|
INSERT INTO `sys_params` (id, param_code, param_value, value_type, param_type, remark) VALUES (114, 'server.vision_explain', 'null', 'string', 1, '视觉分析接口地址,用于下发到设备,多个用;分隔');
|
||||||
|
|
||||||
|
-- 智能体表增加VLLM模型配置
|
||||||
|
ALTER TABLE `ai_agent`
|
||||||
|
ADD COLUMN `vllm_model_id` varchar(32) NULL DEFAULT 'VLLM_ChatGLMVLLM' COMMENT '视觉模型标识' AFTER `llm_model_id`;
|
||||||
|
|
||||||
|
-- 智能体模版表增加VLLM模型配置
|
||||||
|
ALTER TABLE `ai_agent_template`
|
||||||
|
ADD COLUMN `vllm_model_id` varchar(32) NULL DEFAULT 'VLLM_ChatGLMVLLM' COMMENT '视觉模型标识' AFTER `llm_model_id`;
|
||||||
@@ -163,3 +163,17 @@ databaseChangeLog:
|
|||||||
- sqlFile:
|
- sqlFile:
|
||||||
encoding: utf8
|
encoding: utf8
|
||||||
path: classpath:db/changelog/202505151451.sql
|
path: classpath:db/changelog/202505151451.sql
|
||||||
|
- changeSet:
|
||||||
|
id: 202505271414
|
||||||
|
author: hrz
|
||||||
|
changes:
|
||||||
|
- sqlFile:
|
||||||
|
encoding: utf8
|
||||||
|
path: classpath:db/changelog/202505271414.sql
|
||||||
|
- changeSet:
|
||||||
|
id: 202506010920
|
||||||
|
author: hrz
|
||||||
|
changes:
|
||||||
|
- sqlFile:
|
||||||
|
encoding: utf8
|
||||||
|
path: classpath:db/changelog/202506010920.sql
|
||||||
@@ -14,10 +14,10 @@
|
|||||||
</div>
|
</div>
|
||||||
</div>
|
</div>
|
||||||
<div class="device-name">
|
<div class="device-name">
|
||||||
设备型号:{{ device.ttsModelName }}
|
语言模型:{{ device.llmModelName }}
|
||||||
</div>
|
</div>
|
||||||
<div class="device-name">
|
<div class="device-name">
|
||||||
音色模型:{{ device.ttsVoiceName }}
|
音色模型:{{ device.ttsModelName }} ({{ device.ttsVoiceName }})
|
||||||
</div>
|
</div>
|
||||||
<div style="display: flex;gap: 10px;align-items: center;">
|
<div style="display: flex;gap: 10px;align-items: center;">
|
||||||
<div class="settings-btn" @click="handleConfigure">
|
<div class="settings-btn" @click="handleConfigure">
|
||||||
|
|||||||
@@ -51,7 +51,7 @@
|
|||||||
字典管理
|
字典管理
|
||||||
</el-dropdown-item>
|
</el-dropdown-item>
|
||||||
<el-dropdown-item @click.native="goProviderManagement">
|
<el-dropdown-item @click.native="goProviderManagement">
|
||||||
供应器管理
|
字段管理
|
||||||
</el-dropdown-item>
|
</el-dropdown-item>
|
||||||
<el-dropdown-item @click.native="goServerSideManagement">
|
<el-dropdown-item @click.native="goServerSideManagement">
|
||||||
服务端管理
|
服务端管理
|
||||||
|
|||||||
@@ -30,6 +30,9 @@
|
|||||||
<el-menu-item index="llm">
|
<el-menu-item index="llm">
|
||||||
<span class="menu-text">大语言模型</span>
|
<span class="menu-text">大语言模型</span>
|
||||||
</el-menu-item>
|
</el-menu-item>
|
||||||
|
<el-menu-item index="vllm">
|
||||||
|
<span class="menu-text">视觉大语言模型</span>
|
||||||
|
</el-menu-item>
|
||||||
<el-menu-item index="intent">
|
<el-menu-item index="intent">
|
||||||
<span class="menu-text">意图识别</span>
|
<span class="menu-text">意图识别</span>
|
||||||
</el-menu-item>
|
</el-menu-item>
|
||||||
@@ -173,6 +176,7 @@ export default {
|
|||||||
vad: '语言活动检测模型(VAD)',
|
vad: '语言活动检测模型(VAD)',
|
||||||
asr: '语音识别模型(ASR)',
|
asr: '语音识别模型(ASR)',
|
||||||
llm: '大语言模型(LLM)',
|
llm: '大语言模型(LLM)',
|
||||||
|
vllm: '视觉大语言模型(VLLM)',
|
||||||
intent: '意图识别模型(Intent)',
|
intent: '意图识别模型(Intent)',
|
||||||
tts: '语音合成模型(TTS)',
|
tts: '语音合成模型(TTS)',
|
||||||
memory: '记忆模型(Memory)'
|
memory: '记忆模型(Memory)'
|
||||||
@@ -467,7 +471,7 @@ export default {
|
|||||||
.main-wrapper {
|
.main-wrapper {
|
||||||
margin: 5px 22px;
|
margin: 5px 22px;
|
||||||
border-radius: 15px;
|
border-radius: 15px;
|
||||||
min-height: calc(100vh - 24vh);
|
min-height: calc(100vh - 26vh);
|
||||||
height: auto;
|
height: auto;
|
||||||
max-height: 80vh;
|
max-height: 80vh;
|
||||||
box-shadow: 0 2px 12px rgba(0, 0, 0, 0.1);
|
box-shadow: 0 2px 12px rgba(0, 0, 0, 0.1);
|
||||||
|
|||||||
@@ -3,7 +3,7 @@
|
|||||||
<HeaderBar />
|
<HeaderBar />
|
||||||
|
|
||||||
<div class="operation-bar">
|
<div class="operation-bar">
|
||||||
<h2 class="page-title">供应器管理</h2>
|
<h2 class="page-title">字段管理</h2>
|
||||||
<div class="right-operations">
|
<div class="right-operations">
|
||||||
<el-dropdown trigger="click" @command="handleSelectModelType" @visible-change="handleDropdownVisibleChange">
|
<el-dropdown trigger="click" @command="handleSelectModelType" @visible-change="handleDropdownVisibleChange">
|
||||||
<el-button class="category-btn">
|
<el-button class="category-btn">
|
||||||
|
|||||||
@@ -64,7 +64,27 @@
|
|||||||
</el-form-item>
|
</el-form-item>
|
||||||
</div>
|
</div>
|
||||||
<div class="form-column">
|
<div class="form-column">
|
||||||
<el-form-item v-for="(model, index) in models" :key="`model-${index}`" :label="model.label"
|
<div class="model-row">
|
||||||
|
<el-form-item label="语音活动检测(VAD)" class="model-item">
|
||||||
|
<div class="model-select-wrapper">
|
||||||
|
<el-select v-model="form.model.vadModelId" filterable placeholder="请选择" class="form-select"
|
||||||
|
@change="handleModelChange('VAD', $event)">
|
||||||
|
<el-option v-for="(item, optionIndex) in modelOptions['VAD']"
|
||||||
|
:key="`option-vad-${optionIndex}`" :label="item.label" :value="item.value" />
|
||||||
|
</el-select>
|
||||||
|
</div>
|
||||||
|
</el-form-item>
|
||||||
|
<el-form-item label="语音识别(ASR)" class="model-item">
|
||||||
|
<div class="model-select-wrapper">
|
||||||
|
<el-select v-model="form.model.asrModelId" filterable placeholder="请选择" class="form-select"
|
||||||
|
@change="handleModelChange('ASR', $event)">
|
||||||
|
<el-option v-for="(item, optionIndex) in modelOptions['ASR']"
|
||||||
|
:key="`option-asr-${optionIndex}`" :label="item.label" :value="item.value" />
|
||||||
|
</el-select>
|
||||||
|
</div>
|
||||||
|
</el-form-item>
|
||||||
|
</div>
|
||||||
|
<el-form-item v-for="(model, index) in models.slice(2)" :key="`model-${index}`" :label="model.label"
|
||||||
class="model-item">
|
class="model-item">
|
||||||
<div class="model-select-wrapper">
|
<div class="model-select-wrapper">
|
||||||
<el-select v-model="form.model[model.key]" filterable placeholder="请选择" class="form-select"
|
<el-select v-model="form.model[model.key]" filterable placeholder="请选择" class="form-select"
|
||||||
@@ -148,6 +168,7 @@ export default {
|
|||||||
vadModelId: "",
|
vadModelId: "",
|
||||||
asrModelId: "",
|
asrModelId: "",
|
||||||
llmModelId: "",
|
llmModelId: "",
|
||||||
|
vllmModelId: "",
|
||||||
memModelId: "",
|
memModelId: "",
|
||||||
intentModelId: "",
|
intentModelId: "",
|
||||||
}
|
}
|
||||||
@@ -156,6 +177,7 @@ export default {
|
|||||||
{ label: '语音活动检测(VAD)', key: 'vadModelId', type: 'VAD' },
|
{ label: '语音活动检测(VAD)', key: 'vadModelId', type: 'VAD' },
|
||||||
{ label: '语音识别(ASR)', key: 'asrModelId', type: 'ASR' },
|
{ label: '语音识别(ASR)', key: 'asrModelId', type: 'ASR' },
|
||||||
{ label: '大语言模型(LLM)', key: 'llmModelId', type: 'LLM' },
|
{ label: '大语言模型(LLM)', key: 'llmModelId', type: 'LLM' },
|
||||||
|
{ label: '视觉大语言模型(VLLM)', key: 'vllmModelId', type: 'VLLM' },
|
||||||
{ label: '意图识别(Intent)', key: 'intentModelId', type: 'Intent' },
|
{ label: '意图识别(Intent)', key: 'intentModelId', type: 'Intent' },
|
||||||
{ label: '记忆(Memory)', key: 'memModelId', type: 'Memory' },
|
{ label: '记忆(Memory)', key: 'memModelId', type: 'Memory' },
|
||||||
{ label: '语音合成(TTS)', key: 'ttsModelId', type: 'TTS' },
|
{ label: '语音合成(TTS)', key: 'ttsModelId', type: 'TTS' },
|
||||||
@@ -189,6 +211,7 @@ export default {
|
|||||||
asrModelId: this.form.model.asrModelId,
|
asrModelId: this.form.model.asrModelId,
|
||||||
vadModelId: this.form.model.vadModelId,
|
vadModelId: this.form.model.vadModelId,
|
||||||
llmModelId: this.form.model.llmModelId,
|
llmModelId: this.form.model.llmModelId,
|
||||||
|
vllmModelId: this.form.model.vllmModelId,
|
||||||
ttsModelId: this.form.model.ttsModelId,
|
ttsModelId: this.form.model.ttsModelId,
|
||||||
ttsVoiceId: this.form.ttsVoiceId,
|
ttsVoiceId: this.form.ttsVoiceId,
|
||||||
chatHistoryConf: this.form.chatHistoryConf,
|
chatHistoryConf: this.form.chatHistoryConf,
|
||||||
@@ -236,6 +259,7 @@ export default {
|
|||||||
vadModelId: "",
|
vadModelId: "",
|
||||||
asrModelId: "",
|
asrModelId: "",
|
||||||
llmModelId: "",
|
llmModelId: "",
|
||||||
|
vllmModelId: "",
|
||||||
memModelId: "",
|
memModelId: "",
|
||||||
intentModelId: "",
|
intentModelId: "",
|
||||||
}
|
}
|
||||||
@@ -289,6 +313,7 @@ export default {
|
|||||||
vadModelId: templateData.vadModelId || this.form.model.vadModelId,
|
vadModelId: templateData.vadModelId || this.form.model.vadModelId,
|
||||||
asrModelId: templateData.asrModelId || this.form.model.asrModelId,
|
asrModelId: templateData.asrModelId || this.form.model.asrModelId,
|
||||||
llmModelId: templateData.llmModelId || this.form.model.llmModelId,
|
llmModelId: templateData.llmModelId || this.form.model.llmModelId,
|
||||||
|
vllmModelId: templateData.vllmModelId || this.form.model.vllmModelId,
|
||||||
memModelId: templateData.memModelId || this.form.model.memModelId,
|
memModelId: templateData.memModelId || this.form.model.memModelId,
|
||||||
intentModelId: templateData.intentModelId || this.form.model.intentModelId
|
intentModelId: templateData.intentModelId || this.form.model.intentModelId
|
||||||
}
|
}
|
||||||
@@ -305,6 +330,7 @@ export default {
|
|||||||
vadModelId: data.data.vadModelId,
|
vadModelId: data.data.vadModelId,
|
||||||
asrModelId: data.data.asrModelId,
|
asrModelId: data.data.asrModelId,
|
||||||
llmModelId: data.data.llmModelId,
|
llmModelId: data.data.llmModelId,
|
||||||
|
vllmModelId: data.data.vllmModelId,
|
||||||
memModelId: data.data.memModelId,
|
memModelId: data.data.memModelId,
|
||||||
intentModelId: data.data.intentModelId
|
intentModelId: data.data.intentModelId
|
||||||
}
|
}
|
||||||
@@ -587,6 +613,25 @@ export default {
|
|||||||
width: 100%;
|
width: 100%;
|
||||||
}
|
}
|
||||||
|
|
||||||
|
.model-row {
|
||||||
|
display: flex;
|
||||||
|
gap: 20px;
|
||||||
|
margin-bottom: 6px;
|
||||||
|
}
|
||||||
|
|
||||||
|
.model-row .model-item {
|
||||||
|
flex: 1;
|
||||||
|
margin-bottom: 0;
|
||||||
|
}
|
||||||
|
|
||||||
|
.model-row .el-form-item__label {
|
||||||
|
font-size: 12px !important;
|
||||||
|
color: #3d4566 !important;
|
||||||
|
font-weight: 400;
|
||||||
|
line-height: 22px;
|
||||||
|
padding-bottom: 2px;
|
||||||
|
}
|
||||||
|
|
||||||
.function-icons {
|
.function-icons {
|
||||||
display: flex;
|
display: flex;
|
||||||
align-items: center;
|
align-items: center;
|
||||||
|
|||||||
+19
-11
@@ -1,13 +1,14 @@
|
|||||||
import asyncio
|
|
||||||
import sys
|
import sys
|
||||||
|
import uuid
|
||||||
import signal
|
import signal
|
||||||
|
import asyncio
|
||||||
|
from aioconsole import ainput
|
||||||
from config.settings import load_config
|
from config.settings import load_config
|
||||||
from core.websocket_server import WebSocketServer
|
|
||||||
from core.ota_server import SimpleOtaServer
|
|
||||||
from core.utils.util import check_ffmpeg_installed
|
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
from core.utils.util import get_local_ip
|
from core.utils.util import get_local_ip
|
||||||
from aioconsole import ainput
|
from core.http_server import SimpleHttpServer
|
||||||
|
from core.websocket_server import WebSocketServer
|
||||||
|
from core.utils.util import check_ffmpeg_installed
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
logger = setup_logging()
|
logger = setup_logging()
|
||||||
@@ -45,25 +46,32 @@ async def main():
|
|||||||
check_ffmpeg_installed()
|
check_ffmpeg_installed()
|
||||||
config = load_config()
|
config = load_config()
|
||||||
|
|
||||||
|
# 生成随机密钥并添加到配置中
|
||||||
|
config["server"]["auth_key"] = str(uuid.uuid4().hex)
|
||||||
|
|
||||||
# 添加 stdin 监控任务
|
# 添加 stdin 监控任务
|
||||||
stdin_task = asyncio.create_task(monitor_stdin())
|
stdin_task = asyncio.create_task(monitor_stdin())
|
||||||
|
|
||||||
# 启动 WebSocket 服务器
|
# 启动 WebSocket 服务器
|
||||||
ws_server = WebSocketServer(config)
|
ws_server = WebSocketServer(config)
|
||||||
ws_task = asyncio.create_task(ws_server.start())
|
ws_task = asyncio.create_task(ws_server.start())
|
||||||
ota_task = None
|
# 启动 Simple http 服务器
|
||||||
|
ota_server = SimpleHttpServer(config)
|
||||||
|
ota_task = asyncio.create_task(ota_server.start())
|
||||||
|
|
||||||
read_config_from_api = config.get("read_config_from_api", False)
|
read_config_from_api = config.get("read_config_from_api", False)
|
||||||
|
port = int(config["server"].get("http_port", 8003))
|
||||||
if not read_config_from_api:
|
if not read_config_from_api:
|
||||||
# 启动 Simple OTA 服务器
|
|
||||||
ota_server = SimpleOtaServer(config)
|
|
||||||
ota_task = asyncio.create_task(ota_server.start())
|
|
||||||
|
|
||||||
logger.bind(tag=TAG).info(
|
logger.bind(tag=TAG).info(
|
||||||
"OTA接口是\t\thttp://{}:{}/xiaozhi/ota/",
|
"OTA接口是\t\thttp://{}:{}/xiaozhi/ota/",
|
||||||
get_local_ip(),
|
get_local_ip(),
|
||||||
config["server"]["ota_port"],
|
port,
|
||||||
)
|
)
|
||||||
|
logger.bind(tag=TAG).info(
|
||||||
|
"视觉分析接口是\thttp://{}:{}/mcp/vision/explain",
|
||||||
|
get_local_ip(),
|
||||||
|
port,
|
||||||
|
)
|
||||||
|
|
||||||
# 获取WebSocket配置,使用安全的默认值
|
# 获取WebSocket配置,使用安全的默认值
|
||||||
websocket_port = 8000
|
websocket_port = 8000
|
||||||
|
|||||||
@@ -2,6 +2,7 @@
|
|||||||
# 然后你想修改覆盖修改什么配置,就修改【.config.yaml】文件,而不是修改【config.yaml】文件
|
# 然后你想修改覆盖修改什么配置,就修改【.config.yaml】文件,而不是修改【config.yaml】文件
|
||||||
# 系统会优先读取【data/.config.yaml】文件的配置,如果【.config.yaml】文件里的配置不存在,系统会自动去读取【config.yaml】文件的配置。
|
# 系统会优先读取【data/.config.yaml】文件的配置,如果【.config.yaml】文件里的配置不存在,系统会自动去读取【config.yaml】文件的配置。
|
||||||
# 这样做,可以最简化配置,保护您的密钥安全。
|
# 这样做,可以最简化配置,保护您的密钥安全。
|
||||||
|
# 如果你使用了智控台,那么以下所有配置,都不会生效,请在智控台中修改配置
|
||||||
|
|
||||||
# #####################################################################################
|
# #####################################################################################
|
||||||
# #############################以下是服务器基本运行配置####################################
|
# #############################以下是服务器基本运行配置####################################
|
||||||
@@ -9,13 +10,21 @@ server:
|
|||||||
# 服务器监听地址和端口(Server listening address and port)
|
# 服务器监听地址和端口(Server listening address and port)
|
||||||
ip: 0.0.0.0
|
ip: 0.0.0.0
|
||||||
port: 8000
|
port: 8000
|
||||||
# OTA接口的端口号
|
# http服务的端口,用于简单OTA接口(单服务部署),以及视觉分析接口
|
||||||
ota_port: 8002
|
http_port: 8003
|
||||||
# 这个websocket配置是指ota接口向设备发送的websocket地址
|
# 这个websocket配置是指ota接口向设备发送的websocket地址
|
||||||
# 如果按默认的写法,ota接口会自动生成websocket地址。这个地址你可以直接用浏览器访问ota接口确认一下
|
# 如果按默认的写法,ota接口会自动生成websocket地址,并输出在启动日志里,这个地址你可以直接用浏览器访问ota接口确认一下
|
||||||
# 当你使用docker部署或使用公网部署(使用ssl、域名)时,不一定准确
|
# 当你使用docker部署或使用公网部署(使用ssl、域名)时,不一定准确
|
||||||
# 所以如果你使用docker部署或使用公网部署时,请设置正确的websocket地址
|
# 所以如果你使用docker部署时,将websocket设置成局域网地址
|
||||||
|
# 如果你使用公网部署时,将vwebsocket设置成公网地址
|
||||||
websocket: ws://你的ip或者域名:端口号/xiaozhi/v1/
|
websocket: ws://你的ip或者域名:端口号/xiaozhi/v1/
|
||||||
|
# 视觉分析接口地址
|
||||||
|
# 向设备发送的视觉分析的接口地址
|
||||||
|
# 如果按下面默认的写法,系统会自动生成视觉识别地址,并输出在启动日志里,这个地址你可以直接用浏览器访问确认一下
|
||||||
|
# 当你使用docker部署或使用公网部署(使用ssl、域名)时,不一定准确
|
||||||
|
# 所以如果你使用docker部署时,将vision_explain设置成局域网地址
|
||||||
|
# 如果你使用公网部署时,将vision_explain设置成公网地址
|
||||||
|
vision_explain: http://你的ip或者域名:端口号/mcp/vision/explain
|
||||||
# OTA返回信息时区偏移量
|
# OTA返回信息时区偏移量
|
||||||
timezone_offset: +8
|
timezone_offset: +8
|
||||||
# 认证配置
|
# 认证配置
|
||||||
@@ -85,6 +94,7 @@ module_test:
|
|||||||
# 唤醒词,用于识别唤醒词还是讲话内容
|
# 唤醒词,用于识别唤醒词还是讲话内容
|
||||||
wakeup_words:
|
wakeup_words:
|
||||||
- "你好小智"
|
- "你好小智"
|
||||||
|
- "嘿你好呀"
|
||||||
- "你好小志"
|
- "你好小志"
|
||||||
- "小爱同学"
|
- "小爱同学"
|
||||||
- "你好小鑫"
|
- "你好小鑫"
|
||||||
@@ -159,6 +169,8 @@ selected_module:
|
|||||||
ASR: FunASR
|
ASR: FunASR
|
||||||
# 将根据配置名称对应的type调用实际的LLM适配器
|
# 将根据配置名称对应的type调用实际的LLM适配器
|
||||||
LLM: ChatGLMLLM
|
LLM: ChatGLMLLM
|
||||||
|
# 视觉语言大模型
|
||||||
|
VLLM: ChatGLMVLLM
|
||||||
# TTS将根据配置名称对应的type调用实际的TTS适配器
|
# TTS将根据配置名称对应的type调用实际的TTS适配器
|
||||||
TTS: EdgeTTS
|
TTS: EdgeTTS
|
||||||
# 记忆模块,默认不开启记忆;如果想使用超长记忆,推荐使用mem0ai;如果注重隐私,请使用本地的mem_local_short
|
# 记忆模块,默认不开启记忆;如果想使用超长记忆,推荐使用mem0ai;如果注重隐私,请使用本地的mem_local_short
|
||||||
@@ -218,8 +230,12 @@ Memory:
|
|||||||
# 不想使用记忆功能,可以使用nomem
|
# 不想使用记忆功能,可以使用nomem
|
||||||
type: nomem
|
type: nomem
|
||||||
mem_local_short:
|
mem_local_short:
|
||||||
# 本地记忆功能,通过selected_module的llm总结,数据保存在本地,不会上传到服务器
|
# 本地记忆功能,通过selected_module的llm总结,数据保存在本地服务器,不会上传到外部服务器
|
||||||
type: mem_local_short
|
type: mem_local_short
|
||||||
|
# 配备记忆存储独立的思考模型
|
||||||
|
# 如果这里不填,则会默认使用selected_module.LLM的模型作为意图识别的思考模型
|
||||||
|
# 如果你的不想使用selected_module.LLM记忆存储,这里最好使用独立的LLM作为意图识别,例如使用免费的ChatGLMLLM
|
||||||
|
llm: ChatGLMLLM
|
||||||
|
|
||||||
ASR:
|
ASR:
|
||||||
FunASR:
|
FunASR:
|
||||||
@@ -325,7 +341,7 @@ LLM:
|
|||||||
# 定义LLM API类型
|
# 定义LLM API类型
|
||||||
type: openai
|
type: openai
|
||||||
# 先开通服务,打开以下网址,开通的服务搜索Doubao-1.5-pro,开通它
|
# 先开通服务,打开以下网址,开通的服务搜索Doubao-1.5-pro,开通它
|
||||||
# 开通改地址:https://console.volcengine.com/ark/region:ark+cn-beijing/openManagement?LLM=%7B%7D&OpenTokenDrawer=false
|
# 开通地址:https://console.volcengine.com/ark/region:ark+cn-beijing/openManagement?LLM=%7B%7D&OpenTokenDrawer=false
|
||||||
# 免费额度500000token
|
# 免费额度500000token
|
||||||
# 开通后,进入这里获取密钥:https://console.volcengine.com/ark/region:ark+cn-beijing/apiKey?apikey=%7B%7D
|
# 开通后,进入这里获取密钥:https://console.volcengine.com/ark/region:ark+cn-beijing/apiKey?apikey=%7B%7D
|
||||||
base_url: https://ark.cn-beijing.volces.com/api/v3
|
base_url: https://ark.cn-beijing.volces.com/api/v3
|
||||||
@@ -427,6 +443,15 @@ LLM:
|
|||||||
# Xinference服务地址和模型名称
|
# Xinference服务地址和模型名称
|
||||||
model_name: qwen2.5:3b-AWQ # 使用的小模型名称,用于意图识别
|
model_name: qwen2.5:3b-AWQ # 使用的小模型名称,用于意图识别
|
||||||
base_url: http://localhost:9997 # Xinference服务地址
|
base_url: http://localhost:9997 # Xinference服务地址
|
||||||
|
# VLLM配置(视觉语言大模型)
|
||||||
|
VLLM:
|
||||||
|
ChatGLMVLLM:
|
||||||
|
type: openai
|
||||||
|
# glm-4v-flash是智谱免费AI的视觉模型,需要先在智谱AI平台创建API密钥并获取api_key
|
||||||
|
# 可在这里找到你的api key https://bigmodel.cn/usercenter/proj-mgmt/apikeys
|
||||||
|
model_name: glm-4v-flash # 智谱AI的视觉模型
|
||||||
|
url: https://open.bigmodel.cn/api/paas/v4/
|
||||||
|
api_key: 你的api_key
|
||||||
TTS:
|
TTS:
|
||||||
# 当前支持的type为edge、doubao,可自行适配
|
# 当前支持的type为edge、doubao,可自行适配
|
||||||
EdgeTTS:
|
EdgeTTS:
|
||||||
@@ -452,6 +477,19 @@ TTS:
|
|||||||
speed_ratio: 1.0
|
speed_ratio: 1.0
|
||||||
volume_ratio: 1.0
|
volume_ratio: 1.0
|
||||||
pitch_ratio: 1.0
|
pitch_ratio: 1.0
|
||||||
|
#火山tts,支持双向流式tts
|
||||||
|
HuoshanDoubleStreamTTS:
|
||||||
|
type: huoshan_double_stream
|
||||||
|
# 访问 https://console.volcengine.com/speech/service/10007 开通语音合成大模型,购买音色
|
||||||
|
# 在页面底部获取appid和access_token
|
||||||
|
# 资源ID固定为:volc.service_type.10029(大模型语音合成及混音)
|
||||||
|
# 如果是机智云,把接口地址换成wss://bytedance.gizwitsapi.com/api/v3/tts/bidirection
|
||||||
|
# 机智云不需要天填 appid
|
||||||
|
ws_url: wss://openspeech.bytedance.com/api/v3/tts/bidirection
|
||||||
|
appid: 你的火山引擎语音合成服务appid
|
||||||
|
access_token: 你的火山引擎语音合成服务access_token
|
||||||
|
resource_id: volc.service_type.10029
|
||||||
|
speaker: zh_female_wanwanxiaohe_moon_bigtts
|
||||||
CosyVoiceSiliconflow:
|
CosyVoiceSiliconflow:
|
||||||
type: siliconflow
|
type: siliconflow
|
||||||
# 硅基流动TTS
|
# 硅基流动TTS
|
||||||
|
|||||||
@@ -59,10 +59,14 @@ def get_config_from_api(config):
|
|||||||
"url": config["manager-api"].get("url", ""),
|
"url": config["manager-api"].get("url", ""),
|
||||||
"secret": config["manager-api"].get("secret", ""),
|
"secret": config["manager-api"].get("secret", ""),
|
||||||
}
|
}
|
||||||
|
# server的配置以本地为准
|
||||||
if config.get("server"):
|
if config.get("server"):
|
||||||
config_data["server"] = {
|
config_data["server"] = {
|
||||||
"ip": config["server"].get("ip", ""),
|
"ip": config["server"].get("ip", ""),
|
||||||
"port": config["server"].get("port", ""),
|
"port": config["server"].get("port", ""),
|
||||||
|
"http_port": config["server"].get("http_port", ""),
|
||||||
|
"vision_explain": config["server"].get("vision_explain", ""),
|
||||||
|
"auth_key": config["server"].get("auth_key", ""),
|
||||||
}
|
}
|
||||||
return config_data
|
return config_data
|
||||||
|
|
||||||
|
|||||||
@@ -4,7 +4,8 @@ from loguru import logger
|
|||||||
from config.config_loader import load_config
|
from config.config_loader import load_config
|
||||||
from config.settings import check_config_file
|
from config.settings import check_config_file
|
||||||
|
|
||||||
SERVER_VERSION = "0.4.4"
|
SERVER_VERSION = "0.5.2"
|
||||||
|
_logger_initialized = False
|
||||||
|
|
||||||
|
|
||||||
def get_module_abbreviation(module_name, module_dict):
|
def get_module_abbreviation(module_name, module_dict):
|
||||||
@@ -43,41 +44,105 @@ def setup_logging():
|
|||||||
"""从配置文件中读取日志配置,并设置日志输出格式和级别"""
|
"""从配置文件中读取日志配置,并设置日志输出格式和级别"""
|
||||||
config = load_config()
|
config = load_config()
|
||||||
log_config = config["log"]
|
log_config = config["log"]
|
||||||
log_format = log_config.get(
|
global _logger_initialized
|
||||||
"log_format",
|
|
||||||
"<green>{time:YYMMDD HH:mm:ss}</green>[{version}_{selected_module}][<light-blue>{extra[tag]}</light-blue>]-<level>{level}</level>-<light-green>{message}</light-green>",
|
|
||||||
)
|
|
||||||
log_format_file = log_config.get(
|
|
||||||
"log_format_file",
|
|
||||||
"{time:YYYY-MM-DD HH:mm:ss} - {version_{selected_module}} - {name} - {level} - {extra[tag]} - {message}",
|
|
||||||
)
|
|
||||||
selected_module_str = build_module_string(config.get("selected_module", {}))
|
|
||||||
|
|
||||||
log_format = log_format.replace("{version}", SERVER_VERSION)
|
# 第一次初始化时配置日志
|
||||||
log_format = log_format.replace("{selected_module}", selected_module_str)
|
if not _logger_initialized:
|
||||||
log_format_file = log_format_file.replace("{version}", SERVER_VERSION)
|
logger.configure(
|
||||||
log_format_file = log_format_file.replace("{selected_module}", selected_module_str)
|
extra={
|
||||||
|
"selected_module": log_config.get("selected_module", "00000000000000")
|
||||||
|
}
|
||||||
|
) # 新增配置
|
||||||
|
log_format = log_config.get(
|
||||||
|
"log_format",
|
||||||
|
"<green>{time:YYMMDD HH:mm:ss}</green>[{version}_{extra[selected_module]}][<light-blue>{extra[tag]}</light-blue>]-<level>{level}</level>-<light-green>{message}</light-green>",
|
||||||
|
)
|
||||||
|
log_format_file = log_config.get(
|
||||||
|
"log_format_file",
|
||||||
|
"{time:YYYY-MM-DD HH:mm:ss} - {version_{extra[selected_module]}} - {name} - {level} - {extra[tag]} - {message}",
|
||||||
|
)
|
||||||
|
selected_module_str = logger._core.extra["selected_module"]
|
||||||
|
|
||||||
log_level = log_config.get("log_level", "INFO")
|
log_format = log_format.replace("{version}", SERVER_VERSION)
|
||||||
log_dir = log_config.get("log_dir", "tmp")
|
log_format = log_format.replace("{selected_module}", selected_module_str)
|
||||||
log_file = log_config.get("log_file", "server.log")
|
log_format_file = log_format_file.replace("{version}", SERVER_VERSION)
|
||||||
data_dir = log_config.get("data_dir", "data")
|
log_format_file = log_format_file.replace(
|
||||||
|
"{selected_module}", selected_module_str
|
||||||
|
)
|
||||||
|
|
||||||
os.makedirs(log_dir, exist_ok=True)
|
log_level = log_config.get("log_level", "INFO")
|
||||||
os.makedirs(data_dir, exist_ok=True)
|
log_dir = log_config.get("log_dir", "tmp")
|
||||||
|
log_file = log_config.get("log_file", "server.log")
|
||||||
|
data_dir = log_config.get("data_dir", "data")
|
||||||
|
|
||||||
# 配置日志输出
|
os.makedirs(log_dir, exist_ok=True)
|
||||||
logger.remove()
|
os.makedirs(data_dir, exist_ok=True)
|
||||||
|
|
||||||
# 输出到控制台
|
# 配置日志输出
|
||||||
logger.add(sys.stdout, format=log_format, level=log_level, filter=formatter)
|
logger.remove()
|
||||||
|
|
||||||
# 输出到文件
|
# 输出到控制台
|
||||||
logger.add(
|
logger.add(sys.stdout, format=log_format, level=log_level, filter=formatter)
|
||||||
os.path.join(log_dir, log_file),
|
|
||||||
format=log_format_file,
|
# 输出到文件
|
||||||
level=log_level,
|
logger.add(
|
||||||
filter=formatter,
|
os.path.join(log_dir, log_file),
|
||||||
)
|
format=log_format_file,
|
||||||
|
level=log_level,
|
||||||
|
filter=formatter,
|
||||||
|
)
|
||||||
|
_logger_initialized = True # 标记为已初始化
|
||||||
|
|
||||||
return logger
|
return logger
|
||||||
|
|
||||||
|
|
||||||
|
def update_module_string(selected_module_str):
|
||||||
|
"""更新模块字符串并重新配置日志处理器"""
|
||||||
|
logger.debug(f"更新日志配置组件")
|
||||||
|
current_module = logger._core.extra["selected_module"]
|
||||||
|
|
||||||
|
if current_module == selected_module_str:
|
||||||
|
return
|
||||||
|
|
||||||
|
try:
|
||||||
|
logger.configure(extra={"selected_module": selected_module_str})
|
||||||
|
|
||||||
|
config = load_config()
|
||||||
|
log_config = config["log"]
|
||||||
|
|
||||||
|
log_format = log_config.get(
|
||||||
|
"log_format",
|
||||||
|
"<green>{time:YYMMDD HH:mm:ss}</green>[{version}_{extra[selected_module]}][<light-blue>{extra[tag]}</light-blue>]-<level>{level}</level>-<light-green>{message}</light-green>",
|
||||||
|
)
|
||||||
|
log_format_file = log_config.get(
|
||||||
|
"log_format_file",
|
||||||
|
"{time:YYYY-MM-DD HH:mm:ss} - {version_{extra[selected_module]}} - {name} - {level} - {extra[tag]} - {message}",
|
||||||
|
)
|
||||||
|
|
||||||
|
log_format = log_format.replace("{version}", SERVER_VERSION)
|
||||||
|
log_format = log_format.replace("{selected_module}", selected_module_str)
|
||||||
|
log_format_file = log_format_file.replace("{version}", SERVER_VERSION)
|
||||||
|
log_format_file = log_format_file.replace(
|
||||||
|
"{selected_module}", selected_module_str
|
||||||
|
)
|
||||||
|
|
||||||
|
logger.remove()
|
||||||
|
logger.add(
|
||||||
|
sys.stdout,
|
||||||
|
format=log_format,
|
||||||
|
level=log_config.get("log_level", "INFO"),
|
||||||
|
filter=formatter,
|
||||||
|
)
|
||||||
|
logger.add(
|
||||||
|
os.path.join(
|
||||||
|
log_config.get("log_dir", "tmp"),
|
||||||
|
log_config.get("log_file", "server.log"),
|
||||||
|
),
|
||||||
|
format=log_format_file,
|
||||||
|
level=log_config.get("log_level", "INFO"),
|
||||||
|
filter=formatter,
|
||||||
|
)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.error(f"日志配置更新失败: {str(e)}")
|
||||||
|
raise
|
||||||
|
|||||||
@@ -160,7 +160,7 @@ def save_mem_local_short(mac_address: str, short_momery: str) -> Optional[Dict]:
|
|||||||
|
|
||||||
|
|
||||||
def report(
|
def report(
|
||||||
mac_address: str, session_id: str, chat_type: int, content: str, audio
|
mac_address: str, session_id: str, chat_type: int, content: str, audio, report_time
|
||||||
) -> Optional[Dict]:
|
) -> Optional[Dict]:
|
||||||
"""带熔断的业务方法示例"""
|
"""带熔断的业务方法示例"""
|
||||||
if not content or not ManageApiClient._instance:
|
if not content or not ManageApiClient._instance:
|
||||||
@@ -174,6 +174,7 @@ def report(
|
|||||||
"sessionId": session_id,
|
"sessionId": session_id,
|
||||||
"chatType": chat_type,
|
"chatType": chat_type,
|
||||||
"content": content,
|
"content": content,
|
||||||
|
"reportTime": report_time,
|
||||||
"audioBase64": (
|
"audioBase64": (
|
||||||
base64.b64encode(audio).decode("utf-8") if audio else None
|
base64.b64encode(audio).decode("utf-8") if audio else None
|
||||||
),
|
),
|
||||||
|
|||||||
@@ -8,6 +8,15 @@
|
|||||||
server:
|
server:
|
||||||
ip: 0.0.0.0
|
ip: 0.0.0.0
|
||||||
port: 8000
|
port: 8000
|
||||||
|
# http服务的端口,用于视觉分析接口
|
||||||
|
http_port: 8003
|
||||||
|
# 视觉分析接口地址
|
||||||
|
# 向设备发送的视觉分析的接口地址
|
||||||
|
# 如果按下面默认的写法,系统会自动生成视觉识别地址,并输出在启动日志里,这个地址你可以直接用浏览器访问确认一下
|
||||||
|
# 当你使用docker部署或使用公网部署(使用ssl、域名)时,不一定准确
|
||||||
|
# 所以如果你使用docker部署时,将vision_explain设置成局域网地址
|
||||||
|
# 如果你使用公网部署时,将vision_explain设置成公网地址
|
||||||
|
vision_explain: http://你的ip或者域名:端口号/mcp/vision/explain
|
||||||
manager-api:
|
manager-api:
|
||||||
# 你的manager-api的地址,最好使用局域网ip
|
# 你的manager-api的地址,最好使用局域网ip
|
||||||
url: http://127.0.0.1:8002/xiaozhi
|
url: http://127.0.0.1:8002/xiaozhi
|
||||||
|
|||||||
@@ -0,0 +1,16 @@
|
|||||||
|
from aiohttp import web
|
||||||
|
from config.logger import setup_logging
|
||||||
|
|
||||||
|
|
||||||
|
class BaseHandler:
|
||||||
|
def __init__(self, config: dict):
|
||||||
|
self.config = config
|
||||||
|
self.logger = setup_logging()
|
||||||
|
|
||||||
|
def _add_cors_headers(self, response):
|
||||||
|
"""添加CORS头信息"""
|
||||||
|
response.headers["Access-Control-Allow-Headers"] = (
|
||||||
|
"client-id, content-type, device-id"
|
||||||
|
)
|
||||||
|
response.headers["Access-Control-Allow-Credentials"] = "true"
|
||||||
|
response.headers["Access-Control-Allow-Origin"] = "*"
|
||||||
+12
-54
@@ -1,18 +1,15 @@
|
|||||||
import json
|
import json
|
||||||
import time
|
import time
|
||||||
import asyncio
|
|
||||||
from aiohttp import web
|
from aiohttp import web
|
||||||
from config.logger import setup_logging
|
from core.utils.util import get_local_ip
|
||||||
from core.connection import ConnectionHandler
|
from core.api.base_handler import BaseHandler
|
||||||
from core.utils.util import get_local_ip, initialize_modules
|
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
|
|
||||||
|
|
||||||
class SimpleOtaServer:
|
class OTAHandler(BaseHandler):
|
||||||
def __init__(self, config: dict):
|
def __init__(self, config: dict):
|
||||||
self.config = config
|
super().__init__(config)
|
||||||
self.logger = setup_logging()
|
|
||||||
|
|
||||||
def _get_websocket_url(self, local_ip: str, port: int) -> str:
|
def _get_websocket_url(self, local_ip: str, port: int) -> str:
|
||||||
"""获取websocket地址
|
"""获取websocket地址
|
||||||
@@ -25,41 +22,15 @@ class SimpleOtaServer:
|
|||||||
str: websocket地址
|
str: websocket地址
|
||||||
"""
|
"""
|
||||||
server_config = self.config["server"]
|
server_config = self.config["server"]
|
||||||
websocket_config = server_config.get("websocket")
|
websocket_config = server_config.get("websocket", "")
|
||||||
|
|
||||||
if websocket_config and "你" not in websocket_config:
|
if "你的" not in websocket_config:
|
||||||
return websocket_config
|
return websocket_config
|
||||||
else:
|
else:
|
||||||
return f"ws://{local_ip}:{port}/xiaozhi/v1/"
|
return f"ws://{local_ip}:{port}/xiaozhi/v1/"
|
||||||
|
|
||||||
async def start(self):
|
async def handle_post(self, request):
|
||||||
server_config = self.config["server"]
|
"""处理 OTA POST 请求"""
|
||||||
host = server_config.get("ip", "0.0.0.0")
|
|
||||||
port = int(server_config.get("ota_port"))
|
|
||||||
|
|
||||||
if port:
|
|
||||||
app = web.Application()
|
|
||||||
# 添加路由
|
|
||||||
app.add_routes(
|
|
||||||
[
|
|
||||||
web.get("/xiaozhi/ota/", self._handle_ota_get_request),
|
|
||||||
web.post("/xiaozhi/ota/", self._handle_ota_request),
|
|
||||||
web.options("/xiaozhi/ota/", self._handle_ota_request),
|
|
||||||
]
|
|
||||||
)
|
|
||||||
|
|
||||||
# 运行服务
|
|
||||||
runner = web.AppRunner(app)
|
|
||||||
await runner.setup()
|
|
||||||
site = web.TCPSite(runner, host, port)
|
|
||||||
await site.start()
|
|
||||||
|
|
||||||
# 保持服务运行
|
|
||||||
while True:
|
|
||||||
await asyncio.sleep(3600) # 每隔 1 小时检查一次
|
|
||||||
|
|
||||||
async def _handle_ota_request(self, request):
|
|
||||||
"""处理 /xiaozhi/ota/ 的 POST 请求"""
|
|
||||||
try:
|
try:
|
||||||
data = await request.text()
|
data = await request.text()
|
||||||
self.logger.bind(tag=TAG).debug(f"OTA请求方法: {request.method}")
|
self.logger.bind(tag=TAG).debug(f"OTA请求方法: {request.method}")
|
||||||
@@ -75,11 +46,9 @@ class SimpleOtaServer:
|
|||||||
data_json = json.loads(data)
|
data_json = json.loads(data)
|
||||||
|
|
||||||
server_config = self.config["server"]
|
server_config = self.config["server"]
|
||||||
host = server_config.get("ip", "0.0.0.0")
|
|
||||||
port = int(server_config.get("port", 8000))
|
port = int(server_config.get("port", 8000))
|
||||||
local_ip = get_local_ip()
|
local_ip = get_local_ip()
|
||||||
|
|
||||||
# OTA基础信息
|
|
||||||
return_json = {
|
return_json = {
|
||||||
"server_time": {
|
"server_time": {
|
||||||
"timestamp": int(round(time.time() * 1000)),
|
"timestamp": int(round(time.time() * 1000)),
|
||||||
@@ -98,23 +67,17 @@ class SimpleOtaServer:
|
|||||||
content_type="application/json",
|
content_type="application/json",
|
||||||
)
|
)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
self.logger.bind(tag=TAG).error(f"OTA请求异常: {e}")
|
|
||||||
return_json = {"success": False, "message": "request error."}
|
return_json = {"success": False, "message": "request error."}
|
||||||
response = web.Response(
|
response = web.Response(
|
||||||
text=json.dumps(return_json, separators=(",", ":")),
|
text=json.dumps(return_json, separators=(",", ":")),
|
||||||
content_type="application/json",
|
content_type="application/json",
|
||||||
)
|
)
|
||||||
finally:
|
finally:
|
||||||
# 添加header,允许跨域访问
|
self._add_cors_headers(response)
|
||||||
response.headers["Access-Control-Allow-Headers"] = (
|
|
||||||
"client-id, content-type, device-id"
|
|
||||||
)
|
|
||||||
response.headers["Access-Control-Allow-Credentials"] = "true"
|
|
||||||
response.headers["Access-Control-Allow-Origin"] = "*"
|
|
||||||
return response
|
return response
|
||||||
|
|
||||||
async def _handle_ota_get_request(self, request):
|
async def handle_get(self, request):
|
||||||
"""处理 /xiaozhi/ota/ 的 GET 请求"""
|
"""处理 OTA GET 请求"""
|
||||||
try:
|
try:
|
||||||
server_config = self.config["server"]
|
server_config = self.config["server"]
|
||||||
local_ip = get_local_ip()
|
local_ip = get_local_ip()
|
||||||
@@ -126,10 +89,5 @@ class SimpleOtaServer:
|
|||||||
self.logger.bind(tag=TAG).error(f"OTA GET请求异常: {e}")
|
self.logger.bind(tag=TAG).error(f"OTA GET请求异常: {e}")
|
||||||
response = web.Response(text="OTA接口异常", content_type="text/plain")
|
response = web.Response(text="OTA接口异常", content_type="text/plain")
|
||||||
finally:
|
finally:
|
||||||
# 添加header,允许跨域访问
|
self._add_cors_headers(response)
|
||||||
response.headers["Access-Control-Allow-Headers"] = (
|
|
||||||
"client-id, content-type, device-id"
|
|
||||||
)
|
|
||||||
response.headers["Access-Control-Allow-Credentials"] = "true"
|
|
||||||
response.headers["Access-Control-Allow-Origin"] = "*"
|
|
||||||
return response
|
return response
|
||||||
@@ -0,0 +1,181 @@
|
|||||||
|
import json
|
||||||
|
import copy
|
||||||
|
from aiohttp import web
|
||||||
|
from config.logger import setup_logging
|
||||||
|
from core.utils.util import get_vision_url, is_valid_image_file
|
||||||
|
from core.utils.vllm import create_instance
|
||||||
|
from config.config_loader import get_private_config_from_api
|
||||||
|
from core.utils.auth import AuthToken
|
||||||
|
import base64
|
||||||
|
from typing import Tuple, Optional
|
||||||
|
|
||||||
|
TAG = __name__
|
||||||
|
|
||||||
|
# 设置最大文件大小为5MB
|
||||||
|
MAX_FILE_SIZE = 5 * 1024 * 1024
|
||||||
|
|
||||||
|
|
||||||
|
class VisionHandler:
|
||||||
|
def __init__(self, config: dict):
|
||||||
|
self.config = config
|
||||||
|
self.logger = setup_logging()
|
||||||
|
# 初始化认证工具
|
||||||
|
self.auth = AuthToken(config["server"]["auth_key"])
|
||||||
|
|
||||||
|
def _create_error_response(self, message: str) -> dict:
|
||||||
|
"""创建统一的错误响应格式"""
|
||||||
|
return {"success": False, "message": message}
|
||||||
|
|
||||||
|
def _verify_auth_token(self, request) -> Tuple[bool, Optional[str]]:
|
||||||
|
"""验证认证token"""
|
||||||
|
auth_header = request.headers.get("Authorization", "")
|
||||||
|
if not auth_header.startswith("Bearer "):
|
||||||
|
return False, None
|
||||||
|
|
||||||
|
token = auth_header[7:] # 移除"Bearer "前缀
|
||||||
|
return self.auth.verify_token(token)
|
||||||
|
|
||||||
|
async def handle_post(self, request):
|
||||||
|
"""处理 MCP Vision POST 请求"""
|
||||||
|
try:
|
||||||
|
# 验证token
|
||||||
|
is_valid, token_device_id = self._verify_auth_token(request)
|
||||||
|
if not is_valid:
|
||||||
|
return web.Response(
|
||||||
|
text=json.dumps(
|
||||||
|
self._create_error_response("无效的认证token或token已过期")
|
||||||
|
),
|
||||||
|
content_type="application/json",
|
||||||
|
status=401,
|
||||||
|
)
|
||||||
|
|
||||||
|
# 获取请求头信息
|
||||||
|
device_id = request.headers.get("Device-Id", "")
|
||||||
|
client_id = request.headers.get("Client-Id", "")
|
||||||
|
if device_id != token_device_id:
|
||||||
|
return web.Response(
|
||||||
|
text=json.dumps(self._create_error_response("设备ID与token不匹配")),
|
||||||
|
content_type="application/json",
|
||||||
|
status=401,
|
||||||
|
)
|
||||||
|
# 解析multipart/form-data请求
|
||||||
|
reader = await request.multipart()
|
||||||
|
|
||||||
|
# 读取question字段
|
||||||
|
question_field = await reader.next()
|
||||||
|
if question_field is None:
|
||||||
|
raise ValueError("缺少问题字段")
|
||||||
|
question = await question_field.text()
|
||||||
|
self.logger.bind(tag=TAG).debug(f"Question: {question}")
|
||||||
|
|
||||||
|
# 读取图片文件
|
||||||
|
image_field = await reader.next()
|
||||||
|
if image_field is None:
|
||||||
|
raise ValueError("缺少图片文件")
|
||||||
|
|
||||||
|
# 读取图片数据
|
||||||
|
image_data = await image_field.read()
|
||||||
|
if not image_data:
|
||||||
|
raise ValueError("图片数据为空")
|
||||||
|
|
||||||
|
# 检查文件大小
|
||||||
|
if len(image_data) > MAX_FILE_SIZE:
|
||||||
|
raise ValueError(
|
||||||
|
f"图片大小超过限制,最大允许{MAX_FILE_SIZE/1024/1024}MB"
|
||||||
|
)
|
||||||
|
|
||||||
|
# 检查文件格式
|
||||||
|
if not is_valid_image_file(image_data):
|
||||||
|
raise ValueError(
|
||||||
|
"不支持的文件格式,请上传有效的图片文件(支持JPEG、PNG、GIF、BMP、TIFF、WEBP格式)"
|
||||||
|
)
|
||||||
|
|
||||||
|
# 将图片转换为base64编码
|
||||||
|
image_base64 = base64.b64encode(image_data).decode("utf-8")
|
||||||
|
|
||||||
|
# 如果开启了智控台,则从智控台获取模型配置
|
||||||
|
current_config = copy.deepcopy(self.config)
|
||||||
|
read_config_from_api = current_config.get("read_config_from_api", False)
|
||||||
|
if read_config_from_api:
|
||||||
|
current_config = get_private_config_from_api(
|
||||||
|
current_config,
|
||||||
|
device_id,
|
||||||
|
client_id,
|
||||||
|
)
|
||||||
|
|
||||||
|
select_vllm_module = current_config["selected_module"].get("VLLM")
|
||||||
|
if not select_vllm_module:
|
||||||
|
raise ValueError("您还未设置默认的视觉分析模块")
|
||||||
|
|
||||||
|
vllm_type = (
|
||||||
|
select_vllm_module
|
||||||
|
if "type" not in current_config["VLLM"][select_vllm_module]
|
||||||
|
else current_config["VLLM"][select_vllm_module]["type"]
|
||||||
|
)
|
||||||
|
|
||||||
|
if not vllm_type:
|
||||||
|
raise ValueError(f"无法找到VLLM模块对应的供应器{vllm_type}")
|
||||||
|
|
||||||
|
vllm = create_instance(
|
||||||
|
vllm_type, current_config["VLLM"][select_vllm_module]
|
||||||
|
)
|
||||||
|
|
||||||
|
response = vllm.response(question, image_base64)
|
||||||
|
|
||||||
|
return_json = {
|
||||||
|
"success": True,
|
||||||
|
"result": response,
|
||||||
|
}
|
||||||
|
|
||||||
|
response = web.Response(
|
||||||
|
text=json.dumps(return_json, separators=(",", ":")),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
except ValueError as e:
|
||||||
|
self.logger.bind(tag=TAG).error(f"MCP Vision POST请求异常: {e}")
|
||||||
|
return_json = self._create_error_response(str(e))
|
||||||
|
response = web.Response(
|
||||||
|
text=json.dumps(return_json, separators=(",", ":")),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
self.logger.bind(tag=TAG).error(f"MCP Vision POST请求异常: {e}")
|
||||||
|
return_json = self._create_error_response("处理请求时发生错误")
|
||||||
|
response = web.Response(
|
||||||
|
text=json.dumps(return_json, separators=(",", ":")),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
finally:
|
||||||
|
self._add_cors_headers(response)
|
||||||
|
return response
|
||||||
|
|
||||||
|
async def handle_get(self, request):
|
||||||
|
"""处理 MCP Vision GET 请求"""
|
||||||
|
try:
|
||||||
|
vision_explain = get_vision_url(self.config)
|
||||||
|
if vision_explain and len(vision_explain) > 0 and "null" != vision_explain:
|
||||||
|
message = (
|
||||||
|
f"MCP Vision 接口运行正常,视觉解释接口地址是:{vision_explain}"
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
message = "MCP Vision 接口运行不正常,请打开data目录下的.config.yaml文件,找到【server.vision_explain】,设置好地址"
|
||||||
|
|
||||||
|
response = web.Response(text=message, content_type="text/plain")
|
||||||
|
except Exception as e:
|
||||||
|
self.logger.bind(tag=TAG).error(f"MCP Vision GET请求异常: {e}")
|
||||||
|
return_json = self._create_error_response("服务器内部错误")
|
||||||
|
response = web.Response(
|
||||||
|
text=json.dumps(return_json, separators=(",", ":")),
|
||||||
|
content_type="application/json",
|
||||||
|
)
|
||||||
|
finally:
|
||||||
|
self._add_cors_headers(response)
|
||||||
|
return response
|
||||||
|
|
||||||
|
def _add_cors_headers(self, response):
|
||||||
|
"""添加CORS头信息"""
|
||||||
|
response.headers["Access-Control-Allow-Headers"] = (
|
||||||
|
"client-id, content-type, device-id"
|
||||||
|
)
|
||||||
|
response.headers["Access-Control-Allow-Credentials"] = "true"
|
||||||
|
response.headers["Access-Control-Allow-Origin"] = "*"
|
||||||
@@ -1,40 +1,45 @@
|
|||||||
import os
|
import os
|
||||||
|
import sys
|
||||||
import copy
|
import copy
|
||||||
import json
|
import json
|
||||||
import subprocess
|
|
||||||
import sys
|
|
||||||
import uuid
|
import uuid
|
||||||
import time
|
import time
|
||||||
import queue
|
import queue
|
||||||
import asyncio
|
import asyncio
|
||||||
import traceback
|
|
||||||
|
|
||||||
import threading
|
import threading
|
||||||
|
import traceback
|
||||||
|
import subprocess
|
||||||
import websockets
|
import websockets
|
||||||
from typing import Dict, Any
|
from core.handle.mcpHandle import call_mcp_tool
|
||||||
from plugins_func.loadplugins import auto_import_modules
|
|
||||||
from config.logger import setup_logging
|
|
||||||
from core.utils.dialogue import Message, Dialogue
|
|
||||||
from core.handle.textHandle import handleTextMessage
|
|
||||||
from core.utils.util import (
|
from core.utils.util import (
|
||||||
get_string_no_punctuation_or_emoji,
|
|
||||||
extract_json_from_string,
|
extract_json_from_string,
|
||||||
initialize_modules,
|
|
||||||
check_vad_update,
|
check_vad_update,
|
||||||
check_asr_update,
|
check_asr_update,
|
||||||
filter_sensitive_info,
|
filter_sensitive_info,
|
||||||
)
|
)
|
||||||
from concurrent.futures import ThreadPoolExecutor, TimeoutError
|
from typing import Dict, Any
|
||||||
from core.handle.sendAudioHandle import sendAudioMessage
|
from core.mcp.manager import MCPManager
|
||||||
from core.handle.receiveAudioHandle import handleAudioMessage
|
from core.utils.modules_initialize import (
|
||||||
|
initialize_modules,
|
||||||
|
initialize_tts,
|
||||||
|
initialize_asr,
|
||||||
|
)
|
||||||
|
from core.handle.reportHandle import report
|
||||||
|
from core.providers.tts.default import DefaultTTS
|
||||||
|
from concurrent.futures import ThreadPoolExecutor
|
||||||
|
from core.utils.dialogue import Message, Dialogue
|
||||||
|
from core.providers.asr.dto.dto import InterfaceType
|
||||||
|
from core.handle.textHandle import handleTextMessage
|
||||||
from core.handle.functionHandler import FunctionHandler
|
from core.handle.functionHandler import FunctionHandler
|
||||||
|
from plugins_func.loadplugins import auto_import_modules
|
||||||
from plugins_func.register import Action, ActionResponse
|
from plugins_func.register import Action, ActionResponse
|
||||||
from core.auth import AuthMiddleware, AuthenticationError
|
from core.auth import AuthMiddleware, AuthenticationError
|
||||||
from core.mcp.manager import MCPManager
|
|
||||||
from config.config_loader import get_private_config_from_api
|
from config.config_loader import get_private_config_from_api
|
||||||
|
from core.handle.receiveAudioHandle import handleAudioMessage
|
||||||
|
from core.providers.tts.dto.dto import ContentType, TTSMessageDTO, SentenceType
|
||||||
|
from config.logger import setup_logging, build_module_string, update_module_string
|
||||||
from config.manage_api_client import DeviceNotFoundException, DeviceBindException
|
from config.manage_api_client import DeviceNotFoundException, DeviceBindException
|
||||||
from core.utils.output_counter import add_device_output
|
|
||||||
from core.handle.reportHandle import enqueue_tts_report, report
|
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
|
|
||||||
@@ -52,7 +57,6 @@ class ConnectionHandler:
|
|||||||
_vad,
|
_vad,
|
||||||
_asr,
|
_asr,
|
||||||
_llm,
|
_llm,
|
||||||
_tts,
|
|
||||||
_memory,
|
_memory,
|
||||||
_intent,
|
_intent,
|
||||||
server=None,
|
server=None,
|
||||||
@@ -80,29 +84,28 @@ class ConnectionHandler:
|
|||||||
|
|
||||||
# 客户端状态相关
|
# 客户端状态相关
|
||||||
self.client_abort = False
|
self.client_abort = False
|
||||||
|
self.client_is_speaking = False
|
||||||
self.client_listen_mode = "auto"
|
self.client_listen_mode = "auto"
|
||||||
|
|
||||||
# 线程任务相关
|
# 线程任务相关
|
||||||
self.loop = asyncio.get_event_loop()
|
self.loop = asyncio.get_event_loop()
|
||||||
self.stop_event = threading.Event()
|
self.stop_event = threading.Event()
|
||||||
self.tts_queue = queue.Queue()
|
self.executor = ThreadPoolExecutor(max_workers=5)
|
||||||
self.audio_play_queue = queue.Queue()
|
|
||||||
self.executor = ThreadPoolExecutor(max_workers=10)
|
|
||||||
|
|
||||||
# 上报线程
|
# 添加上报线程池
|
||||||
self.report_queue = queue.Queue()
|
self.report_queue = queue.Queue()
|
||||||
self.report_thread = None
|
self.report_thread = None
|
||||||
# TODO(haotian): 2025/5/12 可以通过修改此处,调节asr的上报和tts的上报
|
# 未来可以通过修改此处,调节asr的上报和tts的上报,目前默认都开启
|
||||||
self.report_asr_enable = self.read_config_from_api
|
self.report_asr_enable = self.read_config_from_api
|
||||||
self.report_tts_enable = self.read_config_from_api
|
self.report_tts_enable = self.read_config_from_api
|
||||||
|
|
||||||
# 依赖的组件
|
# 依赖的组件
|
||||||
self.vad = None
|
self.vad = None
|
||||||
self.asr = None
|
self.asr = None
|
||||||
|
self.tts = None
|
||||||
self._asr = _asr
|
self._asr = _asr
|
||||||
self._vad = _vad
|
self._vad = _vad
|
||||||
self.llm = _llm
|
self.llm = _llm
|
||||||
self.tts = _tts
|
|
||||||
self.memory = _memory
|
self.memory = _memory
|
||||||
self.intent = _intent
|
self.intent = _intent
|
||||||
|
|
||||||
@@ -115,15 +118,13 @@ class ConnectionHandler:
|
|||||||
|
|
||||||
# asr相关变量
|
# asr相关变量
|
||||||
self.asr_audio = []
|
self.asr_audio = []
|
||||||
self.asr_server_receive = True
|
|
||||||
|
|
||||||
# llm相关变量
|
# llm相关变量
|
||||||
self.llm_finish_task = False
|
self.llm_finish_task = True
|
||||||
self.dialogue = Dialogue()
|
self.dialogue = Dialogue()
|
||||||
|
|
||||||
# tts相关变量
|
# tts相关变量
|
||||||
self.tts_first_text_index = -1
|
self.sentence_id = None
|
||||||
self.tts_last_text_index = -1
|
|
||||||
|
|
||||||
# iot相关变量
|
# iot相关变量
|
||||||
self.iot_descriptors = {}
|
self.iot_descriptors = {}
|
||||||
@@ -146,6 +147,8 @@ class ConnectionHandler:
|
|||||||
) # 在原来第一道关闭的基础上加60秒,进行二道关闭
|
) # 在原来第一道关闭的基础上加60秒,进行二道关闭
|
||||||
|
|
||||||
self.audio_format = "opus"
|
self.audio_format = "opus"
|
||||||
|
# {"mcp":true} 表示启用MCP功能
|
||||||
|
self.features = None
|
||||||
|
|
||||||
async def handle_connection(self, ws):
|
async def handle_connection(self, ws):
|
||||||
try:
|
try:
|
||||||
@@ -194,17 +197,6 @@ class ConnectionHandler:
|
|||||||
self._initialize_private_config()
|
self._initialize_private_config()
|
||||||
# 异步初始化
|
# 异步初始化
|
||||||
self.executor.submit(self._initialize_components)
|
self.executor.submit(self._initialize_components)
|
||||||
# tts 消化线程
|
|
||||||
self.tts_priority_thread = threading.Thread(
|
|
||||||
target=self._tts_priority_thread, daemon=True
|
|
||||||
)
|
|
||||||
self.tts_priority_thread.start()
|
|
||||||
|
|
||||||
# 音频播放 消化线程
|
|
||||||
self.audio_play_priority_thread = threading.Thread(
|
|
||||||
target=self._audio_play_priority_thread, daemon=True
|
|
||||||
)
|
|
||||||
self.audio_play_priority_thread.start()
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
async for message in self.websocket:
|
async for message in self.websocket:
|
||||||
@@ -313,25 +305,43 @@ class ConnectionHandler:
|
|||||||
)
|
)
|
||||||
|
|
||||||
def _initialize_components(self):
|
def _initialize_components(self):
|
||||||
"""初始化组件"""
|
try:
|
||||||
if self.config.get("prompt") is not None:
|
self.selected_module_str = build_module_string(
|
||||||
self.prompt = self.config["prompt"]
|
self.config.get("selected_module", {})
|
||||||
self.change_system_prompt(self.prompt)
|
)
|
||||||
self.logger.bind(tag=TAG).info(
|
update_module_string(self.selected_module_str)
|
||||||
f"初始化组件: prompt成功 {self.prompt[:50]}..."
|
"""初始化组件"""
|
||||||
|
if self.config.get("prompt") is not None:
|
||||||
|
self.prompt = self.config["prompt"]
|
||||||
|
self.change_system_prompt(self.prompt)
|
||||||
|
self.logger.bind(tag=TAG).info(
|
||||||
|
f"初始化组件: prompt成功 {self.prompt[:50]}..."
|
||||||
|
)
|
||||||
|
|
||||||
|
"""初始化本地组件"""
|
||||||
|
if self.vad is None:
|
||||||
|
self.vad = self._vad
|
||||||
|
if self.asr is None:
|
||||||
|
self.asr = self._initialize_asr()
|
||||||
|
# 打开语音识别通道
|
||||||
|
asyncio.run_coroutine_threadsafe(
|
||||||
|
self.asr.open_audio_channels(self), self.loop
|
||||||
|
)
|
||||||
|
if self.tts is None:
|
||||||
|
self.tts = self._initialize_tts()
|
||||||
|
# 打开语音合成通道
|
||||||
|
asyncio.run_coroutine_threadsafe(
|
||||||
|
self.tts.open_audio_channels(self), self.loop
|
||||||
)
|
)
|
||||||
|
|
||||||
"""初始化本地组件"""
|
"""加载记忆"""
|
||||||
if self.vad is None:
|
self._initialize_memory()
|
||||||
self.vad = self._vad
|
"""加载意图识别"""
|
||||||
if self.asr is None:
|
self._initialize_intent()
|
||||||
self.asr = self._asr
|
"""初始化上报线程"""
|
||||||
"""加载记忆"""
|
self._init_report_threads()
|
||||||
self._initialize_memory()
|
except Exception as e:
|
||||||
"""加载意图识别"""
|
self.logger.bind(tag=TAG).error(f"实例化组件失败: {e}")
|
||||||
self._initialize_intent()
|
|
||||||
"""初始化上报线程"""
|
|
||||||
self._init_report_threads()
|
|
||||||
|
|
||||||
def _init_report_threads(self):
|
def _init_report_threads(self):
|
||||||
"""初始化ASR和TTS上报线程"""
|
"""初始化ASR和TTS上报线程"""
|
||||||
@@ -346,6 +356,30 @@ class ConnectionHandler:
|
|||||||
self.report_thread.start()
|
self.report_thread.start()
|
||||||
self.logger.bind(tag=TAG).info("TTS上报线程已启动")
|
self.logger.bind(tag=TAG).info("TTS上报线程已启动")
|
||||||
|
|
||||||
|
def _initialize_tts(self):
|
||||||
|
"""初始化TTS"""
|
||||||
|
tts = None
|
||||||
|
if not self.need_bind:
|
||||||
|
tts = initialize_tts(self.config)
|
||||||
|
|
||||||
|
if tts is None:
|
||||||
|
tts = DefaultTTS(self.config, delete_audio_file=True)
|
||||||
|
|
||||||
|
return tts
|
||||||
|
|
||||||
|
def _initialize_asr(self):
|
||||||
|
"""初始化ASR"""
|
||||||
|
if self._asr.interface_type == InterfaceType.LOCAL:
|
||||||
|
# 如果公共ASR是本地服务,则直接返回
|
||||||
|
# 因为本地一个实例ASR,可以被多个连接共享
|
||||||
|
asr = self._asr
|
||||||
|
else:
|
||||||
|
# 如果公共ASR是远程服务,则初始化一个新实例
|
||||||
|
# 因为远程ASR,涉及到websocket连接和接收线程,需要每个连接一个实例
|
||||||
|
asr = initialize_asr(self.config)
|
||||||
|
|
||||||
|
return asr
|
||||||
|
|
||||||
def _initialize_private_config(self):
|
def _initialize_private_config(self):
|
||||||
"""如果是从配置文件获取,则进行二次实例化"""
|
"""如果是从配置文件获取,则进行二次实例化"""
|
||||||
if not self.read_config_from_api:
|
if not self.read_config_from_api:
|
||||||
@@ -444,6 +478,8 @@ class ConnectionHandler:
|
|||||||
self.memory = modules["memory"]
|
self.memory = modules["memory"]
|
||||||
|
|
||||||
def _initialize_memory(self):
|
def _initialize_memory(self):
|
||||||
|
if self.memory is None:
|
||||||
|
return
|
||||||
"""初始化记忆模块"""
|
"""初始化记忆模块"""
|
||||||
self.memory.init_memory(
|
self.memory.init_memory(
|
||||||
role_id=self.device_id,
|
role_id=self.device_id,
|
||||||
@@ -452,7 +488,40 @@ class ConnectionHandler:
|
|||||||
save_to_file=not self.read_config_from_api,
|
save_to_file=not self.read_config_from_api,
|
||||||
)
|
)
|
||||||
|
|
||||||
|
# 获取记忆总结配置
|
||||||
|
memory_config = self.config["Memory"]
|
||||||
|
memory_type = self.config["Memory"][self.config["selected_module"]["Memory"]][
|
||||||
|
"type"
|
||||||
|
]
|
||||||
|
# 如果使用 nomen,直接返回
|
||||||
|
if memory_type == "nomem":
|
||||||
|
return
|
||||||
|
# 使用 mem_local_short 模式
|
||||||
|
elif memory_type == "mem_local_short":
|
||||||
|
memory_llm_name = memory_config[self.config["selected_module"]["Memory"]][
|
||||||
|
"llm"
|
||||||
|
]
|
||||||
|
if memory_llm_name and memory_llm_name in self.config["LLM"]:
|
||||||
|
# 如果配置了专用LLM,则创建独立的LLM实例
|
||||||
|
from core.utils import llm as llm_utils
|
||||||
|
|
||||||
|
memory_llm_config = self.config["LLM"][memory_llm_name]
|
||||||
|
memory_llm_type = memory_llm_config.get("type", memory_llm_name)
|
||||||
|
memory_llm = llm_utils.create_instance(
|
||||||
|
memory_llm_type, memory_llm_config
|
||||||
|
)
|
||||||
|
self.logger.bind(tag=TAG).info(
|
||||||
|
f"为记忆总结创建了专用LLM: {memory_llm_name}, 类型: {memory_llm_type}"
|
||||||
|
)
|
||||||
|
self.memory.set_llm(memory_llm)
|
||||||
|
else:
|
||||||
|
# 否则使用主LLM
|
||||||
|
self.memory.set_llm(self.llm)
|
||||||
|
self.logger.bind(tag=TAG).info("使用主LLM作为意图识别模型")
|
||||||
|
|
||||||
def _initialize_intent(self):
|
def _initialize_intent(self):
|
||||||
|
if self.intent is None:
|
||||||
|
return
|
||||||
self.intent_type = self.config["Intent"][
|
self.intent_type = self.config["Intent"][
|
||||||
self.config["selected_module"]["Intent"]
|
self.config["selected_module"]["Intent"]
|
||||||
]["type"]
|
]["type"]
|
||||||
@@ -506,106 +575,26 @@ class ConnectionHandler:
|
|||||||
# 更新系统prompt至上下文
|
# 更新系统prompt至上下文
|
||||||
self.dialogue.update_system_message(self.prompt)
|
self.dialogue.update_system_message(self.prompt)
|
||||||
|
|
||||||
def chat(self, query):
|
def chat(self, query, tool_call=False):
|
||||||
|
self.logger.bind(tag=TAG).info(f"大模型收到用户消息: {query}")
|
||||||
self.dialogue.put(Message(role="user", content=query))
|
|
||||||
|
|
||||||
response_message = []
|
|
||||||
processed_chars = 0 # 跟踪已处理的字符位置
|
|
||||||
try:
|
|
||||||
# 使用带记忆的对话
|
|
||||||
memory_str = None
|
|
||||||
if self.memory is not None:
|
|
||||||
future = asyncio.run_coroutine_threadsafe(
|
|
||||||
self.memory.query_memory(query), self.loop
|
|
||||||
)
|
|
||||||
memory_str = future.result()
|
|
||||||
|
|
||||||
self.logger.bind(tag=TAG).debug(f"记忆内容: {memory_str}")
|
|
||||||
llm_responses = self.llm.response(
|
|
||||||
self.session_id, self.dialogue.get_llm_dialogue_with_memory(memory_str)
|
|
||||||
)
|
|
||||||
except Exception as e:
|
|
||||||
self.logger.bind(tag=TAG).error(f"LLM 处理出错 {query}: {e}")
|
|
||||||
return None
|
|
||||||
|
|
||||||
self.llm_finish_task = False
|
self.llm_finish_task = False
|
||||||
text_index = 0
|
|
||||||
for content in llm_responses:
|
|
||||||
response_message.append(content)
|
|
||||||
if self.client_abort:
|
|
||||||
break
|
|
||||||
|
|
||||||
# 合并当前全部文本并处理未分割部分
|
|
||||||
full_text = "".join(response_message)
|
|
||||||
current_text = full_text[processed_chars:] # 从未处理的位置开始
|
|
||||||
|
|
||||||
# 查找最后一个有效标点
|
|
||||||
punctuations = ("。", ".", "?", "?", "!", "!", ";", ";", ":")
|
|
||||||
last_punct_pos = -1
|
|
||||||
number_flag = True
|
|
||||||
for punct in punctuations:
|
|
||||||
pos = current_text.rfind(punct)
|
|
||||||
prev_char = current_text[pos - 1] if pos - 1 >= 0 else ""
|
|
||||||
# 如果.前面是数字统一判断为小数
|
|
||||||
if prev_char.isdigit() and punct == ".":
|
|
||||||
number_flag = False
|
|
||||||
if pos > last_punct_pos and number_flag:
|
|
||||||
last_punct_pos = pos
|
|
||||||
|
|
||||||
# 找到分割点则处理
|
|
||||||
if last_punct_pos != -1:
|
|
||||||
segment_text_raw = current_text[: last_punct_pos + 1]
|
|
||||||
segment_text = get_string_no_punctuation_or_emoji(segment_text_raw)
|
|
||||||
if segment_text:
|
|
||||||
# 强制设置空字符,测试TTS出错返回语音的健壮性
|
|
||||||
# if text_index % 2 == 0:
|
|
||||||
# segment_text = " "
|
|
||||||
text_index += 1
|
|
||||||
self.recode_first_last_text(segment_text, text_index)
|
|
||||||
future = self.executor.submit(
|
|
||||||
self.speak_and_play, segment_text, text_index
|
|
||||||
)
|
|
||||||
self.tts_queue.put((future, text_index))
|
|
||||||
processed_chars += len(segment_text_raw) # 更新已处理字符位置
|
|
||||||
|
|
||||||
# 处理最后剩余的文本
|
|
||||||
full_text = "".join(response_message)
|
|
||||||
remaining_text = full_text[processed_chars:]
|
|
||||||
if remaining_text:
|
|
||||||
segment_text = get_string_no_punctuation_or_emoji(remaining_text)
|
|
||||||
if segment_text:
|
|
||||||
text_index += 1
|
|
||||||
self.recode_first_last_text(segment_text, text_index)
|
|
||||||
future = self.executor.submit(
|
|
||||||
self.speak_and_play, segment_text, text_index
|
|
||||||
)
|
|
||||||
self.tts_queue.put((future, text_index))
|
|
||||||
|
|
||||||
self.llm_finish_task = True
|
|
||||||
self.dialogue.put(Message(role="assistant", content="".join(response_message)))
|
|
||||||
self.logger.bind(tag=TAG).debug(
|
|
||||||
json.dumps(self.dialogue.get_llm_dialogue(), indent=4, ensure_ascii=False)
|
|
||||||
)
|
|
||||||
return True
|
|
||||||
|
|
||||||
def chat_with_function_calling(self, query, tool_call=False):
|
|
||||||
self.logger.bind(tag=TAG).debug(f"Chat with function calling start: {query}")
|
|
||||||
"""Chat with function calling for intent detection using streaming"""
|
|
||||||
|
|
||||||
if not tool_call:
|
if not tool_call:
|
||||||
self.dialogue.put(Message(role="user", content=query))
|
self.dialogue.put(Message(role="user", content=query))
|
||||||
|
|
||||||
# Define intent functions
|
# Define intent functions
|
||||||
functions = None
|
functions = None
|
||||||
if hasattr(self, "func_handler"):
|
if self.intent_type == "function_call" and hasattr(self, "func_handler"):
|
||||||
functions = self.func_handler.get_functions()
|
functions = self.func_handler.get_functions()
|
||||||
|
if hasattr(self, "mcp_client"):
|
||||||
|
mcp_tools = self.mcp_client.get_available_tools()
|
||||||
|
if mcp_tools is not None and len(mcp_tools) > 0:
|
||||||
|
if functions is None:
|
||||||
|
functions = []
|
||||||
|
functions.extend(mcp_tools)
|
||||||
response_message = []
|
response_message = []
|
||||||
processed_chars = 0 # 跟踪已处理的字符位置
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
start_time = time.time()
|
|
||||||
|
|
||||||
# 使用带记忆的对话
|
# 使用带记忆的对话
|
||||||
memory_str = None
|
memory_str = None
|
||||||
if self.memory is not None:
|
if self.memory is not None:
|
||||||
@@ -614,94 +603,78 @@ class ConnectionHandler:
|
|||||||
)
|
)
|
||||||
memory_str = future.result()
|
memory_str = future.result()
|
||||||
|
|
||||||
# self.logger.bind(tag=TAG).info(f"对话记录: {self.dialogue.get_llm_dialogue_with_memory(memory_str)}")
|
uuid_str = str(uuid.uuid4()).replace("-", "")
|
||||||
|
self.sentence_id = uuid_str
|
||||||
|
|
||||||
# 使用支持functions的streaming接口
|
if functions is not None:
|
||||||
llm_responses = self.llm.response_with_functions(
|
# 使用支持functions的streaming接口
|
||||||
self.session_id,
|
llm_responses = self.llm.response_with_functions(
|
||||||
self.dialogue.get_llm_dialogue_with_memory(memory_str),
|
self.session_id,
|
||||||
functions=functions,
|
self.dialogue.get_llm_dialogue_with_memory(memory_str),
|
||||||
)
|
functions=functions,
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
llm_responses = self.llm.response(
|
||||||
|
self.session_id,
|
||||||
|
self.dialogue.get_llm_dialogue_with_memory(memory_str),
|
||||||
|
)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
self.logger.bind(tag=TAG).error(f"LLM 处理出错 {query}: {e}")
|
self.logger.bind(tag=TAG).error(f"LLM 处理出错 {query}: {e}")
|
||||||
return None
|
return None
|
||||||
|
|
||||||
self.llm_finish_task = False
|
|
||||||
text_index = 0
|
|
||||||
|
|
||||||
# 处理流式响应
|
# 处理流式响应
|
||||||
tool_call_flag = False
|
tool_call_flag = False
|
||||||
function_name = None
|
function_name = None
|
||||||
function_id = None
|
function_id = None
|
||||||
function_arguments = ""
|
function_arguments = ""
|
||||||
content_arguments = ""
|
content_arguments = ""
|
||||||
|
text_index = 0
|
||||||
|
self.client_abort = False
|
||||||
for response in llm_responses:
|
for response in llm_responses:
|
||||||
content, tools_call = response
|
if self.client_abort:
|
||||||
|
break
|
||||||
|
if functions is not None:
|
||||||
|
content, tools_call = response
|
||||||
|
if "content" in response:
|
||||||
|
content = response["content"]
|
||||||
|
tools_call = None
|
||||||
|
if content is not None and len(content) > 0:
|
||||||
|
content_arguments += content
|
||||||
|
|
||||||
if "content" in response:
|
if not tool_call_flag and content_arguments.startswith("<tool_call>"):
|
||||||
content = response["content"]
|
# print("content_arguments", content_arguments)
|
||||||
tools_call = None
|
tool_call_flag = True
|
||||||
if content is not None and len(content) > 0:
|
|
||||||
content_arguments += content
|
|
||||||
|
|
||||||
if not tool_call_flag and content_arguments.startswith("<tool_call>"):
|
|
||||||
# print("content_arguments", content_arguments)
|
|
||||||
tool_call_flag = True
|
|
||||||
|
|
||||||
if tools_call is not None:
|
|
||||||
tool_call_flag = True
|
|
||||||
if tools_call[0].id is not None:
|
|
||||||
function_id = tools_call[0].id
|
|
||||||
if tools_call[0].function.name is not None:
|
|
||||||
function_name = tools_call[0].function.name
|
|
||||||
if tools_call[0].function.arguments is not None:
|
|
||||||
function_arguments += tools_call[0].function.arguments
|
|
||||||
|
|
||||||
|
if tools_call is not None:
|
||||||
|
tool_call_flag = True
|
||||||
|
if tools_call[0].id is not None:
|
||||||
|
function_id = tools_call[0].id
|
||||||
|
if tools_call[0].function.name is not None:
|
||||||
|
function_name = tools_call[0].function.name
|
||||||
|
if tools_call[0].function.arguments is not None:
|
||||||
|
function_arguments += tools_call[0].function.arguments
|
||||||
|
else:
|
||||||
|
content = response
|
||||||
if content is not None and len(content) > 0:
|
if content is not None and len(content) > 0:
|
||||||
if not tool_call_flag:
|
if not tool_call_flag:
|
||||||
response_message.append(content)
|
response_message.append(content)
|
||||||
|
if text_index == 0:
|
||||||
if self.client_abort:
|
self.tts.tts_text_queue.put(
|
||||||
break
|
TTSMessageDTO(
|
||||||
|
sentence_id=self.sentence_id,
|
||||||
end_time = time.time()
|
sentence_type=SentenceType.FIRST,
|
||||||
# self.logger.bind(tag=TAG).debug(f"大模型返回时间: {end_time - start_time} 秒, 生成token={content}")
|
content_type=ContentType.ACTION,
|
||||||
|
|
||||||
# 处理文本分段和TTS逻辑
|
|
||||||
# 合并当前全部文本并处理未分割部分
|
|
||||||
full_text = "".join(response_message)
|
|
||||||
current_text = full_text[processed_chars:] # 从未处理的位置开始
|
|
||||||
|
|
||||||
# 查找最后一个有效标点
|
|
||||||
punctuations = ("。", ".", "?", "?", "!", "!", ";", ";", ":")
|
|
||||||
last_punct_pos = -1
|
|
||||||
number_flag = True
|
|
||||||
for punct in punctuations:
|
|
||||||
pos = current_text.rfind(punct)
|
|
||||||
prev_char = current_text[pos - 1] if pos - 1 >= 0 else ""
|
|
||||||
# 如果.前面是数字统一判断为小数
|
|
||||||
if prev_char.isdigit() and punct == ".":
|
|
||||||
number_flag = False
|
|
||||||
if pos > last_punct_pos and number_flag:
|
|
||||||
last_punct_pos = pos
|
|
||||||
|
|
||||||
# 找到分割点则处理
|
|
||||||
if last_punct_pos != -1:
|
|
||||||
segment_text_raw = current_text[: last_punct_pos + 1]
|
|
||||||
segment_text = get_string_no_punctuation_or_emoji(
|
|
||||||
segment_text_raw
|
|
||||||
)
|
|
||||||
if segment_text:
|
|
||||||
text_index += 1
|
|
||||||
self.recode_first_last_text(segment_text, text_index)
|
|
||||||
future = self.executor.submit(
|
|
||||||
self.speak_and_play, segment_text, text_index
|
|
||||||
)
|
)
|
||||||
self.tts_queue.put((future, text_index))
|
)
|
||||||
# 更新已处理字符位置
|
self.tts.tts_text_queue.put(
|
||||||
processed_chars += len(segment_text_raw)
|
TTSMessageDTO(
|
||||||
|
sentence_id=self.sentence_id,
|
||||||
|
sentence_type=SentenceType.MIDDLE,
|
||||||
|
content_type=ContentType.TEXT,
|
||||||
|
content_detail=content,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
text_index += 1
|
||||||
# 处理function call
|
# 处理function call
|
||||||
if tool_call_flag:
|
if tool_call_flag:
|
||||||
bHasError = False
|
bHasError = False
|
||||||
@@ -736,35 +709,52 @@ class ConnectionHandler:
|
|||||||
"arguments": function_arguments,
|
"arguments": function_arguments,
|
||||||
}
|
}
|
||||||
|
|
||||||
# 处理MCP工具调用
|
# 处理Server端MCP工具调用
|
||||||
if self.mcp_manager.is_mcp_tool(function_name):
|
if self.mcp_manager.is_mcp_tool(function_name):
|
||||||
result = self._handle_mcp_tool_call(function_call_data)
|
result = self._handle_mcp_tool_call(function_call_data)
|
||||||
|
elif hasattr(self, "mcp_client") and self.mcp_client.has_tool(
|
||||||
|
function_name
|
||||||
|
):
|
||||||
|
# 如果是小智端MCP工具调用
|
||||||
|
self.logger.bind(tag=TAG).debug(
|
||||||
|
f"调用小智端MCP工具: {function_name}, 参数: {function_arguments}"
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
result = asyncio.run_coroutine_threadsafe(
|
||||||
|
call_mcp_tool(
|
||||||
|
self, self.mcp_client, function_name, function_arguments
|
||||||
|
),
|
||||||
|
self.loop,
|
||||||
|
).result()
|
||||||
|
self.logger.bind(tag=TAG).debug(f"MCP工具调用结果: {result}")
|
||||||
|
result = ActionResponse(
|
||||||
|
action=Action.REQLLM, result=result, response=""
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
self.logger.bind(tag=TAG).error(f"MCP工具调用失败: {e}")
|
||||||
|
result = ActionResponse(
|
||||||
|
action=Action.REQLLM, result="MCP工具调用失败", response=""
|
||||||
|
)
|
||||||
else:
|
else:
|
||||||
# 处理系统函数
|
# 处理系统函数
|
||||||
result = self.func_handler.handle_llm_function_call(
|
result = self.func_handler.handle_llm_function_call(
|
||||||
self, function_call_data
|
self, function_call_data
|
||||||
)
|
)
|
||||||
self._handle_function_result(result, function_call_data, text_index + 1)
|
self._handle_function_result(result, function_call_data)
|
||||||
|
|
||||||
# 处理最后剩余的文本
|
|
||||||
full_text = "".join(response_message)
|
|
||||||
remaining_text = full_text[processed_chars:]
|
|
||||||
if remaining_text:
|
|
||||||
segment_text = get_string_no_punctuation_or_emoji(remaining_text)
|
|
||||||
if segment_text:
|
|
||||||
text_index += 1
|
|
||||||
self.recode_first_last_text(segment_text, text_index)
|
|
||||||
future = self.executor.submit(
|
|
||||||
self.speak_and_play, segment_text, text_index
|
|
||||||
)
|
|
||||||
self.tts_queue.put((future, text_index))
|
|
||||||
|
|
||||||
# 存储对话内容
|
# 存储对话内容
|
||||||
if len(response_message) > 0:
|
if len(response_message) > 0:
|
||||||
self.dialogue.put(
|
self.dialogue.put(
|
||||||
Message(role="assistant", content="".join(response_message))
|
Message(role="assistant", content="".join(response_message))
|
||||||
)
|
)
|
||||||
|
if text_index > 0:
|
||||||
|
self.tts.tts_text_queue.put(
|
||||||
|
TTSMessageDTO(
|
||||||
|
sentence_id=self.sentence_id,
|
||||||
|
sentence_type=SentenceType.LAST,
|
||||||
|
content_type=ContentType.ACTION,
|
||||||
|
)
|
||||||
|
)
|
||||||
self.llm_finish_task = True
|
self.llm_finish_task = True
|
||||||
self.logger.bind(tag=TAG).debug(
|
self.logger.bind(tag=TAG).debug(
|
||||||
json.dumps(self.dialogue.get_llm_dialogue(), indent=4, ensure_ascii=False)
|
json.dumps(self.dialogue.get_llm_dialogue(), indent=4, ensure_ascii=False)
|
||||||
@@ -814,12 +804,10 @@ class ConnectionHandler:
|
|||||||
|
|
||||||
return ActionResponse(action=Action.REQLLM, result="工具调用出错", response="")
|
return ActionResponse(action=Action.REQLLM, result="工具调用出错", response="")
|
||||||
|
|
||||||
def _handle_function_result(self, result, function_call_data, text_index):
|
def _handle_function_result(self, result, function_call_data):
|
||||||
if result.action == Action.RESPONSE: # 直接回复前端
|
if result.action == Action.RESPONSE: # 直接回复前端
|
||||||
text = result.response
|
text = result.response
|
||||||
self.recode_first_last_text(text, text_index)
|
self.tts.tts_one_sentence(self, ContentType.TEXT, content_detail=text)
|
||||||
future = self.executor.submit(self.speak_and_play, text, text_index)
|
|
||||||
self.tts_queue.put((future, text_index))
|
|
||||||
self.dialogue.put(Message(role="assistant", content=text))
|
self.dialogue.put(Message(role="assistant", content=text))
|
||||||
elif result.action == Action.REQLLM: # 调用函数后再请求llm生成回复
|
elif result.action == Action.REQLLM: # 调用函数后再请求llm生成回复
|
||||||
text = result.result
|
text = result.result
|
||||||
@@ -853,111 +841,14 @@ class ConnectionHandler:
|
|||||||
content=text,
|
content=text,
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
self.chat_with_function_calling(text, tool_call=True)
|
self.chat(text, tool_call=True)
|
||||||
elif result.action == Action.NOTFOUND or result.action == Action.ERROR:
|
elif result.action == Action.NOTFOUND or result.action == Action.ERROR:
|
||||||
text = result.result
|
text = result.result
|
||||||
self.recode_first_last_text(text, text_index)
|
self.tts.tts_one_sentence(self, ContentType.TEXT, content_detail=text)
|
||||||
future = self.executor.submit(self.speak_and_play, text, text_index)
|
|
||||||
self.tts_queue.put((future, text_index))
|
|
||||||
self.dialogue.put(Message(role="assistant", content=text))
|
self.dialogue.put(Message(role="assistant", content=text))
|
||||||
else:
|
else:
|
||||||
pass
|
pass
|
||||||
|
|
||||||
def _tts_priority_thread(self):
|
|
||||||
while not self.stop_event.is_set():
|
|
||||||
text = None
|
|
||||||
try:
|
|
||||||
try:
|
|
||||||
item = self.tts_queue.get(timeout=1)
|
|
||||||
if item is None:
|
|
||||||
continue
|
|
||||||
future, text_index = item # 解包获取 Future 和 text_index
|
|
||||||
except queue.Empty:
|
|
||||||
if self.stop_event.is_set():
|
|
||||||
break
|
|
||||||
continue
|
|
||||||
if future is None:
|
|
||||||
continue
|
|
||||||
text = None
|
|
||||||
audio_datas, tts_file = [], None
|
|
||||||
try:
|
|
||||||
self.logger.bind(tag=TAG).debug("正在处理TTS任务...")
|
|
||||||
tts_timeout = int(self.config.get("tts_timeout", 10))
|
|
||||||
tts_file, text, _ = future.result(timeout=tts_timeout)
|
|
||||||
if text is None or len(text) <= 0:
|
|
||||||
self.logger.bind(tag=TAG).error(
|
|
||||||
f"TTS出错:{text_index}: tts text is empty"
|
|
||||||
)
|
|
||||||
elif tts_file is None:
|
|
||||||
self.logger.bind(tag=TAG).error(
|
|
||||||
f"TTS出错: file is empty: {text_index}: {text}"
|
|
||||||
)
|
|
||||||
else:
|
|
||||||
self.logger.bind(tag=TAG).debug(
|
|
||||||
f"TTS生成:文件路径: {tts_file}"
|
|
||||||
)
|
|
||||||
if os.path.exists(tts_file):
|
|
||||||
if self.audio_format == "pcm":
|
|
||||||
audio_datas, _ = self.tts.audio_to_pcm_data(tts_file)
|
|
||||||
else:
|
|
||||||
audio_datas, _ = self.tts.audio_to_opus_data(tts_file)
|
|
||||||
# 在这里上报TTS数据
|
|
||||||
enqueue_tts_report(self, text, audio_datas)
|
|
||||||
else:
|
|
||||||
self.logger.bind(tag=TAG).error(
|
|
||||||
f"TTS出错:文件不存在{tts_file}"
|
|
||||||
)
|
|
||||||
except TimeoutError:
|
|
||||||
self.logger.bind(tag=TAG).error("TTS超时")
|
|
||||||
except Exception as e:
|
|
||||||
self.logger.bind(tag=TAG).error(f"TTS出错: {e}")
|
|
||||||
if not self.client_abort:
|
|
||||||
# 如果没有中途打断就发送语音
|
|
||||||
self.audio_play_queue.put((audio_datas, text, text_index))
|
|
||||||
if (
|
|
||||||
self.tts.delete_audio_file
|
|
||||||
and tts_file is not None
|
|
||||||
and os.path.exists(tts_file)
|
|
||||||
):
|
|
||||||
os.remove(tts_file)
|
|
||||||
except Exception as e:
|
|
||||||
self.logger.bind(tag=TAG).error(f"TTS任务处理错误: {e}")
|
|
||||||
self.clearSpeakStatus()
|
|
||||||
asyncio.run_coroutine_threadsafe(
|
|
||||||
self.websocket.send(
|
|
||||||
json.dumps(
|
|
||||||
{
|
|
||||||
"type": "tts",
|
|
||||||
"state": "stop",
|
|
||||||
"session_id": self.session_id,
|
|
||||||
}
|
|
||||||
)
|
|
||||||
),
|
|
||||||
self.loop,
|
|
||||||
)
|
|
||||||
self.logger.bind(tag=TAG).error(
|
|
||||||
f"tts_priority priority_thread: {text} {e}"
|
|
||||||
)
|
|
||||||
|
|
||||||
def _audio_play_priority_thread(self):
|
|
||||||
while not self.stop_event.is_set():
|
|
||||||
text = None
|
|
||||||
try:
|
|
||||||
try:
|
|
||||||
audio_datas, text, text_index = self.audio_play_queue.get(timeout=1)
|
|
||||||
except queue.Empty:
|
|
||||||
if self.stop_event.is_set():
|
|
||||||
break
|
|
||||||
continue
|
|
||||||
future = asyncio.run_coroutine_threadsafe(
|
|
||||||
sendAudioMessage(self, audio_datas, text, text_index), self.loop
|
|
||||||
)
|
|
||||||
future.result()
|
|
||||||
except Exception as e:
|
|
||||||
self.logger.bind(tag=TAG).error(
|
|
||||||
f"audio_play_priority priority_thread: {text} {e}"
|
|
||||||
)
|
|
||||||
|
|
||||||
def _report_worker(self):
|
def _report_worker(self):
|
||||||
"""聊天记录上报工作线程"""
|
"""聊天记录上报工作线程"""
|
||||||
while not self.stop_event.is_set():
|
while not self.stop_event.is_set():
|
||||||
@@ -966,17 +857,17 @@ class ConnectionHandler:
|
|||||||
item = self.report_queue.get(timeout=1)
|
item = self.report_queue.get(timeout=1)
|
||||||
if item is None: # 检测毒丸对象
|
if item is None: # 检测毒丸对象
|
||||||
break
|
break
|
||||||
|
type, text, audio_data, report_time = item
|
||||||
type, text, audio_data = item
|
|
||||||
|
|
||||||
try:
|
try:
|
||||||
# 执行上报(传入二进制数据)
|
# 检查线程池状态
|
||||||
report(self, type, text, audio_data)
|
if self.executor is None:
|
||||||
|
continue
|
||||||
|
# 提交任务到线程池
|
||||||
|
self.executor.submit(
|
||||||
|
self._process_report, type, text, audio_data, report_time
|
||||||
|
)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
self.logger.bind(tag=TAG).error(f"聊天记录上报线程异常: {e}")
|
self.logger.bind(tag=TAG).error(f"聊天记录上报线程异常: {e}")
|
||||||
finally:
|
|
||||||
# 标记任务完成
|
|
||||||
self.report_queue.task_done()
|
|
||||||
except queue.Empty:
|
except queue.Empty:
|
||||||
continue
|
continue
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
@@ -984,82 +875,79 @@ class ConnectionHandler:
|
|||||||
|
|
||||||
self.logger.bind(tag=TAG).info("聊天记录上报线程已退出")
|
self.logger.bind(tag=TAG).info("聊天记录上报线程已退出")
|
||||||
|
|
||||||
def speak_and_play(self, text, text_index=0):
|
def _process_report(self, type, text, audio_data, report_time):
|
||||||
if text is None or len(text) <= 0:
|
"""处理上报任务"""
|
||||||
self.logger.bind(tag=TAG).info(f"无需tts转换,query为空,{text}")
|
try:
|
||||||
return None, text, text_index
|
# 执行上报(传入二进制数据)
|
||||||
tts_file = self.tts.to_tts(text)
|
report(self, type, text, audio_data, report_time)
|
||||||
if tts_file is None:
|
except Exception as e:
|
||||||
self.logger.bind(tag=TAG).error(f"tts转换失败,{text}")
|
self.logger.bind(tag=TAG).error(f"上报处理异常: {e}")
|
||||||
return None, text, text_index
|
finally:
|
||||||
self.logger.bind(tag=TAG).debug(f"TTS 文件生成完毕: {tts_file}")
|
# 标记任务完成
|
||||||
if self.max_output_size > 0:
|
self.report_queue.task_done()
|
||||||
add_device_output(self.headers.get("device-id"), len(text))
|
|
||||||
return tts_file, text, text_index
|
|
||||||
|
|
||||||
def clearSpeakStatus(self):
|
def clearSpeakStatus(self):
|
||||||
|
self.client_is_speaking = False
|
||||||
self.logger.bind(tag=TAG).debug(f"清除服务端讲话状态")
|
self.logger.bind(tag=TAG).debug(f"清除服务端讲话状态")
|
||||||
self.asr_server_receive = True
|
|
||||||
self.tts_last_text_index = -1
|
|
||||||
self.tts_first_text_index = -1
|
|
||||||
|
|
||||||
def recode_first_last_text(self, text, text_index=0):
|
|
||||||
if self.tts_first_text_index == -1:
|
|
||||||
self.logger.bind(tag=TAG).info(f"大模型说出第一句话: {text}")
|
|
||||||
self.tts_first_text_index = text_index
|
|
||||||
self.tts_last_text_index = text_index
|
|
||||||
|
|
||||||
async def close(self, ws=None):
|
async def close(self, ws=None):
|
||||||
"""资源清理方法"""
|
"""资源清理方法"""
|
||||||
|
try:
|
||||||
|
# 取消超时任务
|
||||||
|
if self.timeout_task:
|
||||||
|
self.timeout_task.cancel()
|
||||||
|
self.timeout_task = None
|
||||||
|
|
||||||
# 取消超时任务
|
# 清理MCP资源
|
||||||
if self.timeout_task:
|
if hasattr(self, "mcp_manager") and self.mcp_manager:
|
||||||
self.timeout_task.cancel()
|
await self.mcp_manager.cleanup_all()
|
||||||
self.timeout_task = None
|
|
||||||
|
|
||||||
# 清理MCP资源
|
# 触发停止事件
|
||||||
if hasattr(self, "mcp_manager") and self.mcp_manager:
|
if self.stop_event:
|
||||||
await self.mcp_manager.cleanup_all()
|
self.stop_event.set()
|
||||||
|
|
||||||
# 触发停止事件
|
# 清空任务队列
|
||||||
if self.stop_event:
|
self.clear_queues()
|
||||||
self.stop_event.set()
|
|
||||||
|
|
||||||
# 清空任务队列
|
# 关闭WebSocket连接
|
||||||
self.clear_queues()
|
if ws:
|
||||||
|
await ws.close()
|
||||||
|
elif self.websocket:
|
||||||
|
await self.websocket.close()
|
||||||
|
|
||||||
# 关闭WebSocket连接
|
# 最后关闭线程池(避免阻塞)
|
||||||
if ws:
|
if self.executor:
|
||||||
await ws.close()
|
self.executor.shutdown(wait=False)
|
||||||
elif self.websocket:
|
self.executor = None
|
||||||
await self.websocket.close()
|
|
||||||
|
|
||||||
# 最后关闭线程池(避免阻塞)
|
self.logger.bind(tag=TAG).info("连接资源已释放")
|
||||||
if self.executor:
|
except Exception as e:
|
||||||
self.executor.shutdown(wait=False)
|
self.logger.bind(tag=TAG).error(f"关闭连接时出错: {e}")
|
||||||
self.executor = None
|
|
||||||
|
|
||||||
self.logger.bind(tag=TAG).info("连接资源已释放")
|
|
||||||
|
|
||||||
def clear_queues(self):
|
def clear_queues(self):
|
||||||
"""清空所有任务队列"""
|
"""清空所有任务队列"""
|
||||||
self.logger.bind(tag=TAG).debug(
|
if self.tts:
|
||||||
f"开始清理: TTS队列大小={self.tts_queue.qsize()}, 音频队列大小={self.audio_play_queue.qsize()}"
|
self.logger.bind(tag=TAG).debug(
|
||||||
)
|
f"开始清理: TTS队列大小={self.tts.tts_text_queue.qsize()}, 音频队列大小={self.tts.tts_audio_queue.qsize()}"
|
||||||
|
)
|
||||||
|
|
||||||
# 使用非阻塞方式清空队列
|
# 使用非阻塞方式清空队列
|
||||||
for q in [self.tts_queue, self.audio_play_queue]:
|
for q in [
|
||||||
if not q:
|
self.tts.tts_text_queue,
|
||||||
continue
|
self.tts.tts_audio_queue,
|
||||||
while True:
|
self.report_queue,
|
||||||
try:
|
]:
|
||||||
q.get_nowait()
|
if not q:
|
||||||
except queue.Empty:
|
continue
|
||||||
break
|
while True:
|
||||||
|
try:
|
||||||
|
q.get_nowait()
|
||||||
|
except queue.Empty:
|
||||||
|
break
|
||||||
|
|
||||||
self.logger.bind(tag=TAG).debug(
|
self.logger.bind(tag=TAG).debug(
|
||||||
f"清理结束: TTS队列大小={self.tts_queue.qsize()}, 音频队列大小={self.audio_play_queue.qsize()}"
|
f"清理结束: TTS队列大小={self.tts.tts_text_queue.qsize()}, 音频队列大小={self.tts.tts_audio_queue.qsize()}"
|
||||||
)
|
)
|
||||||
|
|
||||||
def reset_vad_states(self):
|
def reset_vad_states(self):
|
||||||
self.client_audio_buffer = bytearray()
|
self.client_audio_buffer = bytearray()
|
||||||
|
|||||||
@@ -61,7 +61,7 @@ class FunctionHandler:
|
|||||||
self.function_registry.register_function("plugin_loader")
|
self.function_registry.register_function("plugin_loader")
|
||||||
self.function_registry.register_function("get_time")
|
self.function_registry.register_function("get_time")
|
||||||
self.function_registry.register_function("get_lunar")
|
self.function_registry.register_function("get_lunar")
|
||||||
self.function_registry.register_function("handle_speaker_volume_or_screen_brightness")
|
# self.function_registry.register_function("handle_speaker_volume_or_screen_brightness")
|
||||||
|
|
||||||
def register_config_functions(self):
|
def register_config_functions(self):
|
||||||
"""注册配置中的函数,可以不同客户端使用不同的配置"""
|
"""注册配置中的函数,可以不同客户端使用不同的配置"""
|
||||||
|
|||||||
@@ -1,11 +1,18 @@
|
|||||||
|
import os
|
||||||
|
import time
|
||||||
import json
|
import json
|
||||||
from core.handle.sendAudioHandle import send_stt_message
|
import random
|
||||||
from core.utils.util import remove_punctuation_and_length
|
|
||||||
import shutil
|
import shutil
|
||||||
import asyncio
|
import asyncio
|
||||||
import os
|
from core.handle.sendAudioHandle import send_stt_message
|
||||||
import random
|
from core.utils.util import remove_punctuation_and_length
|
||||||
import time
|
from core.providers.tts.dto.dto import ContentType, InterfaceType
|
||||||
|
from core.handle.mcpHandle import (
|
||||||
|
MCPClient,
|
||||||
|
send_mcp_initialize_message,
|
||||||
|
send_mcp_tools_list_request,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
|
|
||||||
@@ -26,8 +33,20 @@ async def handleHelloMessage(conn, msg_json):
|
|||||||
format = audio_params.get("format")
|
format = audio_params.get("format")
|
||||||
conn.logger.bind(tag=TAG).info(f"客户端音频格式: {format}")
|
conn.logger.bind(tag=TAG).info(f"客户端音频格式: {format}")
|
||||||
conn.audio_format = format
|
conn.audio_format = format
|
||||||
conn.asr.set_audio_format(format)
|
if conn.asr is not None:
|
||||||
|
conn.asr.set_audio_format(format)
|
||||||
conn.welcome_msg["audio_params"] = audio_params
|
conn.welcome_msg["audio_params"] = audio_params
|
||||||
|
features = msg_json.get("features")
|
||||||
|
if features:
|
||||||
|
conn.logger.bind(tag=TAG).info(f"客户端特性: {features}")
|
||||||
|
conn.features = features
|
||||||
|
if features.get("mcp"):
|
||||||
|
conn.logger.bind(tag=TAG).info("客户端支持MCP")
|
||||||
|
conn.mcp_client = MCPClient()
|
||||||
|
# 发送初始化
|
||||||
|
asyncio.create_task(send_mcp_initialize_message(conn))
|
||||||
|
# 发送mcp消息,获取tools列表
|
||||||
|
asyncio.create_task(send_mcp_tools_list_request(conn))
|
||||||
|
|
||||||
await conn.websocket.send(json.dumps(conn.welcome_msg))
|
await conn.websocket.send(json.dumps(conn.welcome_msg))
|
||||||
|
|
||||||
@@ -36,6 +55,10 @@ async def checkWakeupWords(conn, text):
|
|||||||
enable_wakeup_words_response_cache = conn.config[
|
enable_wakeup_words_response_cache = conn.config[
|
||||||
"enable_wakeup_words_response_cache"
|
"enable_wakeup_words_response_cache"
|
||||||
]
|
]
|
||||||
|
"""是否用的是非流式tts"""
|
||||||
|
if conn.tts and conn.tts.interface_type != InterfaceType.NON_STREAM:
|
||||||
|
return False
|
||||||
|
|
||||||
"""是否开启唤醒词加速"""
|
"""是否开启唤醒词加速"""
|
||||||
if not enable_wakeup_words_response_cache:
|
if not enable_wakeup_words_response_cache:
|
||||||
return False
|
return False
|
||||||
@@ -43,19 +66,17 @@ async def checkWakeupWords(conn, text):
|
|||||||
_, filtered_text = remove_punctuation_and_length(text)
|
_, filtered_text = remove_punctuation_and_length(text)
|
||||||
if filtered_text in conn.config.get("wakeup_words"):
|
if filtered_text in conn.config.get("wakeup_words"):
|
||||||
await send_stt_message(conn, text)
|
await send_stt_message(conn, text)
|
||||||
conn.tts_first_text_index = 0
|
|
||||||
conn.tts_last_text_index = 0
|
|
||||||
conn.llm_finish_task = True
|
|
||||||
|
|
||||||
file = getWakeupWordFile(WAKEUP_CONFIG["file_name"])
|
file = getWakeupWordFile(WAKEUP_CONFIG["file_name"])
|
||||||
if file is None:
|
if file is None:
|
||||||
asyncio.create_task(wakeupWordsResponse(conn))
|
asyncio.create_task(wakeupWordsResponse(conn))
|
||||||
return False
|
return False
|
||||||
opus_packets, _ = conn.tts.audio_to_opus_data(file)
|
|
||||||
text_hello = WAKEUP_CONFIG["text"]
|
text_hello = WAKEUP_CONFIG["text"]
|
||||||
if not text_hello:
|
if not text_hello:
|
||||||
text_hello = text
|
text_hello = text
|
||||||
conn.audio_play_queue.put((opus_packets, text_hello, 0))
|
conn.tts.tts_one_sentence(
|
||||||
|
conn, ContentType.FILE, content_file=file, content_detail=text_hello
|
||||||
|
)
|
||||||
if time.time() - WAKEUP_CONFIG["create_time"] > WAKEUP_CONFIG["refresh_time"]:
|
if time.time() - WAKEUP_CONFIG["create_time"] > WAKEUP_CONFIG["refresh_time"]:
|
||||||
asyncio.create_task(wakeupWordsResponse(conn))
|
asyncio.create_task(wakeupWordsResponse(conn))
|
||||||
return True
|
return True
|
||||||
@@ -87,7 +108,13 @@ async def wakeupWordsResponse(conn):
|
|||||||
|
|
||||||
"""唤醒词响应"""
|
"""唤醒词响应"""
|
||||||
wakeup_word = random.choice(WAKEUP_CONFIG["words"])
|
wakeup_word = random.choice(WAKEUP_CONFIG["words"])
|
||||||
result = conn.llm.response_no_stream(conn.config["prompt"], wakeup_word)
|
question = (
|
||||||
|
"此刻用户正在和你说```"
|
||||||
|
+ wakeup_word
|
||||||
|
+ "```。\n请你根据以上用户的内容,进行简短回复,文字内容控制在15个字以内。\n"
|
||||||
|
+ "请勿对这条内容本身进行任何解释和回应,仅返回对用户的内容的回复。"
|
||||||
|
)
|
||||||
|
result = conn.llm.response_no_stream(conn.config["prompt"], question)
|
||||||
if result is None or result == "":
|
if result is None or result == "":
|
||||||
return
|
return
|
||||||
tts_file = await asyncio.to_thread(conn.tts.to_tts, result)
|
tts_file = await asyncio.to_thread(conn.tts.to_tts, result)
|
||||||
|
|||||||
@@ -1,9 +1,9 @@
|
|||||||
from config.logger import setup_logging
|
|
||||||
import json
|
import json
|
||||||
import uuid
|
import uuid
|
||||||
from core.handle.sendAudioHandle import send_stt_message
|
from core.handle.sendAudioHandle import send_stt_message
|
||||||
from core.handle.helloHandle import checkWakeupWords
|
from core.handle.helloHandle import checkWakeupWords
|
||||||
from core.utils.util import remove_punctuation_and_length
|
from core.utils.util import remove_punctuation_and_length
|
||||||
|
from core.providers.tts.dto.dto import ContentType
|
||||||
from core.utils.dialogue import Message
|
from core.utils.dialogue import Message
|
||||||
from plugins_func.register import Action
|
from plugins_func.register import Action
|
||||||
from loguru import logger
|
from loguru import logger
|
||||||
@@ -109,21 +109,21 @@ async def process_intent_result(conn, intent_result, original_text):
|
|||||||
if result.action == Action.RESPONSE: # 直接回复前端
|
if result.action == Action.RESPONSE: # 直接回复前端
|
||||||
text = result.response
|
text = result.response
|
||||||
if text is not None:
|
if text is not None:
|
||||||
speak_and_play(conn, text)
|
speak_txt(conn, text)
|
||||||
elif result.action == Action.REQLLM: # 调用函数后再请求llm生成回复
|
elif result.action == Action.REQLLM: # 调用函数后再请求llm生成回复
|
||||||
text = result.result
|
text = result.result
|
||||||
conn.dialogue.put(Message(role="tool", content=text))
|
conn.dialogue.put(Message(role="tool", content=text))
|
||||||
llm_result = conn.intent.replyResult(text, original_text)
|
llm_result = conn.intent.replyResult(text, original_text)
|
||||||
if llm_result is None:
|
if llm_result is None:
|
||||||
llm_result = text
|
llm_result = text
|
||||||
speak_and_play(conn, llm_result)
|
speak_txt(conn, llm_result)
|
||||||
elif (
|
elif (
|
||||||
result.action == Action.NOTFOUND
|
result.action == Action.NOTFOUND
|
||||||
or result.action == Action.ERROR
|
or result.action == Action.ERROR
|
||||||
):
|
):
|
||||||
text = result.result
|
text = result.result
|
||||||
if text is not None:
|
if text is not None:
|
||||||
speak_and_play(conn, text)
|
speak_txt(conn, text)
|
||||||
elif function_name != "play_music":
|
elif function_name != "play_music":
|
||||||
# For backward compatibility with original code
|
# For backward compatibility with original code
|
||||||
# 获取当前最新的文本索引
|
# 获取当前最新的文本索引
|
||||||
@@ -131,7 +131,7 @@ async def process_intent_result(conn, intent_result, original_text):
|
|||||||
if text is None:
|
if text is None:
|
||||||
text = result.result
|
text = result.result
|
||||||
if text is not None:
|
if text is not None:
|
||||||
speak_and_play(conn, text)
|
speak_txt(conn, text)
|
||||||
|
|
||||||
# 将函数执行放在线程池中
|
# 将函数执行放在线程池中
|
||||||
conn.executor.submit(process_function_call)
|
conn.executor.submit(process_function_call)
|
||||||
@@ -142,12 +142,6 @@ async def process_intent_result(conn, intent_result, original_text):
|
|||||||
return False
|
return False
|
||||||
|
|
||||||
|
|
||||||
def speak_and_play(conn, text):
|
def speak_txt(conn, text):
|
||||||
text_index = (
|
conn.tts.tts_one_sentence(conn, ContentType.TEXT, content_detail=text)
|
||||||
conn.tts_last_text_index + 1 if hasattr(conn, "tts_last_text_index") else 0
|
|
||||||
)
|
|
||||||
conn.recode_first_last_text(text, text_index)
|
|
||||||
future = conn.executor.submit(conn.speak_and_play, text, text_index)
|
|
||||||
conn.llm_finish_task = True
|
|
||||||
conn.tts_queue.put((future, text_index))
|
|
||||||
conn.dialogue.put(Message(role="assistant", content=text))
|
conn.dialogue.put(Message(role="assistant", content=text))
|
||||||
|
|||||||
@@ -0,0 +1,374 @@
|
|||||||
|
import json
|
||||||
|
import asyncio
|
||||||
|
from concurrent.futures import Future
|
||||||
|
from core.utils.util import get_vision_url
|
||||||
|
from core.utils.auth import AuthToken
|
||||||
|
|
||||||
|
TAG = __name__
|
||||||
|
|
||||||
|
|
||||||
|
class MCPClient:
|
||||||
|
"""MCPClient,用于管理MCP状态和工具"""
|
||||||
|
|
||||||
|
def __init__(self):
|
||||||
|
self.tools = {} # Dictionary for O(1) lookup
|
||||||
|
self.ready = False
|
||||||
|
self.call_results = {} # To store Futures for tool call responses
|
||||||
|
self.next_id = 1
|
||||||
|
self.lock = asyncio.Lock()
|
||||||
|
self._cached_available_tools = None # Cache for get_available_tools
|
||||||
|
|
||||||
|
def has_tool(self, name: str) -> bool:
|
||||||
|
return name in self.tools
|
||||||
|
|
||||||
|
def get_available_tools(self) -> list:
|
||||||
|
# Check if the cache is valid
|
||||||
|
if self._cached_available_tools is not None:
|
||||||
|
return self._cached_available_tools
|
||||||
|
|
||||||
|
# If cache is not valid, regenerate the list
|
||||||
|
result = []
|
||||||
|
for tool_name, tool_data in self.tools.items():
|
||||||
|
function_def = {
|
||||||
|
"name": tool_data["name"],
|
||||||
|
"description": tool_data["description"],
|
||||||
|
"parameters": {
|
||||||
|
"type": tool_data["inputSchema"].get("type", "object"),
|
||||||
|
"properties": tool_data["inputSchema"].get("properties", {}),
|
||||||
|
"required": tool_data["inputSchema"].get("required", []),
|
||||||
|
},
|
||||||
|
}
|
||||||
|
result.append({"type": "function", "function": function_def})
|
||||||
|
|
||||||
|
self._cached_available_tools = result # Store the generated list in cache
|
||||||
|
return result
|
||||||
|
|
||||||
|
async def is_ready(self) -> bool:
|
||||||
|
async with self.lock:
|
||||||
|
return self.ready
|
||||||
|
|
||||||
|
async def set_ready(self, status: bool):
|
||||||
|
async with self.lock:
|
||||||
|
self.ready = status
|
||||||
|
|
||||||
|
async def add_tool(self, tool_data: dict):
|
||||||
|
async with self.lock:
|
||||||
|
self.tools[tool_data["name"]] = tool_data
|
||||||
|
self._cached_available_tools = (
|
||||||
|
None # Invalidate the cache when a tool is added
|
||||||
|
)
|
||||||
|
|
||||||
|
async def get_next_id(self) -> int:
|
||||||
|
async with self.lock:
|
||||||
|
current_id = self.next_id
|
||||||
|
self.next_id += 1
|
||||||
|
return current_id
|
||||||
|
|
||||||
|
async def register_call_result_future(self, id: int, future: Future):
|
||||||
|
async with self.lock:
|
||||||
|
self.call_results[id] = future
|
||||||
|
|
||||||
|
async def resolve_call_result(self, id: int, result: any):
|
||||||
|
async with self.lock:
|
||||||
|
if id in self.call_results:
|
||||||
|
future = self.call_results.pop(id)
|
||||||
|
if not future.done():
|
||||||
|
future.set_result(result)
|
||||||
|
|
||||||
|
async def reject_call_result(self, id: int, exception: Exception):
|
||||||
|
async with self.lock:
|
||||||
|
if id in self.call_results:
|
||||||
|
future = self.call_results.pop(id)
|
||||||
|
if not future.done():
|
||||||
|
future.set_exception(exception)
|
||||||
|
|
||||||
|
async def cleanup_call_result(self, id: int):
|
||||||
|
async with self.lock:
|
||||||
|
if id in self.call_results:
|
||||||
|
self.call_results.pop(id)
|
||||||
|
|
||||||
|
|
||||||
|
async def send_mcp_message(conn, payload: dict):
|
||||||
|
"""Helper to send MCP messages, encapsulating common logic."""
|
||||||
|
if not conn.features.get("mcp"):
|
||||||
|
conn.logger.bind(tag=TAG).warning("客户端不支持MCP,无法发送MCP消息")
|
||||||
|
return
|
||||||
|
|
||||||
|
message = json.dumps({"type": "mcp", "payload": payload})
|
||||||
|
|
||||||
|
try:
|
||||||
|
await conn.websocket.send(message)
|
||||||
|
conn.logger.bind(tag=TAG).info(f"成功发送MCP消息: {message}")
|
||||||
|
except Exception as e:
|
||||||
|
conn.logger.bind(tag=TAG).error(f"发送MCP消息失败: {e}")
|
||||||
|
|
||||||
|
|
||||||
|
async def handle_mcp_message(conn, mcp_client: MCPClient, payload: dict):
|
||||||
|
"""处理MCP消息,包括初始化、工具列表和工具调用响应等"""
|
||||||
|
conn.logger.bind(tag=TAG).info(f"处理MCP消息: {payload}")
|
||||||
|
|
||||||
|
if not isinstance(payload, dict):
|
||||||
|
conn.logger.bind(tag=TAG).error("MCP消息缺少payload字段或格式错误")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Handle result
|
||||||
|
if "result" in payload:
|
||||||
|
result = payload["result"]
|
||||||
|
msg_id = int(payload.get("id", 0))
|
||||||
|
|
||||||
|
# Check for tool call response first
|
||||||
|
if msg_id in mcp_client.call_results:
|
||||||
|
conn.logger.bind(tag=TAG).debug(
|
||||||
|
f"收到工具调用响应,ID: {msg_id}, 结果: {result}"
|
||||||
|
)
|
||||||
|
await mcp_client.resolve_call_result(msg_id, result)
|
||||||
|
return
|
||||||
|
|
||||||
|
if msg_id == 1: # mcpInitializeID
|
||||||
|
conn.logger.bind(tag=TAG).debug("收到MCP初始化响应")
|
||||||
|
server_info = result.get("serverInfo")
|
||||||
|
if isinstance(server_info, dict):
|
||||||
|
name = server_info.get("name")
|
||||||
|
version = server_info.get("version")
|
||||||
|
conn.logger.bind(tag=TAG).info(
|
||||||
|
f"客户端MCP服务器信息: name={name}, version={version}"
|
||||||
|
)
|
||||||
|
await send_mcp_tools_list_request(
|
||||||
|
conn
|
||||||
|
) # After initialization, request tool list
|
||||||
|
return
|
||||||
|
|
||||||
|
elif msg_id == 2: # mcpToolsListID
|
||||||
|
conn.logger.bind(tag=TAG).debug("收到MCP工具列表响应")
|
||||||
|
if isinstance(result, dict) and "tools" in result:
|
||||||
|
tools_data = result["tools"]
|
||||||
|
if not isinstance(tools_data, list):
|
||||||
|
conn.logger.bind(tag=TAG).error("工具列表格式错误")
|
||||||
|
return
|
||||||
|
|
||||||
|
conn.logger.bind(tag=TAG).info(
|
||||||
|
f"客户端设备支持的工具数量: {len(tools_data)}"
|
||||||
|
)
|
||||||
|
|
||||||
|
for i, tool in enumerate(tools_data):
|
||||||
|
if not isinstance(tool, dict):
|
||||||
|
continue
|
||||||
|
|
||||||
|
name = tool.get("name", "")
|
||||||
|
description = tool.get("description", "")
|
||||||
|
input_schema = {"type": "object", "properties": {}, "required": []}
|
||||||
|
|
||||||
|
if "inputSchema" in tool and isinstance(tool["inputSchema"], dict):
|
||||||
|
schema = tool["inputSchema"]
|
||||||
|
input_schema["type"] = schema.get("type", "object")
|
||||||
|
input_schema["properties"] = schema.get("properties", {})
|
||||||
|
input_schema["required"] = [
|
||||||
|
s for s in schema.get("required", []) if isinstance(s, str)
|
||||||
|
]
|
||||||
|
|
||||||
|
new_tool = {
|
||||||
|
"name": name,
|
||||||
|
"description": description,
|
||||||
|
"inputSchema": input_schema,
|
||||||
|
}
|
||||||
|
await mcp_client.add_tool(new_tool)
|
||||||
|
conn.logger.bind(tag=TAG).debug(f"客户端工具 #{i+1}: {name}")
|
||||||
|
|
||||||
|
next_cursor = result.get("nextCursor", "")
|
||||||
|
if next_cursor:
|
||||||
|
conn.logger.bind(tag=TAG).info(
|
||||||
|
f"有更多工具,nextCursor: {next_cursor}"
|
||||||
|
)
|
||||||
|
await send_mcp_tools_list_continue_request(conn, next_cursor)
|
||||||
|
else:
|
||||||
|
await mcp_client.set_ready(True)
|
||||||
|
conn.logger.bind(tag=TAG).info("所有工具已获取,MCP客户端准备就绪")
|
||||||
|
return
|
||||||
|
|
||||||
|
# Handle method calls (requests from the client)
|
||||||
|
elif "method" in payload:
|
||||||
|
method = payload["method"]
|
||||||
|
conn.logger.bind(tag=TAG).info(f"收到MCP客户端请求: {method}")
|
||||||
|
|
||||||
|
elif "error" in payload:
|
||||||
|
error_data = payload["error"]
|
||||||
|
error_msg = error_data.get("message", "未知错误")
|
||||||
|
conn.logger.bind(tag=TAG).error(f"收到MCP错误响应: {error_msg}")
|
||||||
|
|
||||||
|
msg_id = int(payload.get("id", 0))
|
||||||
|
if msg_id in mcp_client.call_results:
|
||||||
|
await mcp_client.reject_call_result(
|
||||||
|
msg_id, Exception(f"MCP错误: {error_msg}")
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
# --- Outgoing MCP Messages ---
|
||||||
|
|
||||||
|
|
||||||
|
async def send_mcp_initialize_message(conn):
|
||||||
|
"""发送MCP初始化消息"""
|
||||||
|
|
||||||
|
vision_url = get_vision_url(conn.config)
|
||||||
|
|
||||||
|
# 密钥生成token
|
||||||
|
auth = AuthToken(conn.config["server"]["auth_key"])
|
||||||
|
token = auth.generate_token(conn.headers.get("device-id"))
|
||||||
|
|
||||||
|
vision = {
|
||||||
|
"url": vision_url,
|
||||||
|
"token": token,
|
||||||
|
}
|
||||||
|
|
||||||
|
conn.logger.bind(tag=TAG).info(f"视觉服务信息: {vision}")
|
||||||
|
|
||||||
|
payload = {
|
||||||
|
"jsonrpc": "2.0",
|
||||||
|
"id": 1, # mcpInitializeID
|
||||||
|
"method": "initialize",
|
||||||
|
"params": {
|
||||||
|
"protocolVersion": "2024-11-05",
|
||||||
|
"capabilities": {
|
||||||
|
"roots": {"listChanged": True},
|
||||||
|
"sampling": {},
|
||||||
|
"vision": vision,
|
||||||
|
},
|
||||||
|
"clientInfo": {
|
||||||
|
"name": "XiaozhiClient",
|
||||||
|
"version": "1.0.0",
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
conn.logger.bind(tag=TAG).info("发送MCP初始化消息")
|
||||||
|
await send_mcp_message(conn, payload)
|
||||||
|
|
||||||
|
|
||||||
|
async def send_mcp_tools_list_request(conn):
|
||||||
|
"""发送MCP工具列表请求"""
|
||||||
|
payload = {
|
||||||
|
"jsonrpc": "2.0",
|
||||||
|
"id": 2, # mcpToolsListID
|
||||||
|
"method": "tools/list",
|
||||||
|
}
|
||||||
|
conn.logger.bind(tag=TAG).debug("发送MCP工具列表请求")
|
||||||
|
await send_mcp_message(conn, payload)
|
||||||
|
|
||||||
|
|
||||||
|
async def send_mcp_tools_list_continue_request(conn, cursor: str):
|
||||||
|
"""发送带有cursor的MCP工具列表请求"""
|
||||||
|
payload = {
|
||||||
|
"jsonrpc": "2.0",
|
||||||
|
"id": 2, # mcpToolsListID (same ID for continuation)
|
||||||
|
"method": "tools/list",
|
||||||
|
"params": {"cursor": cursor},
|
||||||
|
}
|
||||||
|
conn.logger.bind(tag=TAG).info(f"发送带cursor的MCP工具列表请求: {cursor}")
|
||||||
|
await send_mcp_message(conn, payload)
|
||||||
|
|
||||||
|
|
||||||
|
async def call_mcp_tool(
|
||||||
|
conn, mcp_client: MCPClient, tool_name: str, args: str = "{}", timeout: int = 30
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
调用指定的工具,并等待响应
|
||||||
|
"""
|
||||||
|
if not await mcp_client.is_ready():
|
||||||
|
raise RuntimeError("MCP客户端尚未准备就绪")
|
||||||
|
|
||||||
|
if not mcp_client.has_tool(tool_name):
|
||||||
|
raise ValueError(f"工具 {tool_name} 不存在")
|
||||||
|
|
||||||
|
tool_call_id = await mcp_client.get_next_id()
|
||||||
|
result_future = asyncio.Future()
|
||||||
|
await mcp_client.register_call_result_future(tool_call_id, result_future)
|
||||||
|
|
||||||
|
# 处理参数
|
||||||
|
try:
|
||||||
|
if isinstance(args, str):
|
||||||
|
# 确保字符串是有效的JSON
|
||||||
|
if not args.strip():
|
||||||
|
arguments = {}
|
||||||
|
else:
|
||||||
|
try:
|
||||||
|
# 尝试直接解析
|
||||||
|
arguments = json.loads(args)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
# 如果解析失败,尝试合并多个JSON对象
|
||||||
|
try:
|
||||||
|
# 使用正则表达式匹配所有JSON对象
|
||||||
|
import re
|
||||||
|
|
||||||
|
json_objects = re.findall(r"\{[^{}]*\}", args)
|
||||||
|
if len(json_objects) > 1:
|
||||||
|
# 合并所有JSON对象
|
||||||
|
merged_dict = {}
|
||||||
|
for json_str in json_objects:
|
||||||
|
try:
|
||||||
|
obj = json.loads(json_str)
|
||||||
|
if isinstance(obj, dict):
|
||||||
|
merged_dict.update(obj)
|
||||||
|
except json.JSONDecodeError:
|
||||||
|
continue
|
||||||
|
if merged_dict:
|
||||||
|
arguments = merged_dict
|
||||||
|
else:
|
||||||
|
raise ValueError(f"无法解析任何有效的JSON对象: {args}")
|
||||||
|
else:
|
||||||
|
raise ValueError(f"参数JSON解析失败: {args}")
|
||||||
|
except Exception as e:
|
||||||
|
conn.logger.bind(tag=TAG).error(
|
||||||
|
f"参数JSON解析失败: {str(e)}, 原始参数: {args}"
|
||||||
|
)
|
||||||
|
raise ValueError(f"参数JSON解析失败: {str(e)}")
|
||||||
|
elif isinstance(args, dict):
|
||||||
|
arguments = args
|
||||||
|
else:
|
||||||
|
raise ValueError(f"参数类型错误,期望字符串或字典,实际类型: {type(args)}")
|
||||||
|
|
||||||
|
# 确保参数是字典类型
|
||||||
|
if not isinstance(arguments, dict):
|
||||||
|
raise ValueError(f"参数必须是字典类型,实际类型: {type(arguments)}")
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
if not isinstance(e, ValueError):
|
||||||
|
raise ValueError(f"参数处理失败: {str(e)}")
|
||||||
|
raise e
|
||||||
|
|
||||||
|
payload = {
|
||||||
|
"jsonrpc": "2.0",
|
||||||
|
"id": tool_call_id,
|
||||||
|
"method": "tools/call",
|
||||||
|
"params": {"name": tool_name, "arguments": arguments},
|
||||||
|
}
|
||||||
|
|
||||||
|
conn.logger.bind(tag=TAG).info(
|
||||||
|
f"发送客户端mcp工具调用请求: {tool_name},参数: {args}"
|
||||||
|
)
|
||||||
|
await send_mcp_message(conn, payload)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# Wait for response or timeout
|
||||||
|
raw_result = await asyncio.wait_for(result_future, timeout=timeout)
|
||||||
|
conn.logger.bind(tag=TAG).info(
|
||||||
|
f"客户端mcp工具调用 {tool_name} 成功,原始结果: {raw_result}"
|
||||||
|
)
|
||||||
|
|
||||||
|
if isinstance(raw_result, dict):
|
||||||
|
if raw_result.get("isError") is True:
|
||||||
|
error_msg = raw_result.get(
|
||||||
|
"error", "工具调用返回错误,但未提供具体错误信息"
|
||||||
|
)
|
||||||
|
raise RuntimeError(f"工具调用错误: {error_msg}")
|
||||||
|
|
||||||
|
content = raw_result.get("content")
|
||||||
|
if isinstance(content, list) and len(content) > 0:
|
||||||
|
if isinstance(content[0], dict) and "text" in content[0]:
|
||||||
|
# 直接返回文本内容,不进行JSON解析
|
||||||
|
return content[0]["text"]
|
||||||
|
# 如果结果不是预期的格式,将其转换为字符串
|
||||||
|
return str(raw_result)
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
await mcp_client.cleanup_call_result(tool_call_id)
|
||||||
|
raise TimeoutError("工具调用请求超时")
|
||||||
|
except Exception as e:
|
||||||
|
await mcp_client.cleanup_call_result(tool_call_id)
|
||||||
|
raise e
|
||||||
@@ -1,10 +1,9 @@
|
|||||||
import time
|
|
||||||
import copy
|
|
||||||
from core.utils.util import remove_punctuation_and_length
|
|
||||||
from core.handle.sendAudioHandle import send_stt_message
|
from core.handle.sendAudioHandle import send_stt_message
|
||||||
from core.handle.intentHandler import handle_user_intent
|
from core.handle.intentHandler import handle_user_intent
|
||||||
from core.utils.output_counter import check_device_output_limit
|
from core.utils.output_counter import check_device_output_limit
|
||||||
from core.handle.reportHandle import enqueue_asr_report
|
from core.handle.abortHandle import handleAbortMessage
|
||||||
|
import time
|
||||||
|
from core.handle.sendAudioHandle import SentenceType
|
||||||
from core.utils.util import audio_to_data
|
from core.utils.util import audio_to_data
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
@@ -12,45 +11,20 @@ TAG = __name__
|
|||||||
|
|
||||||
async def handleAudioMessage(conn, audio):
|
async def handleAudioMessage(conn, audio):
|
||||||
if conn.vad is None:
|
if conn.vad is None:
|
||||||
|
conn.logger.bind(tag=TAG).warning("VAD模块未初始化,继续等待")
|
||||||
return
|
return
|
||||||
if not conn.asr_server_receive:
|
if conn.asr is None or not hasattr(conn.asr, "conn") or conn.asr.conn is None:
|
||||||
conn.logger.bind(tag=TAG).debug(f"前期数据处理中,暂停接收")
|
conn.logger.bind(tag=TAG).warning("ASR模块未初始化或通道未就绪,继续等待")
|
||||||
return
|
return
|
||||||
if conn.client_listen_mode == "auto" or conn.client_listen_mode == "realtime":
|
# 当前片段是否有人说话
|
||||||
have_voice = conn.vad.is_vad(conn, audio)
|
have_voice = conn.vad.is_vad(conn, audio)
|
||||||
else:
|
if have_voice:
|
||||||
have_voice = conn.client_have_voice
|
if conn.client_is_speaking:
|
||||||
|
await handleAbortMessage(conn)
|
||||||
# 如果本次没有声音,本段也没声音,就把声音丢弃了
|
# 设备长时间空闲检测,用于say goodbye
|
||||||
if have_voice == False and conn.client_have_voice == False:
|
await no_voice_close_connect(conn, have_voice)
|
||||||
await no_voice_close_connect(conn)
|
# 接收音频
|
||||||
conn.asr_audio.append(audio)
|
await conn.asr.receive_audio(audio, have_voice)
|
||||||
conn.asr_audio = conn.asr_audio[
|
|
||||||
-10:
|
|
||||||
] # 保留最新的10帧音频内容,解决ASR句首丢字问题
|
|
||||||
return
|
|
||||||
conn.client_no_voice_last_time = 0.0
|
|
||||||
conn.asr_audio.append(audio)
|
|
||||||
# 如果本段有声音,且已经停止了
|
|
||||||
if conn.client_voice_stop:
|
|
||||||
conn.client_abort = False
|
|
||||||
conn.asr_server_receive = False
|
|
||||||
# 音频太短了,无法识别
|
|
||||||
if len(conn.asr_audio) < 15:
|
|
||||||
conn.asr_server_receive = True
|
|
||||||
else:
|
|
||||||
raw_text, _ = await conn.asr.speech_to_text(conn.asr_audio, conn.session_id) # 确保ASR模块返回原始文本
|
|
||||||
conn.logger.bind(tag=TAG).info(f"识别文本: {raw_text}")
|
|
||||||
text_len, _ = remove_punctuation_and_length(raw_text)
|
|
||||||
if text_len > 0:
|
|
||||||
# 使用自定义模块进行上报
|
|
||||||
enqueue_asr_report(conn, raw_text, copy.deepcopy(conn.asr_audio))
|
|
||||||
|
|
||||||
await startToChat(conn, raw_text)
|
|
||||||
else:
|
|
||||||
conn.asr_server_receive = True
|
|
||||||
conn.asr_audio.clear()
|
|
||||||
conn.reset_vad_states()
|
|
||||||
|
|
||||||
|
|
||||||
async def startToChat(conn, text):
|
async def startToChat(conn, text):
|
||||||
@@ -65,25 +39,25 @@ async def startToChat(conn, text):
|
|||||||
):
|
):
|
||||||
await max_out_size(conn)
|
await max_out_size(conn)
|
||||||
return
|
return
|
||||||
|
if conn.client_is_speaking:
|
||||||
|
await handleAbortMessage(conn)
|
||||||
|
|
||||||
# 首先进行意图分析
|
# 首先进行意图分析
|
||||||
intent_handled = await handle_user_intent(conn, text)
|
intent_handled = await handle_user_intent(conn, text)
|
||||||
|
|
||||||
if intent_handled:
|
if intent_handled:
|
||||||
# 如果意图已被处理,不再进行聊天
|
# 如果意图已被处理,不再进行聊天
|
||||||
conn.asr_server_receive = True
|
|
||||||
return
|
return
|
||||||
|
|
||||||
# 意图未被处理,继续常规聊天流程
|
# 意图未被处理,继续常规聊天流程
|
||||||
await send_stt_message(conn, text)
|
await send_stt_message(conn, text)
|
||||||
if conn.intent_type == "function_call":
|
conn.executor.submit(conn.chat, text)
|
||||||
# 使用支持function calling的聊天方法
|
|
||||||
conn.executor.submit(conn.chat_with_function_calling, text)
|
|
||||||
else:
|
|
||||||
conn.executor.submit(conn.chat, text)
|
|
||||||
|
|
||||||
|
|
||||||
async def no_voice_close_connect(conn):
|
async def no_voice_close_connect(conn, have_voice):
|
||||||
|
if have_voice:
|
||||||
|
conn.client_no_voice_last_time = 0.0
|
||||||
|
return
|
||||||
if conn.client_no_voice_last_time == 0.0:
|
if conn.client_no_voice_last_time == 0.0:
|
||||||
conn.client_no_voice_last_time = time.time() * 1000
|
conn.client_no_voice_last_time = time.time() * 1000
|
||||||
else:
|
else:
|
||||||
@@ -97,7 +71,6 @@ async def no_voice_close_connect(conn):
|
|||||||
):
|
):
|
||||||
conn.close_after_chat = True
|
conn.close_after_chat = True
|
||||||
conn.client_abort = False
|
conn.client_abort = False
|
||||||
conn.asr_server_receive = False
|
|
||||||
end_prompt = conn.config.get("end_prompt", {})
|
end_prompt = conn.config.get("end_prompt", {})
|
||||||
if end_prompt and end_prompt.get("enable", True) is False:
|
if end_prompt and end_prompt.get("enable", True) is False:
|
||||||
conn.logger.bind(tag=TAG).info("结束对话,无需发送结束提示语")
|
conn.logger.bind(tag=TAG).info("结束对话,无需发送结束提示语")
|
||||||
@@ -105,19 +78,16 @@ async def no_voice_close_connect(conn):
|
|||||||
return
|
return
|
||||||
prompt = end_prompt.get("prompt")
|
prompt = end_prompt.get("prompt")
|
||||||
if not prompt:
|
if not prompt:
|
||||||
prompt = "请你以“时间过得真快”未来头,用富有感情、依依不舍的话来结束这场对话吧。!"
|
prompt = "请你以```时间过得真快```未来头,用富有感情、依依不舍的话来结束这场对话吧。!"
|
||||||
await startToChat(conn, prompt)
|
await startToChat(conn, prompt)
|
||||||
|
|
||||||
|
|
||||||
async def max_out_size(conn):
|
async def max_out_size(conn):
|
||||||
text = "不好意思,我现在有点事情要忙,明天这个时候我们再聊,约好了哦!明天不见不散,拜拜!"
|
text = "不好意思,我现在有点事情要忙,明天这个时候我们再聊,约好了哦!明天不见不散,拜拜!"
|
||||||
await send_stt_message(conn, text)
|
await send_stt_message(conn, text)
|
||||||
conn.tts_first_text_index = 0
|
|
||||||
conn.tts_last_text_index = 0
|
|
||||||
conn.llm_finish_task = True
|
|
||||||
file_path = "config/assets/max_output_size.wav"
|
file_path = "config/assets/max_output_size.wav"
|
||||||
opus_packets, _ = audio_to_data(file_path)
|
opus_packets, _ = audio_to_data(file_path)
|
||||||
conn.audio_play_queue.put((opus_packets, text, 0))
|
conn.tts.tts_audio_queue.put((SentenceType.LAST, opus_packets, text))
|
||||||
conn.close_after_chat = True
|
conn.close_after_chat = True
|
||||||
|
|
||||||
|
|
||||||
@@ -132,14 +102,11 @@ async def check_bind_device(conn):
|
|||||||
|
|
||||||
text = f"请登录控制面板,输入{conn.bind_code},绑定设备。"
|
text = f"请登录控制面板,输入{conn.bind_code},绑定设备。"
|
||||||
await send_stt_message(conn, text)
|
await send_stt_message(conn, text)
|
||||||
conn.tts_first_text_index = 0
|
|
||||||
conn.tts_last_text_index = 6
|
|
||||||
conn.llm_finish_task = True
|
|
||||||
|
|
||||||
# 播放提示音
|
# 播放提示音
|
||||||
music_path = "config/assets/bind_code.wav"
|
music_path = "config/assets/bind_code.wav"
|
||||||
opus_packets, _ = audio_to_data(music_path)
|
opus_packets, _ = audio_to_data(music_path)
|
||||||
conn.audio_play_queue.put((opus_packets, text, 0))
|
conn.tts.tts_audio_queue.put((SentenceType.FIRST, opus_packets, text))
|
||||||
|
|
||||||
# 逐个播放数字
|
# 逐个播放数字
|
||||||
for i in range(6): # 确保只播放6位数字
|
for i in range(6): # 确保只播放6位数字
|
||||||
@@ -147,16 +114,14 @@ async def check_bind_device(conn):
|
|||||||
digit = conn.bind_code[i]
|
digit = conn.bind_code[i]
|
||||||
num_path = f"config/assets/bind_code/{digit}.wav"
|
num_path = f"config/assets/bind_code/{digit}.wav"
|
||||||
num_packets, _ = audio_to_data(num_path)
|
num_packets, _ = audio_to_data(num_path)
|
||||||
conn.audio_play_queue.put((num_packets, None, i + 1))
|
conn.tts.tts_audio_queue.put((SentenceType.MIDDLE, num_packets, None))
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
conn.logger.bind(tag=TAG).error(f"播放数字音频失败: {e}")
|
conn.logger.bind(tag=TAG).error(f"播放数字音频失败: {e}")
|
||||||
continue
|
continue
|
||||||
|
conn.tts.tts_audio_queue.put((SentenceType.LAST, [], None))
|
||||||
else:
|
else:
|
||||||
text = f"没有找到该设备的版本信息,请正确配置 OTA地址,然后重新编译固件。"
|
text = f"没有找到该设备的版本信息,请正确配置 OTA地址,然后重新编译固件。"
|
||||||
await send_stt_message(conn, text)
|
await send_stt_message(conn, text)
|
||||||
conn.tts_first_text_index = 0
|
|
||||||
conn.tts_last_text_index = 0
|
|
||||||
conn.llm_finish_task = True
|
|
||||||
music_path = "config/assets/bind_not_found.wav"
|
music_path = "config/assets/bind_not_found.wav"
|
||||||
opus_packets, _ = audio_to_data(music_path)
|
opus_packets, _ = audio_to_data(music_path)
|
||||||
conn.audio_play_queue.put((opus_packets, text, 0))
|
conn.tts.tts_audio_queue.put((SentenceType.LAST, opus_packets, text))
|
||||||
|
|||||||
@@ -9,6 +9,8 @@ TTS上报功能已集成到ConnectionHandler类中。
|
|||||||
具体实现请参考core/connection.py中的相关代码。
|
具体实现请参考core/connection.py中的相关代码。
|
||||||
"""
|
"""
|
||||||
|
|
||||||
|
import time
|
||||||
|
|
||||||
import opuslib_next
|
import opuslib_next
|
||||||
|
|
||||||
from config.manage_api_client import report as manage_report
|
from config.manage_api_client import report as manage_report
|
||||||
@@ -16,7 +18,7 @@ from config.manage_api_client import report as manage_report
|
|||||||
TAG = __name__
|
TAG = __name__
|
||||||
|
|
||||||
|
|
||||||
def report(conn, type, text, opus_data):
|
def report(conn, type, text, opus_data, report_time):
|
||||||
"""执行聊天记录上报操作
|
"""执行聊天记录上报操作
|
||||||
|
|
||||||
Args:
|
Args:
|
||||||
@@ -24,6 +26,7 @@ def report(conn, type, text, opus_data):
|
|||||||
type: 上报类型,1为用户,2为智能体
|
type: 上报类型,1为用户,2为智能体
|
||||||
text: 合成文本
|
text: 合成文本
|
||||||
opus_data: opus音频数据
|
opus_data: opus音频数据
|
||||||
|
report_time: 上报时间
|
||||||
"""
|
"""
|
||||||
try:
|
try:
|
||||||
if opus_data:
|
if opus_data:
|
||||||
@@ -37,6 +40,7 @@ def report(conn, type, text, opus_data):
|
|||||||
chat_type=type,
|
chat_type=type,
|
||||||
content=text,
|
content=text,
|
||||||
audio=audio_data,
|
audio=audio_data,
|
||||||
|
report_time=report_time,
|
||||||
)
|
)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
conn.logger.bind(tag=TAG).error(f"聊天记录上报失败: {e}")
|
conn.logger.bind(tag=TAG).error(f"聊天记录上报失败: {e}")
|
||||||
@@ -104,12 +108,12 @@ def enqueue_tts_report(conn, text, opus_data):
|
|||||||
try:
|
try:
|
||||||
# 使用连接对象的队列,传入文本和二进制数据而非文件路径
|
# 使用连接对象的队列,传入文本和二进制数据而非文件路径
|
||||||
if conn.chat_history_conf == 2:
|
if conn.chat_history_conf == 2:
|
||||||
conn.report_queue.put((2, text, opus_data))
|
conn.report_queue.put((2, text, opus_data, int(time.time())))
|
||||||
conn.logger.bind(tag=TAG).debug(
|
conn.logger.bind(tag=TAG).debug(
|
||||||
f"TTS数据已加入上报队列: {conn.device_id}, 音频大小: {len(opus_data)} "
|
f"TTS数据已加入上报队列: {conn.device_id}, 音频大小: {len(opus_data)} "
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
conn.report_queue.put((2, text, None))
|
conn.report_queue.put((2, text, None, int(time.time())))
|
||||||
conn.logger.bind(tag=TAG).debug(
|
conn.logger.bind(tag=TAG).debug(
|
||||||
f"TTS数据已加入上报队列: {conn.device_id}, 不上报音频"
|
f"TTS数据已加入上报队列: {conn.device_id}, 不上报音频"
|
||||||
)
|
)
|
||||||
@@ -132,14 +136,14 @@ def enqueue_asr_report(conn, text, opus_data):
|
|||||||
try:
|
try:
|
||||||
# 使用连接对象的队列,传入文本和二进制数据而非文件路径
|
# 使用连接对象的队列,传入文本和二进制数据而非文件路径
|
||||||
if conn.chat_history_conf == 2:
|
if conn.chat_history_conf == 2:
|
||||||
conn.report_queue.put((1, text, opus_data))
|
conn.report_queue.put((1, text, opus_data, int(time.time())))
|
||||||
conn.logger.bind(tag=TAG).debug(
|
conn.logger.bind(tag=TAG).debug(
|
||||||
f"ASR数据已加入上报队列: {conn.device_id}, 音频大小: {len(opus_data)} "
|
f"ASR数据已加入上报队列: {conn.device_id}, 音频大小: {len(opus_data)} "
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
conn.report_queue.put((1, text, None))
|
conn.report_queue.put((1, text, None, int(time.time())))
|
||||||
conn.logger.bind(tag=TAG).debug(
|
conn.logger.bind(tag=TAG).debug(
|
||||||
f"ASR数据已加入上报队列: {conn.device_id}, 不上报音频"
|
f"ASR数据已加入上报队列: {conn.device_id}, 不上报音频"
|
||||||
)
|
)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
conn.logger.bind(tag=TAG).error(f"加入ASR上报队列失败: {text}, {e}")
|
conn.logger.bind(tag=TAG).debug(f"加入ASR上报队列失败: {text}, {e}")
|
||||||
|
|||||||
@@ -1,7 +1,9 @@
|
|||||||
import json
|
import json
|
||||||
import asyncio
|
import asyncio
|
||||||
import time
|
import time
|
||||||
|
from core.providers.tts.dto.dto import SentenceType
|
||||||
from core.utils.util import get_string_no_punctuation_or_emoji, analyze_emotion
|
from core.utils.util import get_string_no_punctuation_or_emoji, analyze_emotion
|
||||||
|
from loguru import logger
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
|
|
||||||
@@ -30,7 +32,7 @@ emoji_map = {
|
|||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
async def sendAudioMessage(conn, audios, text, text_index=0):
|
async def sendAudioMessage(conn, sentenceType, audios, text):
|
||||||
# 发送句子开始消息
|
# 发送句子开始消息
|
||||||
if text is not None:
|
if text is not None:
|
||||||
emotion = analyze_emotion(text)
|
emotion = analyze_emotion(text)
|
||||||
@@ -45,25 +47,30 @@ async def sendAudioMessage(conn, audios, text, text_index=0):
|
|||||||
}
|
}
|
||||||
)
|
)
|
||||||
)
|
)
|
||||||
|
pre_buffer = False
|
||||||
if text_index == conn.tts_first_text_index:
|
if conn.tts.tts_audio_first_sentence and text is not None:
|
||||||
conn.logger.bind(tag=TAG).info(f"发送第一段语音: {text}")
|
conn.logger.bind(tag=TAG).info(f"发送第一段语音: {text}")
|
||||||
|
conn.tts.tts_audio_first_sentence = False
|
||||||
|
pre_buffer = True
|
||||||
|
|
||||||
await send_tts_message(conn, "sentence_start", text)
|
await send_tts_message(conn, "sentence_start", text)
|
||||||
|
|
||||||
is_first_audio = text_index == conn.tts_first_text_index
|
await sendAudio(conn, audios, pre_buffer)
|
||||||
await sendAudio(conn, audios, pre_buffer=is_first_audio)
|
|
||||||
|
|
||||||
await send_tts_message(conn, "sentence_end", text)
|
await send_tts_message(conn, "sentence_end", text)
|
||||||
|
|
||||||
# 发送结束消息(如果是最后一个文本)
|
# 发送结束消息(如果是最后一个文本)
|
||||||
if conn.llm_finish_task and text_index == conn.tts_last_text_index:
|
if conn.llm_finish_task and sentenceType == SentenceType.LAST:
|
||||||
await send_tts_message(conn, "stop", None)
|
await send_tts_message(conn, "stop", None)
|
||||||
|
conn.client_is_speaking = False
|
||||||
if conn.close_after_chat:
|
if conn.close_after_chat:
|
||||||
await conn.close()
|
await conn.close()
|
||||||
|
|
||||||
|
|
||||||
# 播放音频
|
# 播放音频
|
||||||
async def sendAudio(conn, audios, pre_buffer=True):
|
async def sendAudio(conn, audios, pre_buffer=True):
|
||||||
|
if audios is None or len(audios) == 0:
|
||||||
|
return
|
||||||
# 流控参数优化
|
# 流控参数优化
|
||||||
frame_duration = 60 # 帧时长(毫秒),匹配 Opus 编码
|
frame_duration = 60 # 帧时长(毫秒),匹配 Opus 编码
|
||||||
start_time = time.perf_counter()
|
start_time = time.perf_counter()
|
||||||
@@ -82,6 +89,7 @@ async def sendAudio(conn, audios, pre_buffer=True):
|
|||||||
# 播放剩余音频帧
|
# 播放剩余音频帧
|
||||||
for opus_packet in remaining_audios:
|
for opus_packet in remaining_audios:
|
||||||
if conn.client_abort:
|
if conn.client_abort:
|
||||||
|
conn.client_abort = False
|
||||||
return
|
return
|
||||||
|
|
||||||
# 每分钟重置一次计时器
|
# 每分钟重置一次计时器
|
||||||
@@ -125,9 +133,15 @@ async def send_tts_message(conn, state, text=None):
|
|||||||
|
|
||||||
|
|
||||||
async def send_stt_message(conn, text):
|
async def send_stt_message(conn, text):
|
||||||
|
end_prompt_str = conn.config.get("end_prompt", {}).get("prompt")
|
||||||
|
if end_prompt_str and end_prompt_str == text:
|
||||||
|
await send_tts_message(conn, "start")
|
||||||
|
return
|
||||||
|
|
||||||
"""发送 STT 状态消息"""
|
"""发送 STT 状态消息"""
|
||||||
stt_text = get_string_no_punctuation_or_emoji(text)
|
stt_text = get_string_no_punctuation_or_emoji(text)
|
||||||
await conn.websocket.send(
|
await conn.websocket.send(
|
||||||
json.dumps({"type": "stt", "text": stt_text, "session_id": conn.session_id})
|
json.dumps({"type": "stt", "text": stt_text, "session_id": conn.session_id})
|
||||||
)
|
)
|
||||||
|
conn.client_is_speaking = True
|
||||||
await send_tts_message(conn, "start")
|
await send_tts_message(conn, "start")
|
||||||
|
|||||||
@@ -1,6 +1,7 @@
|
|||||||
import json
|
import json
|
||||||
from core.handle.abortHandle import handleAbortMessage
|
from core.handle.abortHandle import handleAbortMessage
|
||||||
from core.handle.helloHandle import handleHelloMessage
|
from core.handle.helloHandle import handleHelloMessage
|
||||||
|
from core.handle.mcpHandle import handle_mcp_message
|
||||||
from core.utils.util import remove_punctuation_and_length, filter_sensitive_info
|
from core.utils.util import remove_punctuation_and_length, filter_sensitive_info
|
||||||
from core.handle.receiveAudioHandle import startToChat, handleAudioMessage
|
from core.handle.receiveAudioHandle import startToChat, handleAudioMessage
|
||||||
from core.handle.sendAudioHandle import send_stt_message, send_tts_message
|
from core.handle.sendAudioHandle import send_stt_message, send_tts_message
|
||||||
@@ -41,12 +42,13 @@ async def handleTextMessage(conn, message):
|
|||||||
if len(conn.asr_audio) > 0:
|
if len(conn.asr_audio) > 0:
|
||||||
await handleAudioMessage(conn, b"")
|
await handleAudioMessage(conn, b"")
|
||||||
elif msg_json["state"] == "detect":
|
elif msg_json["state"] == "detect":
|
||||||
conn.asr_server_receive = False
|
|
||||||
conn.client_have_voice = False
|
conn.client_have_voice = False
|
||||||
conn.asr_audio.clear()
|
conn.asr_audio.clear()
|
||||||
if "text" in msg_json:
|
if "text" in msg_json:
|
||||||
original_text = msg_json["text"] # 保留原始文本
|
original_text = msg_json["text"] # 保留原始文本
|
||||||
filtered_len, filtered_text = remove_punctuation_and_length(original_text)
|
filtered_len, filtered_text = remove_punctuation_and_length(
|
||||||
|
original_text
|
||||||
|
)
|
||||||
|
|
||||||
# 识别是否是唤醒词
|
# 识别是否是唤醒词
|
||||||
is_wakeup_words = filtered_text in conn.config.get("wakeup_words")
|
is_wakeup_words = filtered_text in conn.config.get("wakeup_words")
|
||||||
@@ -57,6 +59,7 @@ async def handleTextMessage(conn, message):
|
|||||||
# 如果是唤醒词,且关闭了唤醒词回复,就不用回答
|
# 如果是唤醒词,且关闭了唤醒词回复,就不用回答
|
||||||
await send_stt_message(conn, original_text)
|
await send_stt_message(conn, original_text)
|
||||||
await send_tts_message(conn, "stop", None)
|
await send_tts_message(conn, "stop", None)
|
||||||
|
conn.client_is_speaking = False
|
||||||
elif is_wakeup_words:
|
elif is_wakeup_words:
|
||||||
# 上报纯文字数据(复用ASR上报功能,但不提供音频数据)
|
# 上报纯文字数据(复用ASR上报功能,但不提供音频数据)
|
||||||
enqueue_asr_report(conn, "嘿,你好呀", [])
|
enqueue_asr_report(conn, "嘿,你好呀", [])
|
||||||
@@ -72,6 +75,10 @@ async def handleTextMessage(conn, message):
|
|||||||
asyncio.create_task(handleIotDescriptors(conn, msg_json["descriptors"]))
|
asyncio.create_task(handleIotDescriptors(conn, msg_json["descriptors"]))
|
||||||
if "states" in msg_json:
|
if "states" in msg_json:
|
||||||
asyncio.create_task(handleIotStatus(conn, msg_json["states"]))
|
asyncio.create_task(handleIotStatus(conn, msg_json["states"]))
|
||||||
|
elif msg_json["type"] == "mcp":
|
||||||
|
conn.logger.bind(tag=TAG).info(f"收到mcp消息:{message}")
|
||||||
|
if "payload" in msg_json:
|
||||||
|
asyncio.create_task(handle_mcp_message(conn, conn.mcp_client, msg_json["payload"]))
|
||||||
elif msg_json["type"] == "server":
|
elif msg_json["type"] == "server":
|
||||||
# 记录日志时过滤敏感信息
|
# 记录日志时过滤敏感信息
|
||||||
conn.logger.bind(tag=TAG).info(
|
conn.logger.bind(tag=TAG).info(
|
||||||
@@ -151,5 +158,7 @@ async def handleTextMessage(conn, message):
|
|||||||
# 重启服务器
|
# 重启服务器
|
||||||
elif msg_json["action"] == "restart":
|
elif msg_json["action"] == "restart":
|
||||||
await conn.handle_restart(msg_json)
|
await conn.handle_restart(msg_json)
|
||||||
|
else:
|
||||||
|
conn.logger.bind(tag=TAG).error(f"收到未知类型消息:{message}")
|
||||||
except json.JSONDecodeError:
|
except json.JSONDecodeError:
|
||||||
await conn.websocket.send(message)
|
await conn.websocket.send(message)
|
||||||
|
|||||||
@@ -0,0 +1,71 @@
|
|||||||
|
import asyncio
|
||||||
|
from aiohttp import web
|
||||||
|
from config.logger import setup_logging
|
||||||
|
from core.api.ota_handler import OTAHandler
|
||||||
|
from core.api.vision_handler import VisionHandler
|
||||||
|
|
||||||
|
TAG = __name__
|
||||||
|
|
||||||
|
|
||||||
|
class SimpleHttpServer:
|
||||||
|
def __init__(self, config: dict):
|
||||||
|
self.config = config
|
||||||
|
self.logger = setup_logging()
|
||||||
|
self.ota_handler = OTAHandler(config)
|
||||||
|
self.vision_handler = VisionHandler(config)
|
||||||
|
|
||||||
|
def _get_websocket_url(self, local_ip: str, port: int) -> str:
|
||||||
|
"""获取websocket地址
|
||||||
|
|
||||||
|
Args:
|
||||||
|
local_ip: 本地IP地址
|
||||||
|
port: 端口号
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: websocket地址
|
||||||
|
"""
|
||||||
|
server_config = self.config["server"]
|
||||||
|
websocket_config = server_config.get("websocket")
|
||||||
|
|
||||||
|
if websocket_config and "你" not in websocket_config:
|
||||||
|
return websocket_config
|
||||||
|
else:
|
||||||
|
return f"ws://{local_ip}:{port}/xiaozhi/v1/"
|
||||||
|
|
||||||
|
async def start(self):
|
||||||
|
server_config = self.config["server"]
|
||||||
|
host = server_config.get("ip", "0.0.0.0")
|
||||||
|
port = int(server_config.get("http_port", 8003))
|
||||||
|
|
||||||
|
if port:
|
||||||
|
app = web.Application()
|
||||||
|
|
||||||
|
read_config_from_api = server_config.get("read_config_from_api", False)
|
||||||
|
|
||||||
|
if not read_config_from_api:
|
||||||
|
# 如果没有开启智控台,只是单模块运行,就需要再添加简单OTA接口,用于下发websocket接口
|
||||||
|
app.add_routes(
|
||||||
|
[
|
||||||
|
web.get("/xiaozhi/ota/", self.ota_handler.handle_get),
|
||||||
|
web.post("/xiaozhi/ota/", self.ota_handler.handle_post),
|
||||||
|
web.options("/xiaozhi/ota/", self.ota_handler.handle_post),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
# 添加路由
|
||||||
|
app.add_routes(
|
||||||
|
[
|
||||||
|
web.get("/mcp/vision/explain", self.vision_handler.handle_get),
|
||||||
|
web.post("/mcp/vision/explain", self.vision_handler.handle_post),
|
||||||
|
web.options("/mcp/vision/explain", self.vision_handler.handle_post),
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
# 运行服务
|
||||||
|
runner = web.AppRunner(app)
|
||||||
|
await runner.setup()
|
||||||
|
site = web.TCPSite(runner, host, port)
|
||||||
|
await site.start()
|
||||||
|
|
||||||
|
# 保持服务运行
|
||||||
|
while True:
|
||||||
|
await asyncio.sleep(3600) # 每隔 1 小时检查一次
|
||||||
@@ -92,11 +92,21 @@ class MCPClient:
|
|||||||
args=self.config.get("args", []),
|
args=self.config.get("args", []),
|
||||||
env=env,
|
env=env,
|
||||||
)
|
)
|
||||||
stdio_r, stdio_w = await stack.enter_async_context(stdio_client(params))
|
stdio_r, stdio_w = await stack.enter_async_context(
|
||||||
|
stdio_client(params)
|
||||||
|
)
|
||||||
read_stream, write_stream = stdio_r, stdio_w
|
read_stream, write_stream = stdio_r, stdio_w
|
||||||
# 建立SSEClient
|
# 建立SSEClient
|
||||||
elif "url" in self.config:
|
elif "url" in self.config:
|
||||||
sse_r, sse_w = await stack.enter_async_context(sse_client(self.config["url"]))
|
if "API_ACCESS_TOKEN" in self.config:
|
||||||
|
headers = {
|
||||||
|
"Authorization": f"Bearer {self.config['API_ACCESS_TOKEN']}"
|
||||||
|
}
|
||||||
|
else:
|
||||||
|
headers = {}
|
||||||
|
sse_r, sse_w = await stack.enter_async_context(
|
||||||
|
sse_client(self.config["url"], headers=headers)
|
||||||
|
)
|
||||||
read_stream, write_stream = sse_r, sse_w
|
read_stream, write_stream = sse_r, sse_w
|
||||||
|
|
||||||
else:
|
else:
|
||||||
|
|||||||
@@ -2,9 +2,6 @@ import http.client
|
|||||||
import json
|
import json
|
||||||
import asyncio
|
import asyncio
|
||||||
from typing import Optional, Tuple, List
|
from typing import Optional, Tuple, List
|
||||||
import opuslib_next
|
|
||||||
import wave
|
|
||||||
import io
|
|
||||||
import os
|
import os
|
||||||
import uuid
|
import uuid
|
||||||
import hmac
|
import hmac
|
||||||
@@ -16,6 +13,7 @@ import time
|
|||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
from core.providers.asr.base import ASRProviderBase
|
from core.providers.asr.base import ASRProviderBase
|
||||||
|
from core.providers.asr.dto.dto import InterfaceType
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
logger = setup_logging()
|
logger = setup_logging()
|
||||||
@@ -92,6 +90,7 @@ class AccessToken:
|
|||||||
class ASRProvider(ASRProviderBase):
|
class ASRProvider(ASRProviderBase):
|
||||||
def __init__(self, config: dict, delete_audio_file: bool):
|
def __init__(self, config: dict, delete_audio_file: bool):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
self.interface_type = InterfaceType.NON_STREAM
|
||||||
"""阿里云ASR初始化"""
|
"""阿里云ASR初始化"""
|
||||||
# 新增空值判断逻辑
|
# 新增空值判断逻辑
|
||||||
self.access_key_id = config.get("access_key_id")
|
self.access_key_id = config.get("access_key_id")
|
||||||
@@ -155,12 +154,6 @@ class ASRProvider(ASRProviderBase):
|
|||||||
# f"剩余 {remaining:.2f}秒")
|
# f"剩余 {remaining:.2f}秒")
|
||||||
return time.time() > self.expire_time
|
return time.time() > self.expire_time
|
||||||
|
|
||||||
def generate_filename(self, extension=".wav"):
|
|
||||||
return os.path.join(
|
|
||||||
self.output_file,
|
|
||||||
f"tts-{__name__}{datetime.now().date()}@{uuid.uuid4().hex}{extension}",
|
|
||||||
)
|
|
||||||
|
|
||||||
def _construct_request_url(self) -> str:
|
def _construct_request_url(self) -> str:
|
||||||
"""构造请求URL,包含参数"""
|
"""构造请求URL,包含参数"""
|
||||||
request = f"{self.base_url}?appkey={self.app_key}"
|
request = f"{self.base_url}?appkey={self.app_key}"
|
||||||
@@ -171,21 +164,6 @@ class ASRProvider(ASRProviderBase):
|
|||||||
request += "&enable_voice_detection=false"
|
request += "&enable_voice_detection=false"
|
||||||
return request
|
return request
|
||||||
|
|
||||||
def save_audio_to_file(self, pcm_data: List[bytes], session_id: str) -> str:
|
|
||||||
"""PCM数据保存为WAV文件"""
|
|
||||||
module_name = __name__.split(".")[-1]
|
|
||||||
file_name = f"asr_{module_name}_{session_id}_{uuid.uuid4()}.wav"
|
|
||||||
file_path = os.path.join(self.output_dir, file_name)
|
|
||||||
|
|
||||||
with wave.open(file_path, "wb") as wf:
|
|
||||||
wf.setnchannels(1) # 单声道
|
|
||||||
wf.setsampwidth(2) # 16-bit
|
|
||||||
wf.setframerate(self.sample_rate)
|
|
||||||
wf.writeframes(b"".join(pcm_data))
|
|
||||||
|
|
||||||
logger.bind(tag=TAG).debug(f"音频文件已保存至: {file_path}")
|
|
||||||
return file_path
|
|
||||||
|
|
||||||
async def _send_request(self, pcm_data: bytes) -> Optional[str]:
|
async def _send_request(self, pcm_data: bytes) -> Optional[str]:
|
||||||
"""发送请求到阿里云ASR服务"""
|
"""发送请求到阿里云ASR服务"""
|
||||||
try:
|
try:
|
||||||
|
|||||||
@@ -1,18 +1,10 @@
|
|||||||
import base64
|
|
||||||
import hashlib
|
|
||||||
import hmac
|
|
||||||
import json
|
|
||||||
import time
|
import time
|
||||||
from datetime import datetime, timezone
|
|
||||||
import os
|
import os
|
||||||
import uuid
|
|
||||||
from typing import Optional, Tuple, List
|
from typing import Optional, Tuple, List
|
||||||
import wave
|
|
||||||
import opuslib_next
|
|
||||||
|
|
||||||
from aip import AipSpeech
|
from aip import AipSpeech
|
||||||
from core.providers.asr.base import ASRProviderBase
|
from core.providers.asr.base import ASRProviderBase
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
|
from core.providers.asr.dto.dto import InterfaceType
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
logger = setup_logging()
|
logger = setup_logging()
|
||||||
@@ -21,6 +13,7 @@ logger = setup_logging()
|
|||||||
class ASRProvider(ASRProviderBase):
|
class ASRProvider(ASRProviderBase):
|
||||||
def __init__(self, config: dict, delete_audio_file: bool = True):
|
def __init__(self, config: dict, delete_audio_file: bool = True):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
self.interface_type = InterfaceType.NON_STREAM
|
||||||
self.app_id = config.get("app_id")
|
self.app_id = config.get("app_id")
|
||||||
self.api_key = config.get("api_key")
|
self.api_key = config.get("api_key")
|
||||||
self.secret_key = config.get("secret_key")
|
self.secret_key = config.get("secret_key")
|
||||||
@@ -36,20 +29,6 @@ class ASRProvider(ASRProviderBase):
|
|||||||
# 确保输出目录存在
|
# 确保输出目录存在
|
||||||
os.makedirs(self.output_dir, exist_ok=True)
|
os.makedirs(self.output_dir, exist_ok=True)
|
||||||
|
|
||||||
def save_audio_to_file(self, pcm_data: List[bytes], session_id: str) -> str:
|
|
||||||
"""PCM数据保存为WAV文件"""
|
|
||||||
module_name = __name__.split(".")[-1]
|
|
||||||
file_name = f"asr_{module_name}_{session_id}_{uuid.uuid4()}.wav"
|
|
||||||
file_path = os.path.join(self.output_dir, file_name)
|
|
||||||
|
|
||||||
with wave.open(file_path, "wb") as wf:
|
|
||||||
wf.setnchannels(1)
|
|
||||||
wf.setsampwidth(2) # 2 bytes = 16-bit
|
|
||||||
wf.setframerate(16000)
|
|
||||||
wf.writeframes(b"".join(pcm_data))
|
|
||||||
|
|
||||||
return file_path
|
|
||||||
|
|
||||||
async def speech_to_text(
|
async def speech_to_text(
|
||||||
self, opus_data: List[bytes], session_id: str
|
self, opus_data: List[bytes], session_id: str
|
||||||
) -> Tuple[Optional[str], Optional[str]]:
|
) -> Tuple[Optional[str], Optional[str]]:
|
||||||
|
|||||||
@@ -1,7 +1,15 @@
|
|||||||
from abc import ABC, abstractmethod
|
import os
|
||||||
from typing import Optional, Tuple, List
|
import time
|
||||||
|
import copy
|
||||||
|
import uuid
|
||||||
|
import wave
|
||||||
import opuslib_next
|
import opuslib_next
|
||||||
|
from abc import ABC, abstractmethod
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
|
from typing import Optional, Tuple, List
|
||||||
|
from core.utils.util import remove_punctuation_and_length
|
||||||
|
from core.handle.reportHandle import enqueue_asr_report
|
||||||
|
from core.handle.receiveAudioHandle import startToChat
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
logger = setup_logging()
|
logger = setup_logging()
|
||||||
@@ -10,11 +18,66 @@ logger = setup_logging()
|
|||||||
class ASRProviderBase(ABC):
|
class ASRProviderBase(ABC):
|
||||||
def __init__(self):
|
def __init__(self):
|
||||||
self.audio_format = "opus"
|
self.audio_format = "opus"
|
||||||
|
self.conn = None
|
||||||
|
|
||||||
|
# 打开音频通道
|
||||||
|
# 这里默认是非流式的处理方式
|
||||||
|
# 流式处理方式请在子类中重写
|
||||||
|
async def open_audio_channels(self, conn):
|
||||||
|
self.conn = conn
|
||||||
|
|
||||||
|
# 接收音频
|
||||||
|
# 这里默认是非流式的处理方式
|
||||||
|
# 流式处理方式请在子类中重写
|
||||||
|
async def receive_audio(self, audio, audio_have_voice):
|
||||||
|
if (
|
||||||
|
self.conn.client_listen_mode == "auto"
|
||||||
|
or self.conn.client_listen_mode == "realtime"
|
||||||
|
):
|
||||||
|
have_voice = audio_have_voice
|
||||||
|
else:
|
||||||
|
have_voice = self.conn.client_have_voice
|
||||||
|
# 如果本次没有声音,本段也没声音,就把声音丢弃了
|
||||||
|
self.conn.asr_audio.append(audio)
|
||||||
|
if have_voice == False and self.conn.client_have_voice == False:
|
||||||
|
self.conn.asr_audio = self.conn.asr_audio[-10:]
|
||||||
|
return
|
||||||
|
|
||||||
|
# 如果本段有声音,且已经停止了
|
||||||
|
if self.conn.client_voice_stop:
|
||||||
|
asr_audio_task = copy.deepcopy(self.conn.asr_audio)
|
||||||
|
self.conn.asr_audio.clear()
|
||||||
|
|
||||||
|
# 音频太短了,无法识别
|
||||||
|
self.conn.reset_vad_states()
|
||||||
|
if len(asr_audio_task) > 15:
|
||||||
|
await self.handle_voice_stop(asr_audio_task)
|
||||||
|
|
||||||
|
# 处理语音停止
|
||||||
|
async def handle_voice_stop(self, asr_audio_task):
|
||||||
|
raw_text, _ = await self.speech_to_text(
|
||||||
|
asr_audio_task, self.conn.session_id
|
||||||
|
) # 确保ASR模块返回原始文本
|
||||||
|
self.conn.logger.bind(tag=TAG).info(f"识别文本: {raw_text}")
|
||||||
|
text_len, _ = remove_punctuation_and_length(raw_text)
|
||||||
|
if text_len > 0:
|
||||||
|
# 使用自定义模块进行上报
|
||||||
|
await startToChat(self.conn, raw_text)
|
||||||
|
enqueue_asr_report(self.conn, raw_text, asr_audio_task)
|
||||||
|
|
||||||
@abstractmethod
|
|
||||||
def save_audio_to_file(self, pcm_data: List[bytes], session_id: str) -> str:
|
def save_audio_to_file(self, pcm_data: List[bytes], session_id: str) -> str:
|
||||||
"""PCM数据保存为WAV文件"""
|
"""PCM数据保存为WAV文件"""
|
||||||
pass
|
module_name = __name__.split(".")[-1]
|
||||||
|
file_name = f"asr_{module_name}_{session_id}_{uuid.uuid4()}.wav"
|
||||||
|
file_path = os.path.join(self.output_dir, file_name)
|
||||||
|
|
||||||
|
with wave.open(file_path, "wb") as wf:
|
||||||
|
wf.setnchannels(1)
|
||||||
|
wf.setsampwidth(2) # 2 bytes = 16-bit
|
||||||
|
wf.setframerate(16000)
|
||||||
|
wf.writeframes(b"".join(pcm_data))
|
||||||
|
|
||||||
|
return file_path
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
async def speech_to_text(
|
async def speech_to_text(
|
||||||
@@ -30,15 +93,25 @@ class ASRProviderBase(ABC):
|
|||||||
@staticmethod
|
@staticmethod
|
||||||
def decode_opus(opus_data: List[bytes]) -> bytes:
|
def decode_opus(opus_data: List[bytes]) -> bytes:
|
||||||
"""将Opus音频数据解码为PCM数据"""
|
"""将Opus音频数据解码为PCM数据"""
|
||||||
|
try:
|
||||||
|
decoder = opuslib_next.Decoder(16000, 1) # 16kHz, 单声道
|
||||||
|
pcm_data = []
|
||||||
|
buffer_size = 960 # 每次处理960个采样点
|
||||||
|
|
||||||
decoder = opuslib_next.Decoder(16000, 1) # 16kHz, 单声道
|
for opus_packet in opus_data:
|
||||||
pcm_data = []
|
try:
|
||||||
|
# 使用较小的缓冲区大小进行处理
|
||||||
|
pcm_frame = decoder.decode(opus_packet, buffer_size)
|
||||||
|
if pcm_frame:
|
||||||
|
pcm_data.append(pcm_frame)
|
||||||
|
except opuslib_next.OpusError as e:
|
||||||
|
logger.bind(tag=TAG).warning(f"Opus解码错误,跳过当前数据包: {e}")
|
||||||
|
continue
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"音频处理错误: {e}", exc_info=True)
|
||||||
|
continue
|
||||||
|
|
||||||
for opus_packet in opus_data:
|
return pcm_data
|
||||||
try:
|
except Exception as e:
|
||||||
pcm_frame = decoder.decode(opus_packet, 960) # 960 samples = 60ms
|
logger.bind(tag=TAG).error(f"音频解码过程发生错误: {e}", exc_info=True)
|
||||||
pcm_data.append(pcm_frame)
|
return []
|
||||||
except opuslib_next.OpusError as e:
|
|
||||||
logger.bind(tag=TAG).error(f"Opus解码错误: {e}", exc_info=True)
|
|
||||||
|
|
||||||
return pcm_data
|
|
||||||
|
|||||||
@@ -1,283 +1,534 @@
|
|||||||
import time
|
|
||||||
import io
|
|
||||||
import wave
|
|
||||||
import os
|
|
||||||
from typing import Optional, Tuple, List
|
|
||||||
import uuid
|
|
||||||
import websockets
|
|
||||||
import json
|
import json
|
||||||
import gzip
|
import gzip
|
||||||
|
import uuid
|
||||||
|
import asyncio
|
||||||
|
import websockets
|
||||||
import opuslib_next
|
import opuslib_next
|
||||||
from core.providers.asr.base import ASRProviderBase
|
from core.providers.asr.base import ASRProviderBase
|
||||||
|
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
|
from core.providers.asr.dto.dto import InterfaceType
|
||||||
|
import threading
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
logger = setup_logging()
|
logger = setup_logging()
|
||||||
|
|
||||||
CLIENT_FULL_REQUEST = 0b0001
|
CLIENT_FULL_REQUEST = 0b0001
|
||||||
CLIENT_AUDIO_ONLY_REQUEST = 0b0010
|
CLIENT_AUDIO_ONLY_REQUEST = 0b0010
|
||||||
|
|
||||||
NO_SEQUENCE = 0b0000
|
|
||||||
NEG_SEQUENCE = 0b0010
|
|
||||||
|
|
||||||
SERVER_FULL_RESPONSE = 0b1001
|
SERVER_FULL_RESPONSE = 0b1001
|
||||||
SERVER_ACK = 0b1011
|
SERVER_ACK = 0b1011
|
||||||
SERVER_ERROR_RESPONSE = 0b1111
|
SERVER_ERROR_RESPONSE = 0b1111
|
||||||
|
NO_SEQUENCE = 0b0000
|
||||||
NO_SERIALIZATION = 0b0000
|
NEG_SEQUENCE = 0b0010
|
||||||
JSON = 0b0001
|
JSON_SERIALIZATION = 0b0001
|
||||||
THRIFT = 0b0011
|
GZIP_COMPRESSION = 0b0001
|
||||||
CUSTOM_TYPE = 0b1111
|
PROTOCOL_VERSION = 0b0001
|
||||||
NO_COMPRESSION = 0b0000
|
|
||||||
GZIP = 0b0001
|
|
||||||
CUSTOM_COMPRESSION = 0b1111
|
|
||||||
|
|
||||||
|
|
||||||
def parse_response(res):
|
|
||||||
"""
|
|
||||||
protocol_version(4 bits), header_size(4 bits),
|
|
||||||
message_type(4 bits), message_type_specific_flags(4 bits)
|
|
||||||
serialization_method(4 bits) message_compression(4 bits)
|
|
||||||
reserved (8bits) 保留字段
|
|
||||||
header_extensions 扩展头(大小等于 8 * 4 * (header_size - 1) )
|
|
||||||
payload 类似与http 请求体
|
|
||||||
"""
|
|
||||||
protocol_version = res[0] >> 4
|
|
||||||
header_size = res[0] & 0x0F
|
|
||||||
message_type = res[1] >> 4
|
|
||||||
message_type_specific_flags = res[1] & 0x0F
|
|
||||||
serialization_method = res[2] >> 4
|
|
||||||
message_compression = res[2] & 0x0F
|
|
||||||
reserved = res[3]
|
|
||||||
header_extensions = res[4 : header_size * 4]
|
|
||||||
payload = res[header_size * 4 :]
|
|
||||||
result = {}
|
|
||||||
payload_msg = None
|
|
||||||
payload_size = 0
|
|
||||||
if message_type == SERVER_FULL_RESPONSE:
|
|
||||||
payload_size = int.from_bytes(payload[:4], "big", signed=True)
|
|
||||||
payload_msg = payload[4:]
|
|
||||||
elif message_type == SERVER_ACK:
|
|
||||||
seq = int.from_bytes(payload[:4], "big", signed=True)
|
|
||||||
result["seq"] = seq
|
|
||||||
if len(payload) >= 8:
|
|
||||||
payload_size = int.from_bytes(payload[4:8], "big", signed=False)
|
|
||||||
payload_msg = payload[8:]
|
|
||||||
elif message_type == SERVER_ERROR_RESPONSE:
|
|
||||||
code = int.from_bytes(payload[:4], "big", signed=False)
|
|
||||||
result["code"] = code
|
|
||||||
payload_size = int.from_bytes(payload[4:8], "big", signed=False)
|
|
||||||
payload_msg = payload[8:]
|
|
||||||
if payload_msg is None:
|
|
||||||
return result
|
|
||||||
if message_compression == GZIP:
|
|
||||||
payload_msg = gzip.decompress(payload_msg)
|
|
||||||
if serialization_method == JSON:
|
|
||||||
payload_msg = json.loads(str(payload_msg, "utf-8"))
|
|
||||||
elif serialization_method != NO_SERIALIZATION:
|
|
||||||
payload_msg = str(payload_msg, "utf-8")
|
|
||||||
result["payload_msg"] = payload_msg
|
|
||||||
result["payload_size"] = payload_size
|
|
||||||
return result
|
|
||||||
|
|
||||||
|
|
||||||
class ASRProvider(ASRProviderBase):
|
class ASRProvider(ASRProviderBase):
|
||||||
def __init__(self, config: dict, delete_audio_file: bool):
|
def __init__(self, config, delete_audio_file):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
self.appid = config.get("appid")
|
self.interface_type = InterfaceType.STREAM
|
||||||
|
self.config = config
|
||||||
|
self.text = ""
|
||||||
|
self.max_retries = 3
|
||||||
|
self.retry_delay = 2 # 重试延迟秒数
|
||||||
|
self.recv_lock = asyncio.Lock() # 添加接收锁
|
||||||
|
self.reconnect_lock = asyncio.Lock() # 添加重连锁
|
||||||
|
self.last_reconnect_time = 0 # 上次重连时间
|
||||||
|
self.reconnect_cooldown = 1 # 增加重连冷却时间到10秒
|
||||||
|
self.reconnect_count = 0 # 当前重连次数
|
||||||
|
self.max_reconnect_count = 3 # 减少最大重连次数到3次
|
||||||
|
self.asr_thread = None # ASR监听线程
|
||||||
|
self.thread_lock = threading.Lock() # 线程管理锁
|
||||||
|
self.is_reconnecting = False # 添加重连状态标志
|
||||||
|
|
||||||
|
# 添加会话管理相关属性
|
||||||
|
self._session_lock = asyncio.Lock() # 会话操作的并发锁
|
||||||
|
self._current_session_id = None # 当前会话ID
|
||||||
|
self._session_started = False # 会话是否已开始
|
||||||
|
self._session_finished = False # 会话是否已结束
|
||||||
|
self._session_close_event = asyncio.Event() # 添加会话关闭事件
|
||||||
|
|
||||||
|
self.appid = str(config.get("appid"))
|
||||||
self.cluster = config.get("cluster")
|
self.cluster = config.get("cluster")
|
||||||
self.access_token = config.get("access_token")
|
self.access_token = config.get("access_token")
|
||||||
self.boosting_table_name = config.get("boosting_table_name", "")
|
self.boosting_table_name = config.get("boosting_table_name", "")
|
||||||
self.correct_table_name = config.get("correct_table_name", "")
|
self.correct_table_name = config.get("correct_table_name", "")
|
||||||
self.output_dir = config.get("output_dir")
|
self.output_dir = config.get("output_dir", "temp/")
|
||||||
self.delete_audio_file = delete_audio_file
|
self.delete_audio_file = delete_audio_file
|
||||||
|
|
||||||
self.host = "openspeech.bytedance.com"
|
self.ws_url = "wss://openspeech.bytedance.com/api/v2/asr"
|
||||||
self.ws_url = f"wss://{self.host}/api/v2/asr"
|
self.uid = config.get("uid", "streaming_asr_service")
|
||||||
self.success_code = 1000
|
self.workflow = config.get(
|
||||||
self.seg_duration = 15000
|
"workflow", "audio_in,resample,partition,vad,fe,decode,itn,nlu_punctuate"
|
||||||
|
)
|
||||||
|
self.result_type = config.get("result_type", "single")
|
||||||
|
self.format = config.get("format", "raw")
|
||||||
|
self.codec = config.get("codec", "pcm")
|
||||||
|
self.rate = config.get("sample_rate", 16000)
|
||||||
|
self.language = config.get("language", "zh-CN")
|
||||||
|
self.bits = config.get("bits", 16)
|
||||||
|
self.channel = config.get("channel", 1)
|
||||||
|
self.auth_method = config.get("auth_method", "token")
|
||||||
|
self.secret = config.get("secret", "access_secret")
|
||||||
|
self.decoder = opuslib_next.Decoder(16000, 1)
|
||||||
|
self.asr_ws = None
|
||||||
|
self.forward_task = None
|
||||||
|
self.conn = None
|
||||||
|
|
||||||
# 确保输出目录存在
|
###################################################################################
|
||||||
os.makedirs(self.output_dir, exist_ok=True)
|
# 豆包流式ASR重写父类的方法--开始
|
||||||
|
###################################################################################
|
||||||
|
async def open_audio_channels(self, conn):
|
||||||
|
await super().open_audio_channels(conn)
|
||||||
|
|
||||||
def save_audio_to_file(self, pcm_data: List[bytes], session_id: str) -> str:
|
async with self._session_lock:
|
||||||
"""PCM数据保存为WAV文件"""
|
# 如果正在重连,等待重连完成
|
||||||
module_name = __name__.split(".")[-1]
|
if self.is_reconnecting:
|
||||||
file_name = f"asr_{module_name}_{session_id}_{uuid.uuid4()}.wav"
|
logger.bind(tag=TAG).info("等待当前重连完成...")
|
||||||
file_path = os.path.join(self.output_dir, file_name)
|
await self._session_close_event.wait()
|
||||||
|
self._session_close_event.clear()
|
||||||
|
|
||||||
with wave.open(file_path, "wb") as wf:
|
# 如果已有会话未结束,先关闭它
|
||||||
wf.setnchannels(1)
|
if self._session_started and not self._session_finished:
|
||||||
wf.setsampwidth(2) # 2 bytes = 16-bit
|
logger.bind(tag=TAG).warning(
|
||||||
wf.setframerate(16000)
|
f"发现未关闭的会话 {self._current_session_id},正在关闭..."
|
||||||
wf.writeframes(b"".join(pcm_data))
|
)
|
||||||
|
if self.asr_ws is not None:
|
||||||
|
try:
|
||||||
|
await self.asr_ws.close()
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).warning(f"关闭旧连接时发生错误: {e}")
|
||||||
|
finally:
|
||||||
|
self.asr_ws = None
|
||||||
|
self._session_finished = True
|
||||||
|
self._session_close_event.set()
|
||||||
|
|
||||||
return file_path
|
# 重置会话状态
|
||||||
|
self._current_session_id = str(uuid.uuid4())
|
||||||
|
self._session_started = True
|
||||||
|
self._session_finished = False
|
||||||
|
self.is_reconnecting = True
|
||||||
|
|
||||||
@staticmethod
|
try:
|
||||||
def _generate_header(
|
retry_count = 0
|
||||||
message_type=CLIENT_FULL_REQUEST, message_type_specific_flags=NO_SEQUENCE
|
while retry_count < self.max_retries:
|
||||||
) -> bytearray:
|
try:
|
||||||
"""Generate protocol header."""
|
headers = (
|
||||||
header = bytearray()
|
self.token_auth() if self.auth_method == "token" else None
|
||||||
header_size = 1
|
)
|
||||||
header.append((0b0001 << 4) | header_size) # Protocol version
|
self.asr_ws = await websockets.connect(
|
||||||
header.append((message_type << 4) | message_type_specific_flags)
|
self.ws_url,
|
||||||
header.append((0b0001 << 4) | 0b0001) # JSON serialization & GZIP compression
|
additional_headers=headers,
|
||||||
header.append(0x00) # reserved
|
max_size=1000000000,
|
||||||
return header
|
ping_interval=None,
|
||||||
|
ping_timeout=None,
|
||||||
|
close_timeout=10,
|
||||||
|
)
|
||||||
|
|
||||||
def _construct_request(self, reqid) -> dict:
|
# 发送初始化请求
|
||||||
"""Construct the request payload."""
|
request_params = self.construct_request(
|
||||||
return {
|
self._current_session_id
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
payload_bytes = str.encode(json.dumps(request_params))
|
||||||
|
payload_bytes = gzip.compress(payload_bytes)
|
||||||
|
full_client_request = self.generate_header()
|
||||||
|
full_client_request.extend(
|
||||||
|
(len(payload_bytes)).to_bytes(4, "big")
|
||||||
|
)
|
||||||
|
full_client_request.extend(payload_bytes)
|
||||||
|
await self.asr_ws.send(full_client_request)
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"发送初始化请求失败: {e}")
|
||||||
|
raise e
|
||||||
|
|
||||||
|
# 等待初始化响应
|
||||||
|
try:
|
||||||
|
init_res = await self.asr_ws.recv()
|
||||||
|
self.parse_response(init_res)
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"ASR服务初始化失败: {e}")
|
||||||
|
raise e
|
||||||
|
|
||||||
|
# 启动接收ASR结果的异步任务
|
||||||
|
with self.thread_lock:
|
||||||
|
if (
|
||||||
|
self.asr_thread is None
|
||||||
|
or not self.asr_thread.is_alive()
|
||||||
|
):
|
||||||
|
logger.bind(tag=TAG).info("创建新的ASR监听线程...")
|
||||||
|
self.asr_thread = threading.Thread(
|
||||||
|
target=self._start_monitor_asr_response_thread,
|
||||||
|
daemon=True,
|
||||||
|
)
|
||||||
|
self.asr_thread.start()
|
||||||
|
# 等待一小段时间确保线程启动
|
||||||
|
await asyncio.sleep(0.1)
|
||||||
|
if not self.asr_thread.is_alive():
|
||||||
|
logger.bind(tag=TAG).error("ASR监听线程启动失败")
|
||||||
|
raise Exception("ASR监听线程启动失败")
|
||||||
|
logger.bind(tag=TAG).info("ASR监听线程已启动")
|
||||||
|
return
|
||||||
|
|
||||||
|
except websockets.exceptions.WebSocketException as e:
|
||||||
|
retry_count += 1
|
||||||
|
if retry_count < self.max_retries:
|
||||||
|
logger.bind(tag=TAG).warning(
|
||||||
|
f"WebSocket连接失败,正在进行第{retry_count}次重试: {e}"
|
||||||
|
)
|
||||||
|
await asyncio.sleep(self.retry_delay)
|
||||||
|
else:
|
||||||
|
logger.bind(tag=TAG).warning(
|
||||||
|
f"WebSocket连接失败,已达到最大重试次数: {e}"
|
||||||
|
)
|
||||||
|
raise
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"WebSocket连接发生未知错误: {e}")
|
||||||
|
raise
|
||||||
|
finally:
|
||||||
|
self.is_reconnecting = False
|
||||||
|
self._session_close_event.set()
|
||||||
|
|
||||||
|
async def receive_audio(self, audio, _):
|
||||||
|
if not isinstance(audio, bytes):
|
||||||
|
return
|
||||||
|
|
||||||
|
try:
|
||||||
|
# 解码opus得到PCM数据
|
||||||
|
pcm_frame = self.decoder.decode(audio, 960)
|
||||||
|
payload = gzip.compress(pcm_frame)
|
||||||
|
audio_request = bytearray(self.generate_audio_default_header())
|
||||||
|
audio_request.extend(len(payload).to_bytes(4, "big"))
|
||||||
|
audio_request.extend(payload)
|
||||||
|
if self.asr_ws:
|
||||||
|
await self.asr_ws.send(audio_request)
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).debug(f"发送音频数据时发生错误: {e}")
|
||||||
|
|
||||||
|
###################################################################################
|
||||||
|
# 豆包流式ASR重写父类的方法--结束
|
||||||
|
###################################################################################
|
||||||
|
|
||||||
|
def construct_request(self, reqid):
|
||||||
|
req = {
|
||||||
"app": {
|
"app": {
|
||||||
"appid": f"{self.appid}",
|
"appid": self.appid,
|
||||||
"cluster": self.cluster,
|
"cluster": self.cluster,
|
||||||
"token": self.access_token,
|
"token": self.access_token,
|
||||||
},
|
},
|
||||||
"user": {
|
"user": {"uid": self.uid},
|
||||||
"uid": str(uuid.uuid4()),
|
|
||||||
},
|
|
||||||
"request": {
|
"request": {
|
||||||
"reqid": reqid,
|
"reqid": reqid,
|
||||||
"show_utterances": False,
|
"workflow": self.workflow,
|
||||||
|
"show_utterances": True,
|
||||||
|
"result_type": self.result_type,
|
||||||
"sequence": 1,
|
"sequence": 1,
|
||||||
"boosting_table_name": self.boosting_table_name,
|
"boosting_table_name": self.boosting_table_name,
|
||||||
"correct_table_name": self.correct_table_name,
|
"correct_table_name": self.correct_table_name,
|
||||||
},
|
},
|
||||||
"audio": {
|
"audio": {
|
||||||
"format": "raw",
|
"format": self.format,
|
||||||
"rate": 16000,
|
"codec": self.codec,
|
||||||
"language": "zh-CN",
|
"rate": self.rate,
|
||||||
"bits": 16,
|
"language": self.language,
|
||||||
"channel": 1,
|
"bits": self.bits,
|
||||||
"codec": "raw",
|
"channel": self.channel,
|
||||||
},
|
},
|
||||||
}
|
}
|
||||||
|
return req
|
||||||
|
|
||||||
async def _send_request(
|
def token_auth(self):
|
||||||
self, audio_data: List[bytes], segment_size: int
|
return {"Authorization": f"Bearer; {self.access_token}"}
|
||||||
) -> Optional[str]:
|
|
||||||
"""Send request to Volcano ASR service."""
|
def generate_header(
|
||||||
|
self,
|
||||||
|
version=PROTOCOL_VERSION,
|
||||||
|
message_type=CLIENT_FULL_REQUEST,
|
||||||
|
message_type_specific_flags=NO_SEQUENCE,
|
||||||
|
serial_method=JSON_SERIALIZATION,
|
||||||
|
compression_type=GZIP_COMPRESSION,
|
||||||
|
reserved_data=0x00,
|
||||||
|
extension_header: bytes = b"",
|
||||||
|
):
|
||||||
|
"""
|
||||||
|
生成协议头:
|
||||||
|
- 第1字节:高4位:协议版本,低4位:头部大小(单位 4 字节)
|
||||||
|
- 第2字节:高4位:消息类型,低4位:消息类型特定标志
|
||||||
|
- 第3字节:高4位:序列化方式,低4位:压缩方式
|
||||||
|
- 第4字节:保留字段
|
||||||
|
- 后续:扩展头(如果有)
|
||||||
|
"""
|
||||||
|
header = bytearray()
|
||||||
|
header_size = int(len(extension_header) / 4) + 1
|
||||||
|
header.append((version << 4) | header_size)
|
||||||
|
header.append((message_type << 4) | message_type_specific_flags)
|
||||||
|
header.append((serial_method << 4) | compression_type)
|
||||||
|
header.append(reserved_data)
|
||||||
|
header.extend(extension_header)
|
||||||
|
return header
|
||||||
|
|
||||||
|
def generate_full_default_header(self):
|
||||||
|
# full client request 默认头
|
||||||
|
return self.generate_header(
|
||||||
|
version=PROTOCOL_VERSION,
|
||||||
|
message_type=CLIENT_FULL_REQUEST,
|
||||||
|
message_type_specific_flags=NO_SEQUENCE,
|
||||||
|
serial_method=JSON_SERIALIZATION,
|
||||||
|
compression_type=GZIP_COMPRESSION,
|
||||||
|
)
|
||||||
|
|
||||||
|
def generate_audio_default_header(self):
|
||||||
|
# 普通音频片段请求
|
||||||
|
return self.generate_header(
|
||||||
|
version=PROTOCOL_VERSION,
|
||||||
|
message_type=CLIENT_AUDIO_ONLY_REQUEST,
|
||||||
|
message_type_specific_flags=NO_SEQUENCE,
|
||||||
|
serial_method=JSON_SERIALIZATION,
|
||||||
|
compression_type=GZIP_COMPRESSION,
|
||||||
|
)
|
||||||
|
|
||||||
|
def generate_last_audio_default_header(self):
|
||||||
|
# 最后一个音频片段标志
|
||||||
|
return self.generate_header(
|
||||||
|
version=PROTOCOL_VERSION,
|
||||||
|
message_type=CLIENT_AUDIO_ONLY_REQUEST,
|
||||||
|
message_type_specific_flags=NEG_SEQUENCE, # 用 NEG_SEQUENCE 表示结束
|
||||||
|
serial_method=JSON_SERIALIZATION,
|
||||||
|
compression_type=GZIP_COMPRESSION,
|
||||||
|
)
|
||||||
|
|
||||||
|
def _start_monitor_asr_response_thread(self):
|
||||||
|
# 初始化链接
|
||||||
try:
|
try:
|
||||||
auth_header = {"Authorization": "Bearer; {}".format(self.access_token)}
|
with self.thread_lock:
|
||||||
async with websockets.connect(
|
if self.conn is None or self.conn.loop is None:
|
||||||
self.ws_url, additional_headers=auth_header
|
logger.bind(tag=TAG).error(
|
||||||
) as websocket:
|
"无法启动ASR监听线程:conn或loop未初始化"
|
||||||
# Prepare request data
|
)
|
||||||
request_params = self._construct_request(str(uuid.uuid4()))
|
return
|
||||||
payload_bytes = str.encode(json.dumps(request_params))
|
|
||||||
payload_bytes = gzip.compress(payload_bytes)
|
|
||||||
full_client_request = self._generate_header()
|
|
||||||
full_client_request.extend(
|
|
||||||
(len(payload_bytes)).to_bytes(4, "big")
|
|
||||||
) # payload size(4 bytes)
|
|
||||||
full_client_request.extend(payload_bytes) # payload
|
|
||||||
|
|
||||||
# Send header and metadata
|
try:
|
||||||
# full_client_request
|
logger.bind(tag=TAG).info("开始启动ASR监听...")
|
||||||
await websocket.send(full_client_request)
|
asyncio.run_coroutine_threadsafe(
|
||||||
res = await websocket.recv()
|
self._forward_asr_results(), loop=self.conn.loop
|
||||||
result = parse_response(res)
|
)
|
||||||
if (
|
logger.bind(tag=TAG).info("ASR监听已启动")
|
||||||
"payload_msg" in result
|
except Exception as e:
|
||||||
and result["payload_msg"]["code"] != self.success_code
|
logger.bind(tag=TAG).error(f"启动ASR监听线程失败: {e}")
|
||||||
):
|
except Exception as e:
|
||||||
logger.bind(tag=TAG).error(f"ASR error: {result}")
|
logger.bind(tag=TAG).error(f"ASR监听线程发生未预期的错误: {e}")
|
||||||
return None
|
|
||||||
|
|
||||||
for seq, (chunk, last) in enumerate(
|
async def _forward_asr_results(self):
|
||||||
self.slice_data(audio_data, segment_size), 1
|
try:
|
||||||
):
|
while not self.conn.stop_event.is_set():
|
||||||
if last:
|
try:
|
||||||
audio_only_request = self._generate_header(
|
if self.asr_ws is None:
|
||||||
message_type=CLIENT_AUDIO_ONLY_REQUEST,
|
# 检查是否需要重连
|
||||||
message_type_specific_flags=NEG_SEQUENCE,
|
async with self.reconnect_lock:
|
||||||
|
current_time = asyncio.get_event_loop().time()
|
||||||
|
if (
|
||||||
|
current_time - self.last_reconnect_time
|
||||||
|
< self.reconnect_cooldown
|
||||||
|
):
|
||||||
|
await asyncio.sleep(1)
|
||||||
|
continue
|
||||||
|
|
||||||
|
if self.reconnect_count >= self.max_reconnect_count:
|
||||||
|
logger.bind(tag=TAG).error(
|
||||||
|
"达到最大重连次数限制,停止重连"
|
||||||
|
)
|
||||||
|
await asyncio.sleep(self.reconnect_cooldown)
|
||||||
|
self.reconnect_count = 0
|
||||||
|
continue
|
||||||
|
|
||||||
|
self.last_reconnect_time = current_time
|
||||||
|
self.reconnect_count += 1
|
||||||
|
logger.bind(tag=TAG).info(
|
||||||
|
f"尝试重新连接ASR服务... (第{self.reconnect_count}次)"
|
||||||
|
)
|
||||||
|
await self.open_audio_channels(self.conn)
|
||||||
|
continue
|
||||||
|
|
||||||
|
# 使用锁来确保同一时间只有一个协程在接收数据
|
||||||
|
async with self.recv_lock:
|
||||||
|
response = await self.asr_ws.recv()
|
||||||
|
result = self.parse_response(response)
|
||||||
|
|
||||||
|
# 检查是否需要重连
|
||||||
|
if result.get("need_reconnect", False):
|
||||||
|
logger.bind(tag=TAG).info(
|
||||||
|
"检测到需要重连的错误,准备重新连接..."
|
||||||
)
|
)
|
||||||
else:
|
if self.asr_ws is not None:
|
||||||
audio_only_request = self._generate_header(
|
try:
|
||||||
message_type=CLIENT_AUDIO_ONLY_REQUEST
|
await self.asr_ws.close()
|
||||||
)
|
except Exception as e:
|
||||||
payload_bytes = gzip.compress(chunk)
|
logger.bind(tag=TAG).warning(
|
||||||
audio_only_request.extend(
|
f"关闭旧连接时发生错误: {e}"
|
||||||
(len(payload_bytes)).to_bytes(4, "big")
|
)
|
||||||
) # payload size(4 bytes)
|
finally:
|
||||||
audio_only_request.extend(payload_bytes) # payload
|
self.asr_ws = None
|
||||||
# Send audio data
|
continue
|
||||||
await websocket.send(audio_only_request)
|
|
||||||
|
|
||||||
# Receive response
|
if "payload_msg" in result:
|
||||||
response = await websocket.recv()
|
if "result" in result["payload_msg"]:
|
||||||
result = parse_response(response)
|
# 检查是否有utterances并且definite为True
|
||||||
|
utterances = result["payload_msg"]["result"][0].get(
|
||||||
|
"utterances", []
|
||||||
|
)
|
||||||
|
for utterance in utterances:
|
||||||
|
if utterance.get("definite", False):
|
||||||
|
self.text = utterance["text"]
|
||||||
|
await self.handle_voice_stop(None)
|
||||||
|
break
|
||||||
|
|
||||||
if (
|
except websockets.ConnectionClosed:
|
||||||
"payload_msg" in result
|
logger.bind(tag=TAG).debug("ASR服务连接已关闭,准备重连...")
|
||||||
and result["payload_msg"]["code"] == self.success_code
|
# 确保关闭旧连接
|
||||||
):
|
if self.asr_ws is not None:
|
||||||
if len(result["payload_msg"]["result"]) > 0:
|
try:
|
||||||
return result["payload_msg"]["result"][0]["text"]
|
await self.asr_ws.close()
|
||||||
return None
|
except Exception as e:
|
||||||
else:
|
logger.bind(tag=TAG).warning(f"关闭旧连接时发生错误: {e}")
|
||||||
logger.bind(tag=TAG).error(f"ASR error: {result}")
|
finally:
|
||||||
return None
|
self.asr_ws = None
|
||||||
|
|
||||||
|
# 等待冷却时间
|
||||||
|
await asyncio.sleep(self.reconnect_cooldown)
|
||||||
|
continue
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
if not self.conn.stop_event.is_set():
|
||||||
|
logger.bind(tag=TAG).error(f"ASR监听发生错误: {e}")
|
||||||
|
await asyncio.sleep(self.retry_delay)
|
||||||
|
continue
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.bind(tag=TAG).error(f"ASR request failed: {e}", exc_info=True)
|
logger.bind(tag=TAG).error(f"ASR监听线程发生错误: {e}")
|
||||||
return None
|
# 确保在发生严重错误时也能继续尝试重连
|
||||||
|
if not self.conn.stop_event.is_set():
|
||||||
|
await asyncio.sleep(self.retry_delay)
|
||||||
|
await self._forward_asr_results() # 递归重试
|
||||||
|
|
||||||
@staticmethod
|
async def speech_to_text(self, opus_data, session_id):
|
||||||
def slice_data(data: bytes, chunk_size: int) -> (list, bool):
|
result = self.text
|
||||||
|
self.text = "" # 清空text
|
||||||
|
return result, None
|
||||||
|
|
||||||
|
def parse_response(self, res: bytes) -> dict:
|
||||||
"""
|
"""
|
||||||
slice data
|
解析 ASR 服务返回的二进制响应。
|
||||||
:param data: wav data
|
根据协议格式解析头部和 payload,若采用 GZIP 压缩则先解压,再根据 JSON 反序列化。
|
||||||
:param chunk_size: the segment size in one request
|
|
||||||
:return: segment data, last flag
|
|
||||||
"""
|
"""
|
||||||
data_len = len(data)
|
protocol_version = res[0] >> 4
|
||||||
offset = 0
|
header_size = res[0] & 0x0F
|
||||||
while offset + chunk_size < data_len:
|
message_type = res[1] >> 4
|
||||||
yield data[offset : offset + chunk_size], False
|
serialization_method = res[2] >> 4
|
||||||
offset += chunk_size
|
message_compression = res[2] & 0x0F
|
||||||
|
payload = res[header_size * 4 :]
|
||||||
|
result = {}
|
||||||
|
payload_msg = None
|
||||||
|
payload_size = 0
|
||||||
|
|
||||||
|
if message_type == SERVER_FULL_RESPONSE:
|
||||||
|
payload_size = int.from_bytes(payload[:4], "big", signed=True)
|
||||||
|
payload_msg = payload[4:]
|
||||||
|
elif message_type == SERVER_ACK:
|
||||||
|
seq = int.from_bytes(payload[:4], "big", signed=True)
|
||||||
|
result["seq"] = seq
|
||||||
|
if len(payload) >= 8:
|
||||||
|
payload_size = int.from_bytes(payload[4:8], "big", signed=False)
|
||||||
|
payload_msg = payload[8:]
|
||||||
|
elif message_type == SERVER_ERROR_RESPONSE:
|
||||||
|
code = int.from_bytes(payload[:4], "big", signed=False)
|
||||||
|
result["code"] = code
|
||||||
|
payload_size = int.from_bytes(payload[4:8], "big", signed=False)
|
||||||
|
payload_msg = payload[8:]
|
||||||
|
|
||||||
|
if payload_msg is None:
|
||||||
|
return result
|
||||||
|
if message_compression == GZIP_COMPRESSION:
|
||||||
|
payload_msg = gzip.decompress(payload_msg)
|
||||||
|
if serialization_method == JSON_SERIALIZATION:
|
||||||
|
payload_msg = json.loads(payload_msg.decode("utf-8"))
|
||||||
else:
|
else:
|
||||||
yield data[offset:data_len], True
|
payload_msg = payload_msg.decode("utf-8")
|
||||||
|
result["payload_msg"] = payload_msg
|
||||||
|
result["payload_size"] = payload_size
|
||||||
|
|
||||||
async def speech_to_text(
|
# 错误码处理
|
||||||
self, opus_data: List[bytes], session_id: str
|
if "code" in result:
|
||||||
) -> Tuple[Optional[str], Optional[str]]:
|
error_code = result["code"]
|
||||||
"""将语音数据转换为文本"""
|
error_message = ""
|
||||||
|
|
||||||
file_path = None
|
if error_code == 1000:
|
||||||
try:
|
error_message = "成功"
|
||||||
# 合并所有opus数据包
|
elif error_code == 1001:
|
||||||
if self.audio_format == "pcm":
|
error_message = "请求参数无效:请求参数缺失必需字段/字段值无效/重复请求"
|
||||||
pcm_data = opus_data
|
elif error_code == 1002:
|
||||||
|
error_message = "无访问权限:token无效/过期/无权访问指定服务"
|
||||||
|
elif error_code == 1003:
|
||||||
|
error_message = "访问超频:当前appid访问QPS超出设定阈值"
|
||||||
|
elif error_code == 1004:
|
||||||
|
error_message = "访问超额:当前appid访问次数超出限制"
|
||||||
|
elif error_code == 1005:
|
||||||
|
error_message = "服务器繁忙:服务过载,无法处理当前请求"
|
||||||
|
elif error_code == 1010:
|
||||||
|
error_message = "音频过长:音频数据时长超出阈值"
|
||||||
|
elif error_code == 1011:
|
||||||
|
error_message = "音频过大:音频数据大小超出阈值"
|
||||||
|
elif error_code == 1012:
|
||||||
|
error_message = "音频格式无效:音频header有误/无法进行音频解码"
|
||||||
|
elif error_code == 1013:
|
||||||
|
error_message = "音频静音:音频未识别出任何文本结果"
|
||||||
|
elif error_code >= 1020 and error_code <= 1022:
|
||||||
|
error_message = "识别相关错误:需要重连"
|
||||||
|
if error_code == 1020:
|
||||||
|
error_message = "识别等待超时:等待下一包就绪超时"
|
||||||
|
elif error_code == 1021:
|
||||||
|
error_message = "识别处理超时:识别处理过程超时"
|
||||||
|
elif error_code == 1022:
|
||||||
|
error_message = "识别错误:识别过程中发生错误"
|
||||||
else:
|
else:
|
||||||
pcm_data = self.decode_opus(opus_data)
|
error_message = "未知错误:未归类错误"
|
||||||
combined_pcm_data = b"".join(pcm_data)
|
|
||||||
|
|
||||||
# 判断是否保存为WAV文件
|
logger.bind(tag=TAG).debug(
|
||||||
if self.delete_audio_file:
|
f"ASR错误: {error_message} (错误码: {error_code})"
|
||||||
pass
|
)
|
||||||
else:
|
|
||||||
file_path = self.save_audio_to_file(pcm_data, session_id)
|
|
||||||
|
|
||||||
# 直接使用PCM数据
|
# 如果是识别相关错误,标记需要重连
|
||||||
# 计算分段大小 (单声道, 16bit, 16kHz采样率)
|
if error_code >= 1020 or error_code == 1001:
|
||||||
size_per_sec = 1 * 2 * 16000 # nchannels * sampwidth * framerate
|
result["need_reconnect"] = True
|
||||||
segment_size = int(size_per_sec * self.seg_duration / 1000)
|
|
||||||
|
|
||||||
# 语音识别
|
return result
|
||||||
start_time = time.time()
|
|
||||||
text = await self._send_request(combined_pcm_data, segment_size)
|
async def close_session(self):
|
||||||
if text:
|
"""关闭当前会话"""
|
||||||
logger.bind(tag=TAG).debug(
|
async with self._session_lock:
|
||||||
f"语音识别耗时: {time.time() - start_time:.3f}s | 结果: {text}"
|
if not self._session_started:
|
||||||
|
logger.bind(tag=TAG).warning("尝试关闭未开始的会话")
|
||||||
|
return
|
||||||
|
|
||||||
|
if self._session_finished:
|
||||||
|
logger.bind(tag=TAG).warning(
|
||||||
|
f"会话 {self._current_session_id} 已经关闭"
|
||||||
)
|
)
|
||||||
return text, file_path
|
return
|
||||||
return "", file_path
|
|
||||||
|
|
||||||
except Exception as e:
|
try:
|
||||||
logger.bind(tag=TAG).error(f"语音识别失败: {e}", exc_info=True)
|
if self.asr_ws is not None:
|
||||||
return "", file_path
|
await self.asr_ws.close()
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).warning(f"关闭WebSocket连接时发生错误: {e}")
|
||||||
|
finally:
|
||||||
|
self.asr_ws = None
|
||||||
|
self._session_finished = True
|
||||||
|
self._session_started = False
|
||||||
|
self._current_session_id = None
|
||||||
|
# 重置重连计数
|
||||||
|
self.reconnect_count = 0
|
||||||
|
|
||||||
|
async def close(self):
|
||||||
|
"""资源清理方法"""
|
||||||
|
await self.close_session()
|
||||||
|
|||||||
@@ -0,0 +1,9 @@
|
|||||||
|
from enum import Enum
|
||||||
|
from typing import Union, Optional
|
||||||
|
|
||||||
|
|
||||||
|
class InterfaceType(Enum):
|
||||||
|
# 接口类型
|
||||||
|
STREAM = "STREAM" # 流式接口
|
||||||
|
NON_STREAM = "NON_STREAM" # 非流式接口
|
||||||
|
LOCAL = "LOCAL" # 本地服务
|
||||||
@@ -1,18 +1,21 @@
|
|||||||
import time
|
import time
|
||||||
import wave
|
|
||||||
import os
|
import os
|
||||||
import sys
|
import sys
|
||||||
import io
|
import io
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
from typing import Optional, Tuple, List
|
from typing import Optional, Tuple, List
|
||||||
import uuid
|
|
||||||
from core.providers.asr.base import ASRProviderBase
|
from core.providers.asr.base import ASRProviderBase
|
||||||
from funasr import AutoModel
|
from funasr import AutoModel
|
||||||
from funasr.utils.postprocess_utils import rich_transcription_postprocess
|
from funasr.utils.postprocess_utils import rich_transcription_postprocess
|
||||||
|
import shutil
|
||||||
|
from core.providers.asr.dto.dto import InterfaceType
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
logger = setup_logging()
|
logger = setup_logging()
|
||||||
|
|
||||||
|
MAX_RETRIES = 2
|
||||||
|
RETRY_DELAY = 1 # 重试延迟(秒)
|
||||||
|
|
||||||
|
|
||||||
# 捕获标准输出
|
# 捕获标准输出
|
||||||
class CaptureOutput:
|
class CaptureOutput:
|
||||||
@@ -34,6 +37,7 @@ class CaptureOutput:
|
|||||||
class ASRProvider(ASRProviderBase):
|
class ASRProvider(ASRProviderBase):
|
||||||
def __init__(self, config: dict, delete_audio_file: bool):
|
def __init__(self, config: dict, delete_audio_file: bool):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
self.interface_type = InterfaceType.LOCAL
|
||||||
self.model_dir = config.get("model_dir")
|
self.model_dir = config.get("model_dir")
|
||||||
self.output_dir = config.get("output_dir") # 修正配置键名
|
self.output_dir = config.get("output_dir") # 修正配置键名
|
||||||
self.delete_audio_file = delete_audio_file
|
self.delete_audio_file = delete_audio_file
|
||||||
@@ -49,65 +53,74 @@ class ASRProvider(ASRProviderBase):
|
|||||||
# device="cuda:0", # 启用GPU加速
|
# device="cuda:0", # 启用GPU加速
|
||||||
)
|
)
|
||||||
|
|
||||||
def save_audio_to_file(self, pcm_data: List[bytes], session_id: str) -> str:
|
|
||||||
"""PCM数据保存为WAV文件"""
|
|
||||||
module_name = __name__.split(".")[-1]
|
|
||||||
file_name = f"asr_{module_name}_{session_id}_{uuid.uuid4()}.wav"
|
|
||||||
file_path = os.path.join(self.output_dir, file_name)
|
|
||||||
|
|
||||||
with wave.open(file_path, "wb") as wf:
|
|
||||||
wf.setnchannels(1)
|
|
||||||
wf.setsampwidth(2) # 2 bytes = 16-bit
|
|
||||||
wf.setframerate(16000)
|
|
||||||
wf.writeframes(b"".join(pcm_data))
|
|
||||||
|
|
||||||
return file_path
|
|
||||||
|
|
||||||
async def speech_to_text(
|
async def speech_to_text(
|
||||||
self, opus_data: List[bytes], session_id: str
|
self, opus_data: List[bytes], session_id: str
|
||||||
) -> Tuple[Optional[str], Optional[str]]:
|
) -> Tuple[Optional[str], Optional[str]]:
|
||||||
"""语音转文本主处理逻辑"""
|
"""语音转文本主处理逻辑"""
|
||||||
file_path = None
|
file_path = None
|
||||||
try:
|
retry_count = 0
|
||||||
# 合并所有opus数据包
|
|
||||||
if self.audio_format == "pcm":
|
|
||||||
pcm_data = opus_data
|
|
||||||
else:
|
|
||||||
pcm_data = self.decode_opus(opus_data)
|
|
||||||
|
|
||||||
combined_pcm_data = b"".join(pcm_data)
|
while retry_count < MAX_RETRIES:
|
||||||
|
try:
|
||||||
|
# 合并所有opus数据包
|
||||||
|
if self.audio_format == "pcm":
|
||||||
|
pcm_data = opus_data
|
||||||
|
else:
|
||||||
|
pcm_data = self.decode_opus(opus_data)
|
||||||
|
|
||||||
# 判断是否保存为WAV文件
|
combined_pcm_data = b"".join(pcm_data)
|
||||||
if self.delete_audio_file:
|
|
||||||
pass
|
|
||||||
else:
|
|
||||||
file_path = self.save_audio_to_file(pcm_data, session_id)
|
|
||||||
|
|
||||||
# 语音识别
|
# 检查磁盘空间
|
||||||
start_time = time.time()
|
if not self.delete_audio_file:
|
||||||
result = self.model.generate(
|
free_space = shutil.disk_usage(self.output_dir).free
|
||||||
input=combined_pcm_data,
|
if free_space < len(combined_pcm_data) * 2: # 预留2倍空间
|
||||||
cache={},
|
raise OSError("磁盘空间不足")
|
||||||
language="auto",
|
|
||||||
use_itn=True,
|
|
||||||
batch_size_s=60,
|
|
||||||
)
|
|
||||||
text = rich_transcription_postprocess(result[0]["text"])
|
|
||||||
logger.bind(tag=TAG).debug(
|
|
||||||
f"语音识别耗时: {time.time() - start_time:.3f}s | 结果: {text}"
|
|
||||||
)
|
|
||||||
|
|
||||||
return text, file_path
|
# 判断是否保存为WAV文件
|
||||||
|
if self.delete_audio_file:
|
||||||
|
pass
|
||||||
|
else:
|
||||||
|
file_path = self.save_audio_to_file(pcm_data, session_id)
|
||||||
|
|
||||||
except Exception as e:
|
# 语音识别
|
||||||
logger.bind(tag=TAG).error(f"语音识别失败: {e}", exc_info=True)
|
start_time = time.time()
|
||||||
return "", file_path
|
result = self.model.generate(
|
||||||
|
input=combined_pcm_data,
|
||||||
|
cache={},
|
||||||
|
language="auto",
|
||||||
|
use_itn=True,
|
||||||
|
batch_size_s=60,
|
||||||
|
)
|
||||||
|
text = rich_transcription_postprocess(result[0]["text"])
|
||||||
|
logger.bind(tag=TAG).debug(
|
||||||
|
f"语音识别耗时: {time.time() - start_time:.3f}s | 结果: {text}"
|
||||||
|
)
|
||||||
|
|
||||||
# finally:
|
return text, file_path
|
||||||
# # 文件清理逻辑
|
|
||||||
# if self.delete_audio_file and file_path and os.path.exists(file_path):
|
except OSError as e:
|
||||||
# try:
|
retry_count += 1
|
||||||
# os.remove(file_path)
|
if retry_count >= MAX_RETRIES:
|
||||||
# logger.bind(tag=TAG).debug(f"已删除临时音频文件: {file_path}")
|
logger.bind(tag=TAG).error(
|
||||||
# except Exception as e:
|
f"语音识别失败(已重试{retry_count}次): {e}", exc_info=True
|
||||||
# logger.bind(tag=TAG).error(f"文件删除失败: {file_path} | 错误: {e}")
|
)
|
||||||
|
return "", file_path
|
||||||
|
logger.bind(tag=TAG).warning(
|
||||||
|
f"语音识别失败,正在重试({retry_count}/{MAX_RETRIES}): {e}"
|
||||||
|
)
|
||||||
|
time.sleep(RETRY_DELAY)
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"语音识别失败: {e}", exc_info=True)
|
||||||
|
return "", file_path
|
||||||
|
|
||||||
|
finally:
|
||||||
|
# 文件清理逻辑
|
||||||
|
if self.delete_audio_file and file_path and os.path.exists(file_path):
|
||||||
|
try:
|
||||||
|
os.remove(file_path)
|
||||||
|
logger.bind(tag=TAG).debug(f"已删除临时音频文件: {file_path}")
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(
|
||||||
|
f"文件删除失败: {file_path} | 错误: {e}"
|
||||||
|
)
|
||||||
|
|||||||
@@ -1,11 +1,8 @@
|
|||||||
from typing import Optional, Tuple, List
|
from typing import Optional, Tuple, List
|
||||||
import opuslib_next
|
|
||||||
from core.providers.asr.base import ASRProviderBase
|
from core.providers.asr.base import ASRProviderBase
|
||||||
import os
|
from core.providers.asr.dto.dto import InterfaceType
|
||||||
import ssl
|
import ssl
|
||||||
import json
|
import json
|
||||||
import uuid
|
|
||||||
import wave
|
|
||||||
import websockets
|
import websockets
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
import asyncio
|
import asyncio
|
||||||
@@ -23,6 +20,7 @@ class ASRProvider(ASRProviderBase):
|
|||||||
:param delete_audio_file: Boolean to indicate whether to delete audio files after processing.
|
:param delete_audio_file: Boolean to indicate whether to delete audio files after processing.
|
||||||
"""
|
"""
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
self.interface_type = InterfaceType.NON_STREAM
|
||||||
self.host = config.get("host", "localhost")
|
self.host = config.get("host", "localhost")
|
||||||
self.port = config.get("port", 10095)
|
self.port = config.get("port", 10095)
|
||||||
self.api_key = config.get("api_key", "none")
|
self.api_key = config.get("api_key", "none")
|
||||||
@@ -43,20 +41,6 @@ class ASRProvider(ASRProviderBase):
|
|||||||
self.ssl_context.check_hostname = False
|
self.ssl_context.check_hostname = False
|
||||||
self.ssl_context.verify_mode = ssl.CERT_NONE
|
self.ssl_context.verify_mode = ssl.CERT_NONE
|
||||||
|
|
||||||
def save_audio_to_file(self, pcm_data: List[bytes], session_id: str) -> str:
|
|
||||||
"""PCM数据保存为WAV文件"""
|
|
||||||
module_name = __name__.split(".")[-1]
|
|
||||||
file_name = f"asr_{module_name}_{session_id}_{uuid.uuid4()}.wav"
|
|
||||||
file_path = os.path.join(self.output_dir, file_name)
|
|
||||||
|
|
||||||
with wave.open(file_path, "wb") as wf:
|
|
||||||
wf.setnchannels(1)
|
|
||||||
wf.setsampwidth(2) # 2 bytes = 16-bit
|
|
||||||
wf.setframerate(16000)
|
|
||||||
wf.writeframes(b"".join(pcm_data))
|
|
||||||
|
|
||||||
return file_path
|
|
||||||
|
|
||||||
async def _receive_responses(self, ws) -> None:
|
async def _receive_responses(self, ws) -> None:
|
||||||
"""
|
"""
|
||||||
Asynchronous generator to receive messages from the WebSocket.
|
Asynchronous generator to receive messages from the WebSocket.
|
||||||
|
|||||||
@@ -5,8 +5,7 @@ import sys
|
|||||||
import io
|
import io
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
from typing import Optional, Tuple, List
|
from typing import Optional, Tuple, List
|
||||||
import uuid
|
from core.providers.asr.dto.dto import InterfaceType
|
||||||
import opuslib_next
|
|
||||||
from core.providers.asr.base import ASRProviderBase
|
from core.providers.asr.base import ASRProviderBase
|
||||||
|
|
||||||
import numpy as np
|
import numpy as np
|
||||||
@@ -38,6 +37,7 @@ class CaptureOutput:
|
|||||||
class ASRProvider(ASRProviderBase):
|
class ASRProvider(ASRProviderBase):
|
||||||
def __init__(self, config: dict, delete_audio_file: bool):
|
def __init__(self, config: dict, delete_audio_file: bool):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
self.interface_type = InterfaceType.LOCAL
|
||||||
self.model_dir = config.get("model_dir")
|
self.model_dir = config.get("model_dir")
|
||||||
self.output_dir = config.get("output_dir")
|
self.output_dir = config.get("output_dir")
|
||||||
self.delete_audio_file = delete_audio_file
|
self.delete_audio_file = delete_audio_file
|
||||||
@@ -84,20 +84,6 @@ class ASRProvider(ASRProviderBase):
|
|||||||
use_itn=True,
|
use_itn=True,
|
||||||
)
|
)
|
||||||
|
|
||||||
def save_audio_to_file(self, pcm_data: List[bytes], session_id: str) -> str:
|
|
||||||
"""PCM数据保存为WAV文件"""
|
|
||||||
module_name = __name__.split(".")[-1]
|
|
||||||
file_name = f"asr_{module_name}_{session_id}_{uuid.uuid4()}.wav"
|
|
||||||
file_path = os.path.join(self.output_dir, file_name)
|
|
||||||
|
|
||||||
with wave.open(file_path, "wb") as wf:
|
|
||||||
wf.setnchannels(1)
|
|
||||||
wf.setsampwidth(2) # 2 bytes = 16-bit
|
|
||||||
wf.setframerate(16000)
|
|
||||||
wf.writeframes(b"".join(pcm_data))
|
|
||||||
|
|
||||||
return file_path
|
|
||||||
|
|
||||||
def read_wave(self, wave_filename: str) -> Tuple[np.ndarray, int]:
|
def read_wave(self, wave_filename: str) -> Tuple[np.ndarray, int]:
|
||||||
"""
|
"""
|
||||||
Args:
|
Args:
|
||||||
|
|||||||
@@ -5,11 +5,8 @@ import json
|
|||||||
import time
|
import time
|
||||||
from datetime import datetime, timezone
|
from datetime import datetime, timezone
|
||||||
import os
|
import os
|
||||||
import uuid
|
|
||||||
from typing import Optional, Tuple, List
|
from typing import Optional, Tuple, List
|
||||||
import wave
|
from core.providers.asr.dto.dto import InterfaceType
|
||||||
import opuslib_next
|
|
||||||
|
|
||||||
import requests
|
import requests
|
||||||
from core.providers.asr.base import ASRProviderBase
|
from core.providers.asr.base import ASRProviderBase
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
@@ -25,6 +22,7 @@ class ASRProvider(ASRProviderBase):
|
|||||||
|
|
||||||
def __init__(self, config: dict, delete_audio_file: bool = True):
|
def __init__(self, config: dict, delete_audio_file: bool = True):
|
||||||
super().__init__()
|
super().__init__()
|
||||||
|
self.interface_type = InterfaceType.NON_STREAM
|
||||||
self.secret_id = config.get("secret_id")
|
self.secret_id = config.get("secret_id")
|
||||||
self.secret_key = config.get("secret_key")
|
self.secret_key = config.get("secret_key")
|
||||||
self.output_dir = config.get("output_dir")
|
self.output_dir = config.get("output_dir")
|
||||||
@@ -33,20 +31,6 @@ class ASRProvider(ASRProviderBase):
|
|||||||
# 确保输出目录存在
|
# 确保输出目录存在
|
||||||
os.makedirs(self.output_dir, exist_ok=True)
|
os.makedirs(self.output_dir, exist_ok=True)
|
||||||
|
|
||||||
def save_audio_to_file(self, pcm_data: List[bytes], session_id: str) -> str:
|
|
||||||
"""PCM数据保存为WAV文件"""
|
|
||||||
module_name = __name__.split(".")[-1]
|
|
||||||
file_name = f"asr_{module_name}_{session_id}_{uuid.uuid4()}.wav"
|
|
||||||
file_path = os.path.join(self.output_dir, file_name)
|
|
||||||
|
|
||||||
with wave.open(file_path, "wb") as wf:
|
|
||||||
wf.setnchannels(1)
|
|
||||||
wf.setsampwidth(2) # 2 bytes = 16-bit
|
|
||||||
wf.setframerate(16000)
|
|
||||||
wf.writeframes(b"".join(pcm_data))
|
|
||||||
|
|
||||||
return file_path
|
|
||||||
|
|
||||||
async def speech_to_text(
|
async def speech_to_text(
|
||||||
self, opus_data: List[bytes], session_id: str
|
self, opus_data: List[bytes], session_id: str
|
||||||
) -> Tuple[Optional[str], Optional[str]]:
|
) -> Tuple[Optional[str], Optional[str]]:
|
||||||
|
|||||||
@@ -72,6 +72,10 @@ class IntentProvider(IntentProviderBase):
|
|||||||
'返回: {"function_call": {"name": "get_time"}}\n'
|
'返回: {"function_call": {"name": "get_time"}}\n'
|
||||||
"```\n"
|
"```\n"
|
||||||
"```\n"
|
"```\n"
|
||||||
|
"用户: 当前电池电量是多少?\n"
|
||||||
|
'返回: {"function_call": {"name": "get_battery_level", "arguments": {"response_success": "当前电池电量为{value}%", "response_failure": "无法获取Battery的当前电量百分比"}}}\n'
|
||||||
|
"```\n"
|
||||||
|
"```\n"
|
||||||
"用户: 我想结束对话\n"
|
"用户: 我想结束对话\n"
|
||||||
'返回: {"function_call": {"name": "handle_exit_intent", "arguments": {"say_goodbye": "goodbye"}}}\n'
|
'返回: {"function_call": {"name": "handle_exit_intent", "arguments": {"say_goodbye": "goodbye"}}}\n'
|
||||||
"```\n"
|
"```\n"
|
||||||
@@ -224,7 +228,8 @@ class IntentProvider(IntentProviderBase):
|
|||||||
if function_name == "continue_chat":
|
if function_name == "continue_chat":
|
||||||
# 保留非工具相关的消息
|
# 保留非工具相关的消息
|
||||||
clean_history = [
|
clean_history = [
|
||||||
msg for msg in conn.dialogue.dialogue
|
msg
|
||||||
|
for msg in conn.dialogue.dialogue
|
||||||
if msg.role not in ["tool", "function"]
|
if msg.role not in ["tool", "function"]
|
||||||
]
|
]
|
||||||
conn.dialogue.dialogue = clean_history
|
conn.dialogue.dialogue = clean_history
|
||||||
|
|||||||
@@ -10,7 +10,7 @@ class LLMProviderBase(ABC):
|
|||||||
"""LLM response generator"""
|
"""LLM response generator"""
|
||||||
pass
|
pass
|
||||||
|
|
||||||
def response_no_stream(self, system_prompt, user_prompt):
|
def response_no_stream(self, system_prompt, user_prompt, **kwargs):
|
||||||
try:
|
try:
|
||||||
# 构造对话格式
|
# 构造对话格式
|
||||||
dialogue = [
|
dialogue = [
|
||||||
@@ -18,7 +18,7 @@ class LLMProviderBase(ABC):
|
|||||||
{"role": "user", "content": user_prompt}
|
{"role": "user", "content": user_prompt}
|
||||||
]
|
]
|
||||||
result = ""
|
result = ""
|
||||||
for part in self.response("", dialogue):
|
for part in self.response("", dialogue, **kwargs):
|
||||||
result += part
|
result += part
|
||||||
return result
|
return result
|
||||||
|
|
||||||
|
|||||||
@@ -25,7 +25,7 @@ class LLMProvider(LLMProviderBase):
|
|||||||
self.session_conversation_map = {} # 存储session_id和conversation_id的映射
|
self.session_conversation_map = {} # 存储session_id和conversation_id的映射
|
||||||
check_model_key("CozeLLM", self.personal_access_token)
|
check_model_key("CozeLLM", self.personal_access_token)
|
||||||
|
|
||||||
def response(self, session_id, dialogue):
|
def response(self, session_id, dialogue, **kwargs):
|
||||||
coze_api_token = self.personal_access_token
|
coze_api_token = self.personal_access_token
|
||||||
coze_api_base = COZE_CN_BASE_URL
|
coze_api_base = COZE_CN_BASE_URL
|
||||||
|
|
||||||
|
|||||||
@@ -17,7 +17,7 @@ class LLMProvider(LLMProviderBase):
|
|||||||
self.session_conversation_map = {} # 存储session_id和conversation_id的映射
|
self.session_conversation_map = {} # 存储session_id和conversation_id的映射
|
||||||
check_model_key("DifyLLM", self.api_key)
|
check_model_key("DifyLLM", self.api_key)
|
||||||
|
|
||||||
def response(self, session_id, dialogue):
|
def response(self, session_id, dialogue, **kwargs):
|
||||||
try:
|
try:
|
||||||
# 取最后一条用户消息
|
# 取最后一条用户消息
|
||||||
last_msg = next(m for m in reversed(dialogue) if m["role"] == "user")
|
last_msg = next(m for m in reversed(dialogue) if m["role"] == "user")
|
||||||
|
|||||||
@@ -16,7 +16,7 @@ class LLMProvider(LLMProviderBase):
|
|||||||
self.variables = config.get("variables", {})
|
self.variables = config.get("variables", {})
|
||||||
check_model_key("FastGPTLLM", self.api_key)
|
check_model_key("FastGPTLLM", self.api_key)
|
||||||
|
|
||||||
def response(self, session_id, dialogue):
|
def response(self, session_id, dialogue, **kwargs):
|
||||||
try:
|
try:
|
||||||
# 取最后一条用户消息
|
# 取最后一条用户消息
|
||||||
last_msg = next(m for m in reversed(dialogue) if m["role"] == "user")
|
last_msg = next(m for m in reversed(dialogue) if m["role"] == "user")
|
||||||
|
|||||||
@@ -112,7 +112,7 @@ class LLMProvider(LLMProviderBase):
|
|||||||
]
|
]
|
||||||
|
|
||||||
# Gemini文档提到,无需维护session-id,直接用dialogue拼接而成
|
# Gemini文档提到,无需维护session-id,直接用dialogue拼接而成
|
||||||
def response(self, session_id, dialogue):
|
def response(self, session_id, dialogue, **kwargs):
|
||||||
yield from self._generate(dialogue, None)
|
yield from self._generate(dialogue, None)
|
||||||
|
|
||||||
def response_with_functions(self, session_id, dialogue, functions=None):
|
def response_with_functions(self, session_id, dialogue, functions=None):
|
||||||
|
|||||||
@@ -14,7 +14,7 @@ class LLMProvider(LLMProviderBase):
|
|||||||
self.base_url = config.get("base_url", config.get("url")) # 默认使用 base_url
|
self.base_url = config.get("base_url", config.get("url")) # 默认使用 base_url
|
||||||
self.api_url = f"{self.base_url}/api/conversation/process" # 拼接完整的 API URL
|
self.api_url = f"{self.base_url}/api/conversation/process" # 拼接完整的 API URL
|
||||||
|
|
||||||
def response(self, session_id, dialogue):
|
def response(self, session_id, dialogue, **kwargs):
|
||||||
try:
|
try:
|
||||||
# home assistant语音助手自带意图,无需使用xiaozhi ai自带的,只需要把用户说的话传递给home assistant即可
|
# home assistant语音助手自带意图,无需使用xiaozhi ai自带的,只需要把用户说的话传递给home assistant即可
|
||||||
|
|
||||||
|
|||||||
@@ -18,13 +18,13 @@ class LLMProvider(LLMProviderBase):
|
|||||||
|
|
||||||
self.client = OpenAI(
|
self.client = OpenAI(
|
||||||
base_url=self.base_url,
|
base_url=self.base_url,
|
||||||
api_key="ollama" # Ollama doesn't need an API key but OpenAI client requires one
|
api_key="ollama", # Ollama doesn't need an API key but OpenAI client requires one
|
||||||
)
|
)
|
||||||
|
|
||||||
# 检查是否是qwen3模型
|
# 检查是否是qwen3模型
|
||||||
self.is_qwen3 = self.model_name and self.model_name.lower().startswith("qwen3")
|
self.is_qwen3 = self.model_name and self.model_name.lower().startswith("qwen3")
|
||||||
|
|
||||||
def response(self, session_id, dialogue):
|
def response(self, session_id, dialogue, **kwargs):
|
||||||
try:
|
try:
|
||||||
# 如果是qwen3模型,在用户最后一条消息中添加/no_think指令
|
# 如果是qwen3模型,在用户最后一条消息中添加/no_think指令
|
||||||
if self.is_qwen3:
|
if self.is_qwen3:
|
||||||
@@ -35,7 +35,9 @@ class LLMProvider(LLMProviderBase):
|
|||||||
for i in range(len(dialogue_copy) - 1, -1, -1):
|
for i in range(len(dialogue_copy) - 1, -1, -1):
|
||||||
if dialogue_copy[i]["role"] == "user":
|
if dialogue_copy[i]["role"] == "user":
|
||||||
# 在用户消息前添加/no_think指令
|
# 在用户消息前添加/no_think指令
|
||||||
dialogue_copy[i]["content"] = "/no_think " + dialogue_copy[i]["content"]
|
dialogue_copy[i]["content"] = (
|
||||||
|
"/no_think " + dialogue_copy[i]["content"]
|
||||||
|
)
|
||||||
logger.bind(tag=TAG).debug(f"为qwen3模型添加/no_think指令")
|
logger.bind(tag=TAG).debug(f"为qwen3模型添加/no_think指令")
|
||||||
break
|
break
|
||||||
|
|
||||||
@@ -43,9 +45,7 @@ class LLMProvider(LLMProviderBase):
|
|||||||
dialogue = dialogue_copy
|
dialogue = dialogue_copy
|
||||||
|
|
||||||
responses = self.client.chat.completions.create(
|
responses = self.client.chat.completions.create(
|
||||||
model=self.model_name,
|
model=self.model_name, messages=dialogue, stream=True
|
||||||
messages=dialogue,
|
|
||||||
stream=True
|
|
||||||
)
|
)
|
||||||
is_active = True
|
is_active = True
|
||||||
# 用于处理跨chunk的标签
|
# 用于处理跨chunk的标签
|
||||||
@@ -53,29 +53,33 @@ class LLMProvider(LLMProviderBase):
|
|||||||
|
|
||||||
for chunk in responses:
|
for chunk in responses:
|
||||||
try:
|
try:
|
||||||
delta = chunk.choices[0].delta if getattr(chunk, 'choices', None) else None
|
delta = (
|
||||||
content = delta.content if hasattr(delta, 'content') else ''
|
chunk.choices[0].delta
|
||||||
|
if getattr(chunk, "choices", None)
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
content = delta.content if hasattr(delta, "content") else ""
|
||||||
|
|
||||||
if content:
|
if content:
|
||||||
# 将内容添加到缓冲区
|
# 将内容添加到缓冲区
|
||||||
buffer += content
|
buffer += content
|
||||||
|
|
||||||
# 处理缓冲区中的标签
|
# 处理缓冲区中的标签
|
||||||
while '<think>' in buffer and '</think>' in buffer:
|
while "<think>" in buffer and "</think>" in buffer:
|
||||||
# 找到完整的<think></think>标签并移除
|
# 找到完整的<think></think>标签并移除
|
||||||
pre = buffer.split('<think>', 1)[0]
|
pre = buffer.split("<think>", 1)[0]
|
||||||
post = buffer.split('</think>', 1)[1]
|
post = buffer.split("</think>", 1)[1]
|
||||||
buffer = pre + post
|
buffer = pre + post
|
||||||
|
|
||||||
# 处理只有开始标签的情况
|
# 处理只有开始标签的情况
|
||||||
if '<think>' in buffer:
|
if "<think>" in buffer:
|
||||||
is_active = False
|
is_active = False
|
||||||
buffer = buffer.split('<think>', 1)[0]
|
buffer = buffer.split("<think>", 1)[0]
|
||||||
|
|
||||||
# 处理只有结束标签的情况
|
# 处理只有结束标签的情况
|
||||||
if '</think>' in buffer:
|
if "</think>" in buffer:
|
||||||
is_active = True
|
is_active = True
|
||||||
buffer = buffer.split('</think>', 1)[1]
|
buffer = buffer.split("</think>", 1)[1]
|
||||||
|
|
||||||
# 如果当前处于活动状态且缓冲区有内容,则输出
|
# 如果当前处于活动状态且缓冲区有内容,则输出
|
||||||
if is_active and buffer:
|
if is_active and buffer:
|
||||||
@@ -100,7 +104,9 @@ class LLMProvider(LLMProviderBase):
|
|||||||
for i in range(len(dialogue_copy) - 1, -1, -1):
|
for i in range(len(dialogue_copy) - 1, -1, -1):
|
||||||
if dialogue_copy[i]["role"] == "user":
|
if dialogue_copy[i]["role"] == "user":
|
||||||
# 在用户消息前添加/no_think指令
|
# 在用户消息前添加/no_think指令
|
||||||
dialogue_copy[i]["content"] = "/no_think " + dialogue_copy[i]["content"]
|
dialogue_copy[i]["content"] = (
|
||||||
|
"/no_think " + dialogue_copy[i]["content"]
|
||||||
|
)
|
||||||
logger.bind(tag=TAG).debug(f"为qwen3模型添加/no_think指令")
|
logger.bind(tag=TAG).debug(f"为qwen3模型添加/no_think指令")
|
||||||
break
|
break
|
||||||
|
|
||||||
@@ -119,9 +125,15 @@ class LLMProvider(LLMProviderBase):
|
|||||||
|
|
||||||
for chunk in stream:
|
for chunk in stream:
|
||||||
try:
|
try:
|
||||||
delta = chunk.choices[0].delta if getattr(chunk, 'choices', None) else None
|
delta = (
|
||||||
content = delta.content if hasattr(delta, 'content') else None
|
chunk.choices[0].delta
|
||||||
tool_calls = delta.tool_calls if hasattr(delta, 'tool_calls') else None
|
if getattr(chunk, "choices", None)
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
content = delta.content if hasattr(delta, "content") else None
|
||||||
|
tool_calls = (
|
||||||
|
delta.tool_calls if hasattr(delta, "tool_calls") else None
|
||||||
|
)
|
||||||
|
|
||||||
# 如果是工具调用,直接传递
|
# 如果是工具调用,直接传递
|
||||||
if tool_calls:
|
if tool_calls:
|
||||||
@@ -134,21 +146,21 @@ class LLMProvider(LLMProviderBase):
|
|||||||
buffer += content
|
buffer += content
|
||||||
|
|
||||||
# 处理缓冲区中的标签
|
# 处理缓冲区中的标签
|
||||||
while '<think>' in buffer and '</think>' in buffer:
|
while "<think>" in buffer and "</think>" in buffer:
|
||||||
# 找到完整的<think></think>标签并移除
|
# 找到完整的<think></think>标签并移除
|
||||||
pre = buffer.split('<think>', 1)[0]
|
pre = buffer.split("<think>", 1)[0]
|
||||||
post = buffer.split('</think>', 1)[1]
|
post = buffer.split("</think>", 1)[1]
|
||||||
buffer = pre + post
|
buffer = pre + post
|
||||||
|
|
||||||
# 处理只有开始标签的情况
|
# 处理只有开始标签的情况
|
||||||
if '<think>' in buffer:
|
if "<think>" in buffer:
|
||||||
is_active = False
|
is_active = False
|
||||||
buffer = buffer.split('<think>', 1)[0]
|
buffer = buffer.split("<think>", 1)[0]
|
||||||
|
|
||||||
# 处理只有结束标签的情况
|
# 处理只有结束标签的情况
|
||||||
if '</think>' in buffer:
|
if "</think>" in buffer:
|
||||||
is_active = True
|
is_active = True
|
||||||
buffer = buffer.split('</think>', 1)[1]
|
buffer = buffer.split("</think>", 1)[1]
|
||||||
|
|
||||||
# 如果当前处于活动状态且缓冲区有内容,则输出
|
# 如果当前处于活动状态且缓冲区有内容,则输出
|
||||||
if is_active and buffer:
|
if is_active and buffer:
|
||||||
|
|||||||
@@ -16,26 +16,37 @@ class LLMProvider(LLMProviderBase):
|
|||||||
self.base_url = config.get("base_url")
|
self.base_url = config.get("base_url")
|
||||||
else:
|
else:
|
||||||
self.base_url = config.get("url")
|
self.base_url = config.get("url")
|
||||||
max_tokens = config.get("max_tokens")
|
|
||||||
if max_tokens is None or max_tokens == "":
|
|
||||||
max_tokens = 500
|
|
||||||
|
|
||||||
try:
|
param_defaults = {
|
||||||
max_tokens = int(max_tokens)
|
"max_tokens": (500, int),
|
||||||
except (ValueError, TypeError):
|
"temperature": (0.7, lambda x: round(float(x), 1)),
|
||||||
max_tokens = 500
|
"top_p": (1.0, lambda x: round(float(x), 1)),
|
||||||
self.max_tokens = max_tokens
|
"frequency_penalty": (0, lambda x: round(float(x), 1))
|
||||||
|
}
|
||||||
|
|
||||||
|
for param, (default, converter) in param_defaults.items():
|
||||||
|
value = config.get(param)
|
||||||
|
try:
|
||||||
|
setattr(self, param, converter(value) if value not in (None, "") else default)
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
setattr(self, param, default)
|
||||||
|
|
||||||
|
logger.debug(
|
||||||
|
f"意图识别参数初始化: {self.temperature}, {self.max_tokens}, {self.top_p}, {self.frequency_penalty}")
|
||||||
|
|
||||||
check_model_key("LLM", self.api_key)
|
check_model_key("LLM", self.api_key)
|
||||||
self.client = openai.OpenAI(api_key=self.api_key, base_url=self.base_url)
|
self.client = openai.OpenAI(api_key=self.api_key, base_url=self.base_url)
|
||||||
|
|
||||||
def response(self, session_id, dialogue):
|
def response(self, session_id, dialogue, **kwargs):
|
||||||
try:
|
try:
|
||||||
responses = self.client.chat.completions.create(
|
responses = self.client.chat.completions.create(
|
||||||
model=self.model_name,
|
model=self.model_name,
|
||||||
messages=dialogue,
|
messages=dialogue,
|
||||||
stream=True,
|
stream=True,
|
||||||
max_tokens=self.max_tokens,
|
max_tokens=kwargs.get("max_tokens", self.max_tokens),
|
||||||
|
temperature=kwargs.get("temperature", self.temperature),
|
||||||
|
top_p=kwargs.get("top_p", self.top_p),
|
||||||
|
frequency_penalty=kwargs.get("frequency_penalty", self.frequency_penalty),
|
||||||
)
|
)
|
||||||
|
|
||||||
is_active = True
|
is_active = True
|
||||||
|
|||||||
@@ -16,38 +16,44 @@ class LLMProvider(LLMProviderBase):
|
|||||||
if not self.base_url.endswith("/v1"):
|
if not self.base_url.endswith("/v1"):
|
||||||
self.base_url = f"{self.base_url}/v1"
|
self.base_url = f"{self.base_url}/v1"
|
||||||
|
|
||||||
logger.bind(tag=TAG).info(f"Initializing Xinference LLM provider with model: {self.model_name}, base_url: {self.base_url}")
|
logger.bind(tag=TAG).info(
|
||||||
|
f"Initializing Xinference LLM provider with model: {self.model_name}, base_url: {self.base_url}"
|
||||||
|
)
|
||||||
|
|
||||||
try:
|
try:
|
||||||
self.client = OpenAI(
|
self.client = OpenAI(
|
||||||
base_url=self.base_url,
|
base_url=self.base_url,
|
||||||
api_key="xinference" # Xinference has a similar setup to Ollama where it doesn't need an actual key
|
api_key="xinference", # Xinference has a similar setup to Ollama where it doesn't need an actual key
|
||||||
)
|
)
|
||||||
logger.bind(tag=TAG).info("Xinference client initialized successfully")
|
logger.bind(tag=TAG).info("Xinference client initialized successfully")
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.bind(tag=TAG).error(f"Error initializing Xinference client: {e}")
|
logger.bind(tag=TAG).error(f"Error initializing Xinference client: {e}")
|
||||||
raise
|
raise
|
||||||
|
|
||||||
def response(self, session_id, dialogue):
|
def response(self, session_id, dialogue, **kwargs):
|
||||||
try:
|
try:
|
||||||
logger.bind(tag=TAG).debug(f"Sending request to Xinference with model: {self.model_name}, dialogue length: {len(dialogue)}")
|
logger.bind(tag=TAG).debug(
|
||||||
responses = self.client.chat.completions.create(
|
f"Sending request to Xinference with model: {self.model_name}, dialogue length: {len(dialogue)}"
|
||||||
model=self.model_name,
|
|
||||||
messages=dialogue,
|
|
||||||
stream=True
|
|
||||||
)
|
)
|
||||||
is_active=True
|
responses = self.client.chat.completions.create(
|
||||||
|
model=self.model_name, messages=dialogue, stream=True
|
||||||
|
)
|
||||||
|
is_active = True
|
||||||
for chunk in responses:
|
for chunk in responses:
|
||||||
try:
|
try:
|
||||||
delta = chunk.choices[0].delta if getattr(chunk, 'choices', None) else None
|
delta = (
|
||||||
content = delta.content if hasattr(delta, 'content') else ''
|
chunk.choices[0].delta
|
||||||
|
if getattr(chunk, "choices", None)
|
||||||
|
else None
|
||||||
|
)
|
||||||
|
content = delta.content if hasattr(delta, "content") else ""
|
||||||
if content:
|
if content:
|
||||||
if '<think>' in content:
|
if "<think>" in content:
|
||||||
is_active = False
|
is_active = False
|
||||||
content = content.split('<think>')[0]
|
content = content.split("<think>")[0]
|
||||||
if '</think>' in content:
|
if "</think>" in content:
|
||||||
is_active = True
|
is_active = True
|
||||||
content = content.split('</think>')[-1]
|
content = content.split("</think>")[-1]
|
||||||
if is_active:
|
if is_active:
|
||||||
yield content
|
yield content
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
@@ -59,9 +65,13 @@ class LLMProvider(LLMProviderBase):
|
|||||||
|
|
||||||
def response_with_functions(self, session_id, dialogue, functions=None):
|
def response_with_functions(self, session_id, dialogue, functions=None):
|
||||||
try:
|
try:
|
||||||
logger.bind(tag=TAG).debug(f"Sending function call request to Xinference with model: {self.model_name}, dialogue length: {len(dialogue)}")
|
logger.bind(tag=TAG).debug(
|
||||||
|
f"Sending function call request to Xinference with model: {self.model_name}, dialogue length: {len(dialogue)}"
|
||||||
|
)
|
||||||
if functions:
|
if functions:
|
||||||
logger.bind(tag=TAG).debug(f"Function calls enabled with: {[f.get('function', {}).get('name') for f in functions]}")
|
logger.bind(tag=TAG).debug(
|
||||||
|
f"Function calls enabled with: {[f.get('function', {}).get('name') for f in functions]}"
|
||||||
|
)
|
||||||
|
|
||||||
stream = self.client.chat.completions.create(
|
stream = self.client.chat.completions.create(
|
||||||
model=self.model_name,
|
model=self.model_name,
|
||||||
@@ -82,4 +92,7 @@ class LLMProvider(LLMProviderBase):
|
|||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
logger.bind(tag=TAG).error(f"Error in Xinference function call: {e}")
|
logger.bind(tag=TAG).error(f"Error in Xinference function call: {e}")
|
||||||
yield {"type": "content", "content": f"【Xinference服务响应异常: {str(e)}】"}
|
yield {
|
||||||
|
"type": "content",
|
||||||
|
"content": f"【Xinference服务响应异常: {str(e)}】",
|
||||||
|
}
|
||||||
|
|||||||
@@ -9,7 +9,13 @@ class MemoryProviderBase(ABC):
|
|||||||
def __init__(self, config):
|
def __init__(self, config):
|
||||||
self.config = config
|
self.config = config
|
||||||
self.role_id = None
|
self.role_id = None
|
||||||
self.llm = None
|
|
||||||
|
def set_llm(self, llm):
|
||||||
|
self.llm = llm
|
||||||
|
# 获取模型名称和类型信息
|
||||||
|
model_name = getattr(llm, "model_name", str(llm.__class__.__name__))
|
||||||
|
# 记录更详细的日志
|
||||||
|
logger.bind(tag=TAG).info(f"记忆总结设置LLM: {model_name}")
|
||||||
|
|
||||||
@abstractmethod
|
@abstractmethod
|
||||||
async def save_memory(self, msgs):
|
async def save_memory(self, msgs):
|
||||||
|
|||||||
@@ -107,7 +107,7 @@ TAG = __name__
|
|||||||
class MemoryProvider(MemoryProviderBase):
|
class MemoryProvider(MemoryProviderBase):
|
||||||
def __init__(self, config, summary_memory):
|
def __init__(self, config, summary_memory):
|
||||||
super().__init__(config)
|
super().__init__(config)
|
||||||
self.short_momery = ""
|
self.short_memory = ""
|
||||||
self.save_to_file = True
|
self.save_to_file = True
|
||||||
self.memory_path = get_project_dir() + "data/.memory.yaml"
|
self.memory_path = get_project_dir() + "data/.memory.yaml"
|
||||||
self.load_memory(summary_memory)
|
self.load_memory(summary_memory)
|
||||||
@@ -122,7 +122,7 @@ class MemoryProvider(MemoryProviderBase):
|
|||||||
def load_memory(self, summary_memory):
|
def load_memory(self, summary_memory):
|
||||||
# api获取到总结记忆后直接返回
|
# api获取到总结记忆后直接返回
|
||||||
if summary_memory or not self.save_to_file:
|
if summary_memory or not self.save_to_file:
|
||||||
self.short_momery = summary_memory
|
self.short_memory = summary_memory
|
||||||
return
|
return
|
||||||
|
|
||||||
all_memory = {}
|
all_memory = {}
|
||||||
@@ -130,18 +130,21 @@ class MemoryProvider(MemoryProviderBase):
|
|||||||
with open(self.memory_path, "r", encoding="utf-8") as f:
|
with open(self.memory_path, "r", encoding="utf-8") as f:
|
||||||
all_memory = yaml.safe_load(f) or {}
|
all_memory = yaml.safe_load(f) or {}
|
||||||
if self.role_id in all_memory:
|
if self.role_id in all_memory:
|
||||||
self.short_momery = all_memory[self.role_id]
|
self.short_memory = all_memory[self.role_id]
|
||||||
|
|
||||||
def save_memory_to_file(self):
|
def save_memory_to_file(self):
|
||||||
all_memory = {}
|
all_memory = {}
|
||||||
if os.path.exists(self.memory_path):
|
if os.path.exists(self.memory_path):
|
||||||
with open(self.memory_path, "r", encoding="utf-8") as f:
|
with open(self.memory_path, "r", encoding="utf-8") as f:
|
||||||
all_memory = yaml.safe_load(f) or {}
|
all_memory = yaml.safe_load(f) or {}
|
||||||
all_memory[self.role_id] = self.short_momery
|
all_memory[self.role_id] = self.short_memory
|
||||||
with open(self.memory_path, "w", encoding="utf-8") as f:
|
with open(self.memory_path, "w", encoding="utf-8") as f:
|
||||||
yaml.dump(all_memory, f, allow_unicode=True)
|
yaml.dump(all_memory, f, allow_unicode=True)
|
||||||
|
|
||||||
async def save_memory(self, msgs):
|
async def save_memory(self, msgs):
|
||||||
|
# 打印使用的模型信息
|
||||||
|
model_info = getattr(self.llm, "model_name", str(self.llm.__class__.__name__))
|
||||||
|
logger.bind(tag=TAG).debug(f"使用记忆保存模型: {model_info}")
|
||||||
if self.llm is None:
|
if self.llm is None:
|
||||||
logger.bind(tag=TAG).error("LLM is not set for memory provider")
|
logger.bind(tag=TAG).error("LLM is not set for memory provider")
|
||||||
return None
|
return None
|
||||||
@@ -155,31 +158,39 @@ class MemoryProvider(MemoryProviderBase):
|
|||||||
msgStr += f"User: {msg.content}\n"
|
msgStr += f"User: {msg.content}\n"
|
||||||
elif msg.role == "assistant":
|
elif msg.role == "assistant":
|
||||||
msgStr += f"Assistant: {msg.content}\n"
|
msgStr += f"Assistant: {msg.content}\n"
|
||||||
if self.short_momery and len(self.short_momery) > 0:
|
if self.short_memory and len(self.short_memory) > 0:
|
||||||
msgStr += "历史记忆:\n"
|
msgStr += "历史记忆:\n"
|
||||||
msgStr += self.short_momery
|
msgStr += self.short_memory
|
||||||
|
|
||||||
# 当前时间
|
# 当前时间
|
||||||
time_str = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime())
|
time_str = time.strftime("%Y-%m-%d %H:%M:%S", time.localtime())
|
||||||
msgStr += f"当前时间:{time_str}"
|
msgStr += f"当前时间:{time_str}"
|
||||||
|
|
||||||
if self.save_to_file:
|
if self.save_to_file:
|
||||||
result = self.llm.response_no_stream(short_term_memory_prompt, msgStr)
|
result = self.llm.response_no_stream(
|
||||||
|
short_term_memory_prompt,
|
||||||
|
msgStr,
|
||||||
|
max_tokens=2000,
|
||||||
|
temperature=0.2,
|
||||||
|
)
|
||||||
json_str = extract_json_data(result)
|
json_str = extract_json_data(result)
|
||||||
try:
|
try:
|
||||||
json.loads(json_str) # 检查json格式是否正确
|
json.loads(json_str) # 检查json格式是否正确
|
||||||
self.short_momery = json_str
|
self.short_memory = json_str
|
||||||
self.save_memory_to_file()
|
self.save_memory_to_file()
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
print("Error:", e)
|
print("Error:", e)
|
||||||
else:
|
else:
|
||||||
result = self.llm.response_no_stream(
|
result = self.llm.response_no_stream(
|
||||||
short_term_memory_prompt_only_content, msgStr
|
short_term_memory_prompt_only_content,
|
||||||
|
msgStr,
|
||||||
|
max_tokens=2000,
|
||||||
|
temperature=0.2,
|
||||||
)
|
)
|
||||||
save_mem_local_short(self.role_id, result)
|
save_mem_local_short(self.role_id, result)
|
||||||
logger.bind(tag=TAG).info(f"Save memory successful - Role: {self.role_id}")
|
logger.bind(tag=TAG).info(f"Save memory successful - Role: {self.role_id}")
|
||||||
|
|
||||||
return self.short_momery
|
return self.short_memory
|
||||||
|
|
||||||
async def query_memory(self, query: str) -> str:
|
async def query_memory(self, query: str) -> str:
|
||||||
return self.short_momery
|
return self.short_memory
|
||||||
|
|||||||
@@ -1,4 +1,3 @@
|
|||||||
import os
|
|
||||||
import uuid
|
import uuid
|
||||||
import json
|
import json
|
||||||
import hmac
|
import hmac
|
||||||
@@ -8,61 +7,74 @@ import requests
|
|||||||
from datetime import datetime
|
from datetime import datetime
|
||||||
from core.providers.tts.base import TTSProviderBase
|
from core.providers.tts.base import TTSProviderBase
|
||||||
|
|
||||||
import http.client
|
|
||||||
import urllib.parse
|
|
||||||
import time
|
import time
|
||||||
import uuid
|
import uuid
|
||||||
from urllib import parse
|
from urllib import parse
|
||||||
|
|
||||||
|
|
||||||
class AccessToken:
|
class AccessToken:
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _encode_text(text):
|
def _encode_text(text):
|
||||||
encoded_text = parse.quote_plus(text)
|
encoded_text = parse.quote_plus(text)
|
||||||
return encoded_text.replace('+', '%20').replace('*', '%2A').replace('%7E', '~')
|
return encoded_text.replace("+", "%20").replace("*", "%2A").replace("%7E", "~")
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def _encode_dict(dic):
|
def _encode_dict(dic):
|
||||||
keys = dic.keys()
|
keys = dic.keys()
|
||||||
dic_sorted = [(key, dic[key]) for key in sorted(keys)]
|
dic_sorted = [(key, dic[key]) for key in sorted(keys)]
|
||||||
encoded_text = parse.urlencode(dic_sorted)
|
encoded_text = parse.urlencode(dic_sorted)
|
||||||
return encoded_text.replace('+', '%20').replace('*', '%2A').replace('%7E', '~')
|
return encoded_text.replace("+", "%20").replace("*", "%2A").replace("%7E", "~")
|
||||||
|
|
||||||
@staticmethod
|
@staticmethod
|
||||||
def create_token(access_key_id, access_key_secret):
|
def create_token(access_key_id, access_key_secret):
|
||||||
parameters = {'AccessKeyId': access_key_id,
|
parameters = {
|
||||||
'Action': 'CreateToken',
|
"AccessKeyId": access_key_id,
|
||||||
'Format': 'JSON',
|
"Action": "CreateToken",
|
||||||
'RegionId': 'cn-shanghai',
|
"Format": "JSON",
|
||||||
'SignatureMethod': 'HMAC-SHA1',
|
"RegionId": "cn-shanghai",
|
||||||
'SignatureNonce': str(uuid.uuid1()),
|
"SignatureMethod": "HMAC-SHA1",
|
||||||
'SignatureVersion': '1.0',
|
"SignatureNonce": str(uuid.uuid1()),
|
||||||
'Timestamp': time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
"SignatureVersion": "1.0",
|
||||||
'Version': '2019-02-28'}
|
"Timestamp": time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime()),
|
||||||
|
"Version": "2019-02-28",
|
||||||
|
}
|
||||||
# 构造规范化的请求字符串
|
# 构造规范化的请求字符串
|
||||||
query_string = AccessToken._encode_dict(parameters)
|
query_string = AccessToken._encode_dict(parameters)
|
||||||
# print('规范化的请求字符串: %s' % query_string)
|
# print('规范化的请求字符串: %s' % query_string)
|
||||||
# 构造待签名字符串
|
# 构造待签名字符串
|
||||||
string_to_sign = 'GET' + '&' + AccessToken._encode_text('/') + '&' + AccessToken._encode_text(query_string)
|
string_to_sign = (
|
||||||
|
"GET"
|
||||||
|
+ "&"
|
||||||
|
+ AccessToken._encode_text("/")
|
||||||
|
+ "&"
|
||||||
|
+ AccessToken._encode_text(query_string)
|
||||||
|
)
|
||||||
# print('待签名的字符串: %s' % string_to_sign)
|
# print('待签名的字符串: %s' % string_to_sign)
|
||||||
# 计算签名
|
# 计算签名
|
||||||
secreted_string = hmac.new(bytes(access_key_secret + '&', encoding='utf-8'),
|
secreted_string = hmac.new(
|
||||||
bytes(string_to_sign, encoding='utf-8'),
|
bytes(access_key_secret + "&", encoding="utf-8"),
|
||||||
hashlib.sha1).digest()
|
bytes(string_to_sign, encoding="utf-8"),
|
||||||
|
hashlib.sha1,
|
||||||
|
).digest()
|
||||||
signature = base64.b64encode(secreted_string)
|
signature = base64.b64encode(secreted_string)
|
||||||
# print('签名: %s' % signature)
|
# print('签名: %s' % signature)
|
||||||
# 进行URL编码
|
# 进行URL编码
|
||||||
signature = AccessToken._encode_text(signature)
|
signature = AccessToken._encode_text(signature)
|
||||||
# print('URL编码后的签名: %s' % signature)
|
# print('URL编码后的签名: %s' % signature)
|
||||||
# 调用服务
|
# 调用服务
|
||||||
full_url = 'http://nls-meta.cn-shanghai.aliyuncs.com/?Signature=%s&%s' % (signature, query_string)
|
full_url = "http://nls-meta.cn-shanghai.aliyuncs.com/?Signature=%s&%s" % (
|
||||||
|
signature,
|
||||||
|
query_string,
|
||||||
|
)
|
||||||
# print('url: %s' % full_url)
|
# print('url: %s' % full_url)
|
||||||
# 提交HTTP GET请求
|
# 提交HTTP GET请求
|
||||||
response = requests.get(full_url)
|
response = requests.get(full_url)
|
||||||
if response.ok:
|
if response.ok:
|
||||||
root_obj = response.json()
|
root_obj = response.json()
|
||||||
key = 'Token'
|
key = "Token"
|
||||||
if key in root_obj:
|
if key in root_obj:
|
||||||
token = root_obj[key]['Id']
|
token = root_obj[key]["Id"]
|
||||||
expire_time = root_obj[key]['ExpireTime']
|
expire_time = root_obj[key]["ExpireTime"]
|
||||||
return token, expire_time
|
return token, expire_time
|
||||||
# print(response.text)
|
# print(response.text)
|
||||||
return None, None
|
return None, None
|
||||||
@@ -70,7 +82,6 @@ class AccessToken:
|
|||||||
|
|
||||||
class TTSProvider(TTSProviderBase):
|
class TTSProvider(TTSProviderBase):
|
||||||
|
|
||||||
|
|
||||||
def __init__(self, config, delete_audio_file):
|
def __init__(self, config, delete_audio_file):
|
||||||
super().__init__(config, delete_audio_file)
|
super().__init__(config, delete_audio_file)
|
||||||
|
|
||||||
@@ -80,16 +91,27 @@ class TTSProvider(TTSProviderBase):
|
|||||||
|
|
||||||
self.appkey = config.get("appkey")
|
self.appkey = config.get("appkey")
|
||||||
self.format = config.get("format", "wav")
|
self.format = config.get("format", "wav")
|
||||||
self.sample_rate = config.get("sample_rate", 16000)
|
|
||||||
self.voice = config.get("voice", "xiaoyun")
|
sample_rate = config.get("sample_rate", "16000")
|
||||||
self.volume = config.get("volume", 50)
|
self.sample_rate = int(sample_rate) if sample_rate else 16000
|
||||||
self.speech_rate = config.get("speech_rate", 0)
|
|
||||||
self.pitch_rate = config.get("pitch_rate", 0)
|
if config.get("private_voice"):
|
||||||
|
self.voice = config.get("private_voice")
|
||||||
|
else:
|
||||||
|
self.voice = config.get("voice", "xiaoyun")
|
||||||
|
|
||||||
|
volume = config.get("volume", "50")
|
||||||
|
self.volume = int(volume) if volume else 50
|
||||||
|
|
||||||
|
speech_rate = config.get("speech_rate", "0")
|
||||||
|
self.speech_rate = int(speech_rate) if speech_rate else 0
|
||||||
|
|
||||||
|
pitch_rate = config.get("pitch_rate", "0")
|
||||||
|
self.pitch_rate = int(pitch_rate) if pitch_rate else 0
|
||||||
|
|
||||||
self.host = config.get("host", "nls-gateway-cn-shanghai.aliyuncs.com")
|
self.host = config.get("host", "nls-gateway-cn-shanghai.aliyuncs.com")
|
||||||
self.api_url = f"https://{self.host}/stream/v1/tts"
|
self.api_url = f"https://{self.host}/stream/v1/tts"
|
||||||
self.header = {
|
self.header = {"Content-Type": "application/json"}
|
||||||
"Content-Type": "application/json"
|
|
||||||
}
|
|
||||||
|
|
||||||
if self.access_key_id and self.access_key_secret:
|
if self.access_key_id and self.access_key_secret:
|
||||||
# 使用密钥对生成临时token
|
# 使用密钥对生成临时token
|
||||||
@@ -99,28 +121,23 @@ class TTSProvider(TTSProviderBase):
|
|||||||
self.token = config.get("token")
|
self.token = config.get("token")
|
||||||
self.expire_time = None
|
self.expire_time = None
|
||||||
|
|
||||||
|
|
||||||
def _refresh_token(self):
|
def _refresh_token(self):
|
||||||
"""刷新Token并记录过期时间"""
|
"""刷新Token并记录过期时间"""
|
||||||
if self.access_key_id and self.access_key_secret:
|
if self.access_key_id and self.access_key_secret:
|
||||||
self.token, expire_time_str = AccessToken.create_token(
|
self.token, expire_time_str = AccessToken.create_token(
|
||||||
self.access_key_id,
|
self.access_key_id, self.access_key_secret
|
||||||
self.access_key_secret
|
|
||||||
)
|
)
|
||||||
if not expire_time_str:
|
if not expire_time_str:
|
||||||
raise ValueError("无法获取有效的Token过期时间")
|
raise ValueError("无法获取有效的Token过期时间")
|
||||||
|
|
||||||
try:
|
try:
|
||||||
#统一转换为字符串处理
|
# 统一转换为字符串处理
|
||||||
expire_str = str(expire_time_str).strip()
|
expire_str = str(expire_time_str).strip()
|
||||||
|
|
||||||
if expire_str.isdigit():
|
if expire_str.isdigit():
|
||||||
expire_time = datetime.fromtimestamp(int(expire_str))
|
expire_time = datetime.fromtimestamp(int(expire_str))
|
||||||
else:
|
else:
|
||||||
expire_time = datetime.strptime(
|
expire_time = datetime.strptime(expire_str, "%Y-%m-%dT%H:%M:%SZ")
|
||||||
expire_str,
|
|
||||||
"%Y-%m-%dT%H:%M:%SZ"
|
|
||||||
)
|
|
||||||
self.expire_time = expire_time.timestamp() - 60
|
self.expire_time = expire_time.timestamp() - 60
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
raise ValueError(f"无效的过期时间格式: {expire_str}") from e
|
raise ValueError(f"无效的过期时间格式: {expire_str}") from e
|
||||||
@@ -142,8 +159,6 @@ class TTSProvider(TTSProviderBase):
|
|||||||
# f"过期时间 {datetime.fromtimestamp(self.expire_time)} | "
|
# f"过期时间 {datetime.fromtimestamp(self.expire_time)} | "
|
||||||
# f"剩余 {remaining:.2f}秒")
|
# f"剩余 {remaining:.2f}秒")
|
||||||
return time.time() > self.expire_time
|
return time.time() > self.expire_time
|
||||||
def generate_filename(self, extension=".wav"):
|
|
||||||
return os.path.join(self.output_file, f"tts-{__name__}{datetime.now().date()}@{uuid.uuid4().hex}{extension}")
|
|
||||||
|
|
||||||
async def text_to_speak(self, text, output_file):
|
async def text_to_speak(self, text, output_file):
|
||||||
if self._is_token_expired():
|
if self._is_token_expired():
|
||||||
@@ -158,21 +173,27 @@ class TTSProvider(TTSProviderBase):
|
|||||||
"voice": self.voice,
|
"voice": self.voice,
|
||||||
"volume": self.volume,
|
"volume": self.volume,
|
||||||
"speech_rate": self.speech_rate,
|
"speech_rate": self.speech_rate,
|
||||||
"pitch_rate": self.pitch_rate
|
"pitch_rate": self.pitch_rate,
|
||||||
}
|
}
|
||||||
|
|
||||||
# print(self.api_url, json.dumps(request_json, ensure_ascii=False))
|
# print(self.api_url, json.dumps(request_json, ensure_ascii=False))
|
||||||
try:
|
try:
|
||||||
resp = requests.post(self.api_url, json.dumps(request_json), headers=self.header)
|
resp = requests.post(
|
||||||
|
self.api_url, json.dumps(request_json), headers=self.header
|
||||||
|
)
|
||||||
if resp.status_code == 401: # Token过期特殊处理
|
if resp.status_code == 401: # Token过期特殊处理
|
||||||
self._refresh_token()
|
self._refresh_token()
|
||||||
resp = requests.post(self.api_url, json.dumps(request_json), headers=self.header)
|
resp = requests.post(
|
||||||
|
self.api_url, json.dumps(request_json), headers=self.header
|
||||||
|
)
|
||||||
# 检查返回请求数据的mime类型是否是audio/***,是则保存到指定路径下;返回的是binary格式的
|
# 检查返回请求数据的mime类型是否是audio/***,是则保存到指定路径下;返回的是binary格式的
|
||||||
if resp.headers['Content-Type'].startswith('audio/'):
|
if resp.headers["Content-Type"].startswith("audio/"):
|
||||||
with open(output_file, 'wb') as f:
|
with open(output_file, "wb") as f:
|
||||||
f.write(resp.content)
|
f.write(resp.content)
|
||||||
return output_file
|
return output_file
|
||||||
else:
|
else:
|
||||||
raise Exception(f"{__name__} status_code: {resp.status_code} response: {resp.content}")
|
raise Exception(
|
||||||
|
f"{__name__} status_code: {resp.status_code} response: {resp.content}"
|
||||||
|
)
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
raise Exception(f"{__name__} error: {e}")
|
raise Exception(f"{__name__} error: {e}")
|
||||||
|
|||||||
@@ -1,9 +1,27 @@
|
|||||||
import asyncio
|
|
||||||
from config.logger import setup_logging
|
|
||||||
import os
|
import os
|
||||||
|
import queue
|
||||||
|
import uuid
|
||||||
|
import asyncio
|
||||||
|
import threading
|
||||||
|
from core.utils import p3
|
||||||
|
from datetime import datetime
|
||||||
|
from core.utils import textUtils
|
||||||
from abc import ABC, abstractmethod
|
from abc import ABC, abstractmethod
|
||||||
from core.utils.tts import MarkdownCleaner
|
from config.logger import setup_logging
|
||||||
from core.utils.util import audio_to_data
|
from core.utils.util import audio_to_data
|
||||||
|
from core.utils.tts import MarkdownCleaner
|
||||||
|
from core.utils.output_counter import add_device_output
|
||||||
|
from core.handle.reportHandle import enqueue_tts_report
|
||||||
|
from core.handle.sendAudioHandle import sendAudioMessage
|
||||||
|
from core.providers.tts.dto.dto import (
|
||||||
|
TTSMessageDTO,
|
||||||
|
SentenceType,
|
||||||
|
ContentType,
|
||||||
|
InterfaceType,
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
import traceback
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
logger = setup_logging()
|
logger = setup_logging()
|
||||||
@@ -11,12 +29,51 @@ logger = setup_logging()
|
|||||||
|
|
||||||
class TTSProviderBase(ABC):
|
class TTSProviderBase(ABC):
|
||||||
def __init__(self, config, delete_audio_file):
|
def __init__(self, config, delete_audio_file):
|
||||||
|
self.interface_type = InterfaceType.NON_STREAM
|
||||||
|
self.conn = None
|
||||||
|
self.tts_timeout = 10
|
||||||
self.delete_audio_file = delete_audio_file
|
self.delete_audio_file = delete_audio_file
|
||||||
self.output_file = config.get("output_dir")
|
self.output_file = config.get("output_dir", "tmp/")
|
||||||
|
self.tts_text_queue = queue.Queue()
|
||||||
|
self.tts_audio_queue = queue.Queue()
|
||||||
|
self.tts_audio_first_sentence = True
|
||||||
|
|
||||||
@abstractmethod
|
self.tts_text_buff = []
|
||||||
def generate_filename(self):
|
self.punctuations = (
|
||||||
pass
|
"。",
|
||||||
|
".",
|
||||||
|
"?",
|
||||||
|
"?",
|
||||||
|
"!",
|
||||||
|
"!",
|
||||||
|
";",
|
||||||
|
";",
|
||||||
|
":",
|
||||||
|
)
|
||||||
|
self.first_sentence_punctuations = (
|
||||||
|
",",
|
||||||
|
"~",
|
||||||
|
"、",
|
||||||
|
",",
|
||||||
|
"。",
|
||||||
|
".",
|
||||||
|
"?",
|
||||||
|
"?",
|
||||||
|
"!",
|
||||||
|
"!",
|
||||||
|
";",
|
||||||
|
";",
|
||||||
|
":",
|
||||||
|
)
|
||||||
|
self.tts_stop_request = False
|
||||||
|
self.processed_chars = 0
|
||||||
|
self.is_first_sentence = True
|
||||||
|
|
||||||
|
def generate_filename(self, extension=".wav"):
|
||||||
|
return os.path.join(
|
||||||
|
self.output_file,
|
||||||
|
f"tts-{datetime.now().date()}@{uuid.uuid4().hex}{extension}",
|
||||||
|
)
|
||||||
|
|
||||||
def to_tts(self, text):
|
def to_tts(self, text):
|
||||||
tmp_file = self.generate_filename()
|
tmp_file = self.generate_filename()
|
||||||
@@ -60,3 +117,225 @@ class TTSProviderBase(ABC):
|
|||||||
def audio_to_opus_data(self, audio_file_path):
|
def audio_to_opus_data(self, audio_file_path):
|
||||||
"""音频文件转换为Opus编码"""
|
"""音频文件转换为Opus编码"""
|
||||||
return audio_to_data(audio_file_path, is_opus=True)
|
return audio_to_data(audio_file_path, is_opus=True)
|
||||||
|
|
||||||
|
def tts_one_sentence(
|
||||||
|
self,
|
||||||
|
conn,
|
||||||
|
content_type,
|
||||||
|
content_detail=None,
|
||||||
|
content_file=None,
|
||||||
|
sentence_id=None,
|
||||||
|
):
|
||||||
|
"""发送一句话"""
|
||||||
|
if not sentence_id:
|
||||||
|
if conn.sentence_id:
|
||||||
|
sentence_id = conn.sentence_id
|
||||||
|
else:
|
||||||
|
sentence_id = str(uuid.uuid4()).replace("-", "")
|
||||||
|
conn.sentence_id = sentence_id
|
||||||
|
self.tts_text_queue.put(
|
||||||
|
TTSMessageDTO(
|
||||||
|
sentence_id=sentence_id,
|
||||||
|
sentence_type=SentenceType.FIRST,
|
||||||
|
content_type=ContentType.ACTION,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
self.tts_text_queue.put(
|
||||||
|
TTSMessageDTO(
|
||||||
|
sentence_id=sentence_id,
|
||||||
|
sentence_type=SentenceType.MIDDLE,
|
||||||
|
content_type=content_type,
|
||||||
|
content_detail=content_detail,
|
||||||
|
content_file=content_file,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
self.tts_text_queue.put(
|
||||||
|
TTSMessageDTO(
|
||||||
|
sentence_id=sentence_id,
|
||||||
|
sentence_type=SentenceType.LAST,
|
||||||
|
content_type=ContentType.ACTION,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
async def open_audio_channels(self, conn):
|
||||||
|
self.conn = conn
|
||||||
|
self.tts_timeout = conn.config.get("tts_timeout", 10)
|
||||||
|
# tts 消化线程
|
||||||
|
self.tts_priority_thread = threading.Thread(
|
||||||
|
target=self.tts_text_priority_thread, daemon=True
|
||||||
|
)
|
||||||
|
self.tts_priority_thread.start()
|
||||||
|
|
||||||
|
# 音频播放 消化线程
|
||||||
|
self.audio_play_priority_thread = threading.Thread(
|
||||||
|
target=self._audio_play_priority_thread, daemon=True
|
||||||
|
)
|
||||||
|
self.audio_play_priority_thread.start()
|
||||||
|
|
||||||
|
# 这里默认是非流式的处理方式
|
||||||
|
# 流式处理方式请在子类中重写
|
||||||
|
def tts_text_priority_thread(self):
|
||||||
|
while not self.conn.stop_event.is_set():
|
||||||
|
try:
|
||||||
|
message = self.tts_text_queue.get(timeout=1)
|
||||||
|
if message.sentence_type == SentenceType.FIRST:
|
||||||
|
# 初始化参数
|
||||||
|
self.tts_stop_request = False
|
||||||
|
self.processed_chars = 0
|
||||||
|
self.tts_text_buff = []
|
||||||
|
self.is_first_sentence = True
|
||||||
|
self.tts_audio_first_sentence = True
|
||||||
|
elif ContentType.TEXT == message.content_type:
|
||||||
|
self.tts_text_buff.append(message.content_detail)
|
||||||
|
segment_text = self._get_segment_text()
|
||||||
|
if segment_text:
|
||||||
|
tts_file = self.to_tts(segment_text)
|
||||||
|
if tts_file:
|
||||||
|
audio_datas = self._process_audio_file(tts_file)
|
||||||
|
self.tts_audio_queue.put(
|
||||||
|
(message.sentence_type, audio_datas, segment_text)
|
||||||
|
)
|
||||||
|
elif ContentType.FILE == message.content_type:
|
||||||
|
self._process_remaining_text()
|
||||||
|
tts_file = message.content_file
|
||||||
|
if tts_file and os.path.exists(tts_file):
|
||||||
|
audio_datas = self._process_audio_file(tts_file)
|
||||||
|
self.tts_audio_queue.put(
|
||||||
|
(message.sentence_type, audio_datas, message.content_detail)
|
||||||
|
)
|
||||||
|
|
||||||
|
if message.sentence_type == SentenceType.LAST:
|
||||||
|
self._process_remaining_text()
|
||||||
|
self.tts_audio_queue.put(
|
||||||
|
(message.sentence_type, [], message.content_detail)
|
||||||
|
)
|
||||||
|
|
||||||
|
except queue.Empty:
|
||||||
|
continue
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(
|
||||||
|
f"处理TTS文本失败: {str(e)}, 类型: {type(e).__name__}, 堆栈: {traceback.format_exc()}"
|
||||||
|
)
|
||||||
|
continue
|
||||||
|
|
||||||
|
def _audio_play_priority_thread(self):
|
||||||
|
while not self.conn.stop_event.is_set():
|
||||||
|
text = None
|
||||||
|
try:
|
||||||
|
try:
|
||||||
|
sentence_type, audio_datas, text = self.tts_audio_queue.get(
|
||||||
|
timeout=1
|
||||||
|
)
|
||||||
|
except queue.Empty:
|
||||||
|
if self.conn.stop_event.is_set():
|
||||||
|
break
|
||||||
|
continue
|
||||||
|
future = asyncio.run_coroutine_threadsafe(
|
||||||
|
sendAudioMessage(self.conn, sentence_type, audio_datas, text),
|
||||||
|
self.conn.loop,
|
||||||
|
)
|
||||||
|
future.result()
|
||||||
|
if self.conn.max_output_size > 0 and text:
|
||||||
|
add_device_output(self.conn.headers.get("device-id"), len(text))
|
||||||
|
enqueue_tts_report(self.conn, text, audio_datas)
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(
|
||||||
|
f"audio_play_priority priority_thread: {text} {e}"
|
||||||
|
)
|
||||||
|
|
||||||
|
async def start_session(self, session_id):
|
||||||
|
pass
|
||||||
|
|
||||||
|
async def finish_session(self, session_id):
|
||||||
|
pass
|
||||||
|
|
||||||
|
async def close(self):
|
||||||
|
"""资源清理方法"""
|
||||||
|
if hasattr(self, "ws") and self.ws:
|
||||||
|
await self.ws.close()
|
||||||
|
|
||||||
|
def _get_segment_text(self):
|
||||||
|
# 合并当前全部文本并处理未分割部分
|
||||||
|
full_text = "".join(self.tts_text_buff)
|
||||||
|
current_text = full_text[self.processed_chars :] # 从未处理的位置开始
|
||||||
|
last_punct_pos = -1
|
||||||
|
|
||||||
|
# 根据是否是第一句话选择不同的标点符号集合
|
||||||
|
punctuations_to_use = (
|
||||||
|
self.first_sentence_punctuations
|
||||||
|
if self.is_first_sentence
|
||||||
|
else self.punctuations
|
||||||
|
)
|
||||||
|
|
||||||
|
for punct in punctuations_to_use:
|
||||||
|
pos = current_text.rfind(punct)
|
||||||
|
if (pos != -1 and last_punct_pos == -1) or (
|
||||||
|
pos != -1 and pos < last_punct_pos
|
||||||
|
):
|
||||||
|
last_punct_pos = pos
|
||||||
|
|
||||||
|
if last_punct_pos != -1:
|
||||||
|
segment_text_raw = current_text[: last_punct_pos + 1]
|
||||||
|
segment_text = textUtils.get_string_no_punctuation_or_emoji(
|
||||||
|
segment_text_raw
|
||||||
|
)
|
||||||
|
self.processed_chars += len(segment_text_raw) # 更新已处理字符位置
|
||||||
|
|
||||||
|
# 如果是第一句话,在找到第一个逗号后,将标志设置为False
|
||||||
|
if self.is_first_sentence:
|
||||||
|
self.is_first_sentence = False
|
||||||
|
|
||||||
|
return segment_text
|
||||||
|
elif self.tts_stop_request and current_text:
|
||||||
|
segment_text = current_text
|
||||||
|
self.is_first_sentence = True # 重置标志
|
||||||
|
return segment_text
|
||||||
|
else:
|
||||||
|
return None
|
||||||
|
|
||||||
|
def _process_audio_file(self, tts_file):
|
||||||
|
"""处理音频文件并转换为指定格式
|
||||||
|
|
||||||
|
Args:
|
||||||
|
tts_file: 音频文件路径
|
||||||
|
content_detail: 内容详情
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
tuple: (sentence_type, audio_datas, content_detail)
|
||||||
|
"""
|
||||||
|
audio_datas = []
|
||||||
|
if tts_file.endswith(".p3"):
|
||||||
|
audio_datas, _ = p3.decode_opus_from_file(tts_file)
|
||||||
|
elif self.conn.audio_format == "pcm":
|
||||||
|
audio_datas, _ = self.audio_to_pcm_data(tts_file)
|
||||||
|
else:
|
||||||
|
audio_datas, _ = self.audio_to_opus_data(tts_file)
|
||||||
|
|
||||||
|
if (
|
||||||
|
self.delete_audio_file
|
||||||
|
and tts_file is not None
|
||||||
|
and os.path.exists(tts_file)
|
||||||
|
and tts_file.startswith(self.output_file)
|
||||||
|
):
|
||||||
|
os.remove(tts_file)
|
||||||
|
return audio_datas
|
||||||
|
|
||||||
|
def _process_remaining_text(self):
|
||||||
|
"""处理剩余的文本并生成语音
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: 是否成功处理了文本
|
||||||
|
"""
|
||||||
|
full_text = "".join(self.tts_text_buff)
|
||||||
|
remaining_text = full_text[self.processed_chars :]
|
||||||
|
if remaining_text:
|
||||||
|
segment_text = textUtils.get_string_no_punctuation_or_emoji(remaining_text)
|
||||||
|
if segment_text:
|
||||||
|
tts_file = self.to_tts(segment_text)
|
||||||
|
audio_datas = self._process_audio_file(tts_file)
|
||||||
|
self.tts_audio_queue.put(
|
||||||
|
(SentenceType.MIDDLE, audio_datas, segment_text)
|
||||||
|
)
|
||||||
|
self.processed_chars += len(full_text)
|
||||||
|
return True
|
||||||
|
return False
|
||||||
|
|||||||
@@ -1,9 +1,4 @@
|
|||||||
import os
|
|
||||||
import uuid
|
|
||||||
import json
|
|
||||||
import base64
|
|
||||||
import requests
|
import requests
|
||||||
from datetime import datetime
|
|
||||||
from core.providers.tts.base import TTSProviderBase
|
from core.providers.tts.base import TTSProviderBase
|
||||||
|
|
||||||
|
|
||||||
@@ -21,12 +16,6 @@ class TTSProvider(TTSProviderBase):
|
|||||||
self.host = "api.coze.cn"
|
self.host = "api.coze.cn"
|
||||||
self.api_url = f"https://{self.host}/v1/audio/speech"
|
self.api_url = f"https://{self.host}/v1/audio/speech"
|
||||||
|
|
||||||
def generate_filename(self, extension=".wav"):
|
|
||||||
return os.path.join(
|
|
||||||
self.output_file,
|
|
||||||
f"tts-{datetime.now().date()}@{uuid.uuid4().hex}{extension}",
|
|
||||||
)
|
|
||||||
|
|
||||||
async def text_to_speak(self, text, output_file):
|
async def text_to_speak(self, text, output_file):
|
||||||
request_json = {
|
request_json = {
|
||||||
"model": self.model,
|
"model": self.model,
|
||||||
|
|||||||
@@ -0,0 +1,23 @@
|
|||||||
|
import os
|
||||||
|
from config.logger import setup_logging
|
||||||
|
from core.providers.tts.base import TTSProviderBase
|
||||||
|
|
||||||
|
TAG = __name__
|
||||||
|
logger = setup_logging()
|
||||||
|
|
||||||
|
|
||||||
|
class DefaultTTS(TTSProviderBase):
|
||||||
|
def __init__(self, config, delete_audio_file=True):
|
||||||
|
super().__init__(config, delete_audio_file)
|
||||||
|
self.output_dir = config.get("output_dir", "output")
|
||||||
|
if not os.path.exists(self.output_dir):
|
||||||
|
os.makedirs(self.output_dir)
|
||||||
|
|
||||||
|
def generate_filename(self):
|
||||||
|
"""生成唯一的音频文件名"""
|
||||||
|
import uuid
|
||||||
|
|
||||||
|
return os.path.join(self.output_dir, f"{uuid.uuid4()}.wav")
|
||||||
|
|
||||||
|
async def text_to_speak(self, text, output_file):
|
||||||
|
logger.bind(tag=TAG).error(f"无法实例化 TTS 服务,请检查配置")
|
||||||
@@ -1,9 +1,7 @@
|
|||||||
import os
|
|
||||||
import uuid
|
import uuid
|
||||||
import json
|
import json
|
||||||
import base64
|
import base64
|
||||||
import requests
|
import requests
|
||||||
from datetime import datetime
|
|
||||||
from core.utils.util import check_model_key
|
from core.utils.util import check_model_key
|
||||||
from core.providers.tts.base import TTSProviderBase
|
from core.providers.tts.base import TTSProviderBase
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
@@ -41,12 +39,6 @@ class TTSProvider(TTSProviderBase):
|
|||||||
self.header = {"Authorization": f"{self.authorization}{self.access_token}"}
|
self.header = {"Authorization": f"{self.authorization}{self.access_token}"}
|
||||||
check_model_key("TTS", self.access_token)
|
check_model_key("TTS", self.access_token)
|
||||||
|
|
||||||
def generate_filename(self, extension=".wav"):
|
|
||||||
return os.path.join(
|
|
||||||
self.output_file,
|
|
||||||
f"tts-{datetime.now().date()}@{uuid.uuid4().hex}{extension}",
|
|
||||||
)
|
|
||||||
|
|
||||||
async def text_to_speak(self, text, output_file):
|
async def text_to_speak(self, text, output_file):
|
||||||
request_json = {
|
request_json = {
|
||||||
"app": {
|
"app": {
|
||||||
|
|||||||
@@ -0,0 +1,43 @@
|
|||||||
|
from enum import Enum
|
||||||
|
from typing import Union, Optional
|
||||||
|
|
||||||
|
|
||||||
|
class SentenceType(Enum):
|
||||||
|
# 说话阶段
|
||||||
|
FIRST = "FIRST" # 首句话
|
||||||
|
MIDDLE = "MIDDLE" # 说话中
|
||||||
|
LAST = "LAST" # 最后一句
|
||||||
|
|
||||||
|
|
||||||
|
class ContentType(Enum):
|
||||||
|
# 内容类型
|
||||||
|
TEXT = "TEXT" # 文本内容
|
||||||
|
FILE = "FILE" # 文件内容
|
||||||
|
ACTION = "ACTION" # 动作内容
|
||||||
|
|
||||||
|
|
||||||
|
class InterfaceType(Enum):
|
||||||
|
# 接口类型
|
||||||
|
DUAL_STREAM = "DUAL_STREAM" # 双流式
|
||||||
|
SINGLE_STREAM = "SINGLE_STREAM" # 单流式
|
||||||
|
NON_STREAM = "NON_STREAM" # 非流式
|
||||||
|
|
||||||
|
|
||||||
|
class TTSMessageDTO:
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
sentence_id: str,
|
||||||
|
# 说话阶段
|
||||||
|
sentence_type: SentenceType,
|
||||||
|
# 内容类型
|
||||||
|
content_type: ContentType,
|
||||||
|
# 内容详情,一般是需要转换的文本或者音频的歌词
|
||||||
|
content_detail: Optional[str] = None,
|
||||||
|
# 如果内容类型为文件,则需要传入文件路径
|
||||||
|
content_file: Optional[str] = None,
|
||||||
|
):
|
||||||
|
self.sentence_id = sentence_id
|
||||||
|
self.sentence_type = sentence_type
|
||||||
|
self.content_type = content_type
|
||||||
|
self.content_detail = content_detail
|
||||||
|
self.content_file = content_file
|
||||||
@@ -1,12 +1,9 @@
|
|||||||
import base64
|
import base64
|
||||||
import os
|
|
||||||
import uuid
|
|
||||||
import requests
|
import requests
|
||||||
import ormsgpack
|
import ormsgpack
|
||||||
from pathlib import Path
|
from pathlib import Path
|
||||||
from pydantic import BaseModel, Field, conint, model_validator
|
from pydantic import BaseModel, Field, conint, model_validator
|
||||||
from typing_extensions import Annotated
|
from typing_extensions import Annotated
|
||||||
from datetime import datetime
|
|
||||||
from typing import Literal
|
from typing import Literal
|
||||||
from core.utils.util import check_model_key, parse_string_to_list
|
from core.utils.util import check_model_key, parse_string_to_list
|
||||||
from core.providers.tts.base import TTSProviderBase
|
from core.providers.tts.base import TTSProviderBase
|
||||||
@@ -133,12 +130,6 @@ class TTSProvider(TTSProviderBase):
|
|||||||
self.seed = int(config.get("seed")) if config.get("seed") else None
|
self.seed = int(config.get("seed")) if config.get("seed") else None
|
||||||
self.api_url = config.get("api_url", "http://127.0.0.1:8080/v1/tts")
|
self.api_url = config.get("api_url", "http://127.0.0.1:8080/v1/tts")
|
||||||
|
|
||||||
def generate_filename(self, extension=".wav"):
|
|
||||||
return os.path.join(
|
|
||||||
self.output_file,
|
|
||||||
f"tts-{datetime.now().date()}@{uuid.uuid4().hex}{extension}",
|
|
||||||
)
|
|
||||||
|
|
||||||
async def text_to_speak(self, text, output_file):
|
async def text_to_speak(self, text, output_file):
|
||||||
# Prepare reference data
|
# Prepare reference data
|
||||||
byte_audios = [audio_to_bytes(ref_audio) for ref_audio in self.reference_audio]
|
byte_audios = [audio_to_bytes(ref_audio) for ref_audio in self.reference_audio]
|
||||||
|
|||||||
@@ -1,10 +1,5 @@
|
|||||||
import os
|
|
||||||
import uuid
|
|
||||||
import json
|
|
||||||
import base64
|
|
||||||
import requests
|
import requests
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
from datetime import datetime
|
|
||||||
from core.providers.tts.base import TTSProviderBase
|
from core.providers.tts.base import TTSProviderBase
|
||||||
from core.utils.util import parse_string_to_list
|
from core.utils.util import parse_string_to_list
|
||||||
|
|
||||||
@@ -71,12 +66,6 @@ class TTSProvider(TTSProviderBase):
|
|||||||
config.get("aux_ref_audio_paths")
|
config.get("aux_ref_audio_paths")
|
||||||
)
|
)
|
||||||
|
|
||||||
def generate_filename(self, extension=".wav"):
|
|
||||||
return os.path.join(
|
|
||||||
self.output_file,
|
|
||||||
f"tts-{datetime.now().date()}@{uuid.uuid4().hex}{extension}",
|
|
||||||
)
|
|
||||||
|
|
||||||
async def text_to_speak(self, text, output_file):
|
async def text_to_speak(self, text, output_file):
|
||||||
request_json = {
|
request_json = {
|
||||||
"text": text,
|
"text": text,
|
||||||
|
|||||||
@@ -1,8 +1,5 @@
|
|||||||
import os
|
|
||||||
import uuid
|
|
||||||
import requests
|
import requests
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
from datetime import datetime
|
|
||||||
from core.providers.tts.base import TTSProviderBase
|
from core.providers.tts.base import TTSProviderBase
|
||||||
from core.utils.util import parse_string_to_list
|
from core.utils.util import parse_string_to_list
|
||||||
|
|
||||||
@@ -36,12 +33,6 @@ class TTSProvider(TTSProviderBase):
|
|||||||
self.inp_refs = parse_string_to_list(config.get("inp_refs"))
|
self.inp_refs = parse_string_to_list(config.get("inp_refs"))
|
||||||
self.if_sr = str(config.get("if_sr", False)).lower() in ("true", "1", "yes")
|
self.if_sr = str(config.get("if_sr", False)).lower() in ("true", "1", "yes")
|
||||||
|
|
||||||
def generate_filename(self, extension=".wav"):
|
|
||||||
return os.path.join(
|
|
||||||
self.output_file,
|
|
||||||
f"tts-{datetime.now().date()}@{uuid.uuid4().hex}{extension}",
|
|
||||||
)
|
|
||||||
|
|
||||||
async def text_to_speak(self, text, output_file):
|
async def text_to_speak(self, text, output_file):
|
||||||
request_params = {
|
request_params = {
|
||||||
"refer_wav_path": self.refer_wav_path,
|
"refer_wav_path": self.refer_wav_path,
|
||||||
@@ -67,4 +58,3 @@ class TTSProvider(TTSProviderBase):
|
|||||||
error_msg = f"GPT_SoVITS_V3 TTS请求失败: {resp.status_code} - {resp.text}"
|
error_msg = f"GPT_SoVITS_V3 TTS请求失败: {resp.status_code} - {resp.text}"
|
||||||
logger.bind(tag=TAG).error(error_msg)
|
logger.bind(tag=TAG).error(error_msg)
|
||||||
raise Exception(error_msg)
|
raise Exception(error_msg)
|
||||||
|
|
||||||
|
|||||||
@@ -0,0 +1,739 @@
|
|||||||
|
import os
|
||||||
|
import uuid
|
||||||
|
import json
|
||||||
|
import queue
|
||||||
|
import asyncio
|
||||||
|
import threading
|
||||||
|
import traceback
|
||||||
|
import websockets
|
||||||
|
import time
|
||||||
|
from config.logger import setup_logging
|
||||||
|
from core.utils import opus_encoder_utils
|
||||||
|
from core.utils.util import check_model_key
|
||||||
|
from core.providers.tts.base import TTSProviderBase
|
||||||
|
from core.providers.tts.dto.dto import SentenceType, ContentType, InterfaceType
|
||||||
|
|
||||||
|
TAG = __name__
|
||||||
|
logger = setup_logging()
|
||||||
|
|
||||||
|
PROTOCOL_VERSION = 0b0001
|
||||||
|
DEFAULT_HEADER_SIZE = 0b0001
|
||||||
|
|
||||||
|
# Message Type:
|
||||||
|
FULL_CLIENT_REQUEST = 0b0001
|
||||||
|
AUDIO_ONLY_RESPONSE = 0b1011
|
||||||
|
FULL_SERVER_RESPONSE = 0b1001
|
||||||
|
ERROR_INFORMATION = 0b1111
|
||||||
|
|
||||||
|
# Message Type Specific Flags
|
||||||
|
MsgTypeFlagNoSeq = 0b0000 # Non-terminal packet with no sequence
|
||||||
|
MsgTypeFlagPositiveSeq = 0b1 # Non-terminal packet with sequence > 0
|
||||||
|
MsgTypeFlagLastNoSeq = 0b10 # last packet with no sequence
|
||||||
|
MsgTypeFlagNegativeSeq = 0b11 # Payload contains event number (int32)
|
||||||
|
MsgTypeFlagWithEvent = 0b100
|
||||||
|
# Message Serialization
|
||||||
|
NO_SERIALIZATION = 0b0000
|
||||||
|
JSON = 0b0001
|
||||||
|
# Message Compression
|
||||||
|
COMPRESSION_NO = 0b0000
|
||||||
|
COMPRESSION_GZIP = 0b0001
|
||||||
|
|
||||||
|
EVENT_NONE = 0
|
||||||
|
EVENT_Start_Connection = 1
|
||||||
|
|
||||||
|
EVENT_FinishConnection = 2
|
||||||
|
|
||||||
|
EVENT_ConnectionStarted = 50 # 成功建连
|
||||||
|
|
||||||
|
EVENT_ConnectionFailed = 51 # 建连失败(可能是无法通过权限认证)
|
||||||
|
|
||||||
|
EVENT_ConnectionFinished = 52 # 连接结束
|
||||||
|
|
||||||
|
# 上行Session事件
|
||||||
|
EVENT_StartSession = 100
|
||||||
|
|
||||||
|
EVENT_FinishSession = 102
|
||||||
|
# 下行Session事件
|
||||||
|
EVENT_SessionStarted = 150
|
||||||
|
EVENT_SessionFinished = 152
|
||||||
|
|
||||||
|
EVENT_SessionFailed = 153
|
||||||
|
|
||||||
|
# 上行通用事件
|
||||||
|
EVENT_TaskRequest = 200
|
||||||
|
|
||||||
|
# 下行TTS事件
|
||||||
|
EVENT_TTSSentenceStart = 350
|
||||||
|
|
||||||
|
EVENT_TTSSentenceEnd = 351
|
||||||
|
|
||||||
|
EVENT_TTSResponse = 352
|
||||||
|
|
||||||
|
|
||||||
|
class Header:
|
||||||
|
def __init__(
|
||||||
|
self,
|
||||||
|
protocol_version=PROTOCOL_VERSION,
|
||||||
|
header_size=DEFAULT_HEADER_SIZE,
|
||||||
|
message_type: int = 0,
|
||||||
|
message_type_specific_flags: int = 0,
|
||||||
|
serial_method: int = NO_SERIALIZATION,
|
||||||
|
compression_type: int = COMPRESSION_NO,
|
||||||
|
reserved_data=0,
|
||||||
|
):
|
||||||
|
self.header_size = header_size
|
||||||
|
self.protocol_version = protocol_version
|
||||||
|
self.message_type = message_type
|
||||||
|
self.message_type_specific_flags = message_type_specific_flags
|
||||||
|
self.serial_method = serial_method
|
||||||
|
self.compression_type = compression_type
|
||||||
|
self.reserved_data = reserved_data
|
||||||
|
|
||||||
|
def as_bytes(self) -> bytes:
|
||||||
|
return bytes(
|
||||||
|
[
|
||||||
|
(self.protocol_version << 4) | self.header_size,
|
||||||
|
(self.message_type << 4) | self.message_type_specific_flags,
|
||||||
|
(self.serial_method << 4) | self.compression_type,
|
||||||
|
self.reserved_data,
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
|
||||||
|
class Optional:
|
||||||
|
def __init__(
|
||||||
|
self, event: int = EVENT_NONE, sessionId: str = None, sequence: int = None
|
||||||
|
):
|
||||||
|
self.event = event
|
||||||
|
self.sessionId = sessionId
|
||||||
|
self.errorCode: int = 0
|
||||||
|
self.connectionId: str | None = None
|
||||||
|
self.response_meta_json: str | None = None
|
||||||
|
self.sequence = sequence
|
||||||
|
|
||||||
|
# 转成 byte 序列
|
||||||
|
def as_bytes(self) -> bytes:
|
||||||
|
option_bytes = bytearray()
|
||||||
|
if self.event != EVENT_NONE:
|
||||||
|
option_bytes.extend(self.event.to_bytes(4, "big", signed=True))
|
||||||
|
if self.sessionId is not None:
|
||||||
|
session_id_bytes = str.encode(self.sessionId)
|
||||||
|
size = len(session_id_bytes).to_bytes(4, "big", signed=True)
|
||||||
|
option_bytes.extend(size)
|
||||||
|
option_bytes.extend(session_id_bytes)
|
||||||
|
if self.sequence is not None:
|
||||||
|
option_bytes.extend(self.sequence.to_bytes(4, "big", signed=True))
|
||||||
|
return option_bytes
|
||||||
|
|
||||||
|
|
||||||
|
class Response:
|
||||||
|
def __init__(self, header: Header, optional: Optional):
|
||||||
|
self.optional = optional
|
||||||
|
self.header = header
|
||||||
|
self.payload: bytes | None = None
|
||||||
|
|
||||||
|
def __str__(self):
|
||||||
|
return super().__str__()
|
||||||
|
|
||||||
|
|
||||||
|
class TTSProvider(TTSProviderBase):
|
||||||
|
def __init__(self, config, delete_audio_file):
|
||||||
|
super().__init__(config, delete_audio_file)
|
||||||
|
self.ws = None # 初始化ws属性
|
||||||
|
self.interface_type = InterfaceType.DUAL_STREAM
|
||||||
|
self.appId = config.get("appid")
|
||||||
|
self.access_token = config.get("access_token")
|
||||||
|
self.cluster = config.get("cluster")
|
||||||
|
self.resource_id = config.get("resource_id")
|
||||||
|
if config.get("private_voice"):
|
||||||
|
self.speaker = config.get("private_voice")
|
||||||
|
else:
|
||||||
|
self.speaker = config.get("speaker")
|
||||||
|
self.voice = config.get("voice")
|
||||||
|
self.ws_url = config.get("ws_url")
|
||||||
|
self.authorization = config.get("authorization")
|
||||||
|
self.header = {"Authorization": f"{self.authorization}{self.access_token}"}
|
||||||
|
self.enable_two_way = True
|
||||||
|
self.start_connection_flag = False
|
||||||
|
self.tts_text = ""
|
||||||
|
# 合成文字语音后,播放的音频文件列表
|
||||||
|
self.before_stop_play_files = []
|
||||||
|
self.opus_encoder = opus_encoder_utils.OpusEncoderUtils(
|
||||||
|
sample_rate=16000, channels=1, frame_size_ms=60
|
||||||
|
)
|
||||||
|
check_model_key("TTS", self.access_token)
|
||||||
|
|
||||||
|
# 添加会话状态控制
|
||||||
|
self._session_lock = asyncio.Lock() # 会话操作的并发锁
|
||||||
|
self._current_session_id = None # 当前会话ID
|
||||||
|
self._session_started = False # 会话是否已开始
|
||||||
|
self._session_finished = False # 会话是否已结束
|
||||||
|
self._connection_ready = False # 连接是否就绪
|
||||||
|
self._reconnect_attempts = 0 # 重连尝试次数
|
||||||
|
self._max_reconnect_attempts = 3 # 最大重连次数
|
||||||
|
|
||||||
|
###################################################################################
|
||||||
|
# 火山双流式TTS重写父类的方法--开始
|
||||||
|
###################################################################################
|
||||||
|
|
||||||
|
async def open_audio_channels(self, conn):
|
||||||
|
try:
|
||||||
|
await super().open_audio_channels(conn)
|
||||||
|
await self._ensure_connection()
|
||||||
|
tts_priority = threading.Thread(
|
||||||
|
target=self._start_monitor_tts_response_thread, daemon=True
|
||||||
|
)
|
||||||
|
tts_priority.start()
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"Failed to open audio channels: {str(e)}")
|
||||||
|
self.ws = None
|
||||||
|
raise
|
||||||
|
|
||||||
|
async def _ensure_connection(self):
|
||||||
|
"""确保WebSocket连接可用"""
|
||||||
|
try:
|
||||||
|
if self.ws is None:
|
||||||
|
logger.bind(tag=TAG).info("WebSocket连接不存在,开始建立新连接...")
|
||||||
|
ws_header = {
|
||||||
|
"X-Api-App-Key": self.appId,
|
||||||
|
"X-Api-Access-Key": self.access_token,
|
||||||
|
"X-Api-Resource-Id": self.resource_id,
|
||||||
|
"X-Api-Connect-Id": uuid.uuid4(),
|
||||||
|
}
|
||||||
|
self.ws = await websockets.connect(
|
||||||
|
self.ws_url, additional_headers=ws_header, max_size=1000000000
|
||||||
|
)
|
||||||
|
self._connection_ready = True
|
||||||
|
self._reconnect_attempts = 0
|
||||||
|
logger.bind(tag=TAG).info("WebSocket连接建立成功")
|
||||||
|
else:
|
||||||
|
# 尝试发送ping来检查连接是否还活着
|
||||||
|
try:
|
||||||
|
logger.bind(tag=TAG).debug("检查WebSocket连接状态...")
|
||||||
|
pong_waiter = await self.ws.ping()
|
||||||
|
await asyncio.wait_for(pong_waiter, timeout=1.0)
|
||||||
|
logger.bind(tag=TAG).debug("WebSocket连接状态正常")
|
||||||
|
except (asyncio.TimeoutError, websockets.ConnectionClosed):
|
||||||
|
# 如果ping失败,重新建立连接
|
||||||
|
logger.bind(tag=TAG).warning("WebSocket连接已断开,准备重新连接...")
|
||||||
|
try:
|
||||||
|
await self.ws.close()
|
||||||
|
except:
|
||||||
|
pass
|
||||||
|
self.ws = None
|
||||||
|
self._connection_ready = False
|
||||||
|
# 重新建立连接
|
||||||
|
await self._ensure_connection()
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"确保连接失败: {str(e)}")
|
||||||
|
self._connection_ready = False
|
||||||
|
self.ws = None
|
||||||
|
raise
|
||||||
|
|
||||||
|
def tts_text_priority_thread(self):
|
||||||
|
logger.bind(tag=TAG).info("TTS文本处理线程启动")
|
||||||
|
while not self.conn.stop_event.is_set():
|
||||||
|
try:
|
||||||
|
logger.bind(tag=TAG).debug("等待TTS文本队列消息...")
|
||||||
|
message = self.tts_text_queue.get(timeout=1)
|
||||||
|
logger.bind(tag=TAG).info(
|
||||||
|
f"收到TTS任务|{message.sentence_type.name} | {message.content_type.name} | 会话ID: {self.conn.sentence_id}"
|
||||||
|
)
|
||||||
|
if message.sentence_type == SentenceType.FIRST:
|
||||||
|
# 初始化参数
|
||||||
|
try:
|
||||||
|
logger.bind(tag=TAG).info("开始启动TTS会话...")
|
||||||
|
future = asyncio.run_coroutine_threadsafe(
|
||||||
|
self.start_session(self.conn.sentence_id),
|
||||||
|
loop=self.conn.loop,
|
||||||
|
)
|
||||||
|
future.result()
|
||||||
|
self.tts_audio_first_sentence = True
|
||||||
|
self.before_stop_play_files.clear()
|
||||||
|
logger.bind(tag=TAG).info("TTS会话启动成功")
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"启动TTS会话失败: {str(e)}")
|
||||||
|
# 直接跳过当前消息,不重新入队
|
||||||
|
time.sleep(1)
|
||||||
|
continue
|
||||||
|
elif ContentType.TEXT == message.content_type:
|
||||||
|
if message.content_detail:
|
||||||
|
try:
|
||||||
|
logger.bind(tag=TAG).info(
|
||||||
|
f"开始发送TTS文本: {message.content_detail}"
|
||||||
|
)
|
||||||
|
future = asyncio.run_coroutine_threadsafe(
|
||||||
|
self.text_to_speak(message.content_detail, None),
|
||||||
|
loop=self.conn.loop,
|
||||||
|
)
|
||||||
|
future.result()
|
||||||
|
logger.bind(tag=TAG).info("TTS文本发送成功")
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"发送TTS文本失败: {str(e)}")
|
||||||
|
# 直接跳过当前消息,不重新入队
|
||||||
|
time.sleep(1)
|
||||||
|
continue
|
||||||
|
elif ContentType.FILE == message.content_type:
|
||||||
|
logger.bind(tag=TAG).info(
|
||||||
|
f"添加音频文件到待播放列表: {message.content_file}"
|
||||||
|
)
|
||||||
|
self.before_stop_play_files.append(
|
||||||
|
(message.content_file, message.content_detail)
|
||||||
|
)
|
||||||
|
|
||||||
|
if message.sentence_type == SentenceType.LAST:
|
||||||
|
try:
|
||||||
|
logger.bind(tag=TAG).info("开始结束TTS会话...")
|
||||||
|
future = asyncio.run_coroutine_threadsafe(
|
||||||
|
self.finish_session(self.conn.sentence_id),
|
||||||
|
loop=self.conn.loop,
|
||||||
|
)
|
||||||
|
future.result()
|
||||||
|
logger.bind(tag=TAG).info("TTS会话结束成功")
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"结束TTS会话失败: {str(e)}")
|
||||||
|
# 直接跳过当前消息,不重新入队
|
||||||
|
time.sleep(1)
|
||||||
|
continue
|
||||||
|
|
||||||
|
except queue.Empty:
|
||||||
|
continue
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(
|
||||||
|
f"处理TTS文本失败: {str(e)}, 类型: {type(e).__name__}, 堆栈: {traceback.format_exc()}"
|
||||||
|
)
|
||||||
|
# 如果是WebSocket连接关闭错误,等待一段时间后继续
|
||||||
|
if "non-exist session" in str(e):
|
||||||
|
time.sleep(1)
|
||||||
|
continue
|
||||||
|
|
||||||
|
async def text_to_speak(self, text, _):
|
||||||
|
"""发送文本到TTS服务"""
|
||||||
|
try:
|
||||||
|
# 确保WebSocket连接可用
|
||||||
|
if not self._connection_ready or self.ws is None:
|
||||||
|
logger.bind(tag=TAG).warning("WebSocket连接不可用,尝试重新连接...")
|
||||||
|
await self._ensure_connection()
|
||||||
|
|
||||||
|
# 发送文本
|
||||||
|
await self.send_text(self.speaker, text, self.conn.sentence_id)
|
||||||
|
return
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"发送TTS文本失败: {str(e)}")
|
||||||
|
# 如果是连接问题,尝试重新连接
|
||||||
|
if isinstance(e, websockets.ConnectionClosed):
|
||||||
|
self._connection_ready = False
|
||||||
|
self.ws = None
|
||||||
|
await self._handle_connection_error()
|
||||||
|
raise
|
||||||
|
|
||||||
|
###################################################################################
|
||||||
|
# 火山双流式TTS重写父类的方法--结束
|
||||||
|
###################################################################################
|
||||||
|
def _start_monitor_tts_response_thread(self):
|
||||||
|
# 初始化链接
|
||||||
|
asyncio.run_coroutine_threadsafe(
|
||||||
|
self._start_monitor_tts_response(), loop=self.conn.loop
|
||||||
|
)
|
||||||
|
|
||||||
|
async def _start_monitor_tts_response(self):
|
||||||
|
opus_datas_cache = []
|
||||||
|
# 添加标志来区分是否是第一句话
|
||||||
|
is_first_sentence = True
|
||||||
|
while not self.conn.stop_event.is_set():
|
||||||
|
try:
|
||||||
|
# 确保 `recv()` 运行在同一个 event loop
|
||||||
|
msg = await self.ws.recv()
|
||||||
|
res = self.parser_response(msg)
|
||||||
|
self.print_response(res, "send_text res:")
|
||||||
|
|
||||||
|
if res.optional.event == EVENT_TTSSentenceStart:
|
||||||
|
json_data = json.loads(res.payload.decode("utf-8"))
|
||||||
|
self.tts_text = json_data.get("text", "")
|
||||||
|
logger.bind(tag=TAG).debug(f"句子语音生成开始: {self.tts_text}")
|
||||||
|
self.tts_audio_queue.put((SentenceType.FIRST, [], self.tts_text))
|
||||||
|
opus_datas_cache = []
|
||||||
|
elif (
|
||||||
|
res.optional.event == EVENT_TTSResponse
|
||||||
|
and res.header.message_type == AUDIO_ONLY_RESPONSE
|
||||||
|
):
|
||||||
|
logger.bind(tag=TAG).debug(f"推送数据到队列里面~~")
|
||||||
|
opus_datas = self.wav_to_opus_data_audio_raw(res.payload)
|
||||||
|
logger.bind(tag=TAG).debug(
|
||||||
|
f"推送数据到队列里面帧数~~{len(opus_datas)}"
|
||||||
|
)
|
||||||
|
if is_first_sentence:
|
||||||
|
# 第一句话直接发送
|
||||||
|
self.tts_audio_queue.put(
|
||||||
|
(SentenceType.MIDDLE, opus_datas, self.tts_text)
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
# 后续句子缓存
|
||||||
|
opus_datas_cache = opus_datas_cache + opus_datas
|
||||||
|
elif res.optional.event == EVENT_TTSSentenceEnd:
|
||||||
|
logger.bind(tag=TAG).info(f"句子语音生成成功:{self.tts_text}")
|
||||||
|
if not is_first_sentence:
|
||||||
|
# 只有非第一句话才发送缓存的数据
|
||||||
|
self.tts_audio_queue.put(
|
||||||
|
(SentenceType.MIDDLE, opus_datas_cache, self.tts_text)
|
||||||
|
)
|
||||||
|
# 第一句话结束后,将标志设置为False
|
||||||
|
is_first_sentence = False
|
||||||
|
elif res.optional.event == EVENT_SessionFinished:
|
||||||
|
logger.bind(tag=TAG).debug(f"会话结束~~")
|
||||||
|
for tts_file, text in self.before_stop_play_files:
|
||||||
|
if tts_file and os.path.exists(tts_file):
|
||||||
|
audio_datas = self._process_audio_file(tts_file)
|
||||||
|
self.tts_audio_queue.put(
|
||||||
|
(SentenceType.MIDDLE, audio_datas, text)
|
||||||
|
)
|
||||||
|
self.before_stop_play_files.clear()
|
||||||
|
self.tts_audio_queue.put((SentenceType.LAST, [], None))
|
||||||
|
|
||||||
|
opus_datas_cache = []
|
||||||
|
is_first_sentence = True
|
||||||
|
continue
|
||||||
|
except websockets.ConnectionClosed:
|
||||||
|
break # 连接关闭时退出监听
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"Error in _start_monitor_tts_response: {e}")
|
||||||
|
traceback.print_exc()
|
||||||
|
continue
|
||||||
|
|
||||||
|
async def send_event(
|
||||||
|
self, header: bytes, optional: bytes | None = None, payload: bytes = None
|
||||||
|
):
|
||||||
|
try:
|
||||||
|
full_client_request = bytearray(header)
|
||||||
|
if optional is not None:
|
||||||
|
full_client_request.extend(optional)
|
||||||
|
if payload is not None:
|
||||||
|
payload_size = len(payload).to_bytes(4, "big", signed=True)
|
||||||
|
full_client_request.extend(payload_size)
|
||||||
|
full_client_request.extend(payload)
|
||||||
|
await self.ws.send(full_client_request)
|
||||||
|
except websockets.ConnectionClosed:
|
||||||
|
if await self._handle_connection_error():
|
||||||
|
# 重连成功后重试发送
|
||||||
|
await self.ws.send(full_client_request)
|
||||||
|
else:
|
||||||
|
raise
|
||||||
|
|
||||||
|
async def send_text(self, speaker: str, text: str, session_id):
|
||||||
|
header = Header(
|
||||||
|
message_type=FULL_CLIENT_REQUEST,
|
||||||
|
message_type_specific_flags=MsgTypeFlagWithEvent,
|
||||||
|
serial_method=JSON,
|
||||||
|
).as_bytes()
|
||||||
|
optional = Optional(event=EVENT_TaskRequest, sessionId=session_id).as_bytes()
|
||||||
|
payload = self.get_payload_bytes(
|
||||||
|
event=EVENT_TaskRequest, text=text, speaker=speaker
|
||||||
|
)
|
||||||
|
return await self.send_event(header, optional, payload)
|
||||||
|
|
||||||
|
# 读取 res 数组某段 字符串内容
|
||||||
|
def read_res_content(self, res: bytes, offset: int):
|
||||||
|
content_size = int.from_bytes(res[offset : offset + 4], "big", signed=True)
|
||||||
|
offset += 4
|
||||||
|
content = str(res[offset : offset + content_size])
|
||||||
|
offset += content_size
|
||||||
|
return content, offset
|
||||||
|
|
||||||
|
# 读取 payload
|
||||||
|
def read_res_payload(self, res: bytes, offset: int):
|
||||||
|
payload_size = int.from_bytes(res[offset : offset + 4], "big", signed=True)
|
||||||
|
offset += 4
|
||||||
|
payload = res[offset : offset + payload_size]
|
||||||
|
offset += payload_size
|
||||||
|
return payload, offset
|
||||||
|
|
||||||
|
def parser_response(self, res) -> Response:
|
||||||
|
if isinstance(res, str):
|
||||||
|
raise RuntimeError(res)
|
||||||
|
response = Response(Header(), Optional())
|
||||||
|
# 解析结果
|
||||||
|
# header
|
||||||
|
header = response.header
|
||||||
|
num = 0b00001111
|
||||||
|
header.protocol_version = res[0] >> 4 & num
|
||||||
|
header.header_size = res[0] & 0x0F
|
||||||
|
header.message_type = (res[1] >> 4) & num
|
||||||
|
header.message_type_specific_flags = res[1] & 0x0F
|
||||||
|
header.serialization_method = res[2] >> num
|
||||||
|
header.message_compression = res[2] & 0x0F
|
||||||
|
header.reserved = res[3]
|
||||||
|
#
|
||||||
|
offset = 4
|
||||||
|
optional = response.optional
|
||||||
|
if header.message_type == FULL_SERVER_RESPONSE or AUDIO_ONLY_RESPONSE:
|
||||||
|
# read event
|
||||||
|
if header.message_type_specific_flags == MsgTypeFlagWithEvent:
|
||||||
|
optional.event = int.from_bytes(res[offset:8], "big", signed=True)
|
||||||
|
offset += 4
|
||||||
|
if optional.event == EVENT_NONE:
|
||||||
|
return response
|
||||||
|
# read connectionId
|
||||||
|
elif optional.event == EVENT_ConnectionStarted:
|
||||||
|
optional.connectionId, offset = self.read_res_content(res, offset)
|
||||||
|
elif optional.event == EVENT_ConnectionFailed:
|
||||||
|
optional.response_meta_json, offset = self.read_res_content(
|
||||||
|
res, offset
|
||||||
|
)
|
||||||
|
elif (
|
||||||
|
optional.event == EVENT_SessionStarted
|
||||||
|
or optional.event == EVENT_SessionFailed
|
||||||
|
or optional.event == EVENT_SessionFinished
|
||||||
|
):
|
||||||
|
optional.sessionId, offset = self.read_res_content(res, offset)
|
||||||
|
optional.response_meta_json, offset = self.read_res_content(
|
||||||
|
res, offset
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
optional.sessionId, offset = self.read_res_content(res, offset)
|
||||||
|
response.payload, offset = self.read_res_payload(res, offset)
|
||||||
|
|
||||||
|
elif header.message_type == ERROR_INFORMATION:
|
||||||
|
optional.errorCode = int.from_bytes(
|
||||||
|
res[offset : offset + 4], "big", signed=True
|
||||||
|
)
|
||||||
|
offset += 4
|
||||||
|
response.payload, offset = self.read_res_payload(res, offset)
|
||||||
|
return response
|
||||||
|
|
||||||
|
async def start_connection(self):
|
||||||
|
header = Header(
|
||||||
|
message_type=FULL_CLIENT_REQUEST,
|
||||||
|
message_type_specific_flags=MsgTypeFlagWithEvent,
|
||||||
|
).as_bytes()
|
||||||
|
optional = Optional(event=EVENT_Start_Connection).as_bytes()
|
||||||
|
payload = str.encode("{}")
|
||||||
|
return await self.send_event(header, optional, payload)
|
||||||
|
|
||||||
|
def print_response(self, res, tag_msg: str):
|
||||||
|
logger.bind(tag=TAG).debug(f"===>{tag_msg} header:{res.header.__dict__}")
|
||||||
|
logger.bind(tag=TAG).debug(f"===>{tag_msg} optional:{res.optional.__dict__}")
|
||||||
|
|
||||||
|
def get_payload_bytes(
|
||||||
|
self,
|
||||||
|
uid="1234",
|
||||||
|
event=EVENT_NONE,
|
||||||
|
text="",
|
||||||
|
speaker="",
|
||||||
|
audio_format="pcm",
|
||||||
|
audio_sample_rate=16000,
|
||||||
|
):
|
||||||
|
return str.encode(
|
||||||
|
json.dumps(
|
||||||
|
{
|
||||||
|
"user": {"uid": uid},
|
||||||
|
"event": event,
|
||||||
|
"namespace": "BidirectionalTTS",
|
||||||
|
"req_params": {
|
||||||
|
"text": text,
|
||||||
|
"speaker": speaker,
|
||||||
|
"audio_params": {
|
||||||
|
"format": audio_format,
|
||||||
|
"sample_rate": audio_sample_rate,
|
||||||
|
},
|
||||||
|
},
|
||||||
|
}
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
|
async def finish_connection(self):
|
||||||
|
header = Header(
|
||||||
|
message_type=FULL_CLIENT_REQUEST,
|
||||||
|
message_type_specific_flags=MsgTypeFlagWithEvent,
|
||||||
|
serial_method=JSON,
|
||||||
|
).as_bytes()
|
||||||
|
optional = Optional(event=EVENT_FinishConnection).as_bytes()
|
||||||
|
payload = str.encode("{}")
|
||||||
|
await self.send_event(header, optional, payload)
|
||||||
|
return
|
||||||
|
|
||||||
|
async def start_session(self, session_id):
|
||||||
|
logger.bind(tag=TAG).info(f"开始会话~~{session_id}")
|
||||||
|
try:
|
||||||
|
async with self._session_lock:
|
||||||
|
try:
|
||||||
|
# 确保连接可用
|
||||||
|
logger.bind(tag=TAG).info("检查WebSocket连接状态...")
|
||||||
|
await asyncio.wait_for(self._ensure_connection(), timeout=5)
|
||||||
|
|
||||||
|
# 如果已有会话未结束,先关闭它
|
||||||
|
if self._session_started and not self._session_finished:
|
||||||
|
logger.bind(tag=TAG).warning(
|
||||||
|
f"发现未关闭的会话 {self._current_session_id},正在关闭..."
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
await asyncio.wait_for(
|
||||||
|
self.finish_session(self._current_session_id), timeout=5
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"关闭旧会话失败: {str(e)}")
|
||||||
|
# 强制重置会话状态
|
||||||
|
self._session_started = False
|
||||||
|
self._session_finished = True
|
||||||
|
self._current_session_id = None
|
||||||
|
|
||||||
|
# 重置会话状态
|
||||||
|
self._current_session_id = session_id
|
||||||
|
self._session_started = True
|
||||||
|
self._session_finished = False
|
||||||
|
logger.bind(tag=TAG).info(
|
||||||
|
f"会话状态已更新 - 开始: {self._session_started}, 结束: {self._session_finished}"
|
||||||
|
)
|
||||||
|
|
||||||
|
header = Header(
|
||||||
|
message_type=FULL_CLIENT_REQUEST,
|
||||||
|
message_type_specific_flags=MsgTypeFlagWithEvent,
|
||||||
|
serial_method=JSON,
|
||||||
|
).as_bytes()
|
||||||
|
optional = Optional(
|
||||||
|
event=EVENT_StartSession, sessionId=session_id
|
||||||
|
).as_bytes()
|
||||||
|
payload = self.get_payload_bytes(
|
||||||
|
event=EVENT_StartSession, speaker=self.speaker
|
||||||
|
)
|
||||||
|
await asyncio.wait_for(
|
||||||
|
self.send_event(header, optional, payload), timeout=5
|
||||||
|
)
|
||||||
|
logger.bind(tag=TAG).info("会话启动请求已发送")
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"启动会话失败: {str(e)}")
|
||||||
|
self._session_started = False
|
||||||
|
self._session_finished = True
|
||||||
|
self._current_session_id = None
|
||||||
|
raise
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
logger.bind(tag=TAG).error(f"启动会话超时: {session_id}")
|
||||||
|
# 超时后强制重置会话状态
|
||||||
|
self._session_started = False
|
||||||
|
self._session_finished = True
|
||||||
|
self._current_session_id = None
|
||||||
|
# 尝试关闭WebSocket连接
|
||||||
|
if self.ws:
|
||||||
|
try:
|
||||||
|
await self.ws.close()
|
||||||
|
except:
|
||||||
|
pass
|
||||||
|
self.ws = None
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"启动会话时发生未知错误: {str(e)}")
|
||||||
|
# 发生未知错误时也重置会话状态
|
||||||
|
self._session_started = False
|
||||||
|
self._session_finished = True
|
||||||
|
self._current_session_id = None
|
||||||
|
|
||||||
|
async def finish_session(self, session_id):
|
||||||
|
logger.bind(tag=TAG).info(f"关闭会话~~{session_id}")
|
||||||
|
try:
|
||||||
|
async with self._session_lock:
|
||||||
|
try:
|
||||||
|
# 检查会话状态
|
||||||
|
if not self._session_started:
|
||||||
|
logger.bind(tag=TAG).warning(
|
||||||
|
f"尝试关闭未开始的会话 {session_id}"
|
||||||
|
)
|
||||||
|
return
|
||||||
|
|
||||||
|
if self._session_finished:
|
||||||
|
logger.bind(tag=TAG).warning(f"会话 {session_id} 已经关闭")
|
||||||
|
return
|
||||||
|
|
||||||
|
if self._current_session_id != session_id:
|
||||||
|
logger.bind(tag=TAG).warning(
|
||||||
|
f"尝试关闭错误的会话 {session_id},当前会话为 {self._current_session_id}"
|
||||||
|
)
|
||||||
|
# 即使会话ID不匹配,也尝试关闭当前会话
|
||||||
|
if self._current_session_id:
|
||||||
|
session_id = self._current_session_id
|
||||||
|
|
||||||
|
# 确保WebSocket连接可用
|
||||||
|
if self.ws is None:
|
||||||
|
logger.bind(tag=TAG).warning(
|
||||||
|
"WebSocket连接不存在,尝试重新连接..."
|
||||||
|
)
|
||||||
|
await asyncio.wait_for(self._ensure_connection(), timeout=5)
|
||||||
|
|
||||||
|
header = Header(
|
||||||
|
message_type=FULL_CLIENT_REQUEST,
|
||||||
|
message_type_specific_flags=MsgTypeFlagWithEvent,
|
||||||
|
serial_method=JSON,
|
||||||
|
).as_bytes()
|
||||||
|
optional = Optional(
|
||||||
|
event=EVENT_FinishSession, sessionId=session_id
|
||||||
|
).as_bytes()
|
||||||
|
payload = str.encode("{}")
|
||||||
|
await asyncio.wait_for(
|
||||||
|
self.send_event(header, optional, payload), timeout=5
|
||||||
|
)
|
||||||
|
logger.bind(tag=TAG).info("会话结束请求已发送")
|
||||||
|
|
||||||
|
# 更新会话状态
|
||||||
|
self._session_finished = True
|
||||||
|
self._session_started = False
|
||||||
|
self._current_session_id = None
|
||||||
|
logger.bind(tag=TAG).info(
|
||||||
|
"会话状态已更新 - 开始: False, 结束: True"
|
||||||
|
)
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"关闭会话失败: {str(e)}")
|
||||||
|
# 即使发生错误,也要重置会话状态
|
||||||
|
self._session_finished = True
|
||||||
|
self._session_started = False
|
||||||
|
self._current_session_id = None
|
||||||
|
raise
|
||||||
|
except asyncio.TimeoutError:
|
||||||
|
logger.bind(tag=TAG).error(f"关闭会话超时: {session_id}")
|
||||||
|
# 超时后强制重置会话状态
|
||||||
|
self._session_finished = True
|
||||||
|
self._session_started = False
|
||||||
|
self._current_session_id = None
|
||||||
|
# 尝试关闭WebSocket连接
|
||||||
|
if self.ws:
|
||||||
|
try:
|
||||||
|
await self.ws.close()
|
||||||
|
except:
|
||||||
|
pass
|
||||||
|
self.ws = None
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"关闭会话时发生未知错误: {str(e)}")
|
||||||
|
# 发生未知错误时也重置会话状态
|
||||||
|
self._session_finished = True
|
||||||
|
self._session_started = False
|
||||||
|
self._current_session_id = None
|
||||||
|
|
||||||
|
async def reset(self):
|
||||||
|
# 关闭之前的对话
|
||||||
|
if self.start_connection_flag:
|
||||||
|
await self.finish_connection()
|
||||||
|
self.start_connection_flag = False
|
||||||
|
await self.start_connection()
|
||||||
|
self.start_connection_flag = True
|
||||||
|
|
||||||
|
async def close(self):
|
||||||
|
"""资源清理方法"""
|
||||||
|
await self.finish_connection()
|
||||||
|
await self.ws.close()
|
||||||
|
|
||||||
|
def wav_to_opus_data_audio_raw(self, raw_data_var, is_end=False):
|
||||||
|
opus_datas = self.opus_encoder.encode_pcm_to_opus(raw_data_var, is_end)
|
||||||
|
return opus_datas
|
||||||
|
|
||||||
|
async def _handle_connection_error(self):
|
||||||
|
"""处理连接错误"""
|
||||||
|
if self._reconnect_attempts < self._max_reconnect_attempts:
|
||||||
|
self._reconnect_attempts += 1
|
||||||
|
logger.bind(tag=TAG).warning(
|
||||||
|
f"尝试重新连接 (第{self._reconnect_attempts}次)"
|
||||||
|
)
|
||||||
|
try:
|
||||||
|
await self._ensure_connection()
|
||||||
|
return True
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"重新连接失败: {str(e)}")
|
||||||
|
return False
|
||||||
|
else:
|
||||||
|
logger.bind(tag=TAG).error("达到最大重连次数,放弃重连")
|
||||||
|
return False
|
||||||
@@ -1,7 +1,4 @@
|
|||||||
import os
|
|
||||||
import uuid
|
|
||||||
import requests
|
import requests
|
||||||
from datetime import datetime
|
|
||||||
from core.utils.util import check_model_key
|
from core.utils.util import check_model_key
|
||||||
from core.providers.tts.base import TTSProviderBase
|
from core.providers.tts.base import TTSProviderBase
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
@@ -29,12 +26,6 @@ class TTSProvider(TTSProviderBase):
|
|||||||
self.output_file = config.get("output_dir", "tmp/")
|
self.output_file = config.get("output_dir", "tmp/")
|
||||||
check_model_key("TTS", self.api_key)
|
check_model_key("TTS", self.api_key)
|
||||||
|
|
||||||
def generate_filename(self, extension=".wav"):
|
|
||||||
return os.path.join(
|
|
||||||
self.output_file,
|
|
||||||
f"tts-{datetime.now().date()}@{uuid.uuid4().hex}{extension}",
|
|
||||||
)
|
|
||||||
|
|
||||||
async def text_to_speak(self, text, output_file):
|
async def text_to_speak(self, text, output_file):
|
||||||
headers = {
|
headers = {
|
||||||
"Authorization": f"Bearer {self.api_key}",
|
"Authorization": f"Bearer {self.api_key}",
|
||||||
|
|||||||
@@ -1,7 +1,4 @@
|
|||||||
import os
|
|
||||||
import uuid
|
|
||||||
import requests
|
import requests
|
||||||
from datetime import datetime
|
|
||||||
from core.providers.tts.base import TTSProviderBase
|
from core.providers.tts.base import TTSProviderBase
|
||||||
|
|
||||||
|
|
||||||
@@ -22,12 +19,6 @@ class TTSProvider(TTSProviderBase):
|
|||||||
self.host = "api.siliconflow.cn"
|
self.host = "api.siliconflow.cn"
|
||||||
self.api_url = f"https://{self.host}/v1/audio/speech"
|
self.api_url = f"https://{self.host}/v1/audio/speech"
|
||||||
|
|
||||||
def generate_filename(self, extension=".wav"):
|
|
||||||
return os.path.join(
|
|
||||||
self.output_file,
|
|
||||||
f"tts-{datetime.now().date()}@{uuid.uuid4().hex}{extension}",
|
|
||||||
)
|
|
||||||
|
|
||||||
async def text_to_speak(self, text, output_file):
|
async def text_to_speak(self, text, output_file):
|
||||||
request_json = {
|
request_json = {
|
||||||
"model": self.model,
|
"model": self.model,
|
||||||
|
|||||||
@@ -1,6 +1,5 @@
|
|||||||
import hashlib
|
import hashlib
|
||||||
import hmac
|
import hmac
|
||||||
import os
|
|
||||||
import time
|
import time
|
||||||
import uuid
|
import uuid
|
||||||
import json
|
import json
|
||||||
@@ -121,12 +120,6 @@ class TTSProvider(TTSProviderBase):
|
|||||||
msg = msg.encode("utf-8")
|
msg = msg.encode("utf-8")
|
||||||
return hmac.new(key, msg, hashlib.sha256).digest()
|
return hmac.new(key, msg, hashlib.sha256).digest()
|
||||||
|
|
||||||
def generate_filename(self, extension=".wav"):
|
|
||||||
return os.path.join(
|
|
||||||
self.output_file,
|
|
||||||
f"tts-{datetime.now().date()}@{uuid.uuid4().hex}{extension}",
|
|
||||||
)
|
|
||||||
|
|
||||||
async def text_to_speak(self, text, output_file):
|
async def text_to_speak(self, text, output_file):
|
||||||
# 构建请求体
|
# 构建请求体
|
||||||
request_json = {
|
request_json = {
|
||||||
|
|||||||
@@ -12,13 +12,12 @@ logger = setup_logging()
|
|||||||
class VADProvider(VADProviderBase):
|
class VADProvider(VADProviderBase):
|
||||||
def __init__(self, config):
|
def __init__(self, config):
|
||||||
logger.bind(tag=TAG).info("SileroVAD", config)
|
logger.bind(tag=TAG).info("SileroVAD", config)
|
||||||
self.model, self.utils = torch.hub.load(
|
self.model, _ = torch.hub.load(
|
||||||
repo_or_dir=config["model_dir"],
|
repo_or_dir=config["model_dir"],
|
||||||
source="local",
|
source="local",
|
||||||
model="silero_vad",
|
model="silero_vad",
|
||||||
force_reload=False,
|
force_reload=False,
|
||||||
)
|
)
|
||||||
(get_speech_timestamps, _, _, _, _) = self.utils
|
|
||||||
|
|
||||||
self.decoder = opuslib_next.Decoder(16000, 1)
|
self.decoder = opuslib_next.Decoder(16000, 1)
|
||||||
|
|
||||||
@@ -53,7 +52,7 @@ class VADProvider(VADProviderBase):
|
|||||||
speech_prob = self.model(audio_tensor, 16000).item()
|
speech_prob = self.model(audio_tensor, 16000).item()
|
||||||
client_have_voice = speech_prob >= self.vad_threshold
|
client_have_voice = speech_prob >= self.vad_threshold
|
||||||
|
|
||||||
# 如果之前有声音,但本次没有声音,且与上次有声音的时间查已经超过了静默阈值,则认为已经说完一句话
|
# 如果之前有声音,但本次没有声音,且与上次有声音的时间差已经超过了静默阈值,则认为已经说完一句话
|
||||||
if conn.client_have_voice and not client_have_voice:
|
if conn.client_have_voice and not client_have_voice:
|
||||||
stop_duration = (
|
stop_duration = (
|
||||||
time.time() * 1000 - conn.client_have_voice_last_time
|
time.time() * 1000 - conn.client_have_voice_last_time
|
||||||
|
|||||||
@@ -0,0 +1,12 @@
|
|||||||
|
from abc import ABC, abstractmethod
|
||||||
|
from config.logger import setup_logging
|
||||||
|
|
||||||
|
TAG = __name__
|
||||||
|
logger = setup_logging()
|
||||||
|
|
||||||
|
|
||||||
|
class VLLMProviderBase(ABC):
|
||||||
|
@abstractmethod
|
||||||
|
def response(self, question, base64_image):
|
||||||
|
"""VLLM response generator"""
|
||||||
|
pass
|
||||||
@@ -0,0 +1,63 @@
|
|||||||
|
import openai
|
||||||
|
import json
|
||||||
|
from config.logger import setup_logging
|
||||||
|
from core.utils.util import check_model_key
|
||||||
|
from core.providers.vllm.base import VLLMProviderBase
|
||||||
|
|
||||||
|
TAG = __name__
|
||||||
|
logger = setup_logging()
|
||||||
|
|
||||||
|
|
||||||
|
class VLLMProvider(VLLMProviderBase):
|
||||||
|
def __init__(self, config):
|
||||||
|
self.model_name = config.get("model_name")
|
||||||
|
self.api_key = config.get("api_key")
|
||||||
|
if "base_url" in config:
|
||||||
|
self.base_url = config.get("base_url")
|
||||||
|
else:
|
||||||
|
self.base_url = config.get("url")
|
||||||
|
|
||||||
|
param_defaults = {
|
||||||
|
"max_tokens": (500, int),
|
||||||
|
"temperature": (0.7, lambda x: round(float(x), 1)),
|
||||||
|
"top_p": (1.0, lambda x: round(float(x), 1)),
|
||||||
|
}
|
||||||
|
|
||||||
|
for param, (default, converter) in param_defaults.items():
|
||||||
|
value = config.get(param)
|
||||||
|
try:
|
||||||
|
setattr(
|
||||||
|
self,
|
||||||
|
param,
|
||||||
|
converter(value) if value not in (None, "") else default,
|
||||||
|
)
|
||||||
|
except (ValueError, TypeError):
|
||||||
|
setattr(self, param, default)
|
||||||
|
|
||||||
|
check_model_key("VLLM", self.api_key)
|
||||||
|
self.client = openai.OpenAI(api_key=self.api_key, base_url=self.base_url)
|
||||||
|
|
||||||
|
def response(self, question, base64_image):
|
||||||
|
try:
|
||||||
|
messages = [
|
||||||
|
{
|
||||||
|
"role": "user",
|
||||||
|
"content": [
|
||||||
|
{"type": "text", "text": question},
|
||||||
|
{
|
||||||
|
"type": "image_url",
|
||||||
|
"image_url": {"url": f"{base64_image}"},
|
||||||
|
},
|
||||||
|
],
|
||||||
|
}
|
||||||
|
]
|
||||||
|
|
||||||
|
response = self.client.chat.completions.create(
|
||||||
|
model=self.model_name, messages=messages, stream=False
|
||||||
|
)
|
||||||
|
|
||||||
|
return response.choices[0].message.content
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
logger.bind(tag=TAG).error(f"Error in response generation: {e}")
|
||||||
|
raise
|
||||||
@@ -0,0 +1,43 @@
|
|||||||
|
import jwt
|
||||||
|
import time
|
||||||
|
from datetime import datetime, timedelta, timezone
|
||||||
|
from typing import Optional, Tuple
|
||||||
|
|
||||||
|
|
||||||
|
class AuthToken:
|
||||||
|
def __init__(self, secret_key: str):
|
||||||
|
self.secret_key = secret_key
|
||||||
|
|
||||||
|
def generate_token(self, device_id: str) -> str:
|
||||||
|
"""
|
||||||
|
生成JWT token
|
||||||
|
:param device_id: 设备ID
|
||||||
|
:return: JWT token字符串
|
||||||
|
"""
|
||||||
|
# 设置过期时间为1小时后
|
||||||
|
expire_time = datetime.now(timezone.utc) + timedelta(hours=1)
|
||||||
|
|
||||||
|
# 创建payload
|
||||||
|
payload = {"device_id": device_id, "exp": expire_time.timestamp()}
|
||||||
|
|
||||||
|
# 使用JWT进行编码
|
||||||
|
token = jwt.encode(payload, self.secret_key, algorithm="HS256")
|
||||||
|
return token
|
||||||
|
|
||||||
|
def verify_token(self, token: str) -> Tuple[bool, Optional[str]]:
|
||||||
|
"""
|
||||||
|
验证token
|
||||||
|
:param token: JWT token字符串
|
||||||
|
:return: (是否有效, 设备ID)
|
||||||
|
"""
|
||||||
|
try:
|
||||||
|
# 解码token
|
||||||
|
payload = jwt.decode(token, self.secret_key, algorithms=["HS256"])
|
||||||
|
|
||||||
|
# 检查是否过期
|
||||||
|
if payload["exp"] < time.time():
|
||||||
|
return False, None
|
||||||
|
|
||||||
|
return True, payload["device_id"]
|
||||||
|
except jwt.InvalidTokenError:
|
||||||
|
return False, None
|
||||||
@@ -0,0 +1,128 @@
|
|||||||
|
from typing import Dict, Any
|
||||||
|
from config.logger import setup_logging
|
||||||
|
from core.utils import tts, llm, intent, memory, vad, asr
|
||||||
|
|
||||||
|
TAG = __name__
|
||||||
|
logger = setup_logging()
|
||||||
|
|
||||||
|
|
||||||
|
def initialize_modules(
|
||||||
|
logger,
|
||||||
|
config: Dict[str, Any],
|
||||||
|
init_vad=False,
|
||||||
|
init_asr=False,
|
||||||
|
init_llm=False,
|
||||||
|
init_tts=False,
|
||||||
|
init_memory=False,
|
||||||
|
init_intent=False,
|
||||||
|
) -> Dict[str, Any]:
|
||||||
|
"""
|
||||||
|
初始化所有模块组件
|
||||||
|
|
||||||
|
Args:
|
||||||
|
config: 配置字典
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Dict[str, Any]: 包含所有初始化后的模块的字典
|
||||||
|
"""
|
||||||
|
modules = {}
|
||||||
|
|
||||||
|
# 初始化TTS模块
|
||||||
|
if init_tts:
|
||||||
|
select_tts_module = config["selected_module"]["TTS"]
|
||||||
|
modules["tts"] = initialize_tts(config)
|
||||||
|
logger.bind(tag=TAG).info(f"初始化组件: tts成功 {select_tts_module}")
|
||||||
|
|
||||||
|
# 初始化LLM模块
|
||||||
|
if init_llm:
|
||||||
|
select_llm_module = config["selected_module"]["LLM"]
|
||||||
|
llm_type = (
|
||||||
|
select_llm_module
|
||||||
|
if "type" not in config["LLM"][select_llm_module]
|
||||||
|
else config["LLM"][select_llm_module]["type"]
|
||||||
|
)
|
||||||
|
modules["llm"] = llm.create_instance(
|
||||||
|
llm_type,
|
||||||
|
config["LLM"][select_llm_module],
|
||||||
|
)
|
||||||
|
logger.bind(tag=TAG).info(f"初始化组件: llm成功 {select_llm_module}")
|
||||||
|
|
||||||
|
# 初始化Intent模块
|
||||||
|
if init_intent:
|
||||||
|
select_intent_module = config["selected_module"]["Intent"]
|
||||||
|
intent_type = (
|
||||||
|
select_intent_module
|
||||||
|
if "type" not in config["Intent"][select_intent_module]
|
||||||
|
else config["Intent"][select_intent_module]["type"]
|
||||||
|
)
|
||||||
|
modules["intent"] = intent.create_instance(
|
||||||
|
intent_type,
|
||||||
|
config["Intent"][select_intent_module],
|
||||||
|
)
|
||||||
|
logger.bind(tag=TAG).info(f"初始化组件: intent成功 {select_intent_module}")
|
||||||
|
|
||||||
|
# 初始化Memory模块
|
||||||
|
if init_memory:
|
||||||
|
select_memory_module = config["selected_module"]["Memory"]
|
||||||
|
memory_type = (
|
||||||
|
select_memory_module
|
||||||
|
if "type" not in config["Memory"][select_memory_module]
|
||||||
|
else config["Memory"][select_memory_module]["type"]
|
||||||
|
)
|
||||||
|
modules["memory"] = memory.create_instance(
|
||||||
|
memory_type,
|
||||||
|
config["Memory"][select_memory_module],
|
||||||
|
config.get("summaryMemory", None),
|
||||||
|
)
|
||||||
|
logger.bind(tag=TAG).info(f"初始化组件: memory成功 {select_memory_module}")
|
||||||
|
|
||||||
|
# 初始化VAD模块
|
||||||
|
if init_vad:
|
||||||
|
select_vad_module = config["selected_module"]["VAD"]
|
||||||
|
vad_type = (
|
||||||
|
select_vad_module
|
||||||
|
if "type" not in config["VAD"][select_vad_module]
|
||||||
|
else config["VAD"][select_vad_module]["type"]
|
||||||
|
)
|
||||||
|
modules["vad"] = vad.create_instance(
|
||||||
|
vad_type,
|
||||||
|
config["VAD"][select_vad_module],
|
||||||
|
)
|
||||||
|
logger.bind(tag=TAG).info(f"初始化组件: vad成功 {select_vad_module}")
|
||||||
|
|
||||||
|
# 初始化ASR模块
|
||||||
|
if init_asr:
|
||||||
|
select_asr_module = config["selected_module"]["ASR"]
|
||||||
|
modules["asr"] = initialize_asr(config)
|
||||||
|
logger.bind(tag=TAG).info(f"初始化组件: asr成功 {select_asr_module}")
|
||||||
|
return modules
|
||||||
|
|
||||||
|
|
||||||
|
def initialize_tts(config):
|
||||||
|
select_tts_module = config["selected_module"]["TTS"]
|
||||||
|
tts_type = (
|
||||||
|
select_tts_module
|
||||||
|
if "type" not in config["TTS"][select_tts_module]
|
||||||
|
else config["TTS"][select_tts_module]["type"]
|
||||||
|
)
|
||||||
|
new_tts = tts.create_instance(
|
||||||
|
tts_type,
|
||||||
|
config["TTS"][select_tts_module],
|
||||||
|
str(config.get("delete_audio", True)).lower() in ("true", "1", "yes"),
|
||||||
|
)
|
||||||
|
return new_tts
|
||||||
|
|
||||||
|
|
||||||
|
def initialize_asr(config):
|
||||||
|
select_asr_module = config["selected_module"]["ASR"]
|
||||||
|
asr_type = (
|
||||||
|
select_asr_module
|
||||||
|
if "type" not in config["ASR"][select_asr_module]
|
||||||
|
else config["ASR"][select_asr_module]["type"]
|
||||||
|
)
|
||||||
|
new_asr = asr.create_instance(
|
||||||
|
asr_type,
|
||||||
|
config["ASR"][select_asr_module],
|
||||||
|
str(config.get("delete_audio", True)).lower() in ("true", "1", "yes"),
|
||||||
|
)
|
||||||
|
return new_asr
|
||||||
@@ -0,0 +1,136 @@
|
|||||||
|
"""
|
||||||
|
Opus编码工具类
|
||||||
|
将PCM音频数据编码为Opus格式
|
||||||
|
"""
|
||||||
|
|
||||||
|
import logging
|
||||||
|
import traceback
|
||||||
|
|
||||||
|
import numpy as np
|
||||||
|
from typing import List, Optional
|
||||||
|
from opuslib_next import Encoder
|
||||||
|
from opuslib_next import constants
|
||||||
|
|
||||||
|
|
||||||
|
class OpusEncoderUtils:
|
||||||
|
"""PCM到Opus的编码器"""
|
||||||
|
|
||||||
|
def __init__(self, sample_rate: int, channels: int, frame_size_ms: int):
|
||||||
|
"""
|
||||||
|
初始化Opus编码器
|
||||||
|
|
||||||
|
Args:
|
||||||
|
sample_rate: 采样率 (Hz)
|
||||||
|
channels: 通道数 (1=单声道, 2=立体声)
|
||||||
|
frame_size_ms: 帧大小 (毫秒)
|
||||||
|
"""
|
||||||
|
self.sample_rate = sample_rate
|
||||||
|
self.channels = channels
|
||||||
|
self.frame_size_ms = frame_size_ms
|
||||||
|
# 计算每帧样本数 = 采样率 * 帧大小(毫秒) / 1000
|
||||||
|
self.frame_size = (sample_rate * frame_size_ms) // 1000
|
||||||
|
# 总帧大小 = 每帧样本数 * 通道数
|
||||||
|
self.total_frame_size = self.frame_size * channels
|
||||||
|
|
||||||
|
# 比特率和复杂度设置
|
||||||
|
self.bitrate = 24000 # bps
|
||||||
|
self.complexity = 10 # 最高质量
|
||||||
|
|
||||||
|
# 缓冲区初始化为空
|
||||||
|
self.buffer = np.array([], dtype=np.int16)
|
||||||
|
|
||||||
|
try:
|
||||||
|
# 创建Opus编码器
|
||||||
|
self.encoder = Encoder(
|
||||||
|
sample_rate, channels, constants.APPLICATION_AUDIO # 音频优化模式
|
||||||
|
)
|
||||||
|
self.encoder.bitrate = self.bitrate
|
||||||
|
self.encoder.complexity = self.complexity
|
||||||
|
self.encoder.signal = constants.SIGNAL_VOICE # 语音信号优化
|
||||||
|
except Exception as e:
|
||||||
|
logging.error(f"初始化Opus编码器失败: {e}")
|
||||||
|
raise RuntimeError("初始化失败") from e
|
||||||
|
|
||||||
|
def reset_state(self):
|
||||||
|
"""重置编码器状态"""
|
||||||
|
self.encoder.reset_state()
|
||||||
|
self.buffer = np.array([], dtype=np.int16)
|
||||||
|
|
||||||
|
def encode_pcm_to_opus(self, pcm_data: bytes, end_of_stream: bool) -> List[bytes]:
|
||||||
|
"""
|
||||||
|
将PCM数据编码为Opus格式
|
||||||
|
|
||||||
|
Args:
|
||||||
|
pcm_data: PCM字节数据
|
||||||
|
end_of_stream: 是否为流的结束
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
Opus数据包列表
|
||||||
|
"""
|
||||||
|
# 将字节数据转换为short数组
|
||||||
|
new_samples = self._convert_bytes_to_shorts(pcm_data)
|
||||||
|
|
||||||
|
# 校验PCM数据
|
||||||
|
self._validate_pcm_data(new_samples)
|
||||||
|
|
||||||
|
# 将新数据追加到缓冲区
|
||||||
|
self.buffer = np.append(self.buffer, new_samples)
|
||||||
|
|
||||||
|
opus_packets = []
|
||||||
|
offset = 0
|
||||||
|
|
||||||
|
# 处理所有完整帧
|
||||||
|
while offset <= len(self.buffer) - self.total_frame_size:
|
||||||
|
frame = self.buffer[offset : offset + self.total_frame_size]
|
||||||
|
output = self._encode(frame)
|
||||||
|
if output:
|
||||||
|
opus_packets.append(output)
|
||||||
|
offset += self.total_frame_size
|
||||||
|
|
||||||
|
# 保留未处理的样本
|
||||||
|
self.buffer = self.buffer[offset:]
|
||||||
|
|
||||||
|
# 流结束时处理剩余数据
|
||||||
|
if end_of_stream and len(self.buffer) > 0:
|
||||||
|
# 创建最后一帧并用0填充
|
||||||
|
last_frame = np.zeros(self.total_frame_size, dtype=np.int16)
|
||||||
|
last_frame[: len(self.buffer)] = self.buffer
|
||||||
|
|
||||||
|
output = self._encode(last_frame)
|
||||||
|
if output:
|
||||||
|
opus_packets.append(output)
|
||||||
|
self.buffer = np.array([], dtype=np.int16)
|
||||||
|
|
||||||
|
return opus_packets
|
||||||
|
|
||||||
|
def _encode(self, frame: np.ndarray) -> Optional[bytes]:
|
||||||
|
"""编码一帧音频数据"""
|
||||||
|
try:
|
||||||
|
# 将numpy数组转换为bytes
|
||||||
|
frame_bytes = frame.tobytes()
|
||||||
|
# opuslib要求输入字节数必须是channels*2的倍数
|
||||||
|
encoded = self.encoder.encode(frame_bytes, self.frame_size)
|
||||||
|
return encoded
|
||||||
|
except Exception as e:
|
||||||
|
logging.error(f"Opus编码失败: {e}")
|
||||||
|
traceback.print_exc()
|
||||||
|
return None
|
||||||
|
|
||||||
|
def _convert_bytes_to_shorts(self, bytes_data: bytes) -> np.ndarray:
|
||||||
|
"""将字节数组转换为short数组 (16位PCM)"""
|
||||||
|
# 假设输入是小端字节序的16位PCM
|
||||||
|
return np.frombuffer(bytes_data, dtype=np.int16)
|
||||||
|
|
||||||
|
def _validate_pcm_data(self, pcm_shorts: np.ndarray) -> None:
|
||||||
|
"""验证PCM数据是否有效"""
|
||||||
|
# 16位PCM数据范围是 -32768 到 32767
|
||||||
|
if np.any((pcm_shorts < -32768) | (pcm_shorts > 32767)):
|
||||||
|
invalid_samples = pcm_shorts[(pcm_shorts < -32768) | (pcm_shorts > 32767)]
|
||||||
|
logging.warning(f"发现无效PCM样本: {invalid_samples[:5]}...")
|
||||||
|
# 在实际应用中可以选择裁剪而不是抛出异常
|
||||||
|
# np.clip(pcm_shorts, -32768, 32767, out=pcm_shorts)
|
||||||
|
|
||||||
|
def close(self):
|
||||||
|
"""关闭编码器并释放资源"""
|
||||||
|
# opuslib没有明确的关闭方法,Python的垃圾回收会处理
|
||||||
|
pass
|
||||||
@@ -0,0 +1,34 @@
|
|||||||
|
def get_string_no_punctuation_or_emoji(s):
|
||||||
|
"""去除字符串首尾的空格、标点符号和表情符号"""
|
||||||
|
chars = list(s)
|
||||||
|
# 处理开头的字符
|
||||||
|
start = 0
|
||||||
|
while start < len(chars) and is_punctuation_or_emoji(chars[start]):
|
||||||
|
start += 1
|
||||||
|
# 处理结尾的字符
|
||||||
|
end = len(chars) - 1
|
||||||
|
while end >= start and is_punctuation_or_emoji(chars[end]):
|
||||||
|
end -= 1
|
||||||
|
return ''.join(chars[start:end + 1])
|
||||||
|
|
||||||
|
def is_punctuation_or_emoji(char):
|
||||||
|
"""检查字符是否为空格、指定标点或表情符号"""
|
||||||
|
# 定义需要去除的中英文标点(包括全角/半角)
|
||||||
|
punctuation_set = {
|
||||||
|
',', ',', # 中文逗号 + 英文逗号
|
||||||
|
'。', '.', # 中文句号 + 英文句号
|
||||||
|
'!', '!', # 中文感叹号 + 英文感叹号
|
||||||
|
'-', '-', # 英文连字符 + 中文全角横线
|
||||||
|
'、' # 中文顿号
|
||||||
|
}
|
||||||
|
if char.isspace() or char in punctuation_set:
|
||||||
|
return True
|
||||||
|
# 检查表情符号(保留原有逻辑)
|
||||||
|
code_point = ord(char)
|
||||||
|
emoji_ranges = [
|
||||||
|
(0x1F600, 0x1F64F), (0x1F300, 0x1F5FF),
|
||||||
|
(0x1F680, 0x1F6FF), (0x1F900, 0x1F9FF),
|
||||||
|
(0x1FA70, 0x1FAFF), (0x2600, 0x26FF),
|
||||||
|
(0x2700, 0x27BF)
|
||||||
|
]
|
||||||
|
return any(start <= code_point <= end for start, end in emoji_ranges)
|
||||||
@@ -7,8 +7,6 @@ import numpy as np
|
|||||||
import requests
|
import requests
|
||||||
import opuslib_next
|
import opuslib_next
|
||||||
from pydub import AudioSegment
|
from pydub import AudioSegment
|
||||||
from typing import Dict, Any
|
|
||||||
from core.utils import tts, llm, intent, memory, vad, asr
|
|
||||||
import copy
|
import copy
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
@@ -245,116 +243,6 @@ def extract_json_from_string(input_string):
|
|||||||
return None
|
return None
|
||||||
|
|
||||||
|
|
||||||
def initialize_modules(
|
|
||||||
logger,
|
|
||||||
config: Dict[str, Any],
|
|
||||||
init_vad=False,
|
|
||||||
init_asr=False,
|
|
||||||
init_llm=False,
|
|
||||||
init_tts=False,
|
|
||||||
init_memory=False,
|
|
||||||
init_intent=False,
|
|
||||||
) -> Dict[str, Any]:
|
|
||||||
"""
|
|
||||||
初始化所有模块组件
|
|
||||||
|
|
||||||
Args:
|
|
||||||
config: 配置字典
|
|
||||||
|
|
||||||
Returns:
|
|
||||||
Dict[str, Any]: 包含所有初始化后的模块的字典
|
|
||||||
"""
|
|
||||||
modules = {}
|
|
||||||
|
|
||||||
# 初始化TTS模块
|
|
||||||
if init_tts:
|
|
||||||
select_tts_module = config["selected_module"]["TTS"]
|
|
||||||
tts_type = (
|
|
||||||
select_tts_module
|
|
||||||
if "type" not in config["TTS"][select_tts_module]
|
|
||||||
else config["TTS"][select_tts_module]["type"]
|
|
||||||
)
|
|
||||||
modules["tts"] = tts.create_instance(
|
|
||||||
tts_type,
|
|
||||||
config["TTS"][select_tts_module],
|
|
||||||
str(config.get("delete_audio", True)).lower() in ("true", "1", "yes"),
|
|
||||||
)
|
|
||||||
logger.bind(tag=TAG).info(f"初始化组件: tts成功 {select_tts_module}")
|
|
||||||
|
|
||||||
# 初始化LLM模块
|
|
||||||
if init_llm:
|
|
||||||
select_llm_module = config["selected_module"]["LLM"]
|
|
||||||
llm_type = (
|
|
||||||
select_llm_module
|
|
||||||
if "type" not in config["LLM"][select_llm_module]
|
|
||||||
else config["LLM"][select_llm_module]["type"]
|
|
||||||
)
|
|
||||||
modules["llm"] = llm.create_instance(
|
|
||||||
llm_type,
|
|
||||||
config["LLM"][select_llm_module],
|
|
||||||
)
|
|
||||||
logger.bind(tag=TAG).info(f"初始化组件: llm成功 {select_llm_module}")
|
|
||||||
|
|
||||||
# 初始化Intent模块
|
|
||||||
if init_intent:
|
|
||||||
select_intent_module = config["selected_module"]["Intent"]
|
|
||||||
intent_type = (
|
|
||||||
select_intent_module
|
|
||||||
if "type" not in config["Intent"][select_intent_module]
|
|
||||||
else config["Intent"][select_intent_module]["type"]
|
|
||||||
)
|
|
||||||
modules["intent"] = intent.create_instance(
|
|
||||||
intent_type,
|
|
||||||
config["Intent"][select_intent_module],
|
|
||||||
)
|
|
||||||
logger.bind(tag=TAG).info(f"初始化组件: intent成功 {select_intent_module}")
|
|
||||||
|
|
||||||
# 初始化Memory模块
|
|
||||||
if init_memory:
|
|
||||||
select_memory_module = config["selected_module"]["Memory"]
|
|
||||||
memory_type = (
|
|
||||||
select_memory_module
|
|
||||||
if "type" not in config["Memory"][select_memory_module]
|
|
||||||
else config["Memory"][select_memory_module]["type"]
|
|
||||||
)
|
|
||||||
modules["memory"] = memory.create_instance(
|
|
||||||
memory_type,
|
|
||||||
config["Memory"][select_memory_module],
|
|
||||||
config.get("summaryMemory", None),
|
|
||||||
)
|
|
||||||
logger.bind(tag=TAG).info(f"初始化组件: memory成功 {select_memory_module}")
|
|
||||||
|
|
||||||
# 初始化VAD模块
|
|
||||||
if init_vad:
|
|
||||||
select_vad_module = config["selected_module"]["VAD"]
|
|
||||||
vad_type = (
|
|
||||||
select_vad_module
|
|
||||||
if "type" not in config["VAD"][select_vad_module]
|
|
||||||
else config["VAD"][select_vad_module]["type"]
|
|
||||||
)
|
|
||||||
modules["vad"] = vad.create_instance(
|
|
||||||
vad_type,
|
|
||||||
config["VAD"][select_vad_module],
|
|
||||||
)
|
|
||||||
logger.bind(tag=TAG).info(f"初始化组件: vad成功 {select_vad_module}")
|
|
||||||
|
|
||||||
# 初始化ASR模块
|
|
||||||
if init_asr:
|
|
||||||
select_asr_module = config["selected_module"]["ASR"]
|
|
||||||
asr_type = (
|
|
||||||
select_asr_module
|
|
||||||
if "type" not in config["ASR"][select_asr_module]
|
|
||||||
else config["ASR"][select_asr_module]["type"]
|
|
||||||
)
|
|
||||||
modules["asr"] = asr.create_instance(
|
|
||||||
asr_type,
|
|
||||||
config["ASR"][select_asr_module],
|
|
||||||
str(config.get("delete_audio", True)).lower() in ("true", "1", "yes"),
|
|
||||||
)
|
|
||||||
logger.bind(tag=TAG).info(f"初始化组件: asr成功 {select_asr_module}")
|
|
||||||
return modules
|
|
||||||
|
|
||||||
|
|
||||||
def analyze_emotion(text):
|
def analyze_emotion(text):
|
||||||
"""
|
"""
|
||||||
分析文本情感并返回对应的emoji名称(支持中英文)
|
分析文本情感并返回对应的emoji名称(支持中英文)
|
||||||
@@ -882,7 +770,10 @@ def audio_to_data(audio_file_path, is_opus=True):
|
|||||||
|
|
||||||
# 获取原始PCM数据(16位小端)
|
# 获取原始PCM数据(16位小端)
|
||||||
raw_data = audio.raw_data
|
raw_data = audio.raw_data
|
||||||
|
return pcm_to_data(raw_data, is_opus), duration
|
||||||
|
|
||||||
|
|
||||||
|
def pcm_to_data(raw_data, is_opus=True):
|
||||||
# 初始化Opus编码器
|
# 初始化Opus编码器
|
||||||
encoder = opuslib_next.Encoder(16000, 1, opuslib_next.APPLICATION_AUDIO)
|
encoder = opuslib_next.Encoder(16000, 1, opuslib_next.APPLICATION_AUDIO)
|
||||||
|
|
||||||
@@ -910,7 +801,7 @@ def audio_to_data(audio_file_path, is_opus=True):
|
|||||||
|
|
||||||
datas.append(frame_data)
|
datas.append(frame_data)
|
||||||
|
|
||||||
return datas, duration
|
return datas
|
||||||
|
|
||||||
|
|
||||||
def check_vad_update(before_config, new_config):
|
def check_vad_update(before_config, new_config):
|
||||||
@@ -991,3 +882,51 @@ def filter_sensitive_info(config: dict) -> dict:
|
|||||||
return filtered
|
return filtered
|
||||||
|
|
||||||
return _filter_dict(copy.deepcopy(config))
|
return _filter_dict(copy.deepcopy(config))
|
||||||
|
|
||||||
|
|
||||||
|
def get_vision_url(config: dict) -> str:
|
||||||
|
"""获取 vision URL
|
||||||
|
|
||||||
|
Args:
|
||||||
|
config: 配置字典
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
str: vision URL
|
||||||
|
"""
|
||||||
|
server_config = config["server"]
|
||||||
|
vision_explain = server_config.get("vision_explain", "")
|
||||||
|
if "你的" in vision_explain:
|
||||||
|
local_ip = get_local_ip()
|
||||||
|
port = int(server_config.get("http_port", 8003))
|
||||||
|
vision_explain = f"http://{local_ip}:{port}/mcp/vision/explain"
|
||||||
|
return vision_explain
|
||||||
|
|
||||||
|
|
||||||
|
def is_valid_image_file(file_data: bytes) -> bool:
|
||||||
|
"""
|
||||||
|
检查文件数据是否为有效的图片格式
|
||||||
|
|
||||||
|
Args:
|
||||||
|
file_data: 文件的二进制数据
|
||||||
|
|
||||||
|
Returns:
|
||||||
|
bool: 如果是有效的图片格式返回True,否则返回False
|
||||||
|
"""
|
||||||
|
# 常见图片格式的魔数(文件头)
|
||||||
|
image_signatures = {
|
||||||
|
b"\xff\xd8\xff": "JPEG",
|
||||||
|
b"\x89PNG\r\n\x1a\n": "PNG",
|
||||||
|
b"GIF87a": "GIF",
|
||||||
|
b"GIF89a": "GIF",
|
||||||
|
b"BM": "BMP",
|
||||||
|
b"II*\x00": "TIFF",
|
||||||
|
b"MM\x00*": "TIFF",
|
||||||
|
b"RIFF": "WEBP",
|
||||||
|
}
|
||||||
|
|
||||||
|
# 检查文件头是否匹配任何已知的图片格式
|
||||||
|
for signature in image_signatures:
|
||||||
|
if file_data.startswith(signature):
|
||||||
|
return True
|
||||||
|
|
||||||
|
return False
|
||||||
|
|||||||
@@ -0,0 +1,23 @@
|
|||||||
|
import os
|
||||||
|
import sys
|
||||||
|
|
||||||
|
# 添加项目根目录到Python路径
|
||||||
|
current_dir = os.path.dirname(os.path.abspath(__file__))
|
||||||
|
project_root = os.path.abspath(os.path.join(current_dir, "..", ".."))
|
||||||
|
sys.path.insert(0, project_root)
|
||||||
|
|
||||||
|
from config.logger import setup_logging
|
||||||
|
import importlib
|
||||||
|
|
||||||
|
logger = setup_logging()
|
||||||
|
|
||||||
|
|
||||||
|
def create_instance(class_name, *args, **kwargs):
|
||||||
|
# 创建LLM实例
|
||||||
|
if os.path.exists(os.path.join("core", "providers", "vllm", f"{class_name}.py")):
|
||||||
|
lib_name = f"core.providers.vllm.{class_name}"
|
||||||
|
if lib_name not in sys.modules:
|
||||||
|
sys.modules[lib_name] = importlib.import_module(f"{lib_name}")
|
||||||
|
return sys.modules[lib_name].VLLMProvider(*args, **kwargs)
|
||||||
|
|
||||||
|
raise ValueError(f"不支持的VLLM类型: {class_name},请检查该配置的type是否设置正确")
|
||||||
@@ -2,8 +2,9 @@ import asyncio
|
|||||||
import websockets
|
import websockets
|
||||||
from config.logger import setup_logging
|
from config.logger import setup_logging
|
||||||
from core.connection import ConnectionHandler
|
from core.connection import ConnectionHandler
|
||||||
from core.utils.util import initialize_modules, check_vad_update, check_asr_update
|
|
||||||
from config.config_loader import get_config_from_api
|
from config.config_loader import get_config_from_api
|
||||||
|
from core.utils.modules_initialize import initialize_modules
|
||||||
|
from core.utils.util import check_vad_update, check_asr_update
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
|
|
||||||
@@ -19,16 +20,16 @@ class WebSocketServer:
|
|||||||
"VAD" in self.config["selected_module"],
|
"VAD" in self.config["selected_module"],
|
||||||
"ASR" in self.config["selected_module"],
|
"ASR" in self.config["selected_module"],
|
||||||
"LLM" in self.config["selected_module"],
|
"LLM" in self.config["selected_module"],
|
||||||
"TTS" in self.config["selected_module"],
|
False,
|
||||||
"Memory" in self.config["selected_module"],
|
"Memory" in self.config["selected_module"],
|
||||||
"Intent" in self.config["selected_module"],
|
"Intent" in self.config["selected_module"],
|
||||||
)
|
)
|
||||||
self._vad = modules["vad"] if "vad" in modules else None
|
self._vad = modules["vad"] if "vad" in modules else None
|
||||||
self._asr = modules["asr"] if "asr" in modules else None
|
self._asr = modules["asr"] if "asr" in modules else None
|
||||||
self._tts = modules["tts"] if "tts" in modules else None
|
|
||||||
self._llm = modules["llm"] if "llm" in modules else None
|
self._llm = modules["llm"] if "llm" in modules else None
|
||||||
self._intent = modules["intent"] if "intent" in modules else None
|
self._intent = modules["intent"] if "intent" in modules else None
|
||||||
self._memory = modules["memory"] if "memory" in modules else None
|
self._memory = modules["memory"] if "memory" in modules else None
|
||||||
|
|
||||||
self.active_connections = set()
|
self.active_connections = set()
|
||||||
|
|
||||||
async def start(self):
|
async def start(self):
|
||||||
@@ -49,7 +50,6 @@ class WebSocketServer:
|
|||||||
self._vad,
|
self._vad,
|
||||||
self._asr,
|
self._asr,
|
||||||
self._llm,
|
self._llm,
|
||||||
self._tts,
|
|
||||||
self._memory,
|
self._memory,
|
||||||
self._intent,
|
self._intent,
|
||||||
self, # 传入server实例
|
self, # 传入server实例
|
||||||
@@ -98,7 +98,7 @@ class WebSocketServer:
|
|||||||
update_vad,
|
update_vad,
|
||||||
update_asr,
|
update_asr,
|
||||||
"LLM" in new_config["selected_module"],
|
"LLM" in new_config["selected_module"],
|
||||||
"TTS" in new_config["selected_module"],
|
False,
|
||||||
"Memory" in new_config["selected_module"],
|
"Memory" in new_config["selected_module"],
|
||||||
"Intent" in new_config["selected_module"],
|
"Intent" in new_config["selected_module"],
|
||||||
)
|
)
|
||||||
@@ -108,8 +108,6 @@ class WebSocketServer:
|
|||||||
self._vad = modules["vad"]
|
self._vad = modules["vad"]
|
||||||
if "asr" in modules:
|
if "asr" in modules:
|
||||||
self._asr = modules["asr"]
|
self._asr = modules["asr"]
|
||||||
if "tts" in modules:
|
|
||||||
self._tts = modules["tts"]
|
|
||||||
if "llm" in modules:
|
if "llm" in modules:
|
||||||
self._llm = modules["llm"]
|
self._llm = modules["llm"]
|
||||||
if "intent" in modules:
|
if "intent" in modules:
|
||||||
|
|||||||
@@ -13,8 +13,8 @@ services:
|
|||||||
ports:
|
ports:
|
||||||
# ws服务端
|
# ws服务端
|
||||||
- "8000:8000"
|
- "8000:8000"
|
||||||
# ota服务端
|
# http服务的端口,用于简单OTA接口(单服务部署),以及视觉分析接口
|
||||||
- "8002:8002"
|
- "8003:8003"
|
||||||
volumes:
|
volumes:
|
||||||
# 配置文件目录
|
# 配置文件目录
|
||||||
- ./data:/opt/xiaozhi-esp32-server/data
|
- ./data:/opt/xiaozhi-esp32-server/data
|
||||||
|
|||||||
@@ -15,6 +15,8 @@ services:
|
|||||||
ports:
|
ports:
|
||||||
# ws服务端
|
# ws服务端
|
||||||
- "8000:8000"
|
- "8000:8000"
|
||||||
|
# http服务的端口,用于视觉分析接口
|
||||||
|
- "8003:8003"
|
||||||
security_opt:
|
security_opt:
|
||||||
- seccomp:unconfined
|
- seccomp:unconfined
|
||||||
environment:
|
environment:
|
||||||
|
|||||||
@@ -15,15 +15,26 @@ async def _get_device_status(conn, device_name, device_type, property_name):
|
|||||||
return status
|
return status
|
||||||
|
|
||||||
|
|
||||||
async def _set_device_property(conn, device_name, device_type, method_name, property_name, new_value=None, action=None, step=10):
|
async def _set_device_property(
|
||||||
|
conn,
|
||||||
|
device_name,
|
||||||
|
device_type,
|
||||||
|
method_name,
|
||||||
|
property_name,
|
||||||
|
new_value=None,
|
||||||
|
action=None,
|
||||||
|
step=10,
|
||||||
|
):
|
||||||
"""设置设备属性"""
|
"""设置设备属性"""
|
||||||
current_value = await _get_device_status(conn, device_name, device_type, property_name)
|
current_value = await _get_device_status(
|
||||||
|
conn, device_name, device_type, property_name
|
||||||
|
)
|
||||||
|
|
||||||
if action == 'raise':
|
if action == "raise":
|
||||||
current_value += step
|
current_value += step
|
||||||
elif action == 'lower':
|
elif action == "lower":
|
||||||
current_value -= step
|
current_value -= step
|
||||||
elif action == 'set':
|
elif action == "set":
|
||||||
if new_value is None:
|
if new_value is None:
|
||||||
raise Exception(f"缺少{property_name}参数")
|
raise Exception(f"缺少{property_name}参数")
|
||||||
current_value = new_value
|
current_value = new_value
|
||||||
@@ -37,8 +48,7 @@ async def _set_device_property(conn, device_name, device_type, method_name, prop
|
|||||||
|
|
||||||
def _handle_device_action(conn, func, success_message, error_message, *args, **kwargs):
|
def _handle_device_action(conn, func, success_message, error_message, *args, **kwargs):
|
||||||
"""处理设备操作的通用函数"""
|
"""处理设备操作的通用函数"""
|
||||||
future = asyncio.run_coroutine_threadsafe(
|
future = asyncio.run_coroutine_threadsafe(func(conn, *args, **kwargs), conn.loop)
|
||||||
func(conn, *args, **kwargs), conn.loop)
|
|
||||||
try:
|
try:
|
||||||
result = future.result()
|
result = future.result()
|
||||||
logger.bind(tag=TAG).info(f"{success_message}: {result}")
|
logger.bind(tag=TAG).info(f"{success_message}: {result}")
|
||||||
@@ -75,26 +85,41 @@ handle_device_function_desc = {
|
|||||||
"device_type": {
|
"device_type": {
|
||||||
"type": "string",
|
"type": "string",
|
||||||
"description": "设备类型,**严格限定为Speaker(音量)或Screen(亮度)**,其他设备类型禁止调用此函数",
|
"description": "设备类型,**严格限定为Speaker(音量)或Screen(亮度)**,其他设备类型禁止调用此函数",
|
||||||
"enum": ["Speaker", "Screen"]
|
"enum": ["Speaker", "Screen"],
|
||||||
|
|
||||||
},
|
},
|
||||||
"action": {
|
"action": {
|
||||||
"type": "string",
|
"type": "string",
|
||||||
"description": "动作名称,可选值:get(获取),set(设置),raise(提高),lower(降低)"
|
"description": "动作名称,可选值:get(获取),set(设置),raise(提高),lower(降低)",
|
||||||
},
|
},
|
||||||
"value": {
|
"value": {
|
||||||
"type": "integer",
|
"type": "integer",
|
||||||
"description": "值大小,可选值:0-100之间的整数"
|
"description": "值大小,可选值:0-100之间的整数",
|
||||||
}
|
},
|
||||||
},
|
},
|
||||||
"required": ["device_type", "action"]
|
"required": ["device_type", "action"],
|
||||||
}
|
},
|
||||||
}
|
},
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|
||||||
@register_function('handle_speaker_volume_or_screen_brightness', handle_device_function_desc, ToolType.IOT_CTL)
|
@register_function(
|
||||||
def handle_speaker_volume_or_screen_brightness(conn, device_type: str, action: str, value: int = None):
|
"handle_speaker_volume_or_screen_brightness",
|
||||||
|
handle_device_function_desc,
|
||||||
|
ToolType.IOT_CTL,
|
||||||
|
)
|
||||||
|
def handle_speaker_volume_or_screen_brightness(
|
||||||
|
conn, device_type: str, action: str, value: int = None
|
||||||
|
):
|
||||||
|
# 检查value是否为中文值
|
||||||
|
if (
|
||||||
|
value is not None
|
||||||
|
and isinstance(value, str)
|
||||||
|
and any("\u4e00" <= char <= "\u9fff" for char in str(value))
|
||||||
|
):
|
||||||
|
raise Exception(
|
||||||
|
f"请直接告诉我要将{'音量' if device_type=='Speaker' else '亮度'}调整成多少"
|
||||||
|
)
|
||||||
|
|
||||||
if device_type == "Speaker":
|
if device_type == "Speaker":
|
||||||
method_name, property_name, device_name = "SetVolume", "volume", "音量"
|
method_name, property_name, device_name = "SetVolume", "volume", "音量"
|
||||||
elif device_type == "Screen":
|
elif device_type == "Screen":
|
||||||
@@ -108,13 +133,25 @@ def handle_speaker_volume_or_screen_brightness(conn, device_type: str, action: s
|
|||||||
if action == "get":
|
if action == "get":
|
||||||
# get
|
# get
|
||||||
return _handle_device_action(
|
return _handle_device_action(
|
||||||
conn, _get_device_status, f"当前{device_name}", f"获取{device_name}失败",
|
conn,
|
||||||
device_name=device_name, device_type=device_type, property_name=property_name,
|
_get_device_status,
|
||||||
|
f"当前{device_name}",
|
||||||
|
f"获取{device_name}失败",
|
||||||
|
device_name=device_name,
|
||||||
|
device_type=device_type,
|
||||||
|
property_name=property_name,
|
||||||
)
|
)
|
||||||
else:
|
else:
|
||||||
# set, raise, lower
|
# set, raise, lower
|
||||||
return _handle_device_action(
|
return _handle_device_action(
|
||||||
conn, _set_device_property, f"{device_name}已调整到", f"{device_name}调整失败",
|
conn,
|
||||||
device_name=device_name, device_type=device_type, method_name=method_name,
|
_set_device_property,
|
||||||
property_name=property_name, new_value=value, action=action
|
f"{device_name}已调整到",
|
||||||
|
f"{device_name}调整失败",
|
||||||
|
device_name=device_name,
|
||||||
|
device_type=device_type,
|
||||||
|
method_name=method_name,
|
||||||
|
property_name=property_name,
|
||||||
|
new_value=value,
|
||||||
|
action=action,
|
||||||
)
|
)
|
||||||
|
|||||||
@@ -11,6 +11,7 @@ from core.utils import p3
|
|||||||
from core.handle.sendAudioHandle import send_stt_message
|
from core.handle.sendAudioHandle import send_stt_message
|
||||||
from plugins_func.register import register_function, ToolType, ActionResponse, Action
|
from plugins_func.register import register_function, ToolType, ActionResponse, Action
|
||||||
from core.utils.dialogue import Message
|
from core.utils.dialogue import Message
|
||||||
|
from core.providers.tts.dto.dto import TTSMessageDTO, SentenceType, ContentType
|
||||||
|
|
||||||
TAG = __name__
|
TAG = __name__
|
||||||
|
|
||||||
@@ -216,23 +217,37 @@ async def play_local_music(conn, specific_file=None):
|
|||||||
text = _get_random_play_prompt(selected_music)
|
text = _get_random_play_prompt(selected_music)
|
||||||
await send_stt_message(conn, text)
|
await send_stt_message(conn, text)
|
||||||
conn.dialogue.put(Message(role="assistant", content=text))
|
conn.dialogue.put(Message(role="assistant", content=text))
|
||||||
conn.tts_first_text_index = 0
|
|
||||||
conn.tts_last_text_index = 0
|
|
||||||
|
|
||||||
tts_file = await asyncio.to_thread(conn.tts.to_tts, text)
|
conn.tts.tts_text_queue.put(
|
||||||
if tts_file is not None and os.path.exists(tts_file):
|
TTSMessageDTO(
|
||||||
conn.tts_last_text_index = 1
|
sentence_id=conn.sentence_id,
|
||||||
opus_packets, _ = conn.tts.audio_to_opus_data(tts_file)
|
sentence_type=SentenceType.FIRST,
|
||||||
conn.audio_play_queue.put((opus_packets, None, 0))
|
content_type=ContentType.ACTION,
|
||||||
os.remove(tts_file)
|
)
|
||||||
|
)
|
||||||
conn.llm_finish_task = True
|
conn.tts.tts_text_queue.put(
|
||||||
|
TTSMessageDTO(
|
||||||
if music_path.endswith(".p3"):
|
sentence_id=conn.sentence_id,
|
||||||
opus_packets, _ = p3.decode_opus_from_file(music_path)
|
sentence_type=SentenceType.MIDDLE,
|
||||||
else:
|
content_type=ContentType.TEXT,
|
||||||
opus_packets, _ = conn.tts.audio_to_opus_data(music_path)
|
content_detail=text,
|
||||||
conn.audio_play_queue.put((opus_packets, None, conn.tts_last_text_index))
|
)
|
||||||
|
)
|
||||||
|
conn.tts.tts_text_queue.put(
|
||||||
|
TTSMessageDTO(
|
||||||
|
sentence_id=conn.sentence_id,
|
||||||
|
sentence_type=SentenceType.MIDDLE,
|
||||||
|
content_type=ContentType.FILE,
|
||||||
|
content_file=music_path,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
conn.tts.tts_text_queue.put(
|
||||||
|
TTSMessageDTO(
|
||||||
|
sentence_id=conn.sentence_id,
|
||||||
|
sentence_type=SentenceType.LAST,
|
||||||
|
content_type=ContentType.ACTION,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
|
||||||
except Exception as e:
|
except Exception as e:
|
||||||
conn.logger.bind(tag=TAG).error(f"播放音乐失败: {str(e)}")
|
conn.logger.bind(tag=TAG).error(f"播放音乐失败: {str(e)}")
|
||||||
|
|||||||
@@ -31,3 +31,4 @@ chardet==5.2.0
|
|||||||
aioconsole==0.8.1
|
aioconsole==0.8.1
|
||||||
markitdown==0.1.1
|
markitdown==0.1.1
|
||||||
mcp-proxy==0.6.0
|
mcp-proxy==0.6.0
|
||||||
|
PyJWT==2.8.0
|
||||||
@@ -14,7 +14,7 @@
|
|||||||
}
|
}
|
||||||
|
|
||||||
.container {
|
.container {
|
||||||
max-width: 800px;
|
max-width: 1000px;
|
||||||
margin: 0 auto;
|
margin: 0 auto;
|
||||||
background-color: white;
|
background-color: white;
|
||||||
border-radius: 10px;
|
border-radius: 10px;
|
||||||
@@ -482,9 +482,10 @@
|
|||||||
</span>
|
</span>
|
||||||
</h2>
|
</h2>
|
||||||
<div class="connection-controls">
|
<div class="connection-controls">
|
||||||
<input type="text" id="otaUrl" value="http://127.0.0.1:8002/xiaozhi/ota/" placeholder="OTA服务器地址" />
|
<input type="text" id="otaUrl" value="http://127.0.0.1:8002/xiaozhi/ota/"
|
||||||
|
placeholder="OTA服务器地址,如:http://127.0.0.1:8002/xiaozhi/ota/" />
|
||||||
<input type="text" id="serverUrl" value="ws://127.0.0.1:8000/xiaozhi/v1/"
|
<input type="text" id="serverUrl" value="ws://127.0.0.1:8000/xiaozhi/v1/"
|
||||||
placeholder="WebSocket服务器地址" />
|
placeholder="WebSocket服务器地址,如:ws://127.0.0.1:8000/xiaozhi/v1/" />
|
||||||
<button id="connectButton">连接</button>
|
<button id="connectButton">连接</button>
|
||||||
<button id="authTestButton">测试认证</button>
|
<button id="authTestButton">测试认证</button>
|
||||||
</div>
|
</div>
|
||||||
@@ -802,7 +803,10 @@
|
|||||||
if (frameData && frameData.length > 0) {
|
if (frameData && frameData.length > 0) {
|
||||||
// 转换为Float32
|
// 转换为Float32
|
||||||
const floatData = convertInt16ToFloat32(frameData);
|
const floatData = convertInt16ToFloat32(frameData);
|
||||||
decodedSamples.push(...floatData);
|
// 使用循环替代展开运算符
|
||||||
|
for (let i = 0; i < floatData.length; i++) {
|
||||||
|
decodedSamples.push(floatData[i]);
|
||||||
|
}
|
||||||
}
|
}
|
||||||
} catch (error) {
|
} catch (error) {
|
||||||
log("Opus解码失败: " + error.message, 'error');
|
log("Opus解码失败: " + error.message, 'error');
|
||||||
@@ -810,8 +814,10 @@
|
|||||||
}
|
}
|
||||||
|
|
||||||
if (decodedSamples.length > 0) {
|
if (decodedSamples.length > 0) {
|
||||||
// 添加到解码队列
|
// 使用循环替代展开运算符
|
||||||
this.queue.push(...decodedSamples);
|
for (let i = 0; i < decodedSamples.length; i++) {
|
||||||
|
this.queue.push(decodedSamples[i]);
|
||||||
|
}
|
||||||
this.totalSamples += decodedSamples.length;
|
this.totalSamples += decodedSamples.length;
|
||||||
|
|
||||||
// 如果累积了至少0.2秒的音频,开始播放
|
// 如果累积了至少0.2秒的音频,开始播放
|
||||||
@@ -867,35 +873,39 @@
|
|||||||
this.source = null;
|
this.source = null;
|
||||||
this.playing = false;
|
this.playing = false;
|
||||||
|
|
||||||
// 如果队列中还有数据或者缓冲区有新数据,继续播放
|
// 使用setTimeout避免递归调用
|
||||||
if (this.queue.length > 0) {
|
setTimeout(() => {
|
||||||
setTimeout(() => this.startPlaying(), 10);
|
// 如果队列中还有数据,继续播放
|
||||||
} else if (audioBufferQueue.length > 0) {
|
if (this.queue.length > 0) {
|
||||||
// 缓冲区有新数据,进行解码
|
this.startPlaying();
|
||||||
const frames = [...audioBufferQueue];
|
} else if (audioBufferQueue.length > 0) {
|
||||||
audioBufferQueue = [];
|
// 缓冲区有新数据,进行解码
|
||||||
this.decodeOpusFrames(frames);
|
const frames = [...audioBufferQueue];
|
||||||
} else if (this.endOfStream) {
|
audioBufferQueue = [];
|
||||||
// 流已结束且没有更多数据
|
this.decodeOpusFrames(frames);
|
||||||
log("音频播放完成", 'info');
|
} else if (this.endOfStream) {
|
||||||
isAudioPlaying = false;
|
// 流已结束且没有更多数据
|
||||||
streamingContext = null;
|
log("音频播放完成", 'info');
|
||||||
} else {
|
isAudioPlaying = false;
|
||||||
// 等待更多数据
|
this.endOfStream = false;
|
||||||
setTimeout(() => {
|
streamingContext = null;
|
||||||
// 如果仍然没有新数据,但有更多的包到达
|
} else {
|
||||||
if (this.queue.length === 0 && audioBufferQueue.length > 0) {
|
// 等待更多数据
|
||||||
const frames = [...audioBufferQueue];
|
setTimeout(() => {
|
||||||
audioBufferQueue = [];
|
// 如果仍然没有新数据,但有更多的包到达
|
||||||
this.decodeOpusFrames(frames);
|
if (this.queue.length === 0 && audioBufferQueue.length > 0) {
|
||||||
} else if (this.queue.length === 0 && audioBufferQueue.length === 0) {
|
const frames = [...audioBufferQueue];
|
||||||
// 真的没有更多数据了
|
audioBufferQueue = [];
|
||||||
log("音频播放完成 (超时)", 'info');
|
this.decodeOpusFrames(frames);
|
||||||
isAudioPlaying = false;
|
} else if (this.queue.length === 0 && audioBufferQueue.length === 0) {
|
||||||
streamingContext = null;
|
// 真的没有更多数据了
|
||||||
}
|
log("音频播放完成 (超时)", 'info');
|
||||||
}, 500); // 500ms超时
|
isAudioPlaying = false;
|
||||||
}
|
streamingContext = null;
|
||||||
|
}
|
||||||
|
}, 500); // 500ms超时
|
||||||
|
}
|
||||||
|
}, 10); // 10ms延迟,避免立即递归
|
||||||
};
|
};
|
||||||
|
|
||||||
this.source.start();
|
this.source.start();
|
||||||
@@ -1306,7 +1316,8 @@
|
|||||||
// 先检查OTA状态
|
// 先检查OTA状态
|
||||||
log('正在检查OTA状态...', 'info');
|
log('正在检查OTA状态...', 'info');
|
||||||
const otaUrl = document.getElementById('otaUrl').value.trim();
|
const otaUrl = document.getElementById('otaUrl').value.trim();
|
||||||
|
localStorage.setItem('otaUrl', otaUrl);
|
||||||
|
localStorage.setItem('wsUrl', url);
|
||||||
try {
|
try {
|
||||||
const otaResponse = await fetch(otaUrl, {
|
const otaResponse = await fetch(otaUrl, {
|
||||||
method: 'POST',
|
method: 'POST',
|
||||||
@@ -1473,6 +1484,23 @@
|
|||||||
if (message.text && message.text !== '😊') {
|
if (message.text && message.text !== '😊') {
|
||||||
addMessage(message.text);
|
addMessage(message.text);
|
||||||
}
|
}
|
||||||
|
}else if (message.type === 'mcp') {
|
||||||
|
const payload = message.payload || {};
|
||||||
|
log(`服务器下发: ${JSON.stringify(message)}`, 'info');
|
||||||
|
if (payload) {
|
||||||
|
// 模拟小智客户端行为
|
||||||
|
if(payload.method === 'tools/list'){
|
||||||
|
const replay_message = JSON.stringify({"session_id":"","type":"mcp","payload":{"jsonrpc":"2.0","id":2,"result":{"tools":[{"name":"self.get_device_status","description":"Provides the real-time information of the device, including the current status of the audio speaker, screen, battery, network, etc.\nUse this tool for: \n1. Answering questions about current condition (e.g. what is the current volume of the audio speaker?)\n2. As the first step to control the device (e.g. turn up / down the volume of the audio speaker, etc.)","inputSchema":{"type":"object","properties":{}}},{"name":"self.audio_speaker.set_volume","description":"Set the volume of the audio speaker. If the current volume is unknown, you must call `self.get_device_status` tool first and then call this tool.","inputSchema":{"type":"object","properties":{"volume":{"type":"integer","minimum":0,"maximum":100}},"required":["volume"]}},{"name":"self.screen.set_brightness","description":"Set the brightness of the screen.","inputSchema":{"type":"object","properties":{"brightness":{"type":"integer","minimum":0,"maximum":100}},"required":["brightness"]}},{"name":"self.screen.set_theme","description":"Set the theme of the screen. The theme can be 'light' or 'dark'.","inputSchema":{"type":"object","properties":{"theme":{"type":"string"}},"required":["theme"]}}]}}})
|
||||||
|
websocket.send(replay_message);
|
||||||
|
log(`回复MCP消息: ${replay_message}`, 'info');
|
||||||
|
} else if(payload.method === 'tools/call'){
|
||||||
|
// 模拟回复
|
||||||
|
const replay_message = JSON.stringify({"session_id":"9f261599","type":"mcp","payload":{"jsonrpc":"2.0","id": payload.id,"result":{"content":[{"type":"text","text":"true"}],"isError":false}}})
|
||||||
|
websocket.send(replay_message);
|
||||||
|
log(`回复MCP消息: ${replay_message}`, 'info');
|
||||||
|
}
|
||||||
|
}
|
||||||
|
|
||||||
} else {
|
} else {
|
||||||
// 未知消息类型
|
// 未知消息类型
|
||||||
log(`未知消息类型: ${message.type}`, 'info');
|
log(`未知消息类型: ${message.type}`, 'info');
|
||||||
@@ -1512,7 +1540,10 @@
|
|||||||
device_id: config.deviceId,
|
device_id: config.deviceId,
|
||||||
device_name: config.deviceName,
|
device_name: config.deviceName,
|
||||||
device_mac: config.deviceMac,
|
device_mac: config.deviceMac,
|
||||||
token: config.token
|
token: config.token,
|
||||||
|
features: {
|
||||||
|
mcp: true
|
||||||
|
}
|
||||||
};
|
};
|
||||||
|
|
||||||
log('发送hello握手消息', 'info');
|
log('发送hello握手消息', 'info');
|
||||||
@@ -1563,6 +1594,10 @@
|
|||||||
const message = messageInput.value.trim();
|
const message = messageInput.value.trim();
|
||||||
if (message === '' || !websocket || websocket.readyState !== WebSocket.OPEN) return;
|
if (message === '' || !websocket || websocket.readyState !== WebSocket.OPEN) return;
|
||||||
|
|
||||||
|
audioBufferQueue = [];
|
||||||
|
isAudioBuffering = false;
|
||||||
|
isAudioPlaying = false;
|
||||||
|
|
||||||
try {
|
try {
|
||||||
// 直接发送listen消息,不需要重复发送hello
|
// 直接发送listen消息,不需要重复发送hello
|
||||||
const listenMessage = {
|
const listenMessage = {
|
||||||
@@ -1633,6 +1668,16 @@
|
|||||||
// 初始更新显示值
|
// 初始更新显示值
|
||||||
updateDisplayValues();
|
updateDisplayValues();
|
||||||
|
|
||||||
|
const savedOtaUrl = localStorage.getItem('otaUrl');
|
||||||
|
if (savedOtaUrl) {
|
||||||
|
document.getElementById('otaUrl').value = savedOtaUrl;
|
||||||
|
}
|
||||||
|
|
||||||
|
const savedWsUrl = localStorage.getItem('wsUrl');
|
||||||
|
if (savedWsUrl) {
|
||||||
|
document.getElementById('serverUrl').value = savedWsUrl;
|
||||||
|
}
|
||||||
|
|
||||||
// 切换面板显示
|
// 切换面板显示
|
||||||
toggleButton.addEventListener('click', () => {
|
toggleButton.addEventListener('click', () => {
|
||||||
const isExpanded = configPanel.classList.contains('expanded');
|
const isExpanded = configPanel.classList.contains('expanded');
|
||||||
|
|||||||
Reference in New Issue
Block a user