mirror of
https://github.com/xinnan-tech/xiaozhi-esp32-server.git
synced 2026-07-26 00:53:54 +08:00
Merge pull request #1461 from xinnan-tech/vllm-qwen
add:增加千问收费视觉模型,速度更稳定一点
This commit is contained in:
@@ -195,11 +195,22 @@ Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/
|
|||||||
|---------|---------|------|
|
|---------|---------|------|
|
||||||
| ASR(语音识别) | FunASR(本地) | ✅DoubaoASR(火山流式语音识别) |
|
| ASR(语音识别) | FunASR(本地) | ✅DoubaoASR(火山流式语音识别) |
|
||||||
| LLM(大模型) | ChatGLMLLM(智谱glm-4-flash) | ✅DoubaoLLM(火山doubao-1-5-pro-32k-250115) |
|
| LLM(大模型) | ChatGLMLLM(智谱glm-4-flash) | ✅DoubaoLLM(火山doubao-1-5-pro-32k-250115) |
|
||||||
| VLLM(视觉大模型) | ChatGLMVLLM(智谱glm-4v-flash) | ✅ChatGLMVLLM(智谱glm-4v-flash) |
|
| VLLM(视觉大模型) | ChatGLMVLLM(智谱glm-4v-flash) | ✅QwenVLVLLM(千问qwen2.5-vl-3b-instructh) |
|
||||||
| TTS(语音合成) | EdgeTTS(微软语音) | ✅HuoshanDoubleStreamTTS(火山双流式语音合成) |
|
| TTS(语音合成) | EdgeTTS(微软语音) | ✅HuoshanDoubleStreamTTS(火山双流式语音合成) |
|
||||||
| Intent(意图识别) | function_call(函数调用) | ✅function_call(函数调用) |
|
| Intent(意图识别) | function_call(函数调用) | ✅function_call(函数调用) |
|
||||||
| Memory(记忆功能) | mem_local_short(本地短期记忆) | ✅mem_local_short(本地短期记忆) |
|
| Memory(记忆功能) | mem_local_short(本地短期记忆) | ✅mem_local_short(本地短期记忆) |
|
||||||
|
|
||||||
|
#### 🔧 测试工具
|
||||||
|
本项目提供以下测试工具,帮助您验证系统和选择合适的模型:
|
||||||
|
|
||||||
|
| 工具名称 | 位置 | 使用方法 | 功能说明 |
|
||||||
|
|---------|------|---------|---------|
|
||||||
|
| test_page | `main/xiaozhi-server/test/test_page.html` | 使用谷歌浏览器直接打开 | 测试音频播放和接收功能,验证Python端音频处理是否正常 |
|
||||||
|
| performance_tester | `main/xiaozhi-server/performance_tester.py` | 执行 `python performance_tester.py` | 测试ASR(语音识别)、LLM(大模型)、TTS(语音合成)三个核心模块的响应速度 |
|
||||||
|
| performance_tester_vllm | `main/xiaozhi-server/performance_tester_vllm.py` | 执行 `python performance_tester_vllm.py` | 测试VLLM(视觉模型)的响应速度 |
|
||||||
|
|
||||||
|
> 💡 提示:测试模型速度时,只会测试配置了密钥的模型。
|
||||||
|
|
||||||
---
|
---
|
||||||
## 功能清单 ✨
|
## 功能清单 ✨
|
||||||
### 已实现 ✅
|
### 已实现 ✅
|
||||||
@@ -251,6 +262,16 @@ Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/
|
|||||||
|
|
||||||
---
|
---
|
||||||
|
|
||||||
|
### VLLM 视觉模型
|
||||||
|
|
||||||
|
| 使用方式 | 支持平台 | 免费平台 |
|
||||||
|
|:---:|:---:|:---:|
|
||||||
|
| openai 接口调用 | 阿里百炼、智谱ChatGLMVLLM | 智谱ChatGLMVLLM |
|
||||||
|
|
||||||
|
实际上,任何支持 openai 接口调用的 VLLM 均可接入使用。
|
||||||
|
|
||||||
|
---
|
||||||
|
|
||||||
### TTS 语音合成
|
### TTS 语音合成
|
||||||
|
|
||||||
| 使用方式 | 支持平台 | 免费平台 |
|
| 使用方式 | 支持平台 | 免费平台 |
|
||||||
|
|||||||
@@ -0,0 +1,14 @@
|
|||||||
|
-- VLLM模型配置
|
||||||
|
delete from `ai_model_config` where id = 'VLLM_QwenVLVLLM';
|
||||||
|
INSERT INTO `ai_model_config` VALUES ('VLLM_QwenVLVLLM', 'VLLM', 'QwenVLVLLM', '千问视觉模型', 0, 1, '{\"type\": \"openai\", \"model_name\": \"qwen2.5-vl-3b-instruct\", \"base_url\": \"https://dashscope.aliyuncs.com/compatible-mode/v1\", \"api_key\": \"你的api_key\"}', NULL, NULL, 2, NULL, NULL, NULL, NULL);
|
||||||
|
|
||||||
|
-- 更新文档
|
||||||
|
UPDATE `ai_model_config` SET
|
||||||
|
`doc_link` = 'https://bailian.console.aliyun.com/?tab=api#/api/?type=model&url=https%3A%2F%2Fhelp.aliyun.com%2Fdocument_detail%2F2845564.html&renderType=iframe',
|
||||||
|
`remark` = '千问视觉模型配置说明:
|
||||||
|
1. 访问 https://bailian.console.aliyun.com/?tab=model#/api-key
|
||||||
|
2. 注册并获取API密钥
|
||||||
|
3. 填入配置文件中' WHERE `id` = 'VLLM_QwenVLVLLM';
|
||||||
|
|
||||||
|
-- 删除参数,这两个参数已挪至python配置文件
|
||||||
|
delete from `sys_params` where id in (113,114);
|
||||||
@@ -183,4 +183,11 @@ databaseChangeLog:
|
|||||||
changes:
|
changes:
|
||||||
- sqlFile:
|
- sqlFile:
|
||||||
encoding: utf8
|
encoding: utf8
|
||||||
path: classpath:db/changelog/202506031639.sql
|
path: classpath:db/changelog/202506031639.sql
|
||||||
|
- changeSet:
|
||||||
|
id: 202506032232
|
||||||
|
author: hrz
|
||||||
|
changes:
|
||||||
|
- sqlFile:
|
||||||
|
encoding: utf8
|
||||||
|
path: classpath:db/changelog/202506032232.sql
|
||||||
@@ -467,6 +467,12 @@ VLLM:
|
|||||||
model_name: glm-4v-flash # 智谱AI的视觉模型
|
model_name: glm-4v-flash # 智谱AI的视觉模型
|
||||||
url: https://open.bigmodel.cn/api/paas/v4/
|
url: https://open.bigmodel.cn/api/paas/v4/
|
||||||
api_key: 你的api_key
|
api_key: 你的api_key
|
||||||
|
QwenVLVLLM:
|
||||||
|
type: openai
|
||||||
|
model_name: qwen2.5-vl-3b-instruct
|
||||||
|
url: https://dashscope.aliyuncs.com/compatible-mode/v1
|
||||||
|
# 可在这里找到你的api key https://bailian.console.aliyun.com/?apiKey=1#/api-key
|
||||||
|
api_key: 你的api_key
|
||||||
TTS:
|
TTS:
|
||||||
# 当前支持的type为edge、doubao,可自行适配
|
# 当前支持的type为edge、doubao,可自行适配
|
||||||
EdgeTTS:
|
EdgeTTS:
|
||||||
|
|||||||
@@ -46,7 +46,9 @@ class VLLMProvider(VLLMProviderBase):
|
|||||||
{"type": "text", "text": question},
|
{"type": "text", "text": question},
|
||||||
{
|
{
|
||||||
"type": "image_url",
|
"type": "image_url",
|
||||||
"image_url": {"url": f"{base64_image}"},
|
"image_url": {
|
||||||
|
"url": f"data:image/jpeg;base64,{base64_image}"
|
||||||
|
},
|
||||||
},
|
},
|
||||||
],
|
],
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -0,0 +1,189 @@
|
|||||||
|
import time
|
||||||
|
import asyncio
|
||||||
|
import logging
|
||||||
|
import statistics
|
||||||
|
import base64
|
||||||
|
from typing import Dict
|
||||||
|
from tabulate import tabulate
|
||||||
|
from config.settings import load_config
|
||||||
|
from core.utils.vllm import create_instance
|
||||||
|
|
||||||
|
# 设置全局日志级别为WARNING,抑制INFO级别日志
|
||||||
|
logging.basicConfig(level=logging.WARNING)
|
||||||
|
|
||||||
|
|
||||||
|
class AsyncVisionPerformanceTester:
|
||||||
|
def __init__(self):
|
||||||
|
self.config = load_config()
|
||||||
|
self.test_images = [
|
||||||
|
"../../docs/images/demo1.png",
|
||||||
|
"../../docs/images/demo2.png",
|
||||||
|
]
|
||||||
|
self.test_questions = [
|
||||||
|
"这张图片里有什么?",
|
||||||
|
"请详细描述这张图片的内容",
|
||||||
|
]
|
||||||
|
|
||||||
|
# 加载测试图片
|
||||||
|
self.results = {"vllm": {}}
|
||||||
|
|
||||||
|
async def _test_vllm(self, vllm_name: str, config: Dict) -> Dict:
|
||||||
|
"""异步测试单个视觉大模型性能"""
|
||||||
|
try:
|
||||||
|
# 检查API密钥配置
|
||||||
|
if "api_key" in config and any(
|
||||||
|
x in config["api_key"] for x in ["你的", "placeholder", "sk-xxx"]
|
||||||
|
):
|
||||||
|
print(f"⏭️ VLLM {vllm_name} 未配置api_key,已跳过")
|
||||||
|
return {"name": vllm_name, "type": "vllm", "errors": 1}
|
||||||
|
|
||||||
|
# 获取实际类型(兼容旧配置)
|
||||||
|
module_type = config.get("type", vllm_name)
|
||||||
|
vllm = create_instance(module_type, config)
|
||||||
|
|
||||||
|
print(f"🖼️ 测试 VLLM: {vllm_name}")
|
||||||
|
|
||||||
|
# 创建所有测试任务
|
||||||
|
test_tasks = []
|
||||||
|
for question in self.test_questions:
|
||||||
|
for image in self.test_images:
|
||||||
|
test_tasks.append(
|
||||||
|
self._test_single_vision(vllm_name, vllm, question, image)
|
||||||
|
)
|
||||||
|
|
||||||
|
# 并发执行所有测试
|
||||||
|
test_results = await asyncio.gather(*test_tasks)
|
||||||
|
|
||||||
|
# 处理结果
|
||||||
|
valid_results = [r for r in test_results if r is not None]
|
||||||
|
if not valid_results:
|
||||||
|
print(f"⚠️ {vllm_name} 无有效数据,可能配置错误")
|
||||||
|
return {"name": vllm_name, "type": "vllm", "errors": 1}
|
||||||
|
|
||||||
|
response_times = [r["response_time"] for r in valid_results]
|
||||||
|
|
||||||
|
# 过滤异常数据
|
||||||
|
mean = statistics.mean(response_times)
|
||||||
|
stdev = statistics.stdev(response_times) if len(response_times) > 1 else 0
|
||||||
|
filtered_times = [t for t in response_times if t <= mean + 3 * stdev]
|
||||||
|
|
||||||
|
if len(filtered_times) < len(test_tasks) * 0.5:
|
||||||
|
print(f"⚠️ {vllm_name} 有效数据不足,可能网络不稳定")
|
||||||
|
return {"name": vllm_name, "type": "vllm", "errors": 1}
|
||||||
|
|
||||||
|
return {
|
||||||
|
"name": vllm_name,
|
||||||
|
"type": "vllm",
|
||||||
|
"avg_response": sum(response_times) / len(response_times),
|
||||||
|
"std_response": (
|
||||||
|
statistics.stdev(response_times) if len(response_times) > 1 else 0
|
||||||
|
),
|
||||||
|
"errors": 0,
|
||||||
|
}
|
||||||
|
|
||||||
|
except Exception as e:
|
||||||
|
print(f"⚠️ VLLM {vllm_name} 测试失败: {str(e)}")
|
||||||
|
return {"name": vllm_name, "type": "vllm", "errors": 1}
|
||||||
|
|
||||||
|
async def _test_single_vision(
|
||||||
|
self, vllm_name: str, vllm, question: str, image: str
|
||||||
|
) -> Dict:
|
||||||
|
"""测试单个视觉问题的性能"""
|
||||||
|
try:
|
||||||
|
print(f"📝 {vllm_name} 开始测试: {question[:20]}...")
|
||||||
|
start_time = time.time()
|
||||||
|
|
||||||
|
# 读取图片并转换为base64
|
||||||
|
with open(image, "rb") as image_file:
|
||||||
|
image_data = image_file.read()
|
||||||
|
image_base64 = base64.b64encode(image_data).decode("utf-8")
|
||||||
|
|
||||||
|
# 直接获取响应
|
||||||
|
response = vllm.response(question, image_base64)
|
||||||
|
response_time = time.time() - start_time
|
||||||
|
print(f"✓ {vllm_name} 完成响应: {response_time:.3f}s")
|
||||||
|
|
||||||
|
return {
|
||||||
|
"name": vllm_name,
|
||||||
|
"type": "vllm",
|
||||||
|
"response_time": response_time,
|
||||||
|
}
|
||||||
|
except Exception as e:
|
||||||
|
print(f"⚠️ {vllm_name} 测试失败: {str(e)}")
|
||||||
|
return None
|
||||||
|
|
||||||
|
def _print_results(self):
|
||||||
|
"""打印测试结果"""
|
||||||
|
vllm_table = []
|
||||||
|
for name, data in self.results["vllm"].items():
|
||||||
|
if data["errors"] == 0:
|
||||||
|
stability = data["std_response"] / data["avg_response"]
|
||||||
|
vllm_table.append(
|
||||||
|
[
|
||||||
|
name,
|
||||||
|
f"{data['avg_response']:.3f}秒",
|
||||||
|
f"{stability:.3f}",
|
||||||
|
]
|
||||||
|
)
|
||||||
|
|
||||||
|
if vllm_table:
|
||||||
|
print("\n视觉大模型性能排行:\n")
|
||||||
|
print(
|
||||||
|
tabulate(
|
||||||
|
vllm_table,
|
||||||
|
headers=["模型名称", "响应耗时", "稳定性"],
|
||||||
|
tablefmt="github",
|
||||||
|
colalign=("left", "right", "right"),
|
||||||
|
disable_numparse=True,
|
||||||
|
)
|
||||||
|
)
|
||||||
|
else:
|
||||||
|
print("\n⚠️ 没有可用的视觉大模型进行测试。")
|
||||||
|
|
||||||
|
async def run(self):
|
||||||
|
"""执行全量异步测试"""
|
||||||
|
print("🔍 开始筛选可用视觉大模型...")
|
||||||
|
|
||||||
|
if not self.test_images:
|
||||||
|
print(f"\n⚠️ {self.image_root} 路径下没有图片文件,无法进行测试")
|
||||||
|
return
|
||||||
|
|
||||||
|
# 创建所有测试任务
|
||||||
|
all_tasks = []
|
||||||
|
|
||||||
|
# VLLM测试任务
|
||||||
|
if self.config.get("VLLM") is not None:
|
||||||
|
for vllm_name, config in self.config.get("VLLM", {}).items():
|
||||||
|
if "api_key" in config and any(
|
||||||
|
x in config["api_key"] for x in ["你的", "placeholder", "sk-xxx"]
|
||||||
|
):
|
||||||
|
print(f"⏭️ VLLM {vllm_name} 未配置api_key,已跳过")
|
||||||
|
continue
|
||||||
|
print(f"🖼️ 添加VLLM测试任务: {vllm_name}")
|
||||||
|
all_tasks.append(self._test_vllm(vllm_name, config))
|
||||||
|
|
||||||
|
print(f"\n✅ 找到 {len(all_tasks)} 个可用视觉大模型")
|
||||||
|
print(f"✅ 使用 {len(self.test_images)} 张测试图片")
|
||||||
|
print(f"✅ 使用 {len(self.test_questions)} 个测试问题")
|
||||||
|
print("\n⏳ 开始并发测试所有模型...\n")
|
||||||
|
|
||||||
|
# 并发执行所有测试任务
|
||||||
|
all_results = await asyncio.gather(*all_tasks, return_exceptions=True)
|
||||||
|
|
||||||
|
# 处理结果
|
||||||
|
for result in all_results:
|
||||||
|
if isinstance(result, dict) and result["errors"] == 0:
|
||||||
|
self.results["vllm"][result["name"]] = result
|
||||||
|
|
||||||
|
# 打印结果
|
||||||
|
print("\n📊 生成测试报告...")
|
||||||
|
self._print_results()
|
||||||
|
|
||||||
|
|
||||||
|
async def main():
|
||||||
|
tester = AsyncVisionPerformanceTester()
|
||||||
|
await tester.run()
|
||||||
|
|
||||||
|
|
||||||
|
if __name__ == "__main__":
|
||||||
|
asyncio.run(main())
|
||||||
Reference in New Issue
Block a user