diff --git a/README.md b/README.md index 43a8acd4..b1561bbc 100644 --- a/README.md +++ b/README.md @@ -195,11 +195,22 @@ Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/ |---------|---------|------| | ASR(语音识别) | FunASR(本地) | ✅DoubaoASR(火山流式语音识别) | | LLM(大模型) | ChatGLMLLM(智谱glm-4-flash) | ✅DoubaoLLM(火山doubao-1-5-pro-32k-250115) | -| VLLM(视觉大模型) | ChatGLMVLLM(智谱glm-4v-flash) | ✅ChatGLMVLLM(智谱glm-4v-flash) | +| VLLM(视觉大模型) | ChatGLMVLLM(智谱glm-4v-flash) | ✅QwenVLVLLM(千问qwen2.5-vl-3b-instructh) | | TTS(语音合成) | EdgeTTS(微软语音) | ✅HuoshanDoubleStreamTTS(火山双流式语音合成) | | Intent(意图识别) | function_call(函数调用) | ✅function_call(函数调用) | | Memory(记忆功能) | mem_local_short(本地短期记忆) | ✅mem_local_short(本地短期记忆) | +#### 🔧 测试工具 +本项目提供以下测试工具,帮助您验证系统和选择合适的模型: + +| 工具名称 | 位置 | 使用方法 | 功能说明 | +|---------|------|---------|---------| +| test_page | `main/xiaozhi-server/test/test_page.html` | 使用谷歌浏览器直接打开 | 测试音频播放和接收功能,验证Python端音频处理是否正常 | +| performance_tester | `main/xiaozhi-server/performance_tester.py` | 执行 `python performance_tester.py` | 测试ASR(语音识别)、LLM(大模型)、TTS(语音合成)三个核心模块的响应速度 | +| performance_tester_vllm | `main/xiaozhi-server/performance_tester_vllm.py` | 执行 `python performance_tester_vllm.py` | 测试VLLM(视觉模型)的响应速度 | + +> 💡 提示:测试模型速度时,只会测试配置了密钥的模型。 + --- ## 功能清单 ✨ ### 已实现 ✅ @@ -251,6 +262,16 @@ Websocket接口地址: wss://2662r3426b.vicp.fun/xiaozhi/v1/ --- +### VLLM 视觉模型 + +| 使用方式 | 支持平台 | 免费平台 | +|:---:|:---:|:---:| +| openai 接口调用 | 阿里百炼、智谱ChatGLMVLLM | 智谱ChatGLMVLLM | + +实际上,任何支持 openai 接口调用的 VLLM 均可接入使用。 + +--- + ### TTS 语音合成 | 使用方式 | 支持平台 | 免费平台 | diff --git a/main/manager-api/src/main/resources/db/changelog/202506032232.sql b/main/manager-api/src/main/resources/db/changelog/202506032232.sql new file mode 100644 index 00000000..61611981 --- /dev/null +++ b/main/manager-api/src/main/resources/db/changelog/202506032232.sql @@ -0,0 +1,14 @@ +-- VLLM模型配置 +delete from `ai_model_config` where id = 'VLLM_QwenVLVLLM'; +INSERT INTO `ai_model_config` VALUES ('VLLM_QwenVLVLLM', 'VLLM', 'QwenVLVLLM', '千问视觉模型', 0, 1, '{\"type\": \"openai\", \"model_name\": \"qwen2.5-vl-3b-instruct\", \"base_url\": \"https://dashscope.aliyuncs.com/compatible-mode/v1\", \"api_key\": \"你的api_key\"}', NULL, NULL, 2, NULL, NULL, NULL, NULL); + +-- 更新文档 +UPDATE `ai_model_config` SET +`doc_link` = 'https://bailian.console.aliyun.com/?tab=api#/api/?type=model&url=https%3A%2F%2Fhelp.aliyun.com%2Fdocument_detail%2F2845564.html&renderType=iframe', +`remark` = '千问视觉模型配置说明: +1. 访问 https://bailian.console.aliyun.com/?tab=model#/api-key +2. 注册并获取API密钥 +3. 填入配置文件中' WHERE `id` = 'VLLM_QwenVLVLLM'; + +-- 删除参数,这两个参数已挪至python配置文件 +delete from `sys_params` where id in (113,114); diff --git a/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml b/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml index 1bd070aa..0583fce9 100755 --- a/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml +++ b/main/manager-api/src/main/resources/db/changelog/db.changelog-master.yaml @@ -183,4 +183,11 @@ databaseChangeLog: changes: - sqlFile: encoding: utf8 - path: classpath:db/changelog/202506031639.sql \ No newline at end of file + path: classpath:db/changelog/202506031639.sql + - changeSet: + id: 202506032232 + author: hrz + changes: + - sqlFile: + encoding: utf8 + path: classpath:db/changelog/202506032232.sql \ No newline at end of file diff --git a/main/xiaozhi-server/config.yaml b/main/xiaozhi-server/config.yaml index 179d7204..1bc151f0 100644 --- a/main/xiaozhi-server/config.yaml +++ b/main/xiaozhi-server/config.yaml @@ -467,6 +467,12 @@ VLLM: model_name: glm-4v-flash # 智谱AI的视觉模型 url: https://open.bigmodel.cn/api/paas/v4/ api_key: 你的api_key + QwenVLVLLM: + type: openai + model_name: qwen2.5-vl-3b-instruct + url: https://dashscope.aliyuncs.com/compatible-mode/v1 + # 可在这里找到你的api key https://bailian.console.aliyun.com/?apiKey=1#/api-key + api_key: 你的api_key TTS: # 当前支持的type为edge、doubao,可自行适配 EdgeTTS: diff --git a/main/xiaozhi-server/core/providers/vllm/openai.py b/main/xiaozhi-server/core/providers/vllm/openai.py index 5eb8124b..6ef89bca 100644 --- a/main/xiaozhi-server/core/providers/vllm/openai.py +++ b/main/xiaozhi-server/core/providers/vllm/openai.py @@ -46,7 +46,9 @@ class VLLMProvider(VLLMProviderBase): {"type": "text", "text": question}, { "type": "image_url", - "image_url": {"url": f"{base64_image}"}, + "image_url": { + "url": f"data:image/jpeg;base64,{base64_image}" + }, }, ], } diff --git a/main/xiaozhi-server/performance_tester_vllm.py b/main/xiaozhi-server/performance_tester_vllm.py new file mode 100644 index 00000000..4dafddcc --- /dev/null +++ b/main/xiaozhi-server/performance_tester_vllm.py @@ -0,0 +1,189 @@ +import time +import asyncio +import logging +import statistics +import base64 +from typing import Dict +from tabulate import tabulate +from config.settings import load_config +from core.utils.vllm import create_instance + +# 设置全局日志级别为WARNING,抑制INFO级别日志 +logging.basicConfig(level=logging.WARNING) + + +class AsyncVisionPerformanceTester: + def __init__(self): + self.config = load_config() + self.test_images = [ + "../../docs/images/demo1.png", + "../../docs/images/demo2.png", + ] + self.test_questions = [ + "这张图片里有什么?", + "请详细描述这张图片的内容", + ] + + # 加载测试图片 + self.results = {"vllm": {}} + + async def _test_vllm(self, vllm_name: str, config: Dict) -> Dict: + """异步测试单个视觉大模型性能""" + try: + # 检查API密钥配置 + if "api_key" in config and any( + x in config["api_key"] for x in ["你的", "placeholder", "sk-xxx"] + ): + print(f"⏭️ VLLM {vllm_name} 未配置api_key,已跳过") + return {"name": vllm_name, "type": "vllm", "errors": 1} + + # 获取实际类型(兼容旧配置) + module_type = config.get("type", vllm_name) + vllm = create_instance(module_type, config) + + print(f"🖼️ 测试 VLLM: {vllm_name}") + + # 创建所有测试任务 + test_tasks = [] + for question in self.test_questions: + for image in self.test_images: + test_tasks.append( + self._test_single_vision(vllm_name, vllm, question, image) + ) + + # 并发执行所有测试 + test_results = await asyncio.gather(*test_tasks) + + # 处理结果 + valid_results = [r for r in test_results if r is not None] + if not valid_results: + print(f"⚠️ {vllm_name} 无有效数据,可能配置错误") + return {"name": vllm_name, "type": "vllm", "errors": 1} + + response_times = [r["response_time"] for r in valid_results] + + # 过滤异常数据 + mean = statistics.mean(response_times) + stdev = statistics.stdev(response_times) if len(response_times) > 1 else 0 + filtered_times = [t for t in response_times if t <= mean + 3 * stdev] + + if len(filtered_times) < len(test_tasks) * 0.5: + print(f"⚠️ {vllm_name} 有效数据不足,可能网络不稳定") + return {"name": vllm_name, "type": "vllm", "errors": 1} + + return { + "name": vllm_name, + "type": "vllm", + "avg_response": sum(response_times) / len(response_times), + "std_response": ( + statistics.stdev(response_times) if len(response_times) > 1 else 0 + ), + "errors": 0, + } + + except Exception as e: + print(f"⚠️ VLLM {vllm_name} 测试失败: {str(e)}") + return {"name": vllm_name, "type": "vllm", "errors": 1} + + async def _test_single_vision( + self, vllm_name: str, vllm, question: str, image: str + ) -> Dict: + """测试单个视觉问题的性能""" + try: + print(f"📝 {vllm_name} 开始测试: {question[:20]}...") + start_time = time.time() + + # 读取图片并转换为base64 + with open(image, "rb") as image_file: + image_data = image_file.read() + image_base64 = base64.b64encode(image_data).decode("utf-8") + + # 直接获取响应 + response = vllm.response(question, image_base64) + response_time = time.time() - start_time + print(f"✓ {vllm_name} 完成响应: {response_time:.3f}s") + + return { + "name": vllm_name, + "type": "vllm", + "response_time": response_time, + } + except Exception as e: + print(f"⚠️ {vllm_name} 测试失败: {str(e)}") + return None + + def _print_results(self): + """打印测试结果""" + vllm_table = [] + for name, data in self.results["vllm"].items(): + if data["errors"] == 0: + stability = data["std_response"] / data["avg_response"] + vllm_table.append( + [ + name, + f"{data['avg_response']:.3f}秒", + f"{stability:.3f}", + ] + ) + + if vllm_table: + print("\n视觉大模型性能排行:\n") + print( + tabulate( + vllm_table, + headers=["模型名称", "响应耗时", "稳定性"], + tablefmt="github", + colalign=("left", "right", "right"), + disable_numparse=True, + ) + ) + else: + print("\n⚠️ 没有可用的视觉大模型进行测试。") + + async def run(self): + """执行全量异步测试""" + print("🔍 开始筛选可用视觉大模型...") + + if not self.test_images: + print(f"\n⚠️ {self.image_root} 路径下没有图片文件,无法进行测试") + return + + # 创建所有测试任务 + all_tasks = [] + + # VLLM测试任务 + if self.config.get("VLLM") is not None: + for vllm_name, config in self.config.get("VLLM", {}).items(): + if "api_key" in config and any( + x in config["api_key"] for x in ["你的", "placeholder", "sk-xxx"] + ): + print(f"⏭️ VLLM {vllm_name} 未配置api_key,已跳过") + continue + print(f"🖼️ 添加VLLM测试任务: {vllm_name}") + all_tasks.append(self._test_vllm(vllm_name, config)) + + print(f"\n✅ 找到 {len(all_tasks)} 个可用视觉大模型") + print(f"✅ 使用 {len(self.test_images)} 张测试图片") + print(f"✅ 使用 {len(self.test_questions)} 个测试问题") + print("\n⏳ 开始并发测试所有模型...\n") + + # 并发执行所有测试任务 + all_results = await asyncio.gather(*all_tasks, return_exceptions=True) + + # 处理结果 + for result in all_results: + if isinstance(result, dict) and result["errors"] == 0: + self.results["vllm"][result["name"]] = result + + # 打印结果 + print("\n📊 生成测试报告...") + self._print_results() + + +async def main(): + tester = AsyncVisionPerformanceTester() + await tester.run() + + +if __name__ == "__main__": + asyncio.run(main())